/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.81 by wakaba, Tue Mar 4 14:51:54 2008 UTC revision 1.192 by wakaba, Thu Oct 2 10:59:04 2008 UTC
# Line 3  use strict; Line 3  use strict;
3  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};
4  use Error qw(:try);  use Error qw(:try);
5    
6    ## NOTE: This module don't check all HTML5 parse errors; character
7    ## encoding related parse errors are expected to be handled by relevant
8    ## modules.
9    ## Parse errors for control characters that are not allowed in HTML5
10    ## documents, for surrogate code points, and for noncharacter code
11    ## points, as well as U+FFFD substitions for characters whose code points
12    ## is higher than U+10FFFF may be detected by combining the parser with
13    ## the checker implemented by Whatpm::Charset::UnicodeChecker (for its
14    ## usage example, see |t/HTML-tree.t| in the Whatpm package or the
15    ## WebHACC::Language::HTML module in the WebHACC package).
16    
17  ## ISSUE:  ## ISSUE:
18  ## var doc = implementation.createDocument (null, null, null);  ## var doc = implementation.createDocument (null, null, null);
19  ## doc.write ('');  ## doc.write ('');
20  ## alert (doc.compatMode);  ## alert (doc.compatMode);
21    
22  ## TODO: Control charcters and noncharacters are not allowed (HTML5 revision 1263)  require IO::Handle;
23  ## TODO: 1252 parse error (revision 1264)  
24  ## TODO: 8859-11 = 874 (revision 1271)  my $HTML_NS = q<http://www.w3.org/1999/xhtml>;
25    my $MML_NS = q<http://www.w3.org/1998/Math/MathML>;
26  my $permitted_slash_tag_name = {  my $SVG_NS = q<http://www.w3.org/2000/svg>;
27    base => 1,  my $XLINK_NS = q<http://www.w3.org/1999/xlink>;
28    link => 1,  my $XML_NS = q<http://www.w3.org/XML/1998/namespace>;
29    meta => 1,  my $XMLNS_NS = q<http://www.w3.org/2000/xmlns/>;
30    hr => 1,  
31    br => 1,  sub A_EL () { 0b1 }
32    img => 1,  sub ADDRESS_EL () { 0b10 }
33    embed => 1,  sub BODY_EL () { 0b100 }
34    param => 1,  sub BUTTON_EL () { 0b1000 }
35    area => 1,  sub CAPTION_EL () { 0b10000 }
36    col => 1,  sub DD_EL () { 0b100000 }
37    input => 1,  sub DIV_EL () { 0b1000000 }
38    sub DT_EL () { 0b10000000 }
39    sub FORM_EL () { 0b100000000 }
40    sub FORMATTING_EL () { 0b1000000000 }
41    sub FRAMESET_EL () { 0b10000000000 }
42    sub HEADING_EL () { 0b100000000000 }
43    sub HTML_EL () { 0b1000000000000 }
44    sub LI_EL () { 0b10000000000000 }
45    sub NOBR_EL () { 0b100000000000000 }
46    sub OPTION_EL () { 0b1000000000000000 }
47    sub OPTGROUP_EL () { 0b10000000000000000 }
48    sub P_EL () { 0b100000000000000000 }
49    sub SELECT_EL () { 0b1000000000000000000 }
50    sub TABLE_EL () { 0b10000000000000000000 }
51    sub TABLE_CELL_EL () { 0b100000000000000000000 }
52    sub TABLE_ROW_EL () { 0b1000000000000000000000 }
53    sub TABLE_ROW_GROUP_EL () { 0b10000000000000000000000 }
54    sub MISC_SCOPING_EL () { 0b100000000000000000000000 }
55    sub MISC_SPECIAL_EL () { 0b1000000000000000000000000 }
56    sub FOREIGN_EL () { 0b10000000000000000000000000 }
57    sub FOREIGN_FLOW_CONTENT_EL () { 0b100000000000000000000000000 }
58    sub MML_AXML_EL () { 0b1000000000000000000000000000 }
59    sub RUBY_EL () { 0b10000000000000000000000000000 }
60    sub RUBY_COMPONENT_EL () { 0b100000000000000000000000000000 }
61    
62    sub TABLE_ROWS_EL () {
63      TABLE_EL |
64      TABLE_ROW_EL |
65      TABLE_ROW_GROUP_EL
66    }
67    
68    ## NOTE: Used in "generate implied end tags" algorithm.
69    ## NOTE: There is a code where a modified version of END_TAG_OPTIONAL_EL
70    ## is used in "generate implied end tags" implementation (search for the
71    ## function mae).
72    sub END_TAG_OPTIONAL_EL () {
73      DD_EL |
74      DT_EL |
75      LI_EL |
76      P_EL |
77      RUBY_COMPONENT_EL
78    }
79    
80    ## NOTE: Used in </body> and EOF algorithms.
81    sub ALL_END_TAG_OPTIONAL_EL () {
82      DD_EL |
83      DT_EL |
84      LI_EL |
85      P_EL |
86    
87      BODY_EL |
88      HTML_EL |
89      TABLE_CELL_EL |
90      TABLE_ROW_EL |
91      TABLE_ROW_GROUP_EL
92    }
93    
94    sub SCOPING_EL () {
95      BUTTON_EL |
96      CAPTION_EL |
97      HTML_EL |
98      TABLE_EL |
99      TABLE_CELL_EL |
100      MISC_SCOPING_EL
101    }
102    
103    sub TABLE_SCOPING_EL () {
104      HTML_EL |
105      TABLE_EL
106    }
107    
108    sub TABLE_ROWS_SCOPING_EL () {
109      HTML_EL |
110      TABLE_ROW_GROUP_EL
111    }
112    
113    sub TABLE_ROW_SCOPING_EL () {
114      HTML_EL |
115      TABLE_ROW_EL
116    }
117    
118    sub SPECIAL_EL () {
119      ADDRESS_EL |
120      BODY_EL |
121      DIV_EL |
122    
123      DD_EL |
124      DT_EL |
125      LI_EL |
126      P_EL |
127    
128      FORM_EL |
129      FRAMESET_EL |
130      HEADING_EL |
131      OPTION_EL |
132      OPTGROUP_EL |
133      SELECT_EL |
134      TABLE_ROW_EL |
135      TABLE_ROW_GROUP_EL |
136      MISC_SPECIAL_EL
137    }
138    
139    my $el_category = {
140      a => A_EL | FORMATTING_EL,
141      address => ADDRESS_EL,
142      applet => MISC_SCOPING_EL,
143      area => MISC_SPECIAL_EL,
144      b => FORMATTING_EL,
145      base => MISC_SPECIAL_EL,
146      basefont => MISC_SPECIAL_EL,
147      bgsound => MISC_SPECIAL_EL,
148      big => FORMATTING_EL,
149      blockquote => MISC_SPECIAL_EL,
150      body => BODY_EL,
151      br => MISC_SPECIAL_EL,
152      button => BUTTON_EL,
153      caption => CAPTION_EL,
154      center => MISC_SPECIAL_EL,
155      col => MISC_SPECIAL_EL,
156      colgroup => MISC_SPECIAL_EL,
157      dd => DD_EL,
158      dir => MISC_SPECIAL_EL,
159      div => DIV_EL,
160      dl => MISC_SPECIAL_EL,
161      dt => DT_EL,
162      em => FORMATTING_EL,
163      embed => MISC_SPECIAL_EL,
164      fieldset => MISC_SPECIAL_EL,
165      font => FORMATTING_EL,
166      form => FORM_EL,
167      frame => MISC_SPECIAL_EL,
168      frameset => FRAMESET_EL,
169      h1 => HEADING_EL,
170      h2 => HEADING_EL,
171      h3 => HEADING_EL,
172      h4 => HEADING_EL,
173      h5 => HEADING_EL,
174      h6 => HEADING_EL,
175      head => MISC_SPECIAL_EL,
176      hr => MISC_SPECIAL_EL,
177      html => HTML_EL,
178      i => FORMATTING_EL,
179      iframe => MISC_SPECIAL_EL,
180      img => MISC_SPECIAL_EL,
181      input => MISC_SPECIAL_EL,
182      isindex => MISC_SPECIAL_EL,
183      li => LI_EL,
184      link => MISC_SPECIAL_EL,
185      listing => MISC_SPECIAL_EL,
186      marquee => MISC_SCOPING_EL,
187      menu => MISC_SPECIAL_EL,
188      meta => MISC_SPECIAL_EL,
189      nobr => NOBR_EL | FORMATTING_EL,
190      noembed => MISC_SPECIAL_EL,
191      noframes => MISC_SPECIAL_EL,
192      noscript => MISC_SPECIAL_EL,
193      object => MISC_SCOPING_EL,
194      ol => MISC_SPECIAL_EL,
195      optgroup => OPTGROUP_EL,
196      option => OPTION_EL,
197      p => P_EL,
198      param => MISC_SPECIAL_EL,
199      plaintext => MISC_SPECIAL_EL,
200      pre => MISC_SPECIAL_EL,
201      rp => RUBY_COMPONENT_EL,
202      rt => RUBY_COMPONENT_EL,
203      ruby => RUBY_EL,
204      s => FORMATTING_EL,
205      script => MISC_SPECIAL_EL,
206      select => SELECT_EL,
207      small => FORMATTING_EL,
208      spacer => MISC_SPECIAL_EL,
209      strike => FORMATTING_EL,
210      strong => FORMATTING_EL,
211      style => MISC_SPECIAL_EL,
212      table => TABLE_EL,
213      tbody => TABLE_ROW_GROUP_EL,
214      td => TABLE_CELL_EL,
215      textarea => MISC_SPECIAL_EL,
216      tfoot => TABLE_ROW_GROUP_EL,
217      th => TABLE_CELL_EL,
218      thead => TABLE_ROW_GROUP_EL,
219      title => MISC_SPECIAL_EL,
220      tr => TABLE_ROW_EL,
221      tt => FORMATTING_EL,
222      u => FORMATTING_EL,
223      ul => MISC_SPECIAL_EL,
224      wbr => MISC_SPECIAL_EL,
225    };
226    
227    my $el_category_f = {
228      $MML_NS => {
229        'annotation-xml' => MML_AXML_EL,
230        mi => FOREIGN_FLOW_CONTENT_EL,
231        mo => FOREIGN_FLOW_CONTENT_EL,
232        mn => FOREIGN_FLOW_CONTENT_EL,
233        ms => FOREIGN_FLOW_CONTENT_EL,
234        mtext => FOREIGN_FLOW_CONTENT_EL,
235      },
236      $SVG_NS => {
237        foreignObject => FOREIGN_FLOW_CONTENT_EL,
238        desc => FOREIGN_FLOW_CONTENT_EL,
239        title => FOREIGN_FLOW_CONTENT_EL,
240      },
241      ## NOTE: In addition, FOREIGN_EL is set to non-HTML elements.
242    };
243    
244    my $svg_attr_name = {
245      attributename => 'attributeName',
246      attributetype => 'attributeType',
247      basefrequency => 'baseFrequency',
248      baseprofile => 'baseProfile',
249      calcmode => 'calcMode',
250      clippathunits => 'clipPathUnits',
251      contentscripttype => 'contentScriptType',
252      contentstyletype => 'contentStyleType',
253      diffuseconstant => 'diffuseConstant',
254      edgemode => 'edgeMode',
255      externalresourcesrequired => 'externalResourcesRequired',
256      filterres => 'filterRes',
257      filterunits => 'filterUnits',
258      glyphref => 'glyphRef',
259      gradienttransform => 'gradientTransform',
260      gradientunits => 'gradientUnits',
261      kernelmatrix => 'kernelMatrix',
262      kernelunitlength => 'kernelUnitLength',
263      keypoints => 'keyPoints',
264      keysplines => 'keySplines',
265      keytimes => 'keyTimes',
266      lengthadjust => 'lengthAdjust',
267      limitingconeangle => 'limitingConeAngle',
268      markerheight => 'markerHeight',
269      markerunits => 'markerUnits',
270      markerwidth => 'markerWidth',
271      maskcontentunits => 'maskContentUnits',
272      maskunits => 'maskUnits',
273      numoctaves => 'numOctaves',
274      pathlength => 'pathLength',
275      patterncontentunits => 'patternContentUnits',
276      patterntransform => 'patternTransform',
277      patternunits => 'patternUnits',
278      pointsatx => 'pointsAtX',
279      pointsaty => 'pointsAtY',
280      pointsatz => 'pointsAtZ',
281      preservealpha => 'preserveAlpha',
282      preserveaspectratio => 'preserveAspectRatio',
283      primitiveunits => 'primitiveUnits',
284      refx => 'refX',
285      refy => 'refY',
286      repeatcount => 'repeatCount',
287      repeatdur => 'repeatDur',
288      requiredextensions => 'requiredExtensions',
289      requiredfeatures => 'requiredFeatures',
290      specularconstant => 'specularConstant',
291      specularexponent => 'specularExponent',
292      spreadmethod => 'spreadMethod',
293      startoffset => 'startOffset',
294      stddeviation => 'stdDeviation',
295      stitchtiles => 'stitchTiles',
296      surfacescale => 'surfaceScale',
297      systemlanguage => 'systemLanguage',
298      tablevalues => 'tableValues',
299      targetx => 'targetX',
300      targety => 'targetY',
301      textlength => 'textLength',
302      viewbox => 'viewBox',
303      viewtarget => 'viewTarget',
304      xchannelselector => 'xChannelSelector',
305      ychannelselector => 'yChannelSelector',
306      zoomandpan => 'zoomAndPan',
307  };  };
308    
309  my $c1_entity_char = {  my $foreign_attr_xname = {
310      'xlink:actuate' => [$XLINK_NS, ['xlink', 'actuate']],
311      'xlink:arcrole' => [$XLINK_NS, ['xlink', 'arcrole']],
312      'xlink:href' => [$XLINK_NS, ['xlink', 'href']],
313      'xlink:role' => [$XLINK_NS, ['xlink', 'role']],
314      'xlink:show' => [$XLINK_NS, ['xlink', 'show']],
315      'xlink:title' => [$XLINK_NS, ['xlink', 'title']],
316      'xlink:type' => [$XLINK_NS, ['xlink', 'type']],
317      'xml:base' => [$XML_NS, ['xml', 'base']],
318      'xml:lang' => [$XML_NS, ['xml', 'lang']],
319      'xml:space' => [$XML_NS, ['xml', 'space']],
320      'xmlns' => [$XMLNS_NS, [undef, 'xmlns']],
321      'xmlns:xlink' => [$XMLNS_NS, ['xmlns', 'xlink']],
322    };
323    
324    ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.
325    
326    my $charref_map = {
327      0x0D => 0x000A,
328    0x80 => 0x20AC,    0x80 => 0x20AC,
329    0x81 => 0xFFFD,    0x81 => 0xFFFD,
330    0x82 => 0x201A,    0x82 => 0x201A,
# Line 59  my $c1_entity_char = { Line 357  my $c1_entity_char = {
357    0x9D => 0xFFFD,    0x9D => 0xFFFD,
358    0x9E => 0x017E,    0x9E => 0x017E,
359    0x9F => 0x0178,    0x9F => 0x0178,
360  }; # $c1_entity_char  }; # $charref_map
361    $charref_map->{$_} = 0xFFFD
362        for 0x0000..0x0008, 0x000B, 0x000E..0x001F, 0x007F,
363            0xD800..0xDFFF, 0xFDD0..0xFDDF, ## ISSUE: 0xFDEF
364            0xFFFE, 0xFFFF, 0x1FFFE, 0x1FFFF, 0x2FFFE, 0x2FFFF, 0x3FFFE, 0x3FFFF,
365            0x4FFFE, 0x4FFFF, 0x5FFFE, 0x5FFFF, 0x6FFFE, 0x6FFFF, 0x7FFFE,
366            0x7FFFF, 0x8FFFE, 0x8FFFF, 0x9FFFE, 0x9FFFF, 0xAFFFE, 0xAFFFF,
367            0xBFFFE, 0xBFFFF, 0xCFFFE, 0xCFFFF, 0xDFFFE, 0xDFFFF, 0xEFFFE,
368            0xEFFFF, 0xFFFFE, 0xFFFFF, 0x10FFFE, 0x10FFFF;
369    
370  my $special_category = {  ## TODO: Invoke the reset algorithm when a resettable element is
371    address => 1, area => 1, base => 1, basefont => 1, bgsound => 1,  ## created (cf. HTML5 revision 2259).
   blockquote => 1, body => 1, br => 1, center => 1, col => 1, colgroup => 1,  
   dd => 1, dir => 1, div => 1, dl => 1, dt => 1, embed => 1, fieldset => 1,  
   form => 1, frame => 1, frameset => 1, h1 => 1, h2 => 1, h3 => 1,  
   h4 => 1, h5 => 1, h6 => 1, head => 1, hr => 1, iframe => 1, image => 1,  
   img => 1, input => 1, isindex => 1, li => 1, link => 1, listing => 1,  
   menu => 1, meta => 1, noembed => 1, noframes => 1, noscript => 1,  
   ol => 1, optgroup => 1, option => 1, p => 1, param => 1, plaintext => 1,  
   pre => 1, script => 1, select => 1, spacer => 1, style => 1, tbody => 1,  
   textarea => 1, tfoot => 1, thead => 1, title => 1, tr => 1, ul => 1, wbr => 1,  
 };  
 my $scoping_category = {  
   button => 1, caption => 1, html => 1, marquee => 1, object => 1,  
   table => 1, td => 1, th => 1,  
 };  
 my $formatting_category = {  
   a => 1, b => 1, big => 1, em => 1, font => 1, i => 1, nobr => 1,  
   s => 1, small => 1, strile => 1, strong => 1, tt => 1, u => 1,  
 };  
 # $phrasing_category: all other elements  
372    
373  sub parse_byte_string ($$$$;$) {  sub parse_byte_string ($$$$;$) {
374      my $self = shift;
375      my $charset_name = shift;
376      open my $input, '<', ref $_[0] ? $_[0] : \($_[0]);
377      return $self->parse_byte_stream ($charset_name, $input, @_[1..$#_]);
378    } # parse_byte_string
379    
380    sub parse_byte_stream ($$$$;$$) {
381      # my ($self, $charset_name, $byte_stream, $doc, $onerror, $get_wrapper) = @_;
382    my $self = ref $_[0] ? shift : shift->new;    my $self = ref $_[0] ? shift : shift->new;
383    my $charset = shift;    my $charset_name = shift;
384    my $bytes_s = ref $_[0] ? $_[0] : \($_[0]);    my $byte_stream = $_[0];
   my $s;  
     
   if (defined $charset) {  
     require Encode; ## TODO: decode(utf8) don't delete BOM  
     $s = \ (Encode::decode ($charset, $$bytes_s));  
     $self->{input_encoding} = lc $charset; ## TODO: normalize name  
     $self->{confident} = 1;  
   } else {  
     ## TODO: Implement HTML5 detection algorithm  
     require Whatpm::Charset::UniversalCharDet;  
     $charset = Whatpm::Charset::UniversalCharDet->detect_byte_string  
         (substr ($$bytes_s, 0, 1024));  
     $charset ||= 'windows-1252';  
     $s = \ (Encode::decode ($charset, $$bytes_s));  
     $self->{input_encoding} = $charset;  
     $self->{confident} = 0;  
   }  
385    
386    $self->{change_encoding} = sub {    my $onerror = $_[2] || sub {
387      my $self = shift;      my (%opt) = @_;
388      my $charset = lc shift;      warn "Parse error ($opt{type})\n";
389      ## TODO: if $charset is supported    };
390      ## TODO: normalize charset name    $self->{parse_error} = $onerror; # updated later by parse_char_string
391    
392      ## "Change the encoding" algorithm:    my $get_wrapper = $_[3] || sub ($) {
393        return $_[0]; # $_[0] = byte stream handle, returned = arg to char handle
394      ## Step 1        };
395      if ($charset eq 'utf-16') { ## ISSUE: UTF-16BE -> UTF-8? UTF-16LE -> UTF-8?  
396        $charset = 'utf-8';    ## HTML5 encoding sniffing algorithm
397      require Message::Charset::Info;
398      my $charset;
399      my $buffer;
400      my ($char_stream, $e_status);
401    
402      SNIFFING: {
403        ## NOTE: By setting |allow_fallback| option true when the
404        ## |get_decode_handle| method is invoked, we ignore what the HTML5
405        ## spec requires, i.e. unsupported encoding should be ignored.
406          ## TODO: We should not do this unless the parser is invoked
407          ## in the conformance checking mode, in which this behavior
408          ## would be useful.
409    
410        ## Step 1
411        if (defined $charset_name) {
412          $charset = Message::Charset::Info->get_by_html_name ($charset_name);
413              ## TODO: Is this ok?  Transfer protocol's parameter should be
414              ## interpreted in its semantics?
415    
416          ($char_stream, $e_status) = $charset->get_decode_handle
417              ($byte_stream, allow_error_reporting => 1,
418               allow_fallback => 1);
419          if ($char_stream) {
420            $self->{confident} = 1;
421            last SNIFFING;
422          } else {
423            !!!parse-error (type => 'charset:not supported',
424                            layer => 'encode',
425                            line => 1, column => 1,
426                            value => $charset_name,
427                            level => $self->{level}->{uncertain});
428          }
429      }      }
430    
431      ## Step 2      ## Step 2
432      if (defined $self->{input_encoding} and      my $byte_buffer = '';
433          $self->{input_encoding} eq $charset) {      for (1..1024) {
434          my $char = $byte_stream->getc;
435          last unless defined $char;
436          $byte_buffer .= $char;
437        } ## TODO: timeout
438    
439        ## Step 3
440        if ($byte_buffer =~ /^\xFE\xFF/) {
441          $charset = Message::Charset::Info->get_by_html_name ('utf-16be');
442          ($char_stream, $e_status) = $charset->get_decode_handle
443              ($byte_stream, allow_error_reporting => 1,
444               allow_fallback => 1, byte_buffer => \$byte_buffer);
445        $self->{confident} = 1;        $self->{confident} = 1;
446        return;        last SNIFFING;
447        } elsif ($byte_buffer =~ /^\xFF\xFE/) {
448          $charset = Message::Charset::Info->get_by_html_name ('utf-16le');
449          ($char_stream, $e_status) = $charset->get_decode_handle
450              ($byte_stream, allow_error_reporting => 1,
451               allow_fallback => 1, byte_buffer => \$byte_buffer);
452          $self->{confident} = 1;
453          last SNIFFING;
454        } elsif ($byte_buffer =~ /^\xEF\xBB\xBF/) {
455          $charset = Message::Charset::Info->get_by_html_name ('utf-8');
456          ($char_stream, $e_status) = $charset->get_decode_handle
457              ($byte_stream, allow_error_reporting => 1,
458               allow_fallback => 1, byte_buffer => \$byte_buffer);
459          $self->{confident} = 1;
460          last SNIFFING;
461      }      }
462    
463      !!!parse-error (type => 'charset label detected:'.$self->{input_encoding}.      ## Step 4
464          ':'.$charset, level => 'w');      ## TODO: <meta charset>
465    
466      ## Step 3      ## Step 5
467      # if (can) {      ## TODO: from history
       ## change the encoding on the fly.  
       #$self->{confident} = 1;  
       #return;  
     # }  
468    
469      ## Step 4      ## Step 6
470      throw Whatpm::HTML::RestartParser (charset => $charset);      require Whatpm::Charset::UniversalCharDet;
471        $charset_name = Whatpm::Charset::UniversalCharDet->detect_byte_string
472            ($byte_buffer);
473        if (defined $charset_name) {
474          $charset = Message::Charset::Info->get_by_html_name ($charset_name);
475    
476          ## ISSUE: Unsupported encoding is not ignored according to the spec.
477          require Whatpm::Charset::DecodeHandle;
478          $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new
479              ($byte_stream);
480          ($char_stream, $e_status) = $charset->get_decode_handle
481              ($buffer, allow_error_reporting => 1,
482               allow_fallback => 1, byte_buffer => \$byte_buffer);
483          if ($char_stream) {
484            $buffer->{buffer} = $byte_buffer;
485            !!!parse-error (type => 'sniffing:chardet',
486                            text => $charset_name,
487                            level => $self->{level}->{info},
488                            layer => 'encode',
489                            line => 1, column => 1);
490            $self->{confident} = 0;
491            last SNIFFING;
492          }
493        }
494    
495        ## Step 7: default
496        ## TODO: Make this configurable.
497        $charset = Message::Charset::Info->get_by_html_name ('windows-1252');
498            ## NOTE: We choose |windows-1252| here, since |utf-8| should be
499            ## detectable in the step 6.
500        require Whatpm::Charset::DecodeHandle;
501        $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new
502            ($byte_stream);
503        ($char_stream, $e_status)
504            = $charset->get_decode_handle ($buffer,
505                                           allow_error_reporting => 1,
506                                           allow_fallback => 1,
507                                           byte_buffer => \$byte_buffer);
508        $buffer->{buffer} = $byte_buffer;
509        !!!parse-error (type => 'sniffing:default',
510                        text => 'windows-1252',
511                        level => $self->{level}->{info},
512                        line => 1, column => 1,
513                        layer => 'encode');
514        $self->{confident} = 0;
515      } # SNIFFING
516    
517      if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
518        $self->{input_encoding} = $charset->get_iana_name; ## TODO: Should we set actual charset decoder's encoding name?
519        !!!parse-error (type => 'chardecode:fallback',
520                        #text => $self->{input_encoding},
521                        level => $self->{level}->{uncertain},
522                        line => 1, column => 1,
523                        layer => 'encode');
524      } elsif (not ($e_status &
525                    Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) {
526        $self->{input_encoding} = $charset->get_iana_name;
527        !!!parse-error (type => 'chardecode:no error',
528                        text => $self->{input_encoding},
529                        level => $self->{level}->{uncertain},
530                        line => 1, column => 1,
531                        layer => 'encode');
532      } else {
533        $self->{input_encoding} = $charset->get_iana_name;
534      }
535    
536      $self->{change_encoding} = sub {
537        my $self = shift;
538        $charset_name = shift;
539        my $token = shift;
540    
541        $charset = Message::Charset::Info->get_by_html_name ($charset_name);
542        ($char_stream, $e_status) = $charset->get_decode_handle
543            ($byte_stream, allow_error_reporting => 1, allow_fallback => 1,
544             byte_buffer => \ $buffer->{buffer});
545        
546        if ($char_stream) { # if supported
547          ## "Change the encoding" algorithm:
548    
549          ## Step 1    
550          if ($charset->{category} &
551              Message::Charset::Info::CHARSET_CATEGORY_UTF16 ()) {
552            $charset = Message::Charset::Info->get_by_html_name ('utf-8');
553            ($char_stream, $e_status) = $charset->get_decode_handle
554                ($byte_stream,
555                 byte_buffer => \ $buffer->{buffer});
556          }
557          $charset_name = $charset->get_iana_name;
558          
559          ## Step 2
560          if (defined $self->{input_encoding} and
561              $self->{input_encoding} eq $charset_name) {
562            !!!parse-error (type => 'charset label:matching',
563                            text => $charset_name,
564                            level => $self->{level}->{info});
565            $self->{confident} = 1;
566            return;
567          }
568    
569          !!!parse-error (type => 'charset label detected',
570                          text => $self->{input_encoding},
571                          value => $charset_name,
572                          level => $self->{level}->{warn},
573                          token => $token);
574          
575          ## Step 3
576          # if (can) {
577            ## change the encoding on the fly.
578            #$self->{confident} = 1;
579            #return;
580          # }
581          
582          ## Step 4
583          throw Whatpm::HTML::RestartParser ();
584        }
585    }; # $self->{change_encoding}    }; # $self->{change_encoding}
586    
587    my @args = @_; shift @args; # $s    my $char_onerror = sub {
588        my (undef, $type, %opt) = @_;
589        !!!parse-error (layer => 'encode',
590                        line => $self->{line}, column => $self->{column} + 1,
591                        %opt, type => $type);
592        if ($opt{octets}) {
593          ${$opt{octets}} = "\x{FFFD}"; # relacement character
594        }
595      };
596    
597      my $wrapped_char_stream = $get_wrapper->($char_stream);
598      $wrapped_char_stream->onerror ($char_onerror);
599    
600      my @args = ($_[1], $_[2]); # $doc, $onerror - $get_wrapper = undef;
601    my $return;    my $return;
602    try {    try {
603      $return = $self->parse_char_string ($s, @args);        $return = $self->parse_char_stream ($wrapped_char_stream, @args);  
604    } catch Whatpm::HTML::RestartParser with {    } catch Whatpm::HTML::RestartParser with {
605      my $charset = shift->{charset};      ## NOTE: Invoked after {change_encoding}.
606      $s = \ (Encode::decode ($charset, $$bytes_s));      
607      $self->{input_encoding} = $charset; ## TODO: normalize      if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
608          $self->{input_encoding} = $charset->get_iana_name; ## TODO: Should we set actual charset decoder's encoding name?
609          !!!parse-error (type => 'chardecode:fallback',
610                          level => $self->{level}->{uncertain},
611                          #text => $self->{input_encoding},
612                          line => 1, column => 1,
613                          layer => 'encode');
614        } elsif (not ($e_status &
615                      Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) {
616          $self->{input_encoding} = $charset->get_iana_name;
617          !!!parse-error (type => 'chardecode:no error',
618                          text => $self->{input_encoding},
619                          level => $self->{level}->{uncertain},
620                          line => 1, column => 1,
621                          layer => 'encode');
622        } else {
623          $self->{input_encoding} = $charset->get_iana_name;
624        }
625      $self->{confident} = 1;      $self->{confident} = 1;
626      $return = $self->parse_char_string ($s, @args);  
627        $wrapped_char_stream = $get_wrapper->($char_stream);
628        $wrapped_char_stream->onerror ($char_onerror);
629    
630        $return = $self->parse_char_stream ($wrapped_char_stream, @args);
631    };    };
632    return $return;    return $return;
633  } # parse_byte_string  } # parse_byte_stream
634    
635  ## NOTE: HTML5 spec says that the encoding layer MUST NOT strip BOM  ## NOTE: HTML5 spec says that the encoding layer MUST NOT strip BOM
636  ## and the HTML layer MUST ignore it.  However, we does strip BOM in  ## and the HTML layer MUST ignore it.  However, we does strip BOM in
# Line 162  sub parse_byte_string ($$$$;$) { Line 641  sub parse_byte_string ($$$$;$) {
641  ## such as |parse_byte_string| in this module, must ensure that it does  ## such as |parse_byte_string| in this module, must ensure that it does
642  ## strip the BOM and never strip any ZWNBSP.  ## strip the BOM and never strip any ZWNBSP.
643    
644  *parse_char_string = \&parse_string;  sub parse_char_string ($$$;$$) {
645      #my ($self, $s, $doc, $onerror, $get_wrapper) = @_;
646      my $self = shift;
647      my $s = ref $_[0] ? $_[0] : \($_[0]);
648      require Whatpm::Charset::DecodeHandle;
649      my $input = Whatpm::Charset::DecodeHandle::CharString->new ($s);
650      return $self->parse_char_stream ($input, @_[1..$#_]);
651    } # parse_char_string
652    *parse_string = \&parse_char_string; ## NOTE: Alias for backward compatibility.
653    
654  sub parse_string ($$$;$) {  sub parse_char_stream ($$$;$$) {
655    my $self = ref $_[0] ? shift : shift->new;    my $self = ref $_[0] ? shift : shift->new;
656    my $s = ref $_[0] ? $_[0] : \($_[0]);    my $input = $_[0];
657    $self->{document} = $_[1];    $self->{document} = $_[1];
658    @{$self->{document}->child_nodes} = ();    @{$self->{document}->child_nodes} = ();
659    
# Line 175  sub parse_string ($$$;$) { Line 662  sub parse_string ($$$;$) {
662    $self->{confident} = 1 unless exists $self->{confident};    $self->{confident} = 1 unless exists $self->{confident};
663    $self->{document}->input_encoding ($self->{input_encoding})    $self->{document}->input_encoding ($self->{input_encoding})
664        if defined $self->{input_encoding};        if defined $self->{input_encoding};
665    ## TODO: |{input_encoding}| is needless?
666    
667    my $i = 0;    $self->{line_prev} = $self->{line} = 1;
668    my $line = 1;    $self->{column_prev} = -1;
669    my $column = 0;    $self->{column} = 0;
670    $self->{set_next_char} = sub {    $self->{set_nc} = sub {
671      my $self = shift;      my $self = shift;
672    
673      pop @{$self->{prev_char}};      my $char = '';
674      unshift @{$self->{prev_char}}, $self->{next_char};      if (defined $self->{next_nc}) {
675          $char = $self->{next_nc};
676          delete $self->{next_nc};
677          $self->{nc} = ord $char;
678        } else {
679          $self->{char_buffer} = '';
680          $self->{char_buffer_pos} = 0;
681    
682      $self->{next_char} = -1 and return if $i >= length $$s;        my $count = $input->manakai_read_until
683      $self->{next_char} = ord substr $$s, $i++, 1;           ($self->{char_buffer}, qr/[^\x00\x0A\x0D]/, $self->{char_buffer_pos});
684      $column++;        if ($count) {
685            $self->{line_prev} = $self->{line};
686            $self->{column_prev} = $self->{column};
687            $self->{column}++;
688            $self->{nc}
689                = ord substr ($self->{char_buffer}, $self->{char_buffer_pos}++, 1);
690            return;
691          }
692    
693          if ($input->read ($char, 1)) {
694            $self->{nc} = ord $char;
695          } else {
696            $self->{nc} = -1;
697            return;
698          }
699        }
700    
701        ($self->{line_prev}, $self->{column_prev})
702            = ($self->{line}, $self->{column});
703        $self->{column}++;
704            
705      if ($self->{next_char} == 0x000A) { # LF      if ($self->{nc} == 0x000A) { # LF
706        $line++;        !!!cp ('j1');
707        $column = 0;        $self->{line}++;
708      } elsif ($self->{next_char} == 0x000D) { # CR        $self->{column} = 0;
709        $i++ if substr ($$s, $i, 1) eq "\x0A";      } elsif ($self->{nc} == 0x000D) { # CR
710        $self->{next_char} = 0x000A; # LF # MUST        !!!cp ('j2');
711        $line++;  ## TODO: support for abort/streaming
712        $column = 0;        my $next = '';
713      } elsif ($self->{next_char} > 0x10FFFF) {        if ($input->read ($next, 1) and $next ne "\x0A") {
714        $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST          $self->{next_nc} = $next;
715      } elsif ($self->{next_char} == 0x0000) { # NULL        }
716          $self->{nc} = 0x000A; # LF # MUST
717          $self->{line}++;
718          $self->{column} = 0;
719        } elsif ($self->{nc} == 0x0000) { # NULL
720          !!!cp ('j4');
721        !!!parse-error (type => 'NULL');        !!!parse-error (type => 'NULL');
722        $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST        $self->{nc} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
723      }      }
724    };    };
725    $self->{prev_char} = [-1, -1, -1];  
726    $self->{next_char} = -1;    $self->{read_until} = sub {
727        #my ($scalar, $specials_range, $offset) = @_;
728        return 0 if defined $self->{next_nc};
729    
730        my $pattern = qr/[^$_[1]\x00\x0A\x0D]/;
731        my $offset = $_[2] || 0;
732    
733        if ($self->{char_buffer_pos} < length $self->{char_buffer}) {
734          pos ($self->{char_buffer}) = $self->{char_buffer_pos};
735          if ($self->{char_buffer} =~ /\G(?>$pattern)+/) {
736            substr ($_[0], $offset)
737                = substr ($self->{char_buffer}, $-[0], $+[0] - $-[0]);
738            my $count = $+[0] - $-[0];
739            if ($count) {
740              $self->{column} += $count;
741              $self->{char_buffer_pos} += $count;
742              $self->{line_prev} = $self->{line};
743              $self->{column_prev} = $self->{column} - 1;
744              $self->{nc} = -1;
745            }
746            return $count;
747          } else {
748            return 0;
749          }
750        } else {
751          my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]);
752          if ($count) {
753            $self->{column} += $count;
754            $self->{line_prev} = $self->{line};
755            $self->{column_prev} = $self->{column} - 1;
756            $self->{nc} = -1;
757          }
758          return $count;
759        }
760      }; # $self->{read_until}
761    
762    my $onerror = $_[2] || sub {    my $onerror = $_[2] || sub {
763      my (%opt) = @_;      my (%opt) = @_;
764      warn "Parse error ($opt{type}) at line $opt{line} column $opt{column}\n";      my $line = $opt{token} ? $opt{token}->{line} : $opt{line};
765        my $column = $opt{token} ? $opt{token}->{column} : $opt{column};
766        warn "Parse error ($opt{type}) at line $line column $column\n";
767    };    };
768    $self->{parse_error} = sub {    $self->{parse_error} = sub {
769      $onerror->(@_, line => $line, column => $column);      $onerror->(line => $self->{line}, column => $self->{column}, @_);
770    };    };
771    
772      my $char_onerror = sub {
773        my (undef, $type, %opt) = @_;
774        !!!parse-error (layer => 'encode',
775                        line => $self->{line}, column => $self->{column} + 1,
776                        %opt, type => $type);
777      }; # $char_onerror
778    
779      if ($_[3]) {
780        $input = $_[3]->($input);
781        $input->onerror ($char_onerror);
782      } else {
783        $input->onerror ($char_onerror) unless defined $input->onerror;
784      }
785    
786    $self->_initialize_tokenizer;    $self->_initialize_tokenizer;
787    $self->_initialize_tree_constructor;    $self->_initialize_tree_constructor;
788    $self->_construct_tree;    $self->_construct_tree;
789    $self->_terminate_tree_constructor;    $self->_terminate_tree_constructor;
790    
791      delete $self->{parse_error}; # remove loop
792    
793    return $self->{document};    return $self->{document};
794  } # parse_string  } # parse_char_stream
795    
796  sub new ($) {  sub new ($) {
797    my $class = shift;    my $class = shift;
798    my $self = bless {}, $class;    my $self = bless {
799    $self->{set_next_char} = sub {      level => {must => 'm',
800      $self->{next_char} = -1;                should => 's',
801                  warn => 'w',
802                  info => 'i',
803                  uncertain => 'u'},
804      }, $class;
805      $self->{set_nc} = sub {
806        $self->{nc} = -1;
807    };    };
808    $self->{parse_error} = sub {    $self->{parse_error} = sub {
809      #      #
# Line 254  sub RCDATA_CONTENT_MODEL () { CM_ENTITY Line 830  sub RCDATA_CONTENT_MODEL () { CM_ENTITY
830  sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP }  sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP }
831    
832  sub DATA_STATE () { 0 }  sub DATA_STATE () { 0 }
833  sub ENTITY_DATA_STATE () { 1 }  #sub ENTITY_DATA_STATE () { 1 }
834  sub TAG_OPEN_STATE () { 2 }  sub TAG_OPEN_STATE () { 2 }
835  sub CLOSE_TAG_OPEN_STATE () { 3 }  sub CLOSE_TAG_OPEN_STATE () { 3 }
836  sub TAG_NAME_STATE () { 4 }  sub TAG_NAME_STATE () { 4 }
# Line 265  sub BEFORE_ATTRIBUTE_VALUE_STATE () { 8 Line 841  sub BEFORE_ATTRIBUTE_VALUE_STATE () { 8
841  sub ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE () { 9 }  sub ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE () { 9 }
842  sub ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE () { 10 }  sub ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE () { 10 }
843  sub ATTRIBUTE_VALUE_UNQUOTED_STATE () { 11 }  sub ATTRIBUTE_VALUE_UNQUOTED_STATE () { 11 }
844  sub ENTITY_IN_ATTRIBUTE_VALUE_STATE () { 12 }  #sub ENTITY_IN_ATTRIBUTE_VALUE_STATE () { 12 }
845  sub MARKUP_DECLARATION_OPEN_STATE () { 13 }  sub MARKUP_DECLARATION_OPEN_STATE () { 13 }
846  sub COMMENT_START_STATE () { 14 }  sub COMMENT_START_STATE () { 14 }
847  sub COMMENT_START_DASH_STATE () { 15 }  sub COMMENT_START_DASH_STATE () { 15 }
# Line 287  sub DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUO Line 863  sub DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUO
863  sub AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 31 }  sub AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 31 }
864  sub BOGUS_DOCTYPE_STATE () { 32 }  sub BOGUS_DOCTYPE_STATE () { 32 }
865  sub AFTER_ATTRIBUTE_VALUE_QUOTED_STATE () { 33 }  sub AFTER_ATTRIBUTE_VALUE_QUOTED_STATE () { 33 }
866    sub SELF_CLOSING_START_TAG_STATE () { 34 }
867    sub CDATA_SECTION_STATE () { 35 }
868    sub MD_HYPHEN_STATE () { 36 } # "markup declaration open state" in the spec
869    sub MD_DOCTYPE_STATE () { 37 } # "markup declaration open state" in the spec
870    sub MD_CDATA_STATE () { 38 } # "markup declaration open state" in the spec
871    sub CDATA_RCDATA_CLOSE_TAG_STATE () { 39 } # "close tag open state" in the spec
872    sub CDATA_SECTION_MSE1_STATE () { 40 } # "CDATA section state" in the spec
873    sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec
874    sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec
875    sub SYSTEM_STATE () { 43 } # "after DOCTYPE name state" in the spec
876    ## NOTE: "Entity data state", "entity in attribute value state", and
877    ## "consume a character reference" algorithm are jointly implemented
878    ## using the following six states:
879    sub ENTITY_STATE () { 44 }
880    sub ENTITY_HASH_STATE () { 45 }
881    sub NCR_NUM_STATE () { 46 }
882    sub HEXREF_X_STATE () { 47 }
883    sub HEXREF_HEX_STATE () { 48 }
884    sub ENTITY_NAME_STATE () { 49 }
885    sub PCDATA_STATE () { 50 } # "data state" in the spec
886    
887  sub DOCTYPE_TOKEN () { 1 }  sub DOCTYPE_TOKEN () { 1 }
888  sub COMMENT_TOKEN () { 2 }  sub COMMENT_TOKEN () { 2 }
# Line 303  sub TABLE_IMS ()      { 0b1000000 } Line 899  sub TABLE_IMS ()      { 0b1000000 }
899  sub ROW_IMS ()        { 0b10000000 }  sub ROW_IMS ()        { 0b10000000 }
900  sub BODY_AFTER_IMS () { 0b100000000 }  sub BODY_AFTER_IMS () { 0b100000000 }
901  sub FRAME_IMS ()      { 0b1000000000 }  sub FRAME_IMS ()      { 0b1000000000 }
902    sub SELECT_IMS ()     { 0b10000000000 }
903    sub IN_FOREIGN_CONTENT_IM () { 0b100000000000 }
904        ## NOTE: "in foreign content" insertion mode is special; it is combined
905        ## with the secondary insertion mode.  In this parser, they are stored
906        ## together in the bit-or'ed form.
907    
908    ## NOTE: "initial" and "before html" insertion modes have no constants.
909    
910    ## NOTE: "after after body" insertion mode.
911  sub AFTER_HTML_BODY_IM () { AFTER_HTML_IMS | BODY_AFTER_IMS }  sub AFTER_HTML_BODY_IM () { AFTER_HTML_IMS | BODY_AFTER_IMS }
912    
913    ## NOTE: "after after frameset" insertion mode.
914  sub AFTER_HTML_FRAMESET_IM () { AFTER_HTML_IMS | FRAME_IMS }  sub AFTER_HTML_FRAMESET_IM () { AFTER_HTML_IMS | FRAME_IMS }
915    
916  sub IN_HEAD_IM () { HEAD_IMS | 0b00 }  sub IN_HEAD_IM () { HEAD_IMS | 0b00 }
917  sub IN_HEAD_NOSCRIPT_IM () { HEAD_IMS | 0b01 }  sub IN_HEAD_NOSCRIPT_IM () { HEAD_IMS | 0b01 }
918  sub AFTER_HEAD_IM () { HEAD_IMS | 0b10 }  sub AFTER_HEAD_IM () { HEAD_IMS | 0b10 }
# Line 319  sub IN_TABLE_IM () { TABLE_IMS } Line 926  sub IN_TABLE_IM () { TABLE_IMS }
926  sub AFTER_BODY_IM () { BODY_AFTER_IMS }  sub AFTER_BODY_IM () { BODY_AFTER_IMS }
927  sub IN_FRAMESET_IM () { FRAME_IMS | 0b01 }  sub IN_FRAMESET_IM () { FRAME_IMS | 0b01 }
928  sub AFTER_FRAMESET_IM () { FRAME_IMS | 0b10 }  sub AFTER_FRAMESET_IM () { FRAME_IMS | 0b10 }
929  sub IN_SELECT_IM () { 0b01 }  sub IN_SELECT_IM () { SELECT_IMS | 0b01 }
930    sub IN_SELECT_IN_TABLE_IM () { SELECT_IMS | 0b10 }
931  sub IN_COLUMN_GROUP_IM () { 0b10 }  sub IN_COLUMN_GROUP_IM () { 0b10 }
932    
933  ## Implementations MUST act as if state machine in the spec  ## Implementations MUST act as if state machine in the spec
# Line 327  sub IN_COLUMN_GROUP_IM () { 0b10 } Line 935  sub IN_COLUMN_GROUP_IM () { 0b10 }
935  sub _initialize_tokenizer ($) {  sub _initialize_tokenizer ($) {
936    my $self = shift;    my $self = shift;
937    $self->{state} = DATA_STATE; # MUST    $self->{state} = DATA_STATE; # MUST
938      #$self->{s_kwd}; # state keyword - initialized when used
939      #$self->{entity__value}; # initialized when used
940      #$self->{entity__match}; # initialized when used
941    $self->{content_model} = PCDATA_CONTENT_MODEL; # be    $self->{content_model} = PCDATA_CONTENT_MODEL; # be
942    undef $self->{current_token}; # start tag, end tag, comment, or DOCTYPE    undef $self->{ct}; # current token
943    undef $self->{current_attribute};    undef $self->{ca}; # current attribute
944    undef $self->{last_emitted_start_tag_name};    undef $self->{last_stag_name}; # last emitted start tag name
945    undef $self->{last_attribute_value_state};    #$self->{prev_state}; # initialized when used
946    $self->{char} = [];    delete $self->{self_closing};
947    # $self->{next_char}    $self->{char_buffer} = '';
948      $self->{char_buffer_pos} = 0;
949      $self->{nc} = -1; # next input character
950      #$self->{next_nc}
951    !!!next-input-character;    !!!next-input-character;
952    $self->{token} = [];    $self->{token} = [];
953    # $self->{escape}    # $self->{escape}
# Line 344  sub _initialize_tokenizer ($) { Line 958  sub _initialize_tokenizer ($) {
958  ##       CHARACTER_TOKEN, or END_OF_FILE_TOKEN  ##       CHARACTER_TOKEN, or END_OF_FILE_TOKEN
959  ##   ->{name} (DOCTYPE_TOKEN)  ##   ->{name} (DOCTYPE_TOKEN)
960  ##   ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)  ##   ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)
961  ##   ->{public_identifier} (DOCTYPE_TOKEN)  ##   ->{pubid} (DOCTYPE_TOKEN)
962  ##   ->{system_identifier} (DOCTYPE_TOKEN)  ##   ->{sysid} (DOCTYPE_TOKEN)
963  ##   ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag  ##   ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag
964  ##   ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)  ##   ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)
965  ##        ->{name}  ##        ->{name}
966  ##        ->{value}  ##        ->{value}
967  ##        ->{has_reference} == 1 or 0  ##        ->{has_reference} == 1 or 0
968  ##   ->{data} (COMMENT_TOKEN, CHARACTER_TOKEN)  ##   ->{data} (COMMENT_TOKEN, CHARACTER_TOKEN)
969    ## NOTE: The "self-closing flag" is hold as |$self->{self_closing}|.
970    ##     |->{self_closing}| is used to save the value of |$self->{self_closing}|
971    ##     while the token is pushed back to the stack.
972    
973  ## Emitted token MUST immediately be handled by the tree construction state.  ## Emitted token MUST immediately be handled by the tree construction state.
974    
# Line 361  sub _initialize_tokenizer ($) { Line 978  sub _initialize_tokenizer ($) {
978  ## has completed loading.  If one has, then it MUST be executed  ## has completed loading.  If one has, then it MUST be executed
979  ## and removed from the list.  ## and removed from the list.
980    
981  ## NOTE: HTML5 "Writing HTML documents" section, applied to  ## TODO: Polytheistic slash SHOULD NOT be used. (Applied only to atheists.)
982  ## documents and not to user agents and conformance checkers,  ## (This requirement was dropped from HTML5 spec, unfortunately.)
983  ## contains some requirements that are not detected by the  
984  ## parsing algorithm:  my $is_space = {
985  ## - Some requirements on character encoding declarations. ## TODO    0x0009 => 1, # CHARACTER TABULATION (HT)
986  ## - "Elements MUST NOT contain content that their content model disallows."    0x000A => 1, # LINE FEED (LF)
987  ##   ... Some are parse error, some are not (will be reported by c.c.).    #0x000B => 0, # LINE TABULATION (VT)
988  ## - Polytheistic slash SHOULD NOT be used. (Applied only to atheists.) ## TODO    0x000C => 1, # FORM FEED (FF)
989  ## - Text (in elements, attributes, and comments) SHOULD NOT contain    #0x000D => 1, # CARRIAGE RETURN (CR)
990  ##   control characters other than space characters. ## TODO: (what is control character? C0, C1 and DEL?  Unicode control character?)    0x0020 => 1, # SPACE (SP)
991    };
 ## TODO: HTML5 poses authors two SHOULD-level requirements that cannot  
 ## be detected by the HTML5 parsing algorithm:  
 ## - Text,  
992    
993  sub _get_next_token ($) {  sub _get_next_token ($) {
994    my $self = shift;    my $self = shift;
995    
996      if ($self->{self_closing}) {
997        !!!parse-error (type => 'nestc', token => $self->{ct});
998        ## NOTE: The |self_closing| flag is only set by start tag token.
999        ## In addition, when a start tag token is emitted, it is always set to
1000        ## |ct|.
1001        delete $self->{self_closing};
1002      }
1003    
1004    if (@{$self->{token}}) {    if (@{$self->{token}}) {
1005        $self->{self_closing} = $self->{token}->[0]->{self_closing};
1006      return shift @{$self->{token}};      return shift @{$self->{token}};
1007    }    }
1008    
1009    A: {    A: {
1010      if ($self->{state} == DATA_STATE) {      if ($self->{state} == PCDATA_STATE) {
1011        if ($self->{next_char} == 0x0026) { # &        ## NOTE: Same as |DATA_STATE|, but only for |PCDATA| content model.
1012    
1013          if ($self->{nc} == 0x0026) { # &
1014            !!!cp (0.1);
1015            ## NOTE: In the spec, the tokenizer is switched to the
1016            ## "entity data state".  In this implementation, the tokenizer
1017            ## is switched to the |ENTITY_STATE|, which is an implementation
1018            ## of the "consume a character reference" algorithm.
1019            $self->{entity_add} = -1;
1020            $self->{prev_state} = DATA_STATE;
1021            $self->{state} = ENTITY_STATE;
1022            !!!next-input-character;
1023            redo A;
1024          } elsif ($self->{nc} == 0x003C) { # <
1025            !!!cp (0.2);
1026            $self->{state} = TAG_OPEN_STATE;
1027            !!!next-input-character;
1028            redo A;
1029          } elsif ($self->{nc} == -1) {
1030            !!!cp (0.3);
1031            !!!emit ({type => END_OF_FILE_TOKEN,
1032                      line => $self->{line}, column => $self->{column}});
1033            last A; ## TODO: ok?
1034          } else {
1035            !!!cp (0.4);
1036            #
1037          }
1038    
1039          # Anything else
1040          my $token = {type => CHARACTER_TOKEN,
1041                       data => chr $self->{nc},
1042                       line => $self->{line}, column => $self->{column},
1043                      };
1044          $self->{read_until}->($token->{data}, q[<&], length $token->{data});
1045    
1046          ## Stay in the state.
1047          !!!next-input-character;
1048          !!!emit ($token);
1049          redo A;
1050        } elsif ($self->{state} == DATA_STATE) {
1051          $self->{s_kwd} = '' unless defined $self->{s_kwd};
1052          if ($self->{nc} == 0x0026) { # &
1053            $self->{s_kwd} = '';
1054          if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA          if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA
1055              not $self->{escape}) {              not $self->{escape}) {
1056            !!!cp (1);            !!!cp (1);
1057            $self->{state} = ENTITY_DATA_STATE;            ## NOTE: In the spec, the tokenizer is switched to the
1058              ## "entity data state".  In this implementation, the tokenizer
1059              ## is switched to the |ENTITY_STATE|, which is an implementation
1060              ## of the "consume a character reference" algorithm.
1061              $self->{entity_add} = -1;
1062              $self->{prev_state} = DATA_STATE;
1063              $self->{state} = ENTITY_STATE;
1064            !!!next-input-character;            !!!next-input-character;
1065            redo A;            redo A;
1066          } else {          } else {
1067            !!!cp (2);            !!!cp (2);
1068            #            #
1069          }          }
1070        } elsif ($self->{next_char} == 0x002D) { # -        } elsif ($self->{nc} == 0x002D) { # -
1071          if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA          if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1072            unless ($self->{escape}) {            $self->{s_kwd} .= '-';
1073              if ($self->{prev_char}->[0] == 0x002D and # -            
1074                  $self->{prev_char}->[1] == 0x0021 and # !            if ($self->{s_kwd} eq '<!--') {
1075                  $self->{prev_char}->[2] == 0x003C) { # <              !!!cp (3);
1076                !!!cp (3);              $self->{escape} = 1; # unless $self->{escape};
1077                $self->{escape} = 1;              $self->{s_kwd} = '--';
1078              } else {              #
1079                !!!cp (4);            } elsif ($self->{s_kwd} eq '---') {
1080              }              !!!cp (4);
1081                $self->{s_kwd} = '--';
1082                #
1083            } else {            } else {
1084              !!!cp (5);              !!!cp (5);
1085                #
1086            }            }
1087          }          }
1088                    
1089          #          #
1090        } elsif ($self->{next_char} == 0x003C) { # <        } elsif ($self->{nc} == 0x0021) { # !
1091            if (length $self->{s_kwd}) {
1092              !!!cp (5.1);
1093              $self->{s_kwd} .= '!';
1094              #
1095            } else {
1096              !!!cp (5.2);
1097              #$self->{s_kwd} = '';
1098              #
1099            }
1100            #
1101          } elsif ($self->{nc} == 0x003C) { # <
1102          if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA          if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA
1103              (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA              (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA
1104               not $self->{escape})) {               not $self->{escape})) {
# Line 422  sub _get_next_token ($) { Line 1108  sub _get_next_token ($) {
1108            redo A;            redo A;
1109          } else {          } else {
1110            !!!cp (7);            !!!cp (7);
1111              $self->{s_kwd} = '';
1112            #            #
1113          }          }
1114        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1115          if ($self->{escape} and          if ($self->{escape} and
1116              ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA              ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA
1117            if ($self->{prev_char}->[0] == 0x002D and # -            if ($self->{s_kwd} eq '--') {
               $self->{prev_char}->[1] == 0x002D) { # -  
1118              !!!cp (8);              !!!cp (8);
1119              delete $self->{escape};              delete $self->{escape};
1120            } else {            } else {
# Line 438  sub _get_next_token ($) { Line 1124  sub _get_next_token ($) {
1124            !!!cp (10);            !!!cp (10);
1125          }          }
1126                    
1127            $self->{s_kwd} = '';
1128          #          #
1129        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1130          !!!cp (11);          !!!cp (11);
1131          !!!emit ({type => END_OF_FILE_TOKEN});          $self->{s_kwd} = '';
1132            !!!emit ({type => END_OF_FILE_TOKEN,
1133                      line => $self->{line}, column => $self->{column}});
1134          last A; ## TODO: ok?          last A; ## TODO: ok?
1135        } else {        } else {
1136          !!!cp (12);          !!!cp (12);
1137            $self->{s_kwd} = '';
1138            #
1139        }        }
1140    
1141        # Anything else        # Anything else
1142        my $token = {type => CHARACTER_TOKEN,        my $token = {type => CHARACTER_TOKEN,
1143                     data => chr $self->{next_char}};                     data => chr $self->{nc},
1144        ## Stay in the data state                     line => $self->{line}, column => $self->{column},
1145        !!!next-input-character;                    };
1146          if ($self->{read_until}->($token->{data}, q[-!<>&],
1147        !!!emit ($token);                                  length $token->{data})) {
1148            $self->{s_kwd} = '';
1149        redo A;        }
     } elsif ($self->{state} == ENTITY_DATA_STATE) {  
       ## (cannot happen in CDATA state)  
         
       my $token = $self->_tokenize_attempt_to_consume_an_entity (0, -1);  
   
       $self->{state} = DATA_STATE;  
       # next-input-character is already done  
1150    
1151        unless (defined $token) {        ## Stay in the data state.
1152          if ($self->{content_model} == PCDATA_CONTENT_MODEL) {
1153          !!!cp (13);          !!!cp (13);
1154          !!!emit ({type => CHARACTER_TOKEN, data => '&'});          $self->{state} = PCDATA_STATE;
1155        } else {        } else {
1156          !!!cp (14);          !!!cp (14);
1157          !!!emit ($token);          ## Stay in the state.
1158        }        }
1159          !!!next-input-character;
1160          !!!emit ($token);
1161        redo A;        redo A;
1162      } elsif ($self->{state} == TAG_OPEN_STATE) {      } elsif ($self->{state} == TAG_OPEN_STATE) {
1163        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1164          if ($self->{next_char} == 0x002F) { # /          if ($self->{nc} == 0x002F) { # /
1165            !!!cp (15);            !!!cp (15);
1166            !!!next-input-character;            !!!next-input-character;
1167            $self->{state} = CLOSE_TAG_OPEN_STATE;            $self->{state} = CLOSE_TAG_OPEN_STATE;
1168            redo A;            redo A;
1169            } elsif ($self->{nc} == 0x0021) { # !
1170              !!!cp (15.1);
1171              $self->{s_kwd} = '<' unless $self->{escape};
1172              #
1173          } else {          } else {
1174            !!!cp (16);            !!!cp (16);
1175            ## reconsume            #
           $self->{state} = DATA_STATE;  
   
           !!!emit ({type => CHARACTER_TOKEN, data => '<'});  
   
           redo A;  
1176          }          }
1177    
1178            ## reconsume
1179            $self->{state} = DATA_STATE;
1180            !!!emit ({type => CHARACTER_TOKEN, data => '<',
1181                      line => $self->{line_prev},
1182                      column => $self->{column_prev},
1183                     });
1184            redo A;
1185        } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA        } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA
1186          if ($self->{next_char} == 0x0021) { # !          if ($self->{nc} == 0x0021) { # !
1187            !!!cp (17);            !!!cp (17);
1188            $self->{state} = MARKUP_DECLARATION_OPEN_STATE;            $self->{state} = MARKUP_DECLARATION_OPEN_STATE;
1189            !!!next-input-character;            !!!next-input-character;
1190            redo A;            redo A;
1191          } elsif ($self->{next_char} == 0x002F) { # /          } elsif ($self->{nc} == 0x002F) { # /
1192            !!!cp (18);            !!!cp (18);
1193            $self->{state} = CLOSE_TAG_OPEN_STATE;            $self->{state} = CLOSE_TAG_OPEN_STATE;
1194            !!!next-input-character;            !!!next-input-character;
1195            redo A;            redo A;
1196          } elsif (0x0041 <= $self->{next_char} and          } elsif (0x0041 <= $self->{nc} and
1197                   $self->{next_char} <= 0x005A) { # A..Z                   $self->{nc} <= 0x005A) { # A..Z
1198            !!!cp (19);            !!!cp (19);
1199            $self->{current_token}            $self->{ct}
1200              = {type => START_TAG_TOKEN,              = {type => START_TAG_TOKEN,
1201                 tag_name => chr ($self->{next_char} + 0x0020)};                 tag_name => chr ($self->{nc} + 0x0020),
1202                   line => $self->{line_prev},
1203                   column => $self->{column_prev}};
1204            $self->{state} = TAG_NAME_STATE;            $self->{state} = TAG_NAME_STATE;
1205            !!!next-input-character;            !!!next-input-character;
1206            redo A;            redo A;
1207          } elsif (0x0061 <= $self->{next_char} and          } elsif (0x0061 <= $self->{nc} and
1208                   $self->{next_char} <= 0x007A) { # a..z                   $self->{nc} <= 0x007A) { # a..z
1209            !!!cp (20);            !!!cp (20);
1210            $self->{current_token} = {type => START_TAG_TOKEN,            $self->{ct} = {type => START_TAG_TOKEN,
1211                              tag_name => chr ($self->{next_char})};                                      tag_name => chr ($self->{nc}),
1212                                        line => $self->{line_prev},
1213                                        column => $self->{column_prev}};
1214            $self->{state} = TAG_NAME_STATE;            $self->{state} = TAG_NAME_STATE;
1215            !!!next-input-character;            !!!next-input-character;
1216            redo A;            redo A;
1217          } elsif ($self->{next_char} == 0x003E) { # >          } elsif ($self->{nc} == 0x003E) { # >
1218            !!!cp (21);            !!!cp (21);
1219            !!!parse-error (type => 'empty start tag');            !!!parse-error (type => 'empty start tag',
1220                              line => $self->{line_prev},
1221                              column => $self->{column_prev});
1222            $self->{state} = DATA_STATE;            $self->{state} = DATA_STATE;
1223            !!!next-input-character;            !!!next-input-character;
1224    
1225            !!!emit ({type => CHARACTER_TOKEN, data => '<>'});            !!!emit ({type => CHARACTER_TOKEN, data => '<>',
1226                        line => $self->{line_prev},
1227                        column => $self->{column_prev},
1228                       });
1229    
1230            redo A;            redo A;
1231          } elsif ($self->{next_char} == 0x003F) { # ?          } elsif ($self->{nc} == 0x003F) { # ?
1232            !!!cp (22);            !!!cp (22);
1233            !!!parse-error (type => 'pio');            !!!parse-error (type => 'pio',
1234                              line => $self->{line_prev},
1235                              column => $self->{column_prev});
1236            $self->{state} = BOGUS_COMMENT_STATE;            $self->{state} = BOGUS_COMMENT_STATE;
1237            ## $self->{next_char} is intentionally left as is            $self->{ct} = {type => COMMENT_TOKEN, data => '',
1238                                        line => $self->{line_prev},
1239                                        column => $self->{column_prev},
1240                                       };
1241              ## $self->{nc} is intentionally left as is
1242            redo A;            redo A;
1243          } else {          } else {
1244            !!!cp (23);            !!!cp (23);
1245            !!!parse-error (type => 'bare stago');            !!!parse-error (type => 'bare stago',
1246                              line => $self->{line_prev},
1247                              column => $self->{column_prev});
1248            $self->{state} = DATA_STATE;            $self->{state} = DATA_STATE;
1249            ## reconsume            ## reconsume
1250    
1251            !!!emit ({type => CHARACTER_TOKEN, data => '<'});            !!!emit ({type => CHARACTER_TOKEN, data => '<',
1252                        line => $self->{line_prev},
1253                        column => $self->{column_prev},
1254                       });
1255    
1256            redo A;            redo A;
1257          }          }
# Line 545  sub _get_next_token ($) { Line 1259  sub _get_next_token ($) {
1259          die "$0: $self->{content_model} in tag open";          die "$0: $self->{content_model} in tag open";
1260        }        }
1261      } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {      } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {
1262        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA        ## NOTE: The "close tag open state" in the spec is implemented as
1263          if (defined $self->{last_emitted_start_tag_name}) {        ## |CLOSE_TAG_OPEN_STATE| and |CDATA_RCDATA_CLOSE_TAG_STATE|.
           ## NOTE: <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>  
           my @next_char;  
           TAGNAME: for (my $i = 0; $i < length $self->{last_emitted_start_tag_name}; $i++) {  
             push @next_char, $self->{next_char};  
             my $c = ord substr ($self->{last_emitted_start_tag_name}, $i, 1);  
             my $C = 0x0061 <= $c && $c <= 0x007A ? $c - 0x0020 : $c;  
             if ($self->{next_char} == $c or $self->{next_char} == $C) {  
               !!!cp (24);  
               !!!next-input-character;  
               next TAGNAME;  
             } else {  
               !!!cp (25);  
               $self->{next_char} = shift @next_char; # reconsume  
               !!!back-next-input-character (@next_char);  
               $self->{state} = DATA_STATE;  
1264    
1265                !!!emit ({type => CHARACTER_TOKEN, data => '</'});        my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"
1266            if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1267                redo A;          if (defined $self->{last_stag_name}) {
1268              }            $self->{state} = CDATA_RCDATA_CLOSE_TAG_STATE;
1269            }            $self->{s_kwd} = '';
1270            push @next_char, $self->{next_char};            ## Reconsume.
1271                    redo A;
           unless ($self->{next_char} == 0x0009 or # HT  
                   $self->{next_char} == 0x000A or # LF  
                   $self->{next_char} == 0x000B or # VT  
                   $self->{next_char} == 0x000C or # FF  
                   $self->{next_char} == 0x0020 or # SP  
                   $self->{next_char} == 0x003E or # >  
                   $self->{next_char} == 0x002F or # /  
                   $self->{next_char} == -1) {  
             !!!cp (26);  
             $self->{next_char} = shift @next_char; # reconsume  
             !!!back-next-input-character (@next_char);  
             $self->{state} = DATA_STATE;  
             !!!emit ({type => CHARACTER_TOKEN, data => '</'});  
             redo A;  
           } else {  
             !!!cp (27);  
             $self->{next_char} = shift @next_char;  
             !!!back-next-input-character (@next_char);  
             # and consume...  
           }  
1272          } else {          } else {
1273            ## No start tag token has ever been emitted            ## No start tag token has ever been emitted
1274              ## NOTE: See <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>.
1275            !!!cp (28);            !!!cp (28);
           # next-input-character is already done  
1276            $self->{state} = DATA_STATE;            $self->{state} = DATA_STATE;
1277            !!!emit ({type => CHARACTER_TOKEN, data => '</'});            ## Reconsume.
1278              !!!emit ({type => CHARACTER_TOKEN, data => '</',
1279                        line => $l, column => $c,
1280                       });
1281            redo A;            redo A;
1282          }          }
1283        }        }
1284          
1285        if (0x0041 <= $self->{next_char} and        if (0x0041 <= $self->{nc} and
1286            $self->{next_char} <= 0x005A) { # A..Z            $self->{nc} <= 0x005A) { # A..Z
1287          !!!cp (29);          !!!cp (29);
1288          $self->{current_token} = {type => END_TAG_TOKEN,          $self->{ct}
1289                            tag_name => chr ($self->{next_char} + 0x0020)};              = {type => END_TAG_TOKEN,
1290                   tag_name => chr ($self->{nc} + 0x0020),
1291                   line => $l, column => $c};
1292          $self->{state} = TAG_NAME_STATE;          $self->{state} = TAG_NAME_STATE;
1293          !!!next-input-character;          !!!next-input-character;
1294          redo A;          redo A;
1295        } elsif (0x0061 <= $self->{next_char} and        } elsif (0x0061 <= $self->{nc} and
1296                 $self->{next_char} <= 0x007A) { # a..z                 $self->{nc} <= 0x007A) { # a..z
1297          !!!cp (30);          !!!cp (30);
1298          $self->{current_token} = {type => END_TAG_TOKEN,          $self->{ct} = {type => END_TAG_TOKEN,
1299                            tag_name => chr ($self->{next_char})};                                    tag_name => chr ($self->{nc}),
1300                                      line => $l, column => $c};
1301          $self->{state} = TAG_NAME_STATE;          $self->{state} = TAG_NAME_STATE;
1302          !!!next-input-character;          !!!next-input-character;
1303          redo A;          redo A;
1304        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1305          !!!cp (31);          !!!cp (31);
1306          !!!parse-error (type => 'empty end tag');          !!!parse-error (type => 'empty end tag',
1307                            line => $self->{line_prev}, ## "<" in "</>"
1308                            column => $self->{column_prev} - 1);
1309          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1310          !!!next-input-character;          !!!next-input-character;
1311          redo A;          redo A;
1312        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1313          !!!cp (32);          !!!cp (32);
1314          !!!parse-error (type => 'bare etago');          !!!parse-error (type => 'bare etago');
1315          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1316          # reconsume          # reconsume
1317    
1318          !!!emit ({type => CHARACTER_TOKEN, data => '</'});          !!!emit ({type => CHARACTER_TOKEN, data => '</',
1319                      line => $l, column => $c,
1320                     });
1321    
1322          redo A;          redo A;
1323        } else {        } else {
1324          !!!cp (33);          !!!cp (33);
1325          !!!parse-error (type => 'bogus end tag');          !!!parse-error (type => 'bogus end tag');
1326          $self->{state} = BOGUS_COMMENT_STATE;          $self->{state} = BOGUS_COMMENT_STATE;
1327          ## $self->{next_char} is intentionally left as is          $self->{ct} = {type => COMMENT_TOKEN, data => '',
1328          redo A;                                    line => $self->{line_prev}, # "<" of "</"
1329                                      column => $self->{column_prev} - 1,
1330                                     };
1331            ## NOTE: $self->{nc} is intentionally left as is.
1332            ## Although the "anything else" case of the spec not explicitly
1333            ## states that the next input character is to be reconsumed,
1334            ## it will be included to the |data| of the comment token
1335            ## generated from the bogus end tag, as defined in the
1336            ## "bogus comment state" entry.
1337            redo A;
1338          }
1339        } elsif ($self->{state} == CDATA_RCDATA_CLOSE_TAG_STATE) {
1340          my $ch = substr $self->{last_stag_name}, length $self->{s_kwd}, 1;
1341          if (length $ch) {
1342            my $CH = $ch;
1343            $ch =~ tr/a-z/A-Z/;
1344            my $nch = chr $self->{nc};
1345            if ($nch eq $ch or $nch eq $CH) {
1346              !!!cp (24);
1347              ## Stay in the state.
1348              $self->{s_kwd} .= $nch;
1349              !!!next-input-character;
1350              redo A;
1351            } else {
1352              !!!cp (25);
1353              $self->{state} = DATA_STATE;
1354              ## Reconsume.
1355              !!!emit ({type => CHARACTER_TOKEN,
1356                        data => '</' . $self->{s_kwd},
1357                        line => $self->{line_prev},
1358                        column => $self->{column_prev} - 1 - length $self->{s_kwd},
1359                       });
1360              redo A;
1361            }
1362          } else { # after "<{tag-name}"
1363            unless ($is_space->{$self->{nc}} or
1364                    {
1365                     0x003E => 1, # >
1366                     0x002F => 1, # /
1367                     -1 => 1, # EOF
1368                    }->{$self->{nc}}) {
1369              !!!cp (26);
1370              ## Reconsume.
1371              $self->{state} = DATA_STATE;
1372              !!!emit ({type => CHARACTER_TOKEN,
1373                        data => '</' . $self->{s_kwd},
1374                        line => $self->{line_prev},
1375                        column => $self->{column_prev} - 1 - length $self->{s_kwd},
1376                       });
1377              redo A;
1378            } else {
1379              !!!cp (27);
1380              $self->{ct}
1381                  = {type => END_TAG_TOKEN,
1382                     tag_name => $self->{last_stag_name},
1383                     line => $self->{line_prev},
1384                     column => $self->{column_prev} - 1 - length $self->{s_kwd}};
1385              $self->{state} = TAG_NAME_STATE;
1386              ## Reconsume.
1387              redo A;
1388            }
1389        }        }
1390      } elsif ($self->{state} == TAG_NAME_STATE) {      } elsif ($self->{state} == TAG_NAME_STATE) {
1391        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1392          !!!cp (34);          !!!cp (34);
1393          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1394          !!!next-input-character;          !!!next-input-character;
1395          redo A;          redo A;
1396        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1397          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1398            !!!cp (35);            !!!cp (35);
1399            $self->{current_token}->{first_start_tag}            $self->{last_stag_name} = $self->{ct}->{tag_name};
1400                = not defined $self->{last_emitted_start_tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {  
1401            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1402            #if ($self->{current_token}->{attributes}) {            #if ($self->{ct}->{attributes}) {
1403            #  ## NOTE: This should never be reached.            #  ## NOTE: This should never be reached.
1404            #  !!! cp (36);            #  !!! cp (36);
1405            #  !!! parse-error (type => 'end tag attribute');            #  !!! parse-error (type => 'end tag attribute');
# Line 664  sub _get_next_token ($) { Line 1407  sub _get_next_token ($) {
1407              !!!cp (37);              !!!cp (37);
1408            #}            #}
1409          } else {          } else {
1410            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1411          }          }
1412          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1413          !!!next-input-character;          !!!next-input-character;
1414    
1415          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1416    
1417          redo A;          redo A;
1418        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1419                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1420          !!!cp (38);          !!!cp (38);
1421          $self->{current_token}->{tag_name} .= chr ($self->{next_char} + 0x0020);          $self->{ct}->{tag_name} .= chr ($self->{nc} + 0x0020);
1422            # start tag or end tag            # start tag or end tag
1423          ## Stay in this state          ## Stay in this state
1424          !!!next-input-character;          !!!next-input-character;
1425          redo A;          redo A;
1426        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1427          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1428          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1429            !!!cp (39);            !!!cp (39);
1430            $self->{current_token}->{first_start_tag}            $self->{last_stag_name} = $self->{ct}->{tag_name};
1431                = not defined $self->{last_emitted_start_tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {  
1432            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1433            #if ($self->{current_token}->{attributes}) {            #if ($self->{ct}->{attributes}) {
1434            #  ## NOTE: This state should never be reached.            #  ## NOTE: This state should never be reached.
1435            #  !!! cp (40);            #  !!! cp (40);
1436            #  !!! parse-error (type => 'end tag attribute');            #  !!! parse-error (type => 'end tag attribute');
# Line 697  sub _get_next_token ($) { Line 1438  sub _get_next_token ($) {
1438              !!!cp (41);              !!!cp (41);
1439            #}            #}
1440          } else {          } else {
1441            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1442          }          }
1443          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1444          # reconsume          # reconsume
1445    
1446          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1447    
1448          redo A;          redo A;
1449        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1450            !!!cp (42);
1451            $self->{state} = SELF_CLOSING_START_TAG_STATE;
1452          !!!next-input-character;          !!!next-input-character;
         if ($self->{next_char} == 0x003E and # >  
             $self->{current_token}->{type} == START_TAG_TOKEN and  
             $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {  
           # permitted slash  
           !!!cp (42);  
           #  
         } else {  
           !!!cp (43);  
           !!!parse-error (type => 'nestc');  
         }  
         $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;  
         # next-input-character is already done  
1453          redo A;          redo A;
1454        } else {        } else {
1455          !!!cp (44);          !!!cp (44);
1456          $self->{current_token}->{tag_name} .= chr $self->{next_char};          $self->{ct}->{tag_name} .= chr $self->{nc};
1457            # start tag or end tag            # start tag or end tag
1458          ## Stay in the state          ## Stay in the state
1459          !!!next-input-character;          !!!next-input-character;
1460          redo A;          redo A;
1461        }        }
1462      } elsif ($self->{state} == BEFORE_ATTRIBUTE_NAME_STATE) {      } elsif ($self->{state} == BEFORE_ATTRIBUTE_NAME_STATE) {
1463        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1464          !!!cp (45);          !!!cp (45);
1465          ## Stay in the state          ## Stay in the state
1466          !!!next-input-character;          !!!next-input-character;
1467          redo A;          redo A;
1468        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1469          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1470            !!!cp (46);            !!!cp (46);
1471            $self->{current_token}->{first_start_tag}            $self->{last_stag_name} = $self->{ct}->{tag_name};
1472                = not defined $self->{last_emitted_start_tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {  
1473            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1474            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1475              !!!cp (47);              !!!cp (47);
1476              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1477            } else {            } else {
1478              !!!cp (48);              !!!cp (48);
1479            }            }
1480          } else {          } else {
1481            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1482          }          }
1483          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1484          !!!next-input-character;          !!!next-input-character;
1485    
1486          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1487    
1488          redo A;          redo A;
1489        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1490                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1491          !!!cp (49);          !!!cp (49);
1492          $self->{current_attribute} = {name => chr ($self->{next_char} + 0x0020),          $self->{ca}
1493                                value => ''};              = {name => chr ($self->{nc} + 0x0020),
1494                   value => '',
1495                   line => $self->{line}, column => $self->{column}};
1496          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
1497          !!!next-input-character;          !!!next-input-character;
1498          redo A;          redo A;
1499        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1500            !!!cp (50);
1501            $self->{state} = SELF_CLOSING_START_TAG_STATE;
1502          !!!next-input-character;          !!!next-input-character;
         if ($self->{next_char} == 0x003E and # >  
             $self->{current_token}->{type} == START_TAG_TOKEN and  
             $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {  
           # permitted slash  
           !!!cp (50);  
           #  
         } else {  
           !!!cp (51);  
           !!!parse-error (type => 'nestc');  
         }  
         ## Stay in the state  
         # next-input-character is already done  
1503          redo A;          redo A;
1504        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1505          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1506          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1507            !!!cp (52);            !!!cp (52);
1508            $self->{current_token}->{first_start_tag}            $self->{last_stag_name} = $self->{ct}->{tag_name};
1509                = not defined $self->{last_emitted_start_tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {  
1510            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1511            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1512              !!!cp (53);              !!!cp (53);
1513              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1514            } else {            } else {
1515              !!!cp (54);              !!!cp (54);
1516            }            }
1517          } else {          } else {
1518            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1519          }          }
1520          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1521          # reconsume          # reconsume
1522    
1523          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1524    
1525          redo A;          redo A;
1526        } else {        } else {
# Line 813  sub _get_next_token ($) { Line 1528  sub _get_next_token ($) {
1528               0x0022 => 1, # "               0x0022 => 1, # "
1529               0x0027 => 1, # '               0x0027 => 1, # '
1530               0x003D => 1, # =               0x003D => 1, # =
1531              }->{$self->{next_char}}) {              }->{$self->{nc}}) {
1532            !!!cp (55);            !!!cp (55);
1533            !!!parse-error (type => 'bad attribute name');            !!!parse-error (type => 'bad attribute name');
1534          } else {          } else {
1535            !!!cp (56);            !!!cp (56);
1536          }          }
1537          $self->{current_attribute} = {name => chr ($self->{next_char}),          $self->{ca}
1538                                value => ''};              = {name => chr ($self->{nc}),
1539                   value => '',
1540                   line => $self->{line}, column => $self->{column}};
1541          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
1542          !!!next-input-character;          !!!next-input-character;
1543          redo A;          redo A;
1544        }        }
1545      } elsif ($self->{state} == ATTRIBUTE_NAME_STATE) {      } elsif ($self->{state} == ATTRIBUTE_NAME_STATE) {
1546        my $before_leave = sub {        my $before_leave = sub {
1547          if (exists $self->{current_token}->{attributes} # start tag or end tag          if (exists $self->{ct}->{attributes} # start tag or end tag
1548              ->{$self->{current_attribute}->{name}}) { # MUST              ->{$self->{ca}->{name}}) { # MUST
1549            !!!cp (57);            !!!cp (57);
1550            !!!parse-error (type => 'duplicate attribute:'.$self->{current_attribute}->{name});            !!!parse-error (type => 'duplicate attribute', text => $self->{ca}->{name}, line => $self->{ca}->{line}, column => $self->{ca}->{column});
1551            ## Discard $self->{current_attribute} # MUST            ## Discard $self->{ca} # MUST
1552          } else {          } else {
1553            !!!cp (58);            !!!cp (58);
1554            $self->{current_token}->{attributes}->{$self->{current_attribute}->{name}}            $self->{ct}->{attributes}->{$self->{ca}->{name}}
1555              = $self->{current_attribute};              = $self->{ca};
1556          }          }
1557        }; # $before_leave        }; # $before_leave
1558    
1559        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1560          !!!cp (59);          !!!cp (59);
1561          $before_leave->();          $before_leave->();
1562          $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;          $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;
1563          !!!next-input-character;          !!!next-input-character;
1564          redo A;          redo A;
1565        } elsif ($self->{next_char} == 0x003D) { # =        } elsif ($self->{nc} == 0x003D) { # =
1566          !!!cp (60);          !!!cp (60);
1567          $before_leave->();          $before_leave->();
1568          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1569          !!!next-input-character;          !!!next-input-character;
1570          redo A;          redo A;
1571        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1572          $before_leave->();          $before_leave->();
1573          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1574            !!!cp (61);            !!!cp (61);
1575            $self->{current_token}->{first_start_tag}            $self->{last_stag_name} = $self->{ct}->{tag_name};
1576                = not defined $self->{last_emitted_start_tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {  
1577            !!!cp (62);            !!!cp (62);
1578            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1579            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1580              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1581            }            }
1582          } else {          } else {
1583            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1584          }          }
1585          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1586          !!!next-input-character;          !!!next-input-character;
1587    
1588          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1589    
1590          redo A;          redo A;
1591        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1592                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1593          !!!cp (63);          !!!cp (63);
1594          $self->{current_attribute}->{name} .= chr ($self->{next_char} + 0x0020);          $self->{ca}->{name} .= chr ($self->{nc} + 0x0020);
1595          ## Stay in the state          ## Stay in the state
1596          !!!next-input-character;          !!!next-input-character;
1597          redo A;          redo A;
1598        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1599            !!!cp (64);
1600          $before_leave->();          $before_leave->();
1601            $self->{state} = SELF_CLOSING_START_TAG_STATE;
1602          !!!next-input-character;          !!!next-input-character;
         if ($self->{next_char} == 0x003E and # >  
             $self->{current_token}->{type} == START_TAG_TOKEN and  
             $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {  
           # permitted slash  
           !!!cp (64);  
           #  
         } else {  
           !!!cp (65);  
           !!!parse-error (type => 'nestc');  
         }  
         $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;  
         # next-input-character is already done  
1603          redo A;          redo A;
1604        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1605          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1606          $before_leave->();          $before_leave->();
1607          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1608            !!!cp (66);            !!!cp (66);
1609            $self->{current_token}->{first_start_tag}            $self->{last_stag_name} = $self->{ct}->{tag_name};
1610                = not defined $self->{last_emitted_start_tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {  
1611            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1612            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1613              !!!cp (67);              !!!cp (67);
1614              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1615            } else {            } else {
# Line 918  sub _get_next_token ($) { Line 1617  sub _get_next_token ($) {
1617              !!!cp (68);              !!!cp (68);
1618            }            }
1619          } else {          } else {
1620            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1621          }          }
1622          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1623          # reconsume          # reconsume
1624    
1625          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1626    
1627          redo A;          redo A;
1628        } else {        } else {
1629          if ($self->{next_char} == 0x0022 or # "          if ($self->{nc} == 0x0022 or # "
1630              $self->{next_char} == 0x0027) { # '              $self->{nc} == 0x0027) { # '
1631            !!!cp (69);            !!!cp (69);
1632            !!!parse-error (type => 'bad attribute name');            !!!parse-error (type => 'bad attribute name');
1633          } else {          } else {
1634            !!!cp (70);            !!!cp (70);
1635          }          }
1636          $self->{current_attribute}->{name} .= chr ($self->{next_char});          $self->{ca}->{name} .= chr ($self->{nc});
1637          ## Stay in the state          ## Stay in the state
1638          !!!next-input-character;          !!!next-input-character;
1639          redo A;          redo A;
1640        }        }
1641      } elsif ($self->{state} == AFTER_ATTRIBUTE_NAME_STATE) {      } elsif ($self->{state} == AFTER_ATTRIBUTE_NAME_STATE) {
1642        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1643          !!!cp (71);          !!!cp (71);
1644          ## Stay in the state          ## Stay in the state
1645          !!!next-input-character;          !!!next-input-character;
1646          redo A;          redo A;
1647        } elsif ($self->{next_char} == 0x003D) { # =        } elsif ($self->{nc} == 0x003D) { # =
1648          !!!cp (72);          !!!cp (72);
1649          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1650          !!!next-input-character;          !!!next-input-character;
1651          redo A;          redo A;
1652        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1653          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1654            !!!cp (73);            !!!cp (73);
1655            $self->{current_token}->{first_start_tag}            $self->{last_stag_name} = $self->{ct}->{tag_name};
1656                = not defined $self->{last_emitted_start_tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {  
1657            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1658            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1659              !!!cp (74);              !!!cp (74);
1660              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1661            } else {            } else {
# Line 970  sub _get_next_token ($) { Line 1663  sub _get_next_token ($) {
1663              !!!cp (75);              !!!cp (75);
1664            }            }
1665          } else {          } else {
1666            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1667          }          }
1668          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1669          !!!next-input-character;          !!!next-input-character;
1670    
1671          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1672    
1673          redo A;          redo A;
1674        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{nc} and
1675                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1676          !!!cp (76);          !!!cp (76);
1677          $self->{current_attribute} = {name => chr ($self->{next_char} + 0x0020),          $self->{ca}
1678                                value => ''};              = {name => chr ($self->{nc} + 0x0020),
1679                   value => '',
1680                   line => $self->{line}, column => $self->{column}};
1681          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
1682          !!!next-input-character;          !!!next-input-character;
1683          redo A;          redo A;
1684        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1685            !!!cp (77);
1686            $self->{state} = SELF_CLOSING_START_TAG_STATE;
1687          !!!next-input-character;          !!!next-input-character;
         if ($self->{next_char} == 0x003E and # >  
             $self->{current_token}->{type} == START_TAG_TOKEN and  
             $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {  
           # permitted slash  
           !!!cp (77);  
           #  
         } else {  
           !!!cp (78);  
           !!!parse-error (type => 'nestc');  
           ## TODO: Different error type for <aa / bb> than <aa/>  
         }  
         $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;  
         # next-input-character is already done  
1688          redo A;          redo A;
1689        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1690          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1691          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1692            !!!cp (79);            !!!cp (79);
1693            $self->{current_token}->{first_start_tag}            $self->{last_stag_name} = $self->{ct}->{tag_name};
1694                = not defined $self->{last_emitted_start_tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {  
1695            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1696            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1697              !!!cp (80);              !!!cp (80);
1698              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1699            } else {            } else {
# Line 1019  sub _get_next_token ($) { Line 1701  sub _get_next_token ($) {
1701              !!!cp (81);              !!!cp (81);
1702            }            }
1703          } else {          } else {
1704            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1705          }          }
1706          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1707          # reconsume          # reconsume
1708    
1709          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1710    
1711          redo A;          redo A;
1712        } else {        } else {
1713          !!!cp (82);          if ($self->{nc} == 0x0022 or # "
1714          $self->{current_attribute} = {name => chr ($self->{next_char}),              $self->{nc} == 0x0027) { # '
1715                                value => ''};            !!!cp (78);
1716              !!!parse-error (type => 'bad attribute name');
1717            } else {
1718              !!!cp (82);
1719            }
1720            $self->{ca}
1721                = {name => chr ($self->{nc}),
1722                   value => '',
1723                   line => $self->{line}, column => $self->{column}};
1724          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
1725          !!!next-input-character;          !!!next-input-character;
1726          redo A;                  redo A;        
1727        }        }
1728      } elsif ($self->{state} == BEFORE_ATTRIBUTE_VALUE_STATE) {      } elsif ($self->{state} == BEFORE_ATTRIBUTE_VALUE_STATE) {
1729        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP        
1730          !!!cp (83);          !!!cp (83);
1731          ## Stay in the state          ## Stay in the state
1732          !!!next-input-character;          !!!next-input-character;
1733          redo A;          redo A;
1734        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
1735          !!!cp (84);          !!!cp (84);
1736          $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;
1737          !!!next-input-character;          !!!next-input-character;
1738          redo A;          redo A;
1739        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1740          !!!cp (85);          !!!cp (85);
1741          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1742          ## reconsume          ## reconsume
1743          redo A;          redo A;
1744        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
1745          !!!cp (86);          !!!cp (86);
1746          $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;
1747          !!!next-input-character;          !!!next-input-character;
1748          redo A;          redo A;
1749        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1750          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          !!!parse-error (type => 'empty unquoted attribute value');
1751            if ($self->{ct}->{type} == START_TAG_TOKEN) {
1752            !!!cp (87);            !!!cp (87);
1753            $self->{current_token}->{first_start_tag}            $self->{last_stag_name} = $self->{ct}->{tag_name};
1754                = not defined $self->{last_emitted_start_tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {  
1755            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1756            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1757              !!!cp (88);              !!!cp (88);
1758              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1759            } else {            } else {
# Line 1076  sub _get_next_token ($) { Line 1761  sub _get_next_token ($) {
1761              !!!cp (89);              !!!cp (89);
1762            }            }
1763          } else {          } else {
1764            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1765          }          }
1766          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1767          !!!next-input-character;          !!!next-input-character;
1768    
1769          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1770    
1771          redo A;          redo A;
1772        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1773          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1774          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1775            !!!cp (90);            !!!cp (90);
1776            $self->{current_token}->{first_start_tag}            $self->{last_stag_name} = $self->{ct}->{tag_name};
1777                = not defined $self->{last_emitted_start_tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {  
1778            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1779            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1780              !!!cp (91);              !!!cp (91);
1781              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1782            } else {            } else {
# Line 1101  sub _get_next_token ($) { Line 1784  sub _get_next_token ($) {
1784              !!!cp (92);              !!!cp (92);
1785            }            }
1786          } else {          } else {
1787            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1788          }          }
1789          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1790          ## reconsume          ## reconsume
1791    
1792          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1793    
1794          redo A;          redo A;
1795        } else {        } else {
1796          if ($self->{next_char} == 0x003D) { # =          if ($self->{nc} == 0x003D) { # =
1797            !!!cp (93);            !!!cp (93);
1798            !!!parse-error (type => 'bad attribute value');            !!!parse-error (type => 'bad attribute value');
1799          } else {          } else {
1800            !!!cp (94);            !!!cp (94);
1801          }          }
1802          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1803          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1804          !!!next-input-character;          !!!next-input-character;
1805          redo A;          redo A;
1806        }        }
1807      } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {
1808        if ($self->{next_char} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
1809          !!!cp (95);          !!!cp (95);
1810          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1811          !!!next-input-character;          !!!next-input-character;
1812          redo A;          redo A;
1813        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1814          !!!cp (96);          !!!cp (96);
1815          $self->{last_attribute_value_state} = $self->{state};          ## NOTE: In the spec, the tokenizer is switched to the
1816          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          ## "entity in attribute value state".  In this implementation, the
1817            ## tokenizer is switched to the |ENTITY_STATE|, which is an
1818            ## implementation of the "consume a character reference" algorithm.
1819            $self->{prev_state} = $self->{state};
1820            $self->{entity_add} = 0x0022; # "
1821            $self->{state} = ENTITY_STATE;
1822          !!!next-input-character;          !!!next-input-character;
1823          redo A;          redo A;
1824        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1825          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
1826          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1827            !!!cp (97);            !!!cp (97);
1828            $self->{current_token}->{first_start_tag}            $self->{last_stag_name} = $self->{ct}->{tag_name};
1829                = not defined $self->{last_emitted_start_tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {  
1830            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1831            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1832              !!!cp (98);              !!!cp (98);
1833              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1834            } else {            } else {
# Line 1150  sub _get_next_token ($) { Line 1836  sub _get_next_token ($) {
1836              !!!cp (99);              !!!cp (99);
1837            }            }
1838          } else {          } else {
1839            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1840          }          }
1841          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1842          ## reconsume          ## reconsume
1843    
1844          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1845    
1846          redo A;          redo A;
1847        } else {        } else {
1848          !!!cp (100);          !!!cp (100);
1849          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1850            $self->{read_until}->($self->{ca}->{value},
1851                                  q["&],
1852                                  length $self->{ca}->{value});
1853    
1854          ## Stay in the state          ## Stay in the state
1855          !!!next-input-character;          !!!next-input-character;
1856          redo A;          redo A;
1857        }        }
1858      } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {
1859        if ($self->{next_char} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
1860          !!!cp (101);          !!!cp (101);
1861          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1862          !!!next-input-character;          !!!next-input-character;
1863          redo A;          redo A;
1864        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1865          !!!cp (102);          !!!cp (102);
1866          $self->{last_attribute_value_state} = $self->{state};          ## NOTE: In the spec, the tokenizer is switched to the
1867          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          ## "entity in attribute value state".  In this implementation, the
1868            ## tokenizer is switched to the |ENTITY_STATE|, which is an
1869            ## implementation of the "consume a character reference" algorithm.
1870            $self->{entity_add} = 0x0027; # '
1871            $self->{prev_state} = $self->{state};
1872            $self->{state} = ENTITY_STATE;
1873          !!!next-input-character;          !!!next-input-character;
1874          redo A;          redo A;
1875        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1876          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
1877          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1878            !!!cp (103);            !!!cp (103);
1879            $self->{current_token}->{first_start_tag}            $self->{last_stag_name} = $self->{ct}->{tag_name};
1880                = not defined $self->{last_emitted_start_tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {  
1881            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1882            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1883              !!!cp (104);              !!!cp (104);
1884              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1885            } else {            } else {
# Line 1194  sub _get_next_token ($) { Line 1887  sub _get_next_token ($) {
1887              !!!cp (105);              !!!cp (105);
1888            }            }
1889          } else {          } else {
1890            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1891          }          }
1892          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1893          ## reconsume          ## reconsume
1894    
1895          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1896    
1897          redo A;          redo A;
1898        } else {        } else {
1899          !!!cp (106);          !!!cp (106);
1900          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1901            $self->{read_until}->($self->{ca}->{value},
1902                                  q['&],
1903                                  length $self->{ca}->{value});
1904    
1905          ## Stay in the state          ## Stay in the state
1906          !!!next-input-character;          !!!next-input-character;
1907          redo A;          redo A;
1908        }        }
1909      } elsif ($self->{state} == ATTRIBUTE_VALUE_UNQUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_UNQUOTED_STATE) {
1910        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # HT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1911          !!!cp (107);          !!!cp (107);
1912          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1913          !!!next-input-character;          !!!next-input-character;
1914          redo A;          redo A;
1915        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1916          !!!cp (108);          !!!cp (108);
1917          $self->{last_attribute_value_state} = $self->{state};          ## NOTE: In the spec, the tokenizer is switched to the
1918          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          ## "entity in attribute value state".  In this implementation, the
1919            ## tokenizer is switched to the |ENTITY_STATE|, which is an
1920            ## implementation of the "consume a character reference" algorithm.
1921            $self->{entity_add} = -1;
1922            $self->{prev_state} = $self->{state};
1923            $self->{state} = ENTITY_STATE;
1924          !!!next-input-character;          !!!next-input-character;
1925          redo A;          redo A;
1926        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1927          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1928            !!!cp (109);            !!!cp (109);
1929            $self->{current_token}->{first_start_tag}            $self->{last_stag_name} = $self->{ct}->{tag_name};
1930                = not defined $self->{last_emitted_start_tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {  
1931            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1932            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1933              !!!cp (110);              !!!cp (110);
1934              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1935            } else {            } else {
# Line 1241  sub _get_next_token ($) { Line 1937  sub _get_next_token ($) {
1937              !!!cp (111);              !!!cp (111);
1938            }            }
1939          } else {          } else {
1940            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1941          }          }
1942          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1943          !!!next-input-character;          !!!next-input-character;
1944    
1945          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1946    
1947          redo A;          redo A;
1948        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
1949          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1950          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1951            !!!cp (112);            !!!cp (112);
1952            $self->{current_token}->{first_start_tag}            $self->{last_stag_name} = $self->{ct}->{tag_name};
1953                = not defined $self->{last_emitted_start_tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {  
1954            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1955            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1956              !!!cp (113);              !!!cp (113);
1957              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1958            } else {            } else {
# Line 1266  sub _get_next_token ($) { Line 1960  sub _get_next_token ($) {
1960              !!!cp (114);              !!!cp (114);
1961            }            }
1962          } else {          } else {
1963            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1964          }          }
1965          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1966          ## reconsume          ## reconsume
1967    
1968          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1969    
1970          redo A;          redo A;
1971        } else {        } else {
# Line 1279  sub _get_next_token ($) { Line 1973  sub _get_next_token ($) {
1973               0x0022 => 1, # "               0x0022 => 1, # "
1974               0x0027 => 1, # '               0x0027 => 1, # '
1975               0x003D => 1, # =               0x003D => 1, # =
1976              }->{$self->{next_char}}) {              }->{$self->{nc}}) {
1977            !!!cp (115);            !!!cp (115);
1978            !!!parse-error (type => 'bad attribute value');            !!!parse-error (type => 'bad attribute value');
1979          } else {          } else {
1980            !!!cp (116);            !!!cp (116);
1981          }          }
1982          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{ca}->{value} .= chr ($self->{nc});
1983            $self->{read_until}->($self->{ca}->{value},
1984                                  q["'=& >],
1985                                  length $self->{ca}->{value});
1986    
1987          ## Stay in the state          ## Stay in the state
1988          !!!next-input-character;          !!!next-input-character;
1989          redo A;          redo A;
1990        }        }
     } elsif ($self->{state} == ENTITY_IN_ATTRIBUTE_VALUE_STATE) {  
       my $token = $self->_tokenize_attempt_to_consume_an_entity  
           (1,  
            $self->{last_attribute_value_state}  
              == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE ? 0x0022 : # "  
            $self->{last_attribute_value_state}  
              == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE ? 0x0027 : # '  
            -1);  
   
       unless (defined $token) {  
         !!!cp (117);  
         $self->{current_attribute}->{value} .= '&';  
       } else {  
         !!!cp (118);  
         $self->{current_attribute}->{value} .= $token->{data};  
         $self->{current_attribute}->{has_reference} = $token->{has_reference};  
         ## ISSUE: spec says "append the returned character token to the current attribute's value"  
       }  
   
       $self->{state} = $self->{last_attribute_value_state};  
       # next-input-character is already done  
       redo A;  
1991      } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {      } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {
1992        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
1993          !!!cp (118);          !!!cp (118);
1994          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1995          !!!next-input-character;          !!!next-input-character;
1996          redo A;          redo A;
1997        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1998          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1999            !!!cp (119);            !!!cp (119);
2000            $self->{current_token}->{first_start_tag}            $self->{last_stag_name} = $self->{ct}->{tag_name};
2001                = not defined $self->{last_emitted_start_tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {  
2002            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
2003            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
2004              !!!cp (120);              !!!cp (120);
2005              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
2006            } else {            } else {
# Line 1338  sub _get_next_token ($) { Line 2008  sub _get_next_token ($) {
2008              !!!cp (121);              !!!cp (121);
2009            }            }
2010          } else {          } else {
2011            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
2012          }          }
2013          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2014          !!!next-input-character;          !!!next-input-character;
2015    
2016          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
2017    
2018          redo A;          redo A;
2019        } elsif ($self->{next_char} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
2020            !!!cp (122);
2021            $self->{state} = SELF_CLOSING_START_TAG_STATE;
2022          !!!next-input-character;          !!!next-input-character;
2023          if ($self->{next_char} == 0x003E and # >          redo A;
2024              $self->{current_token}->{type} == START_TAG_TOKEN and        } elsif ($self->{nc} == -1) {
2025              $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {          !!!parse-error (type => 'unclosed tag');
2026            # permitted slash          if ($self->{ct}->{type} == START_TAG_TOKEN) {
2027            !!!cp (122);            !!!cp (122.3);
2028            #            $self->{last_stag_name} = $self->{ct}->{tag_name};
2029            } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2030              if ($self->{ct}->{attributes}) {
2031                !!!cp (122.1);
2032                !!!parse-error (type => 'end tag attribute');
2033              } else {
2034                ## NOTE: This state should never be reached.
2035                !!!cp (122.2);
2036              }
2037          } else {          } else {
2038            !!!cp (123);            die "$0: $self->{ct}->{type}: Unknown token type";
           !!!parse-error (type => 'nestc');  
2039          }          }
2040          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = DATA_STATE;
2041          # next-input-character is already done          ## Reconsume.
2042            !!!emit ($self->{ct}); # start tag or end tag
2043          redo A;          redo A;
2044        } else {        } else {
2045          !!!cp (124);          !!!cp ('124.1');
2046          !!!parse-error (type => 'no space between attributes');          !!!parse-error (type => 'no space between attributes');
2047          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
2048          ## reconsume          ## reconsume
2049          redo A;          redo A;
2050        }        }
2051      } elsif ($self->{state} == BOGUS_COMMENT_STATE) {      } elsif ($self->{state} == SELF_CLOSING_START_TAG_STATE) {
2052        ## (only happen if PCDATA state)        if ($self->{nc} == 0x003E) { # >
2053                  if ($self->{ct}->{type} == END_TAG_TOKEN) {
2054        my $token = {type => COMMENT_TOKEN, data => ''};            !!!cp ('124.2');
2055              !!!parse-error (type => 'nestc', token => $self->{ct});
2056        BC: {            ## TODO: Different type than slash in start tag
2057          if ($self->{next_char} == 0x003E) { # >            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
2058            !!!cp (124);            if ($self->{ct}->{attributes}) {
2059            $self->{state} = DATA_STATE;              !!!cp ('124.4');
2060            !!!next-input-character;              !!!parse-error (type => 'end tag attribute');
2061              } else {
2062            !!!emit ($token);              !!!cp ('124.5');
2063              }
2064              ## TODO: Test |<title></title/>|
2065            } else {
2066              !!!cp ('124.3');
2067              $self->{self_closing} = 1;
2068            }
2069    
2070            redo A;          $self->{state} = DATA_STATE;
2071          } elsif ($self->{next_char} == -1) {          !!!next-input-character;
           !!!cp (125);  
           $self->{state} = DATA_STATE;  
           ## reconsume  
2072    
2073            !!!emit ($token);          !!!emit ($self->{ct}); # start tag or end tag
2074    
2075            redo A;          redo A;
2076          } elsif ($self->{nc} == -1) {
2077            !!!parse-error (type => 'unclosed tag');
2078            if ($self->{ct}->{type} == START_TAG_TOKEN) {
2079              !!!cp (124.7);
2080              $self->{last_stag_name} = $self->{ct}->{tag_name};
2081            } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2082              if ($self->{ct}->{attributes}) {
2083                !!!cp (124.5);
2084                !!!parse-error (type => 'end tag attribute');
2085              } else {
2086                ## NOTE: This state should never be reached.
2087                !!!cp (124.6);
2088              }
2089          } else {          } else {
2090            !!!cp (126);            die "$0: $self->{ct}->{type}: Unknown token type";
           $token->{data} .= chr ($self->{next_char});  
           !!!next-input-character;  
           redo BC;  
2091          }          }
2092        } # BC          $self->{state} = DATA_STATE;
2093            ## Reconsume.
2094        die "$0: _get_next_token: unexpected case [BC]";          !!!emit ($self->{ct}); # start tag or end tag
2095      } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {          redo A;
2096          } else {
2097            !!!cp ('124.4');
2098            !!!parse-error (type => 'nestc');
2099            ## TODO: This error type is wrong.
2100            $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
2101            ## Reconsume.
2102            redo A;
2103          }
2104        } elsif ($self->{state} == BOGUS_COMMENT_STATE) {
2105        ## (only happen if PCDATA state)        ## (only happen if PCDATA state)
2106    
2107        my @next_char;        ## NOTE: Unlike spec's "bogus comment state", this implementation
2108        push @next_char, $self->{next_char};        ## consumes characters one-by-one basis.
2109                
2110        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x003E) { # >
2111            !!!cp (124);
2112            $self->{state} = DATA_STATE;
2113          !!!next-input-character;          !!!next-input-character;
2114          push @next_char, $self->{next_char};  
2115          if ($self->{next_char} == 0x002D) { # -          !!!emit ($self->{ct}); # comment
2116            !!!cp (127);          redo A;
2117            $self->{current_token} = {type => COMMENT_TOKEN, data => ''};        } elsif ($self->{nc} == -1) {
2118            $self->{state} = COMMENT_START_STATE;          !!!cp (125);
2119            !!!next-input-character;          $self->{state} = DATA_STATE;
2120            redo A;          ## reconsume
2121          } else {  
2122            !!!cp (128);          !!!emit ($self->{ct}); # comment
2123          }          redo A;
2124        } elsif ($self->{next_char} == 0x0044 or # D        } else {
2125                 $self->{next_char} == 0x0064) { # d          !!!cp (126);
2126            $self->{ct}->{data} .= chr ($self->{nc}); # comment
2127            $self->{read_until}->($self->{ct}->{data},
2128                                  q[>],
2129                                  length $self->{ct}->{data});
2130    
2131            ## Stay in the state.
2132          !!!next-input-character;          !!!next-input-character;
2133          push @next_char, $self->{next_char};          redo A;
2134          if ($self->{next_char} == 0x004F or # O        }
2135              $self->{next_char} == 0x006F) { # o      } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {
2136            !!!next-input-character;        ## (only happen if PCDATA state)
2137            push @next_char, $self->{next_char};        
2138            if ($self->{next_char} == 0x0043 or # C        if ($self->{nc} == 0x002D) { # -
2139                $self->{next_char} == 0x0063) { # c          !!!cp (133);
2140              !!!next-input-character;          $self->{state} = MD_HYPHEN_STATE;
2141              push @next_char, $self->{next_char};          !!!next-input-character;
2142              if ($self->{next_char} == 0x0054 or # T          redo A;
2143                  $self->{next_char} == 0x0074) { # t        } elsif ($self->{nc} == 0x0044 or # D
2144                !!!next-input-character;                 $self->{nc} == 0x0064) { # d
2145                push @next_char, $self->{next_char};          ## ASCII case-insensitive.
2146                if ($self->{next_char} == 0x0059 or # Y          !!!cp (130);
2147                    $self->{next_char} == 0x0079) { # y          $self->{state} = MD_DOCTYPE_STATE;
2148                  !!!next-input-character;          $self->{s_kwd} = chr $self->{nc};
2149                  push @next_char, $self->{next_char};          !!!next-input-character;
2150                  if ($self->{next_char} == 0x0050 or # P          redo A;
2151                      $self->{next_char} == 0x0070) { # p        } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and
2152                    !!!next-input-character;                 $self->{open_elements}->[-1]->[1] & FOREIGN_EL and
2153                    push @next_char, $self->{next_char};                 $self->{nc} == 0x005B) { # [
2154                    if ($self->{next_char} == 0x0045 or # E          !!!cp (135.4);                
2155                        $self->{next_char} == 0x0065) { # e          $self->{state} = MD_CDATA_STATE;
2156                      !!!cp (129);          $self->{s_kwd} = '[';
2157                      ## TODO: What a stupid code this is!          !!!next-input-character;
2158                      $self->{state} = DOCTYPE_STATE;          redo A;
                     !!!next-input-character;  
                     redo A;  
                   } else {  
                     !!!cp (130);  
                   }  
                 } else {  
                   !!!cp (131);  
                 }  
               } else {  
                 !!!cp (132);  
               }  
             } else {  
               !!!cp (133);  
             }  
           } else {  
             !!!cp (134);  
           }  
         } else {  
           !!!cp (135);  
         }  
2159        } else {        } else {
2160          !!!cp (136);          !!!cp (136);
2161        }        }
2162    
2163        !!!parse-error (type => 'bogus comment');        !!!parse-error (type => 'bogus comment',
2164        $self->{next_char} = shift @next_char;                        line => $self->{line_prev},
2165        !!!back-next-input-character (@next_char);                        column => $self->{column_prev} - 1);
2166          ## Reconsume.
2167        $self->{state} = BOGUS_COMMENT_STATE;        $self->{state} = BOGUS_COMMENT_STATE;
2168          $self->{ct} = {type => COMMENT_TOKEN, data => '',
2169                                    line => $self->{line_prev},
2170                                    column => $self->{column_prev} - 1,
2171                                   };
2172        redo A;        redo A;
2173              } elsif ($self->{state} == MD_HYPHEN_STATE) {
2174        ## ISSUE: typos in spec: chacacters, is is a parse error        if ($self->{nc} == 0x002D) { # -
2175        ## ISSUE: spec is somewhat unclear on "is the first character that will be in the comment"; what is "that will be in the comment" is what the algorithm defines, isn't it?          !!!cp (127);
2176            $self->{ct} = {type => COMMENT_TOKEN, data => '',
2177                                      line => $self->{line_prev},
2178                                      column => $self->{column_prev} - 2,
2179                                     };
2180            $self->{state} = COMMENT_START_STATE;
2181            !!!next-input-character;
2182            redo A;
2183          } else {
2184            !!!cp (128);
2185            !!!parse-error (type => 'bogus comment',
2186                            line => $self->{line_prev},
2187                            column => $self->{column_prev} - 2);
2188            $self->{state} = BOGUS_COMMENT_STATE;
2189            ## Reconsume.
2190            $self->{ct} = {type => COMMENT_TOKEN,
2191                                      data => '-',
2192                                      line => $self->{line_prev},
2193                                      column => $self->{column_prev} - 2,
2194                                     };
2195            redo A;
2196          }
2197        } elsif ($self->{state} == MD_DOCTYPE_STATE) {
2198          ## ASCII case-insensitive.
2199          if ($self->{nc} == [
2200                undef,
2201                0x004F, # O
2202                0x0043, # C
2203                0x0054, # T
2204                0x0059, # Y
2205                0x0050, # P
2206              ]->[length $self->{s_kwd}] or
2207              $self->{nc} == [
2208                undef,
2209                0x006F, # o
2210                0x0063, # c
2211                0x0074, # t
2212                0x0079, # y
2213                0x0070, # p
2214              ]->[length $self->{s_kwd}]) {
2215            !!!cp (131);
2216            ## Stay in the state.
2217            $self->{s_kwd} .= chr $self->{nc};
2218            !!!next-input-character;
2219            redo A;
2220          } elsif ((length $self->{s_kwd}) == 6 and
2221                   ($self->{nc} == 0x0045 or # E
2222                    $self->{nc} == 0x0065)) { # e
2223            !!!cp (129);
2224            $self->{state} = DOCTYPE_STATE;
2225            $self->{ct} = {type => DOCTYPE_TOKEN,
2226                                      quirks => 1,
2227                                      line => $self->{line_prev},
2228                                      column => $self->{column_prev} - 7,
2229                                     };
2230            !!!next-input-character;
2231            redo A;
2232          } else {
2233            !!!cp (132);        
2234            !!!parse-error (type => 'bogus comment',
2235                            line => $self->{line_prev},
2236                            column => $self->{column_prev} - 1 - length $self->{s_kwd});
2237            $self->{state} = BOGUS_COMMENT_STATE;
2238            ## Reconsume.
2239            $self->{ct} = {type => COMMENT_TOKEN,
2240                                      data => $self->{s_kwd},
2241                                      line => $self->{line_prev},
2242                                      column => $self->{column_prev} - 1 - length $self->{s_kwd},
2243                                     };
2244            redo A;
2245          }
2246        } elsif ($self->{state} == MD_CDATA_STATE) {
2247          if ($self->{nc} == {
2248                '[' => 0x0043, # C
2249                '[C' => 0x0044, # D
2250                '[CD' => 0x0041, # A
2251                '[CDA' => 0x0054, # T
2252                '[CDAT' => 0x0041, # A
2253              }->{$self->{s_kwd}}) {
2254            !!!cp (135.1);
2255            ## Stay in the state.
2256            $self->{s_kwd} .= chr $self->{nc};
2257            !!!next-input-character;
2258            redo A;
2259          } elsif ($self->{s_kwd} eq '[CDATA' and
2260                   $self->{nc} == 0x005B) { # [
2261            !!!cp (135.2);
2262            $self->{ct} = {type => CHARACTER_TOKEN,
2263                                      data => '',
2264                                      line => $self->{line_prev},
2265                                      column => $self->{column_prev} - 7};
2266            $self->{state} = CDATA_SECTION_STATE;
2267            !!!next-input-character;
2268            redo A;
2269          } else {
2270            !!!cp (135.3);
2271            !!!parse-error (type => 'bogus comment',
2272                            line => $self->{line_prev},
2273                            column => $self->{column_prev} - 1 - length $self->{s_kwd});
2274            $self->{state} = BOGUS_COMMENT_STATE;
2275            ## Reconsume.
2276            $self->{ct} = {type => COMMENT_TOKEN,
2277                                      data => $self->{s_kwd},
2278                                      line => $self->{line_prev},
2279                                      column => $self->{column_prev} - 1 - length $self->{s_kwd},
2280                                     };
2281            redo A;
2282          }
2283      } elsif ($self->{state} == COMMENT_START_STATE) {      } elsif ($self->{state} == COMMENT_START_STATE) {
2284        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2285          !!!cp (137);          !!!cp (137);
2286          $self->{state} = COMMENT_START_DASH_STATE;          $self->{state} = COMMENT_START_DASH_STATE;
2287          !!!next-input-character;          !!!next-input-character;
2288          redo A;          redo A;
2289        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2290          !!!cp (138);          !!!cp (138);
2291          !!!parse-error (type => 'bogus comment');          !!!parse-error (type => 'bogus comment');
2292          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2293          !!!next-input-character;          !!!next-input-character;
2294    
2295          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2296    
2297          redo A;          redo A;
2298        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2299          !!!cp (139);          !!!cp (139);
2300          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2301          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2302          ## reconsume          ## reconsume
2303    
2304          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2305    
2306          redo A;          redo A;
2307        } else {        } else {
2308          !!!cp (140);          !!!cp (140);
2309          $self->{current_token}->{data} # comment          $self->{ct}->{data} # comment
2310              .= chr ($self->{next_char});              .= chr ($self->{nc});
2311          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2312          !!!next-input-character;          !!!next-input-character;
2313          redo A;          redo A;
2314        }        }
2315      } elsif ($self->{state} == COMMENT_START_DASH_STATE) {      } elsif ($self->{state} == COMMENT_START_DASH_STATE) {
2316        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2317          !!!cp (141);          !!!cp (141);
2318          $self->{state} = COMMENT_END_STATE;          $self->{state} = COMMENT_END_STATE;
2319          !!!next-input-character;          !!!next-input-character;
2320          redo A;          redo A;
2321        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2322          !!!cp (142);          !!!cp (142);
2323          !!!parse-error (type => 'bogus comment');          !!!parse-error (type => 'bogus comment');
2324          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2325          !!!next-input-character;          !!!next-input-character;
2326    
2327          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2328    
2329          redo A;          redo A;
2330        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2331          !!!cp (143);          !!!cp (143);
2332          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2333          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2334          ## reconsume          ## reconsume
2335    
2336          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2337    
2338          redo A;          redo A;
2339        } else {        } else {
2340          !!!cp (144);          !!!cp (144);
2341          $self->{current_token}->{data} # comment          $self->{ct}->{data} # comment
2342              .= '-' . chr ($self->{next_char});              .= '-' . chr ($self->{nc});
2343          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2344          !!!next-input-character;          !!!next-input-character;
2345          redo A;          redo A;
2346        }        }
2347      } elsif ($self->{state} == COMMENT_STATE) {      } elsif ($self->{state} == COMMENT_STATE) {
2348        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2349          !!!cp (145);          !!!cp (145);
2350          $self->{state} = COMMENT_END_DASH_STATE;          $self->{state} = COMMENT_END_DASH_STATE;
2351          !!!next-input-character;          !!!next-input-character;
2352          redo A;          redo A;
2353        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2354          !!!cp (146);          !!!cp (146);
2355          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2356          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2357          ## reconsume          ## reconsume
2358    
2359          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2360    
2361          redo A;          redo A;
2362        } else {        } else {
2363          !!!cp (147);          !!!cp (147);
2364          $self->{current_token}->{data} .= chr ($self->{next_char}); # comment          $self->{ct}->{data} .= chr ($self->{nc}); # comment
2365            $self->{read_until}->($self->{ct}->{data},
2366                                  q[-],
2367                                  length $self->{ct}->{data});
2368    
2369          ## Stay in the state          ## Stay in the state
2370          !!!next-input-character;          !!!next-input-character;
2371          redo A;          redo A;
2372        }        }
2373      } elsif ($self->{state} == COMMENT_END_DASH_STATE) {      } elsif ($self->{state} == COMMENT_END_DASH_STATE) {
2374        if ($self->{next_char} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2375          !!!cp (148);          !!!cp (148);
2376          $self->{state} = COMMENT_END_STATE;          $self->{state} = COMMENT_END_STATE;
2377          !!!next-input-character;          !!!next-input-character;
2378          redo A;          redo A;
2379        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2380          !!!cp (149);          !!!cp (149);
2381          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2382          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2383          ## reconsume          ## reconsume
2384    
2385          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2386    
2387          redo A;          redo A;
2388        } else {        } else {
2389          !!!cp (150);          !!!cp (150);
2390          $self->{current_token}->{data} .= '-' . chr ($self->{next_char}); # comment          $self->{ct}->{data} .= '-' . chr ($self->{nc}); # comment
2391          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2392          !!!next-input-character;          !!!next-input-character;
2393          redo A;          redo A;
2394        }        }
2395      } elsif ($self->{state} == COMMENT_END_STATE) {      } elsif ($self->{state} == COMMENT_END_STATE) {
2396        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2397          !!!cp (151);          !!!cp (151);
2398          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2399          !!!next-input-character;          !!!next-input-character;
2400    
2401          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2402    
2403          redo A;          redo A;
2404        } elsif ($self->{next_char} == 0x002D) { # -        } elsif ($self->{nc} == 0x002D) { # -
2405          !!!cp (152);          !!!cp (152);
2406          !!!parse-error (type => 'dash in comment');          !!!parse-error (type => 'dash in comment',
2407          $self->{current_token}->{data} .= '-'; # comment                          line => $self->{line_prev},
2408                            column => $self->{column_prev});
2409            $self->{ct}->{data} .= '-'; # comment
2410          ## Stay in the state          ## Stay in the state
2411          !!!next-input-character;          !!!next-input-character;
2412          redo A;          redo A;
2413        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2414          !!!cp (153);          !!!cp (153);
2415          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2416          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2417          ## reconsume          ## reconsume
2418    
2419          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2420    
2421          redo A;          redo A;
2422        } else {        } else {
2423          !!!cp (154);          !!!cp (154);
2424          !!!parse-error (type => 'dash in comment');          !!!parse-error (type => 'dash in comment',
2425          $self->{current_token}->{data} .= '--' . chr ($self->{next_char}); # comment                          line => $self->{line_prev},
2426                            column => $self->{column_prev});
2427            $self->{ct}->{data} .= '--' . chr ($self->{nc}); # comment
2428          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
2429          !!!next-input-character;          !!!next-input-character;
2430          redo A;          redo A;
2431        }        }
2432      } elsif ($self->{state} == DOCTYPE_STATE) {      } elsif ($self->{state} == DOCTYPE_STATE) {
2433        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2434          !!!cp (155);          !!!cp (155);
2435          $self->{state} = BEFORE_DOCTYPE_NAME_STATE;          $self->{state} = BEFORE_DOCTYPE_NAME_STATE;
2436          !!!next-input-character;          !!!next-input-character;
# Line 1637  sub _get_next_token ($) { Line 2443  sub _get_next_token ($) {
2443          redo A;          redo A;
2444        }        }
2445      } elsif ($self->{state} == BEFORE_DOCTYPE_NAME_STATE) {      } elsif ($self->{state} == BEFORE_DOCTYPE_NAME_STATE) {
2446        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2447          !!!cp (157);          !!!cp (157);
2448          ## Stay in the state          ## Stay in the state
2449          !!!next-input-character;          !!!next-input-character;
2450          redo A;          redo A;
2451        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2452          !!!cp (158);          !!!cp (158);
2453          !!!parse-error (type => 'no DOCTYPE name');          !!!parse-error (type => 'no DOCTYPE name');
2454          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2455          !!!next-input-character;          !!!next-input-character;
2456    
2457          !!!emit ({type => DOCTYPE_TOKEN, quirks => 1});          !!!emit ($self->{ct}); # DOCTYPE (quirks)
2458    
2459          redo A;          redo A;
2460        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2461          !!!cp (159);          !!!cp (159);
2462          !!!parse-error (type => 'no DOCTYPE name');          !!!parse-error (type => 'no DOCTYPE name');
2463          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2464          ## reconsume          ## reconsume
2465    
2466          !!!emit ({type => DOCTYPE_TOKEN, quirks => 1});          !!!emit ($self->{ct}); # DOCTYPE (quirks)
2467    
2468          redo A;          redo A;
2469        } else {        } else {
2470          !!!cp (160);          !!!cp (160);
2471          $self->{current_token}          $self->{ct}->{name} = chr $self->{nc};
2472              = {type => DOCTYPE_TOKEN,          delete $self->{ct}->{quirks};
                name => chr ($self->{next_char}),  
                #quirks => 0,  
               };  
2473  ## ISSUE: "Set the token's name name to the" in the spec  ## ISSUE: "Set the token's name name to the" in the spec
2474          $self->{state} = DOCTYPE_NAME_STATE;          $self->{state} = DOCTYPE_NAME_STATE;
2475          !!!next-input-character;          !!!next-input-character;
# Line 1678  sub _get_next_token ($) { Line 2477  sub _get_next_token ($) {
2477        }        }
2478      } elsif ($self->{state} == DOCTYPE_NAME_STATE) {      } elsif ($self->{state} == DOCTYPE_NAME_STATE) {
2479  ## ISSUE: Redundant "First," in the spec.  ## ISSUE: Redundant "First," in the spec.
2480        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2481          !!!cp (161);          !!!cp (161);
2482          $self->{state} = AFTER_DOCTYPE_NAME_STATE;          $self->{state} = AFTER_DOCTYPE_NAME_STATE;
2483          !!!next-input-character;          !!!next-input-character;
2484          redo A;          redo A;
2485        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2486          !!!cp (162);          !!!cp (162);
2487          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2488          !!!next-input-character;          !!!next-input-character;
2489    
2490          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2491    
2492          redo A;          redo A;
2493        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2494          !!!cp (163);          !!!cp (163);
2495          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2496          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2497          ## reconsume          ## reconsume
2498    
2499          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2500          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2501    
2502          redo A;          redo A;
2503        } else {        } else {
2504          !!!cp (164);          !!!cp (164);
2505          $self->{current_token}->{name}          $self->{ct}->{name}
2506            .= chr ($self->{next_char}); # DOCTYPE            .= chr ($self->{nc}); # DOCTYPE
2507          ## Stay in the state          ## Stay in the state
2508          !!!next-input-character;          !!!next-input-character;
2509          redo A;          redo A;
2510        }        }
2511      } elsif ($self->{state} == AFTER_DOCTYPE_NAME_STATE) {      } elsif ($self->{state} == AFTER_DOCTYPE_NAME_STATE) {
2512        if ($self->{next_char} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
           $self->{next_char} == 0x000A or # LF  
           $self->{next_char} == 0x000B or # VT  
           $self->{next_char} == 0x000C or # FF  
           $self->{next_char} == 0x0020) { # SP  
2513          !!!cp (165);          !!!cp (165);
2514          ## Stay in the state          ## Stay in the state
2515          !!!next-input-character;          !!!next-input-character;
2516          redo A;          redo A;
2517        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2518          !!!cp (166);          !!!cp (166);
2519          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2520          !!!next-input-character;          !!!next-input-character;
2521    
2522          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2523    
2524          redo A;          redo A;
2525        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2526          !!!cp (167);          !!!cp (167);
2527          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2528          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2529          ## reconsume          ## reconsume
2530    
2531          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2532          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2533    
2534          redo A;          redo A;
2535        } elsif ($self->{next_char} == 0x0050 or # P        } elsif ($self->{nc} == 0x0050 or # P
2536                 $self->{next_char} == 0x0070) { # p                 $self->{nc} == 0x0070) { # p
2537            $self->{state} = PUBLIC_STATE;
2538            $self->{s_kwd} = chr $self->{nc};
2539          !!!next-input-character;          !!!next-input-character;
2540          if ($self->{next_char} == 0x0055 or # U          redo A;
2541              $self->{next_char} == 0x0075) { # u        } elsif ($self->{nc} == 0x0053 or # S
2542            !!!next-input-character;                 $self->{nc} == 0x0073) { # s
2543            if ($self->{next_char} == 0x0042 or # B          $self->{state} = SYSTEM_STATE;
2544                $self->{next_char} == 0x0062) { # b          $self->{s_kwd} = chr $self->{nc};
             !!!next-input-character;  
             if ($self->{next_char} == 0x004C or # L  
                 $self->{next_char} == 0x006C) { # l  
               !!!next-input-character;  
               if ($self->{next_char} == 0x0049 or # I  
                   $self->{next_char} == 0x0069) { # i  
                 !!!next-input-character;  
                 if ($self->{next_char} == 0x0043 or # C  
                     $self->{next_char} == 0x0063) { # c  
                   !!!cp (168);  
                   $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;  
                   !!!next-input-character;  
                   redo A;  
                 } else {  
                   !!!cp (169);  
                 }  
               } else {  
                 !!!cp (170);  
               }  
             } else {  
               !!!cp (171);  
             }  
           } else {  
             !!!cp (172);  
           }  
         } else {  
           !!!cp (173);  
         }  
   
         #  
       } elsif ($self->{next_char} == 0x0053 or # S  
                $self->{next_char} == 0x0073) { # s  
2545          !!!next-input-character;          !!!next-input-character;
2546          if ($self->{next_char} == 0x0059 or # Y          redo A;
             $self->{next_char} == 0x0079) { # y  
           !!!next-input-character;  
           if ($self->{next_char} == 0x0053 or # S  
               $self->{next_char} == 0x0073) { # s  
             !!!next-input-character;  
             if ($self->{next_char} == 0x0054 or # T  
                 $self->{next_char} == 0x0074) { # t  
               !!!next-input-character;  
               if ($self->{next_char} == 0x0045 or # E  
                   $self->{next_char} == 0x0065) { # e  
                 !!!next-input-character;  
                 if ($self->{next_char} == 0x004D or # M  
                     $self->{next_char} == 0x006D) { # m  
                   !!!cp (174);  
                   $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;  
                   !!!next-input-character;  
                   redo A;  
                 } else {  
                   !!!cp (175);  
                 }  
               } else {  
                 !!!cp (176);  
               }  
             } else {  
               !!!cp (177);  
             }  
           } else {  
             !!!cp (178);  
           }  
         } else {  
           !!!cp (179);  
         }  
   
         #  
2547        } else {        } else {
2548          !!!cp (180);          !!!cp (180);
2549            !!!parse-error (type => 'string after DOCTYPE name');
2550            $self->{ct}->{quirks} = 1;
2551    
2552            $self->{state} = BOGUS_DOCTYPE_STATE;
2553          !!!next-input-character;          !!!next-input-character;
2554          #          redo A;
2555        }        }
2556        } elsif ($self->{state} == PUBLIC_STATE) {
2557          ## ASCII case-insensitive
2558          if ($self->{nc} == [
2559                undef,
2560                0x0055, # U
2561                0x0042, # B
2562                0x004C, # L
2563                0x0049, # I
2564              ]->[length $self->{s_kwd}] or
2565              $self->{nc} == [
2566                undef,
2567                0x0075, # u
2568                0x0062, # b
2569                0x006C, # l
2570                0x0069, # i
2571              ]->[length $self->{s_kwd}]) {
2572            !!!cp (175);
2573            ## Stay in the state.
2574            $self->{s_kwd} .= chr $self->{nc};
2575            !!!next-input-character;
2576            redo A;
2577          } elsif ((length $self->{s_kwd}) == 5 and
2578                   ($self->{nc} == 0x0043 or # C
2579                    $self->{nc} == 0x0063)) { # c
2580            !!!cp (168);
2581            $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2582            !!!next-input-character;
2583            redo A;
2584          } else {
2585            !!!cp (169);
2586            !!!parse-error (type => 'string after DOCTYPE name',
2587                            line => $self->{line_prev},
2588                            column => $self->{column_prev} + 1 - length $self->{s_kwd});
2589            $self->{ct}->{quirks} = 1;
2590    
2591        !!!parse-error (type => 'string after DOCTYPE name');          $self->{state} = BOGUS_DOCTYPE_STATE;
2592        $self->{current_token}->{quirks} = 1;          ## Reconsume.
2593            redo A;
2594          }
2595        } elsif ($self->{state} == SYSTEM_STATE) {
2596          ## ASCII case-insensitive
2597          if ($self->{nc} == [
2598                undef,
2599                0x0059, # Y
2600                0x0053, # S
2601                0x0054, # T
2602                0x0045, # E
2603              ]->[length $self->{s_kwd}] or
2604              $self->{nc} == [
2605                undef,
2606                0x0079, # y
2607                0x0073, # s
2608                0x0074, # t
2609                0x0065, # e
2610              ]->[length $self->{s_kwd}]) {
2611            !!!cp (170);
2612            ## Stay in the state.
2613            $self->{s_kwd} .= chr $self->{nc};
2614            !!!next-input-character;
2615            redo A;
2616          } elsif ((length $self->{s_kwd}) == 5 and
2617                   ($self->{nc} == 0x004D or # M
2618                    $self->{nc} == 0x006D)) { # m
2619            !!!cp (171);
2620            $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2621            !!!next-input-character;
2622            redo A;
2623          } else {
2624            !!!cp (172);
2625            !!!parse-error (type => 'string after DOCTYPE name',
2626                            line => $self->{line_prev},
2627                            column => $self->{column_prev} + 1 - length $self->{s_kwd});
2628            $self->{ct}->{quirks} = 1;
2629    
2630        $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2631        # next-input-character is already done          ## Reconsume.
2632        redo A;          redo A;
2633          }
2634      } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {      } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2635        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2636          !!!cp (181);          !!!cp (181);
2637          ## Stay in the state          ## Stay in the state
2638          !!!next-input-character;          !!!next-input-character;
2639          redo A;          redo A;
2640        } elsif ($self->{next_char} eq 0x0022) { # "        } elsif ($self->{nc} eq 0x0022) { # "
2641          !!!cp (182);          !!!cp (182);
2642          $self->{current_token}->{public_identifier} = ''; # DOCTYPE          $self->{ct}->{pubid} = ''; # DOCTYPE
2643          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;
2644          !!!next-input-character;          !!!next-input-character;
2645          redo A;          redo A;
2646        } elsif ($self->{next_char} eq 0x0027) { # '        } elsif ($self->{nc} eq 0x0027) { # '
2647          !!!cp (183);          !!!cp (183);
2648          $self->{current_token}->{public_identifier} = ''; # DOCTYPE          $self->{ct}->{pubid} = ''; # DOCTYPE
2649          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;
2650          !!!next-input-character;          !!!next-input-character;
2651          redo A;          redo A;
2652        } elsif ($self->{next_char} eq 0x003E) { # >        } elsif ($self->{nc} eq 0x003E) { # >
2653          !!!cp (184);          !!!cp (184);
2654          !!!parse-error (type => 'no PUBLIC literal');          !!!parse-error (type => 'no PUBLIC literal');
2655    
2656          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2657          !!!next-input-character;          !!!next-input-character;
2658    
2659          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2660          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2661    
2662          redo A;          redo A;
2663        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2664          !!!cp (185);          !!!cp (185);
2665          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2666    
2667          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2668          ## reconsume          ## reconsume
2669    
2670          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2671          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2672    
2673          redo A;          redo A;
2674        } else {        } else {
2675          !!!cp (186);          !!!cp (186);
2676          !!!parse-error (type => 'string after PUBLIC');          !!!parse-error (type => 'string after PUBLIC');
2677          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2678    
2679          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2680          !!!next-input-character;          !!!next-input-character;
2681          redo A;          redo A;
2682        }        }
2683      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2684        if ($self->{next_char} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
2685          !!!cp (187);          !!!cp (187);
2686          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2687          !!!next-input-character;          !!!next-input-character;
2688          redo A;          redo A;
2689        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2690          !!!cp (188);          !!!cp (188);
2691          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2692    
2693          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2694          !!!next-input-character;          !!!next-input-character;
2695    
2696          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2697          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2698    
2699          redo A;          redo A;
2700        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2701          !!!cp (189);          !!!cp (189);
2702          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2703    
2704          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2705          ## reconsume          ## reconsume
2706    
2707          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2708          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2709    
2710          redo A;          redo A;
2711        } else {        } else {
2712          !!!cp (190);          !!!cp (190);
2713          $self->{current_token}->{public_identifier} # DOCTYPE          $self->{ct}->{pubid} # DOCTYPE
2714              .= chr $self->{next_char};              .= chr $self->{nc};
2715            $self->{read_until}->($self->{ct}->{pubid}, q[">],
2716                                  length $self->{ct}->{pubid});
2717    
2718          ## Stay in the state          ## Stay in the state
2719          !!!next-input-character;          !!!next-input-character;
2720          redo A;          redo A;
2721        }        }
2722      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {
2723        if ($self->{next_char} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
2724          !!!cp (191);          !!!cp (191);
2725          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2726          !!!next-input-character;          !!!next-input-character;
2727          redo A;          redo A;
2728        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2729          !!!cp (192);          !!!cp (192);
2730          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2731    
2732          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2733          !!!next-input-character;          !!!next-input-character;
2734    
2735          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2736          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2737    
2738          redo A;          redo A;
2739        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2740          !!!cp (193);          !!!cp (193);
2741          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2742    
2743          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2744          ## reconsume          ## reconsume
2745    
2746          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2747          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2748    
2749          redo A;          redo A;
2750        } else {        } else {
2751          !!!cp (194);          !!!cp (194);
2752          $self->{current_token}->{public_identifier} # DOCTYPE          $self->{ct}->{pubid} # DOCTYPE
2753              .= chr $self->{next_char};              .= chr $self->{nc};
2754            $self->{read_until}->($self->{ct}->{pubid}, q['>],
2755                                  length $self->{ct}->{pubid});
2756    
2757          ## Stay in the state          ## Stay in the state
2758          !!!next-input-character;          !!!next-input-character;
2759          redo A;          redo A;
2760        }        }
2761      } elsif ($self->{state} == AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {      } elsif ($self->{state} == AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2762        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2763          !!!cp (195);          !!!cp (195);
2764          ## Stay in the state          ## Stay in the state
2765          !!!next-input-character;          !!!next-input-character;
2766          redo A;          redo A;
2767        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
2768          !!!cp (196);          !!!cp (196);
2769          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2770          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2771          !!!next-input-character;          !!!next-input-character;
2772          redo A;          redo A;
2773        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
2774          !!!cp (197);          !!!cp (197);
2775          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2776          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2777          !!!next-input-character;          !!!next-input-character;
2778          redo A;          redo A;
2779        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2780          !!!cp (198);          !!!cp (198);
2781          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2782          !!!next-input-character;          !!!next-input-character;
2783    
2784          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2785    
2786          redo A;          redo A;
2787        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2788          !!!cp (199);          !!!cp (199);
2789          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2790    
2791          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2792          ## reconsume          ## reconsume
2793    
2794          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2795          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2796    
2797          redo A;          redo A;
2798        } else {        } else {
2799          !!!cp (200);          !!!cp (200);
2800          !!!parse-error (type => 'string after PUBLIC literal');          !!!parse-error (type => 'string after PUBLIC literal');
2801          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2802    
2803          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2804          !!!next-input-character;          !!!next-input-character;
2805          redo A;          redo A;
2806        }        }
2807      } elsif ($self->{state} == BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {      } elsif ($self->{state} == BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2808        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2809          !!!cp (201);          !!!cp (201);
2810          ## Stay in the state          ## Stay in the state
2811          !!!next-input-character;          !!!next-input-character;
2812          redo A;          redo A;
2813        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
2814          !!!cp (202);          !!!cp (202);
2815          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2816          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2817          !!!next-input-character;          !!!next-input-character;
2818          redo A;          redo A;
2819        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
2820          !!!cp (203);          !!!cp (203);
2821          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{ct}->{sysid} = ''; # DOCTYPE
2822          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2823          !!!next-input-character;          !!!next-input-character;
2824          redo A;          redo A;
2825        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2826          !!!cp (204);          !!!cp (204);
2827          !!!parse-error (type => 'no SYSTEM literal');          !!!parse-error (type => 'no SYSTEM literal');
2828          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2829          !!!next-input-character;          !!!next-input-character;
2830    
2831          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2832          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2833    
2834          redo A;          redo A;
2835        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2836          !!!cp (205);          !!!cp (205);
2837          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2838    
2839          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2840          ## reconsume          ## reconsume
2841    
2842          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2843          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2844    
2845          redo A;          redo A;
2846        } else {        } else {
2847          !!!cp (206);          !!!cp (206);
2848          !!!parse-error (type => 'string after SYSTEM');          !!!parse-error (type => 'string after SYSTEM');
2849          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2850    
2851          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2852          !!!next-input-character;          !!!next-input-character;
2853          redo A;          redo A;
2854        }        }
2855      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2856        if ($self->{next_char} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
2857          !!!cp (207);          !!!cp (207);
2858          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2859          !!!next-input-character;          !!!next-input-character;
2860          redo A;          redo A;
2861        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2862          !!!cp (208);          !!!cp (208);
2863          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2864    
2865          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2866          !!!next-input-character;          !!!next-input-character;
2867    
2868          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2869          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2870    
2871          redo A;          redo A;
2872        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2873          !!!cp (209);          !!!cp (209);
2874          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2875    
2876          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2877          ## reconsume          ## reconsume
2878    
2879          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2880          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2881    
2882          redo A;          redo A;
2883        } else {        } else {
2884          !!!cp (210);          !!!cp (210);
2885          $self->{current_token}->{system_identifier} # DOCTYPE          $self->{ct}->{sysid} # DOCTYPE
2886              .= chr $self->{next_char};              .= chr $self->{nc};
2887            $self->{read_until}->($self->{ct}->{sysid}, q[">],
2888                                  length $self->{ct}->{sysid});
2889    
2890          ## Stay in the state          ## Stay in the state
2891          !!!next-input-character;          !!!next-input-character;
2892          redo A;          redo A;
2893        }        }
2894      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {
2895        if ($self->{next_char} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
2896          !!!cp (211);          !!!cp (211);
2897          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2898          !!!next-input-character;          !!!next-input-character;
2899          redo A;          redo A;
2900        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2901          !!!cp (212);          !!!cp (212);
2902          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2903    
2904          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2905          !!!next-input-character;          !!!next-input-character;
2906    
2907          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2908          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2909    
2910          redo A;          redo A;
2911        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2912          !!!cp (213);          !!!cp (213);
2913          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2914    
2915          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2916          ## reconsume          ## reconsume
2917    
2918          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2919          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2920    
2921          redo A;          redo A;
2922        } else {        } else {
2923          !!!cp (214);          !!!cp (214);
2924          $self->{current_token}->{system_identifier} # DOCTYPE          $self->{ct}->{sysid} # DOCTYPE
2925              .= chr $self->{next_char};              .= chr $self->{nc};
2926            $self->{read_until}->($self->{ct}->{sysid}, q['>],
2927                                  length $self->{ct}->{sysid});
2928    
2929          ## Stay in the state          ## Stay in the state
2930          !!!next-input-character;          !!!next-input-character;
2931          redo A;          redo A;
2932        }        }
2933      } elsif ($self->{state} == AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {      } elsif ($self->{state} == AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2934        if ({        if ($is_space->{$self->{nc}}) {
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_char}}) {  
2935          !!!cp (215);          !!!cp (215);
2936          ## Stay in the state          ## Stay in the state
2937          !!!next-input-character;          !!!next-input-character;
2938          redo A;          redo A;
2939        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2940          !!!cp (216);          !!!cp (216);
2941          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2942          !!!next-input-character;          !!!next-input-character;
2943    
2944          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2945    
2946          redo A;          redo A;
2947        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2948          !!!cp (217);          !!!cp (217);
2949          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
   
2950          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2951          ## reconsume          ## reconsume
2952    
2953          $self->{current_token}->{quirks} = 1;          $self->{ct}->{quirks} = 1;
2954          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2955    
2956          redo A;          redo A;
2957        } else {        } else {
2958          !!!cp (218);          !!!cp (218);
2959          !!!parse-error (type => 'string after SYSTEM literal');          !!!parse-error (type => 'string after SYSTEM literal');
2960          #$self->{current_token}->{quirks} = 1;          #$self->{ct}->{quirks} = 1;
2961    
2962          $self->{state} = BOGUS_DOCTYPE_STATE;          $self->{state} = BOGUS_DOCTYPE_STATE;
2963          !!!next-input-character;          !!!next-input-character;
2964          redo A;          redo A;
2965        }        }
2966      } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {      } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {
2967        if ($self->{next_char} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2968          !!!cp (219);          !!!cp (219);
2969          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2970          !!!next-input-character;          !!!next-input-character;
2971    
2972          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2973    
2974          redo A;          redo A;
2975        } elsif ($self->{next_char} == -1) {        } elsif ($self->{nc} == -1) {
2976          !!!cp (220);          !!!cp (220);
2977          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2978          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2979          ## reconsume          ## reconsume
2980    
2981          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2982    
2983          redo A;          redo A;
2984        } else {        } else {
2985          !!!cp (221);          !!!cp (221);
2986            my $s = '';
2987            $self->{read_until}->($s, q[>], 0);
2988    
2989          ## Stay in the state          ## Stay in the state
2990          !!!next-input-character;          !!!next-input-character;
2991          redo A;          redo A;
2992        }        }
2993      } else {      } elsif ($self->{state} == CDATA_SECTION_STATE) {
2994        die "$0: $self->{state}: Unknown state";        ## NOTE: "CDATA section state" in the state is jointly implemented
2995      }        ## by three states, |CDATA_SECTION_STATE|, |CDATA_SECTION_MSE1_STATE|,
2996    } # A          ## and |CDATA_SECTION_MSE2_STATE|.
2997          
2998    die "$0: _get_next_token: unexpected case";        if ($self->{nc} == 0x005D) { # ]
2999  } # _get_next_token          !!!cp (221.1);
3000            $self->{state} = CDATA_SECTION_MSE1_STATE;
3001            !!!next-input-character;
3002            redo A;
3003          } elsif ($self->{nc} == -1) {
3004            $self->{state} = DATA_STATE;
3005            !!!next-input-character;
3006            if (length $self->{ct}->{data}) { # character
3007              !!!cp (221.2);
3008              !!!emit ($self->{ct}); # character
3009            } else {
3010              !!!cp (221.3);
3011              ## No token to emit. $self->{ct} is discarded.
3012            }        
3013            redo A;
3014          } else {
3015            !!!cp (221.4);
3016            $self->{ct}->{data} .= chr $self->{nc};
3017            $self->{read_until}->($self->{ct}->{data},
3018                                  q<]>,
3019                                  length $self->{ct}->{data});
3020    
3021  sub _tokenize_attempt_to_consume_an_entity ($$$) {          ## Stay in the state.
3022    my ($self, $in_attr, $additional) = @_;          !!!next-input-character;
3023            redo A;
3024          }
3025    
3026    if ({        ## ISSUE: "text tokens" in spec.
3027         0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, # HT, LF, VT, FF,      } elsif ($self->{state} == CDATA_SECTION_MSE1_STATE) {
3028         0x0020 => 1, 0x003C => 1, 0x0026 => 1, -1 => 1, # SP, <, & # 0x000D # CR        if ($self->{nc} == 0x005D) { # ]
3029         $additional => 1,          !!!cp (221.5);
3030        }->{$self->{next_char}}) {          $self->{state} = CDATA_SECTION_MSE2_STATE;
3031      !!!cp (1001);          !!!next-input-character;
3032      ## Don't consume          redo A;
3033      ## No error        } else {
3034      return undef;          !!!cp (221.6);
3035    } elsif ($self->{next_char} == 0x0023) { # #          $self->{ct}->{data} .= ']';
3036      !!!next-input-character;          $self->{state} = CDATA_SECTION_STATE;
3037      if ($self->{next_char} == 0x0078 or # x          ## Reconsume.
3038          $self->{next_char} == 0x0058) { # X          redo A;
3039        my $code;        }
3040        X: {      } elsif ($self->{state} == CDATA_SECTION_MSE2_STATE) {
3041          my $x_char = $self->{next_char};        if ($self->{nc} == 0x003E) { # >
3042          !!!next-input-character;          $self->{state} = DATA_STATE;
3043          if (0x0030 <= $self->{next_char} and          !!!next-input-character;
3044              $self->{next_char} <= 0x0039) { # 0..9          if (length $self->{ct}->{data}) { # character
3045            !!!cp (1002);            !!!cp (221.7);
3046            $code ||= 0;            !!!emit ($self->{ct}); # character
           $code *= 0x10;  
           $code += $self->{next_char} - 0x0030;  
           redo X;  
         } elsif (0x0061 <= $self->{next_char} and  
                  $self->{next_char} <= 0x0066) { # a..f  
           !!!cp (1003);  
           $code ||= 0;  
           $code *= 0x10;  
           $code += $self->{next_char} - 0x0060 + 9;  
           redo X;  
         } elsif (0x0041 <= $self->{next_char} and  
                  $self->{next_char} <= 0x0046) { # A..F  
           !!!cp (1004);  
           $code ||= 0;  
           $code *= 0x10;  
           $code += $self->{next_char} - 0x0040 + 9;  
           redo X;  
         } elsif (not defined $code) { # no hexadecimal digit  
           !!!cp (1005);  
           !!!parse-error (type => 'bare hcro');  
           !!!back-next-input-character ($x_char, $self->{next_char});  
           $self->{next_char} = 0x0023; # #  
           return undef;  
         } elsif ($self->{next_char} == 0x003B) { # ;  
           !!!cp (1006);  
           !!!next-input-character;  
3047          } else {          } else {
3048            !!!cp (1007);            !!!cp (221.8);
3049            !!!parse-error (type => 'no refc');            ## No token to emit. $self->{ct} is discarded.
3050          }          }
3051            redo A;
3052          } elsif ($self->{nc} == 0x005D) { # ]
3053            !!!cp (221.9); # character
3054            $self->{ct}->{data} .= ']'; ## Add first "]" of "]]]".
3055            ## Stay in the state.
3056            !!!next-input-character;
3057            redo A;
3058          } else {
3059            !!!cp (221.11);
3060            $self->{ct}->{data} .= ']]'; # character
3061            $self->{state} = CDATA_SECTION_STATE;
3062            ## Reconsume.
3063            redo A;
3064          }
3065        } elsif ($self->{state} == ENTITY_STATE) {
3066          if ($is_space->{$self->{nc}} or
3067              {
3068                0x003C => 1, 0x0026 => 1, -1 => 1, # <, &
3069                $self->{entity_add} => 1,
3070              }->{$self->{nc}}) {
3071            !!!cp (1001);
3072            ## Don't consume
3073            ## No error
3074            ## Return nothing.
3075            #
3076          } elsif ($self->{nc} == 0x0023) { # #
3077            !!!cp (999);
3078            $self->{state} = ENTITY_HASH_STATE;
3079            $self->{s_kwd} = '#';
3080            !!!next-input-character;
3081            redo A;
3082          } elsif ((0x0041 <= $self->{nc} and
3083                    $self->{nc} <= 0x005A) or # A..Z
3084                   (0x0061 <= $self->{nc} and
3085                    $self->{nc} <= 0x007A)) { # a..z
3086            !!!cp (998);
3087            require Whatpm::_NamedEntityList;
3088            $self->{state} = ENTITY_NAME_STATE;
3089            $self->{s_kwd} = chr $self->{nc};
3090            $self->{entity__value} = $self->{s_kwd};
3091            $self->{entity__match} = 0;
3092            !!!next-input-character;
3093            redo A;
3094          } else {
3095            !!!cp (1027);
3096            !!!parse-error (type => 'bare ero');
3097            ## Return nothing.
3098            #
3099          }
3100    
3101          if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {        ## NOTE: No character is consumed by the "consume a character
3102            !!!cp (1008);        ## reference" algorithm.  In other word, there is an "&" character
3103            !!!parse-error (type => sprintf 'invalid character reference:U+%04X', $code);        ## that does not introduce a character reference, which would be
3104            $code = 0xFFFD;        ## appended to the parent element or the attribute value in later
3105          } elsif ($code > 0x10FFFF) {        ## process of the tokenizer.
3106            !!!cp (1009);  
3107            !!!parse-error (type => sprintf 'invalid character reference:U-%08X', $code);        if ($self->{prev_state} == DATA_STATE) {
3108            $code = 0xFFFD;          !!!cp (997);
3109          } elsif ($code == 0x000D) {          $self->{state} = $self->{prev_state};
3110            !!!cp (1010);          ## Reconsume.
3111            !!!parse-error (type => 'CR character reference');          !!!emit ({type => CHARACTER_TOKEN, data => '&',
3112            $code = 0x000A;                    line => $self->{line_prev},
3113          } elsif (0x80 <= $code and $code <= 0x9F) {                    column => $self->{column_prev},
3114            !!!cp (1011);                   });
3115            !!!parse-error (type => sprintf 'C1 character reference:U+%04X', $code);          redo A;
3116            $code = $c1_entity_char->{$code};        } else {
3117          }          !!!cp (996);
3118            $self->{ca}->{value} .= '&';
3119          return {type => CHARACTER_TOKEN, data => chr $code,          $self->{state} = $self->{prev_state};
3120                  has_reference => 1};          ## Reconsume.
3121        } # X          redo A;
3122      } elsif (0x0030 <= $self->{next_char} and        }
3123               $self->{next_char} <= 0x0039) { # 0..9      } elsif ($self->{state} == ENTITY_HASH_STATE) {
3124        my $code = $self->{next_char} - 0x0030;        if ($self->{nc} == 0x0078 or # x
3125        !!!next-input-character;            $self->{nc} == 0x0058) { # X
3126                  !!!cp (995);
3127        while (0x0030 <= $self->{next_char} and          $self->{state} = HEXREF_X_STATE;
3128                  $self->{next_char} <= 0x0039) { # 0..9          $self->{s_kwd} .= chr $self->{nc};
3129            !!!next-input-character;
3130            redo A;
3131          } elsif (0x0030 <= $self->{nc} and
3132                   $self->{nc} <= 0x0039) { # 0..9
3133            !!!cp (994);
3134            $self->{state} = NCR_NUM_STATE;
3135            $self->{s_kwd} = $self->{nc} - 0x0030;
3136            !!!next-input-character;
3137            redo A;
3138          } else {
3139            !!!parse-error (type => 'bare nero',
3140                            line => $self->{line_prev},
3141                            column => $self->{column_prev} - 1);
3142    
3143            ## NOTE: According to the spec algorithm, nothing is returned,
3144            ## and then "&#" is appended to the parent element or the attribute
3145            ## value in the later processing.
3146    
3147            if ($self->{prev_state} == DATA_STATE) {
3148              !!!cp (1019);
3149              $self->{state} = $self->{prev_state};
3150              ## Reconsume.
3151              !!!emit ({type => CHARACTER_TOKEN,
3152                        data => '&#',
3153                        line => $self->{line_prev},
3154                        column => $self->{column_prev} - 1,
3155                       });
3156              redo A;
3157            } else {
3158              !!!cp (993);
3159              $self->{ca}->{value} .= '&#';
3160              $self->{state} = $self->{prev_state};
3161              ## Reconsume.
3162              redo A;
3163            }
3164          }
3165        } elsif ($self->{state} == NCR_NUM_STATE) {
3166          if (0x0030 <= $self->{nc} and
3167              $self->{nc} <= 0x0039) { # 0..9
3168          !!!cp (1012);          !!!cp (1012);
3169          $code *= 10;          $self->{s_kwd} *= 10;
3170          $code += $self->{next_char} - 0x0030;          $self->{s_kwd} += $self->{nc} - 0x0030;
3171                    
3172            ## Stay in the state.
3173          !!!next-input-character;          !!!next-input-character;
3174        }          redo A;
3175          } elsif ($self->{nc} == 0x003B) { # ;
       if ($self->{next_char} == 0x003B) { # ;  
3176          !!!cp (1013);          !!!cp (1013);
3177          !!!next-input-character;          !!!next-input-character;
3178            #
3179        } else {        } else {
3180          !!!cp (1014);          !!!cp (1014);
3181          !!!parse-error (type => 'no refc');          !!!parse-error (type => 'no refc');
3182            ## Reconsume.
3183            #
3184        }        }
3185    
3186        if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {        my $code = $self->{s_kwd};
3187          my $l = $self->{line_prev};
3188          my $c = $self->{column_prev};
3189          if ($charref_map->{$code}) {
3190          !!!cp (1015);          !!!cp (1015);
3191          !!!parse-error (type => sprintf 'invalid character reference:U+%04X', $code);          !!!parse-error (type => 'invalid character reference',
3192          $code = 0xFFFD;                          text => (sprintf 'U+%04X', $code),
3193                            line => $l, column => $c);
3194            $code = $charref_map->{$code};
3195        } elsif ($code > 0x10FFFF) {        } elsif ($code > 0x10FFFF) {
3196          !!!cp (1016);          !!!cp (1016);
3197          !!!parse-error (type => sprintf 'invalid character reference:U-%08X', $code);          !!!parse-error (type => 'invalid character reference',
3198                            text => (sprintf 'U-%08X', $code),
3199                            line => $l, column => $c);
3200          $code = 0xFFFD;          $code = 0xFFFD;
       } elsif ($code == 0x000D) {  
         !!!cp (1017);  
         !!!parse-error (type => 'CR character reference');  
         $code = 0x000A;  
       } elsif (0x80 <= $code and $code <= 0x9F) {  
         !!!cp (1018);  
         !!!parse-error (type => sprintf 'C1 character reference:U+%04X', $code);  
         $code = $c1_entity_char->{$code};  
3201        }        }
3202          
3203        return {type => CHARACTER_TOKEN, data => chr $code, has_reference => 1};        if ($self->{prev_state} == DATA_STATE) {
3204      } else {          !!!cp (992);
3205        !!!cp (1019);          $self->{state} = $self->{prev_state};
3206        !!!parse-error (type => 'bare nero');          ## Reconsume.
3207        !!!back-next-input-character ($self->{next_char});          !!!emit ({type => CHARACTER_TOKEN, data => chr $code,
3208        $self->{next_char} = 0x0023; # #                    line => $l, column => $c,
3209        return undef;                   });
3210      }          redo A;
3211    } elsif ((0x0041 <= $self->{next_char} and        } else {
3212              $self->{next_char} <= 0x005A) or          !!!cp (991);
3213             (0x0061 <= $self->{next_char} and          $self->{ca}->{value} .= chr $code;
3214              $self->{next_char} <= 0x007A)) {          $self->{ca}->{has_reference} = 1;
3215      my $entity_name = chr $self->{next_char};          $self->{state} = $self->{prev_state};
3216      !!!next-input-character;          ## Reconsume.
3217            redo A;
3218      my $value = $entity_name;        }
3219      my $match = 0;      } elsif ($self->{state} == HEXREF_X_STATE) {
3220      require Whatpm::_NamedEntityList;        if ((0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) or
3221      our $EntityChar;            (0x0041 <= $self->{nc} and $self->{nc} <= 0x0046) or
3222              (0x0061 <= $self->{nc} and $self->{nc} <= 0x0066)) {
3223      while (length $entity_name < 10 and          # 0..9, A..F, a..f
3224             ## NOTE: Some number greater than the maximum length of entity name          !!!cp (990);
3225             ((0x0041 <= $self->{next_char} and # a          $self->{state} = HEXREF_HEX_STATE;
3226               $self->{next_char} <= 0x005A) or # x          $self->{s_kwd} = 0;
3227              (0x0061 <= $self->{next_char} and # a          ## Reconsume.
3228               $self->{next_char} <= 0x007A) or # z          redo A;
3229              (0x0030 <= $self->{next_char} and # 0        } else {
3230               $self->{next_char} <= 0x0039) or # 9          !!!parse-error (type => 'bare hcro',
3231              $self->{next_char} == 0x003B)) { # ;                          line => $self->{line_prev},
3232        $entity_name .= chr $self->{next_char};                          column => $self->{column_prev} - 2);
3233        if (defined $EntityChar->{$entity_name}) {  
3234          if ($self->{next_char} == 0x003B) { # ;          ## NOTE: According to the spec algorithm, nothing is returned,
3235            !!!cp (1020);          ## and then "&#" followed by "X" or "x" is appended to the parent
3236            $value = $EntityChar->{$entity_name};          ## element or the attribute value in the later processing.
3237            $match = 1;  
3238            !!!next-input-character;          if ($self->{prev_state} == DATA_STATE) {
3239            last;            !!!cp (1005);
3240              $self->{state} = $self->{prev_state};
3241              ## Reconsume.
3242              !!!emit ({type => CHARACTER_TOKEN,
3243                        data => '&' . $self->{s_kwd},
3244                        line => $self->{line_prev},
3245                        column => $self->{column_prev} - length $self->{s_kwd},
3246                       });
3247              redo A;
3248          } else {          } else {
3249            !!!cp (1021);            !!!cp (989);
3250            $value = $EntityChar->{$entity_name};            $self->{ca}->{value} .= '&' . $self->{s_kwd};
3251            $match = -1;            $self->{state} = $self->{prev_state};
3252            !!!next-input-character;            ## Reconsume.
3253              redo A;
3254          }          }
3255        } else {        }
3256          !!!cp (1022);      } elsif ($self->{state} == HEXREF_HEX_STATE) {
3257          $value .= chr $self->{next_char};        if (0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) {
3258          $match *= 2;          # 0..9
3259            !!!cp (1002);
3260            $self->{s_kwd} *= 0x10;
3261            $self->{s_kwd} += $self->{nc} - 0x0030;
3262            ## Stay in the state.
3263            !!!next-input-character;
3264            redo A;
3265          } elsif (0x0061 <= $self->{nc} and
3266                   $self->{nc} <= 0x0066) { # a..f
3267            !!!cp (1003);
3268            $self->{s_kwd} *= 0x10;
3269            $self->{s_kwd} += $self->{nc} - 0x0060 + 9;
3270            ## Stay in the state.
3271            !!!next-input-character;
3272            redo A;
3273          } elsif (0x0041 <= $self->{nc} and
3274                   $self->{nc} <= 0x0046) { # A..F
3275            !!!cp (1004);
3276            $self->{s_kwd} *= 0x10;
3277            $self->{s_kwd} += $self->{nc} - 0x0040 + 9;
3278            ## Stay in the state.
3279            !!!next-input-character;
3280            redo A;
3281          } elsif ($self->{nc} == 0x003B) { # ;
3282            !!!cp (1006);
3283          !!!next-input-character;          !!!next-input-character;
3284            #
3285          } else {
3286            !!!cp (1007);
3287            !!!parse-error (type => 'no refc',
3288                            line => $self->{line},
3289                            column => $self->{column});
3290            ## Reconsume.
3291            #
3292        }        }
3293      }  
3294              my $code = $self->{s_kwd};
3295      if ($match > 0) {        my $l = $self->{line_prev};
3296        !!!cp (1023);        my $c = $self->{column_prev};
3297        return {type => CHARACTER_TOKEN, data => $value, has_reference => 1};        if ($charref_map->{$code}) {
3298      } elsif ($match < 0) {          !!!cp (1008);
3299        !!!parse-error (type => 'no refc');          !!!parse-error (type => 'invalid character reference',
3300        if ($in_attr and $match < -1) {                          text => (sprintf 'U+%04X', $code),
3301          !!!cp (1024);                          line => $l, column => $c);
3302          return {type => CHARACTER_TOKEN, data => '&'.$entity_name};          $code = $charref_map->{$code};
3303          } elsif ($code > 0x10FFFF) {
3304            !!!cp (1009);
3305            !!!parse-error (type => 'invalid character reference',
3306                            text => (sprintf 'U-%08X', $code),
3307                            line => $l, column => $c);
3308            $code = 0xFFFD;
3309          }
3310    
3311          if ($self->{prev_state} == DATA_STATE) {
3312            !!!cp (988);
3313            $self->{state} = $self->{prev_state};
3314            ## Reconsume.
3315            !!!emit ({type => CHARACTER_TOKEN, data => chr $code,
3316                      line => $l, column => $c,
3317                     });
3318            redo A;
3319          } else {
3320            !!!cp (987);
3321            $self->{ca}->{value} .= chr $code;
3322            $self->{ca}->{has_reference} = 1;
3323            $self->{state} = $self->{prev_state};
3324            ## Reconsume.
3325            redo A;
3326          }
3327        } elsif ($self->{state} == ENTITY_NAME_STATE) {
3328          if (length $self->{s_kwd} < 30 and
3329              ## NOTE: Some number greater than the maximum length of entity name
3330              ((0x0041 <= $self->{nc} and # a
3331                $self->{nc} <= 0x005A) or # x
3332               (0x0061 <= $self->{nc} and # a
3333                $self->{nc} <= 0x007A) or # z
3334               (0x0030 <= $self->{nc} and # 0
3335                $self->{nc} <= 0x0039) or # 9
3336               $self->{nc} == 0x003B)) { # ;
3337            our $EntityChar;
3338            $self->{s_kwd} .= chr $self->{nc};
3339            if (defined $EntityChar->{$self->{s_kwd}}) {
3340              if ($self->{nc} == 0x003B) { # ;
3341                !!!cp (1020);
3342                $self->{entity__value} = $EntityChar->{$self->{s_kwd}};
3343                $self->{entity__match} = 1;
3344                !!!next-input-character;
3345                #
3346              } else {
3347                !!!cp (1021);
3348                $self->{entity__value} = $EntityChar->{$self->{s_kwd}};
3349                $self->{entity__match} = -1;
3350                ## Stay in the state.
3351                !!!next-input-character;
3352                redo A;
3353              }
3354            } else {
3355              !!!cp (1022);
3356              $self->{entity__value} .= chr $self->{nc};
3357              $self->{entity__match} *= 2;
3358              ## Stay in the state.
3359              !!!next-input-character;
3360              redo A;
3361            }
3362          }
3363    
3364          my $data;
3365          my $has_ref;
3366          if ($self->{entity__match} > 0) {
3367            !!!cp (1023);
3368            $data = $self->{entity__value};
3369            $has_ref = 1;
3370            #
3371          } elsif ($self->{entity__match} < 0) {
3372            !!!parse-error (type => 'no refc');
3373            if ($self->{prev_state} != DATA_STATE and # in attribute
3374                $self->{entity__match} < -1) {
3375              !!!cp (1024);
3376              $data = '&' . $self->{s_kwd};
3377              #
3378            } else {
3379              !!!cp (1025);
3380              $data = $self->{entity__value};
3381              $has_ref = 1;
3382              #
3383            }
3384        } else {        } else {
3385          !!!cp (1025);          !!!cp (1026);
3386          return {type => CHARACTER_TOKEN, data => $value, has_reference => 1};          !!!parse-error (type => 'bare ero',
3387                            line => $self->{line_prev},
3388                            column => $self->{column_prev} - length $self->{s_kwd});
3389            $data = '&' . $self->{s_kwd};
3390            #
3391          }
3392      
3393          ## NOTE: In these cases, when a character reference is found,
3394          ## it is consumed and a character token is returned, or, otherwise,
3395          ## nothing is consumed and returned, according to the spec algorithm.
3396          ## In this implementation, anything that has been examined by the
3397          ## tokenizer is appended to the parent element or the attribute value
3398          ## as string, either literal string when no character reference or
3399          ## entity-replaced string otherwise, in this stage, since any characters
3400          ## that would not be consumed are appended in the data state or in an
3401          ## appropriate attribute value state anyway.
3402    
3403          if ($self->{prev_state} == DATA_STATE) {
3404            !!!cp (986);
3405            $self->{state} = $self->{prev_state};
3406            ## Reconsume.
3407            !!!emit ({type => CHARACTER_TOKEN,
3408                      data => $data,
3409                      line => $self->{line_prev},
3410                      column => $self->{column_prev} + 1 - length $self->{s_kwd},
3411                     });
3412            redo A;
3413          } else {
3414            !!!cp (985);
3415            $self->{ca}->{value} .= $data;
3416            $self->{ca}->{has_reference} = 1 if $has_ref;
3417            $self->{state} = $self->{prev_state};
3418            ## Reconsume.
3419            redo A;
3420        }        }
3421      } else {      } else {
3422        !!!cp (1026);        die "$0: $self->{state}: Unknown state";
       !!!parse-error (type => 'bare ero');  
       ## NOTE: "No characters are consumed" in the spec.  
       return {type => CHARACTER_TOKEN, data => '&'.$value};  
3423      }      }
3424    } else {    } # A  
3425      !!!cp (1027);  
3426      ## no characters are consumed    die "$0: _get_next_token: unexpected case";
3427      !!!parse-error (type => 'bare ero');  } # _get_next_token
     return undef;  
   }  
 } # _tokenize_attempt_to_consume_an_entity  
3428    
3429  sub _initialize_tree_constructor ($) {  sub _initialize_tree_constructor ($) {
3430    my $self = shift;    my $self = shift;
# Line 2394  sub _initialize_tree_constructor ($) { Line 3433  sub _initialize_tree_constructor ($) {
3433    ## TODO: Turn mutation events off # MUST    ## TODO: Turn mutation events off # MUST
3434    ## TODO: Turn loose Document option (manakai extension) on    ## TODO: Turn loose Document option (manakai extension) on
3435    $self->{document}->manakai_is_html (1); # MUST    $self->{document}->manakai_is_html (1); # MUST
3436      $self->{document}->set_user_data (manakai_source_line => 1);
3437      $self->{document}->set_user_data (manakai_source_column => 1);
3438  } # _initialize_tree_constructor  } # _initialize_tree_constructor
3439    
3440  sub _terminate_tree_constructor ($) {  sub _terminate_tree_constructor ($) {
# Line 2420  sub _construct_tree ($) { Line 3461  sub _construct_tree ($) {
3461        
3462    !!!next-token;    !!!next-token;
3463    
   $self->{insertion_mode} = BEFORE_HEAD_IM;  
3464    undef $self->{form_element};    undef $self->{form_element};
3465    undef $self->{head_element};    undef $self->{head_element};
3466    $self->{open_elements} = [];    $self->{open_elements} = [];
3467    undef $self->{inner_html_node};    undef $self->{inner_html_node};
3468    
3469      ## NOTE: The "initial" insertion mode.
3470    $self->_tree_construction_initial; # MUST    $self->_tree_construction_initial; # MUST
3471    
3472      ## NOTE: The "before html" insertion mode.
3473    $self->_tree_construction_root_element;    $self->_tree_construction_root_element;
3474      $self->{insertion_mode} = BEFORE_HEAD_IM;
3475    
3476      ## NOTE: The "before head" insertion mode and so on.
3477    $self->_tree_construction_main;    $self->_tree_construction_main;
3478  } # _construct_tree  } # _construct_tree
3479    
3480  sub _tree_construction_initial ($) {  sub _tree_construction_initial ($) {
3481    my $self = shift;    my $self = shift;
3482    
3483      ## NOTE: "initial" insertion mode
3484    
3485    INITIAL: {    INITIAL: {
3486      if ($token->{type} == DOCTYPE_TOKEN) {      if ($token->{type} == DOCTYPE_TOKEN) {
3487        ## NOTE: Conformance checkers MAY, instead of reporting "not HTML5"        ## NOTE: Conformance checkers MAY, instead of reporting "not HTML5"
# Line 2440  sub _tree_construction_initial ($) { Line 3489  sub _tree_construction_initial ($) {
3489        ## language.        ## language.
3490        my $doctype_name = $token->{name};        my $doctype_name = $token->{name};
3491        $doctype_name = '' unless defined $doctype_name;        $doctype_name = '' unless defined $doctype_name;
3492        $doctype_name =~ tr/a-z/A-Z/;        $doctype_name =~ tr/a-z/A-Z/; # ASCII case-insensitive
3493        if (not defined $token->{name} or # <!DOCTYPE>        if (not defined $token->{name} or # <!DOCTYPE>
3494            defined $token->{public_identifier} or            defined $token->{sysid}) {
           defined $token->{system_identifier}) {  
3495          !!!cp ('t1');          !!!cp ('t1');
3496          !!!parse-error (type => 'not HTML5');          !!!parse-error (type => 'not HTML5', token => $token);
3497        } elsif ($doctype_name ne 'HTML') {        } elsif ($doctype_name ne 'HTML') {
3498          !!!cp ('t2');          !!!cp ('t2');
3499          ## ISSUE: ASCII case-insensitive? (in fact it does not matter)          !!!parse-error (type => 'not HTML5', token => $token);
3500          !!!parse-error (type => 'not HTML5');        } elsif (defined $token->{pubid}) {
3501            if ($token->{pubid} eq 'XSLT-compat') {
3502              !!!cp ('t1.2');
3503              !!!parse-error (type => 'XSLT-compat', token => $token,
3504                              level => $self->{level}->{should});
3505            } else {
3506              !!!parse-error (type => 'not HTML5', token => $token);
3507            }
3508        } else {        } else {
3509          !!!cp ('t3');          !!!cp ('t3');
3510            #
3511        }        }
3512                
3513        my $doctype = $self->{document}->create_document_type_definition        my $doctype = $self->{document}->create_document_type_definition
3514          ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?          ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?
3515        $doctype->public_id ($token->{public_identifier})        ## NOTE: Default value for both |public_id| and |system_id| attributes
3516            if defined $token->{public_identifier};        ## are empty strings, so that we don't set any value in missing cases.
3517        $doctype->system_id ($token->{system_identifier})        $doctype->public_id ($token->{pubid}) if defined $token->{pubid};
3518            if defined $token->{system_identifier};        $doctype->system_id ($token->{sysid}) if defined $token->{sysid};
3519        ## NOTE: Other DocumentType attributes are null or empty lists.        ## NOTE: Other DocumentType attributes are null or empty lists.
3520        ## ISSUE: internalSubset = null??        ## ISSUE: internalSubset = null??
3521        $self->{document}->append_child ($doctype);        $self->{document}->append_child ($doctype);
# Line 2467  sub _tree_construction_initial ($) { Line 3523  sub _tree_construction_initial ($) {
3523        if ($token->{quirks} or $doctype_name ne 'HTML') {        if ($token->{quirks} or $doctype_name ne 'HTML') {
3524          !!!cp ('t4');          !!!cp ('t4');
3525          $self->{document}->manakai_compat_mode ('quirks');          $self->{document}->manakai_compat_mode ('quirks');
3526        } elsif (defined $token->{public_identifier}) {        } elsif (defined $token->{pubid}) {
3527          my $pubid = $token->{public_identifier};          my $pubid = $token->{pubid};
3528          $pubid =~ tr/a-z/A-z/;          $pubid =~ tr/a-z/A-z/;
3529          if ({          my $prefix = [
3530            "+//SILMARIL//DTD HTML PRO V0R11 19970101//EN" => 1,            "+//SILMARIL//DTD HTML PRO V0R11 19970101//",
3531            "-//ADVASOFT LTD//DTD HTML 3.0 ASWEDIT + EXTENSIONS//EN" => 1,            "-//ADVASOFT LTD//DTD HTML 3.0 ASWEDIT + EXTENSIONS//",
3532            "-//AS//DTD HTML 3.0 ASWEDIT + EXTENSIONS//EN" => 1,            "-//AS//DTD HTML 3.0 ASWEDIT + EXTENSIONS//",
3533            "-//IETF//DTD HTML 2.0 LEVEL 1//EN" => 1,            "-//IETF//DTD HTML 2.0 LEVEL 1//",
3534            "-//IETF//DTD HTML 2.0 LEVEL 2//EN" => 1,            "-//IETF//DTD HTML 2.0 LEVEL 2//",
3535            "-//IETF//DTD HTML 2.0 STRICT LEVEL 1//EN" => 1,            "-//IETF//DTD HTML 2.0 STRICT LEVEL 1//",
3536            "-//IETF//DTD HTML 2.0 STRICT LEVEL 2//EN" => 1,            "-//IETF//DTD HTML 2.0 STRICT LEVEL 2//",
3537            "-//IETF//DTD HTML 2.0 STRICT//EN" => 1,            "-//IETF//DTD HTML 2.0 STRICT//",
3538            "-//IETF//DTD HTML 2.0//EN" => 1,            "-//IETF//DTD HTML 2.0//",
3539            "-//IETF//DTD HTML 2.1E//EN" => 1,            "-//IETF//DTD HTML 2.1E//",
3540            "-//IETF//DTD HTML 3.0//EN" => 1,            "-//IETF//DTD HTML 3.0//",
3541            "-//IETF//DTD HTML 3.0//EN//" => 1,            "-//IETF//DTD HTML 3.2 FINAL//",
3542            "-//IETF//DTD HTML 3.2 FINAL//EN" => 1,            "-//IETF//DTD HTML 3.2//",
3543            "-//IETF//DTD HTML 3.2//EN" => 1,            "-//IETF//DTD HTML 3//",
3544            "-//IETF//DTD HTML 3//EN" => 1,            "-//IETF//DTD HTML LEVEL 0//",
3545            "-//IETF//DTD HTML LEVEL 0//EN" => 1,            "-//IETF//DTD HTML LEVEL 1//",
3546            "-//IETF//DTD HTML LEVEL 0//EN//2.0" => 1,            "-//IETF//DTD HTML LEVEL 2//",
3547            "-//IETF//DTD HTML LEVEL 1//EN" => 1,            "-//IETF//DTD HTML LEVEL 3//",
3548            "-//IETF//DTD HTML LEVEL 1//EN//2.0" => 1,            "-//IETF//DTD HTML STRICT LEVEL 0//",
3549            "-//IETF//DTD HTML LEVEL 2//EN" => 1,            "-//IETF//DTD HTML STRICT LEVEL 1//",
3550            "-//IETF//DTD HTML LEVEL 2//EN//2.0" => 1,            "-//IETF//DTD HTML STRICT LEVEL 2//",
3551            "-//IETF//DTD HTML LEVEL 3//EN" => 1,            "-//IETF//DTD HTML STRICT LEVEL 3//",
3552            "-//IETF//DTD HTML LEVEL 3//EN//3.0" => 1,            "-//IETF//DTD HTML STRICT//",
3553            "-//IETF//DTD HTML STRICT LEVEL 0//EN" => 1,            "-//IETF//DTD HTML//",
3554            "-//IETF//DTD HTML STRICT LEVEL 0//EN//2.0" => 1,            "-//METRIUS//DTD METRIUS PRESENTATIONAL//",
3555            "-//IETF//DTD HTML STRICT LEVEL 1//EN" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML STRICT//",
3556            "-//IETF//DTD HTML STRICT LEVEL 1//EN//2.0" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML//",
3557            "-//IETF//DTD HTML STRICT LEVEL 2//EN" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 TABLES//",
3558            "-//IETF//DTD HTML STRICT LEVEL 2//EN//2.0" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML STRICT//",
3559            "-//IETF//DTD HTML STRICT LEVEL 3//EN" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML//",
3560            "-//IETF//DTD HTML STRICT LEVEL 3//EN//3.0" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 TABLES//",
3561            "-//IETF//DTD HTML STRICT//EN" => 1,            "-//NETSCAPE COMM. CORP.//DTD HTML//",
3562            "-//IETF//DTD HTML STRICT//EN//2.0" => 1,            "-//NETSCAPE COMM. CORP.//DTD STRICT HTML//",
3563            "-//IETF//DTD HTML STRICT//EN//3.0" => 1,            "-//O'REILLY AND ASSOCIATES//DTD HTML 2.0//",
3564            "-//IETF//DTD HTML//EN" => 1,            "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED 1.0//",
3565            "-//IETF//DTD HTML//EN//2.0" => 1,            "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED RELAXED 1.0//",
3566            "-//IETF//DTD HTML//EN//3.0" => 1,            "-//SOFTQUAD SOFTWARE//DTD HOTMETAL PRO 6.0::19990601::EXTENSIONS TO HTML 4.0//",
3567            "-//METRIUS//DTD METRIUS PRESENTATIONAL//EN" => 1,            "-//SOFTQUAD//DTD HOTMETAL PRO 4.0::19971010::EXTENSIONS TO HTML 4.0//",
3568            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML STRICT//EN" => 1,            "-//SPYGLASS//DTD HTML 2.0 EXTENDED//",
3569            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML//EN" => 1,            "-//SQ//DTD HTML 2.0 HOTMETAL + EXTENSIONS//",
3570            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 TABLES//EN" => 1,            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA HTML//",
3571            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML STRICT//EN" => 1,            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA STRICT HTML//",
3572            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML//EN" => 1,            "-//W3C//DTD HTML 3 1995-03-24//",
3573            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 TABLES//EN" => 1,            "-//W3C//DTD HTML 3.2 DRAFT//",
3574            "-//NETSCAPE COMM. CORP.//DTD HTML//EN" => 1,            "-//W3C//DTD HTML 3.2 FINAL//",
3575            "-//NETSCAPE COMM. CORP.//DTD STRICT HTML//EN" => 1,            "-//W3C//DTD HTML 3.2//",
3576            "-//O'REILLY AND ASSOCIATES//DTD HTML 2.0//EN" => 1,            "-//W3C//DTD HTML 3.2S DRAFT//",
3577            "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED 1.0//EN" => 1,            "-//W3C//DTD HTML 4.0 FRAMESET//",
3578            "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED RELAXED 1.0//EN" => 1,            "-//W3C//DTD HTML 4.0 TRANSITIONAL//",
3579            "-//SOFTQUAD SOFTWARE//DTD HOTMETAL PRO 6.0::19990601::EXTENSIONS TO HTML 4.0//EN" => 1,            "-//W3C//DTD HTML EXPERIMETNAL 19960712//",
3580            "-//SOFTQUAD//DTD HOTMETAL PRO 4.0::19971010::EXTENSIONS TO HTML 4.0//EN" => 1,            "-//W3C//DTD HTML EXPERIMENTAL 970421//",
3581            "-//SPYGLASS//DTD HTML 2.0 EXTENDED//EN" => 1,            "-//W3C//DTD W3 HTML//",
3582            "-//SQ//DTD HTML 2.0 HOTMETAL + EXTENSIONS//EN" => 1,            "-//W3O//DTD W3 HTML 3.0//",
3583            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA HTML//EN" => 1,            "-//WEBTECHS//DTD MOZILLA HTML 2.0//",
3584            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA STRICT HTML//EN" => 1,            "-//WEBTECHS//DTD MOZILLA HTML//",
3585            "-//W3C//DTD HTML 3 1995-03-24//EN" => 1,          ]; # $prefix
3586            "-//W3C//DTD HTML 3.2 DRAFT//EN" => 1,          my $match;
3587            "-//W3C//DTD HTML 3.2 FINAL//EN" => 1,          for (@$prefix) {
3588            "-//W3C//DTD HTML 3.2//EN" => 1,            if (substr ($prefix, 0, length $_) eq $_) {
3589            "-//W3C//DTD HTML 3.2S DRAFT//EN" => 1,              $match = 1;
3590            "-//W3C//DTD HTML 4.0 FRAMESET//EN" => 1,              last;
3591            "-//W3C//DTD HTML 4.0 TRANSITIONAL//EN" => 1,            }
3592            "-//W3C//DTD HTML EXPERIMETNAL 19960712//EN" => 1,          }
3593            "-//W3C//DTD HTML EXPERIMENTAL 970421//EN" => 1,          if ($match or
3594            "-//W3C//DTD W3 HTML//EN" => 1,              $pubid eq "-//W3O//DTD W3 HTML STRICT 3.0//EN//" or
3595            "-//W3O//DTD W3 HTML 3.0//EN" => 1,              $pubid eq "-/W3C/DTD HTML 4.0 TRANSITIONAL/EN" or
3596            "-//W3O//DTD W3 HTML 3.0//EN//" => 1,              $pubid eq "HTML") {
           "-//W3O//DTD W3 HTML STRICT 3.0//EN//" => 1,  
           "-//WEBTECHS//DTD MOZILLA HTML 2.0//EN" => 1,  
           "-//WEBTECHS//DTD MOZILLA HTML//EN" => 1,  
           "-/W3C/DTD HTML 4.0 TRANSITIONAL/EN" => 1,  
           "HTML" => 1,  
         }->{$pubid}) {  
3597            !!!cp ('t5');            !!!cp ('t5');
3598            $self->{document}->manakai_compat_mode ('quirks');            $self->{document}->manakai_compat_mode ('quirks');
3599          } elsif ($pubid eq "-//W3C//DTD HTML 4.01 FRAMESET//EN" or          } elsif ($pubid =~ m[^-//W3C//DTD HTML 4.01 FRAMESET//] or
3600                   $pubid eq "-//W3C//DTD HTML 4.01 TRANSITIONAL//EN") {                   $pubid =~ m[^-//W3C//DTD HTML 4.01 TRANSITIONAL//]) {
3601            if (defined $token->{system_identifier}) {            if (defined $token->{sysid}) {
3602              !!!cp ('t6');              !!!cp ('t6');
3603              $self->{document}->manakai_compat_mode ('quirks');              $self->{document}->manakai_compat_mode ('quirks');
3604            } else {            } else {
3605              !!!cp ('t7');              !!!cp ('t7');
3606              $self->{document}->manakai_compat_mode ('limited quirks');              $self->{document}->manakai_compat_mode ('limited quirks');
3607            }            }
3608          } elsif ($pubid eq "-//W3C//DTD XHTML 1.0 FRAMESET//EN" or          } elsif ($pubid =~ m[^-//W3C//DTD XHTML 1.0 FRAMESET//] or
3609                   $pubid eq "-//W3C//DTD XHTML 1.0 TRANSITIONAL//EN") {                   $pubid =~ m[^-//W3C//DTD XHTML 1.0 TRANSITIONAL//]) {
3610            !!!cp ('t8');            !!!cp ('t8');
3611            $self->{document}->manakai_compat_mode ('limited quirks');            $self->{document}->manakai_compat_mode ('limited quirks');
3612          } else {          } else {
# Line 2565  sub _tree_construction_initial ($) { Line 3615  sub _tree_construction_initial ($) {
3615        } else {        } else {
3616          !!!cp ('t10');          !!!cp ('t10');
3617        }        }
3618        if (defined $token->{system_identifier}) {        if (defined $token->{sysid}) {
3619          my $sysid = $token->{system_identifier};          my $sysid = $token->{sysid};
3620          $sysid =~ tr/A-Z/a-z/;          $sysid =~ tr/A-Z/a-z/;
3621          if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {          if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {
3622            ## TODO: Check the spec: PUBLIC "(limited quirks)" "(quirks)"            ## NOTE: Ensure that |PUBLIC "(limited quirks)" "(quirks)"| is
3623              ## marked as quirks.
3624            $self->{document}->manakai_compat_mode ('quirks');            $self->{document}->manakai_compat_mode ('quirks');
3625            !!!cp ('t11');            !!!cp ('t11');
3626          } else {          } else {
# Line 2579  sub _tree_construction_initial ($) { Line 3630  sub _tree_construction_initial ($) {
3630          !!!cp ('t13');          !!!cp ('t13');
3631        }        }
3632                
3633        ## Go to the root element phase.        ## Go to the "before html" insertion mode.
3634        !!!next-token;        !!!next-token;
3635        return;        return;
3636      } elsif ({      } elsif ({
# Line 2588  sub _tree_construction_initial ($) { Line 3639  sub _tree_construction_initial ($) {
3639                END_OF_FILE_TOKEN, 1,                END_OF_FILE_TOKEN, 1,
3640               }->{$token->{type}}) {               }->{$token->{type}}) {
3641        !!!cp ('t14');        !!!cp ('t14');
3642        !!!parse-error (type => 'no DOCTYPE');        !!!parse-error (type => 'no DOCTYPE', token => $token);
3643        $self->{document}->manakai_compat_mode ('quirks');        $self->{document}->manakai_compat_mode ('quirks');
3644        ## Go to the root element phase        ## Go to the "before html" insertion mode.
3645        ## reprocess        ## reprocess
3646          !!!ack-later;
3647        return;        return;
3648      } elsif ($token->{type} == CHARACTER_TOKEN) {      } elsif ($token->{type} == CHARACTER_TOKEN) {
3649        if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D        if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3650          ## Ignore the token          ## Ignore the token
3651    
3652          unless (length $token->{data}) {          unless (length $token->{data}) {
3653            !!!cp ('t15');            !!!cp ('t15');
3654            ## Stay in the phase            ## Stay in the insertion mode.
3655            !!!next-token;            !!!next-token;
3656            redo INITIAL;            redo INITIAL;
3657          } else {          } else {
# Line 2609  sub _tree_construction_initial ($) { Line 3661  sub _tree_construction_initial ($) {
3661          !!!cp ('t17');          !!!cp ('t17');
3662        }        }
3663    
3664        !!!parse-error (type => 'no DOCTYPE');        !!!parse-error (type => 'no DOCTYPE', token => $token);
3665        $self->{document}->manakai_compat_mode ('quirks');        $self->{document}->manakai_compat_mode ('quirks');
3666        ## Go to the root element phase        ## Go to the "before html" insertion mode.
3667        ## reprocess        ## reprocess
3668        return;        return;
3669      } elsif ($token->{type} == COMMENT_TOKEN) {      } elsif ($token->{type} == COMMENT_TOKEN) {
# Line 2619  sub _tree_construction_initial ($) { Line 3671  sub _tree_construction_initial ($) {
3671        my $comment = $self->{document}->create_comment ($token->{data});        my $comment = $self->{document}->create_comment ($token->{data});
3672        $self->{document}->append_child ($comment);        $self->{document}->append_child ($comment);
3673                
3674        ## Stay in the phase.        ## Stay in the insertion mode.
3675        !!!next-token;        !!!next-token;
3676        redo INITIAL;        redo INITIAL;
3677      } else {      } else {
# Line 2632  sub _tree_construction_initial ($) { Line 3684  sub _tree_construction_initial ($) {
3684    
3685  sub _tree_construction_root_element ($) {  sub _tree_construction_root_element ($) {
3686    my $self = shift;    my $self = shift;
3687    
3688      ## NOTE: "before html" insertion mode.
3689        
3690    B: {    B: {
3691        if ($token->{type} == DOCTYPE_TOKEN) {        if ($token->{type} == DOCTYPE_TOKEN) {
3692          !!!cp ('t19');          !!!cp ('t19');
3693          !!!parse-error (type => 'in html:#DOCTYPE');          !!!parse-error (type => 'in html:#DOCTYPE', token => $token);
3694          ## Ignore the token          ## Ignore the token
3695          ## Stay in the phase          ## Stay in the insertion mode.
3696          !!!next-token;          !!!next-token;
3697          redo B;          redo B;
3698        } elsif ($token->{type} == COMMENT_TOKEN) {        } elsif ($token->{type} == COMMENT_TOKEN) {
3699          !!!cp ('t20');          !!!cp ('t20');
3700          my $comment = $self->{document}->create_comment ($token->{data});          my $comment = $self->{document}->create_comment ($token->{data});
3701          $self->{document}->append_child ($comment);          $self->{document}->append_child ($comment);
3702          ## Stay in the phase          ## Stay in the insertion mode.
3703          !!!next-token;          !!!next-token;
3704          redo B;          redo B;
3705        } elsif ($token->{type} == CHARACTER_TOKEN) {        } elsif ($token->{type} == CHARACTER_TOKEN) {
3706          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3707            ## Ignore the token.            ## Ignore the token.
3708    
3709            unless (length $token->{data}) {            unless (length $token->{data}) {
3710              !!!cp ('t21');              !!!cp ('t21');
3711              ## Stay in the phase              ## Stay in the insertion mode.
3712              !!!next-token;              !!!next-token;
3713              redo B;              redo B;
3714            } else {            } else {
# Line 2668  sub _tree_construction_root_element ($) Line 3722  sub _tree_construction_root_element ($)
3722    
3723          #          #
3724        } elsif ($token->{type} == START_TAG_TOKEN) {        } elsif ($token->{type} == START_TAG_TOKEN) {
3725          if ($token->{tag_name} eq 'html' and          if ($token->{tag_name} eq 'html') {
3726              $token->{attributes}->{manifest}) {            my $root_element;
3727            !!!cp ('t24');            !!!create-element ($root_element, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);
3728            $self->{application_cache_selection}            $self->{document}->append_child ($root_element);
3729                 ->($token->{attributes}->{manifest}->{value});            push @{$self->{open_elements}},
3730            ## ISSUE: No relative reference resolution?                [$root_element, $el_category->{html}];
3731    
3732              if ($token->{attributes}->{manifest}) {
3733                !!!cp ('t24');
3734                $self->{application_cache_selection}
3735                    ->($token->{attributes}->{manifest}->{value});
3736                ## ISSUE: Spec is unclear on relative references.
3737                ## According to Hixie (#whatwg 2008-03-19), it should be
3738                ## resolved against the base URI of the document in HTML
3739                ## or xml:base of the element in XHTML.
3740              } else {
3741                !!!cp ('t25');
3742                $self->{application_cache_selection}->(undef);
3743              }
3744    
3745              !!!nack ('t25c');
3746    
3747              !!!next-token;
3748              return; ## Go to the "before head" insertion mode.
3749          } else {          } else {
3750            !!!cp ('t25');            !!!cp ('t25.1');
3751            $self->{application_cache_selection}->(undef);            #
3752          }          }
   
         ## ISSUE: There is an issue in the spec  
         #  
3753        } elsif ({        } elsif ({
3754                  END_TAG_TOKEN, 1,                  END_TAG_TOKEN, 1,
3755                  END_OF_FILE_TOKEN, 1,                  END_OF_FILE_TOKEN, 1,
3756                 }->{$token->{type}}) {                 }->{$token->{type}}) {
3757          !!!cp ('t26');          !!!cp ('t26');
         $self->{application_cache_selection}->(undef);  
   
         ## ISSUE: There is an issue in the spec  
3758          #          #
3759        } else {        } else {
3760          die "$0: $token->{type}: Unknown token type";          die "$0: $token->{type}: Unknown token type";
3761        }        }
3762    
3763        my $root_element; !!!create-element ($root_element, 'html');      my $root_element;
3764        $self->{document}->append_child ($root_element);      !!!create-element ($root_element, $HTML_NS, 'html',, $token);
3765        push @{$self->{open_elements}}, [$root_element, 'html'];      $self->{document}->append_child ($root_element);
3766        ## reprocess      push @{$self->{open_elements}}, [$root_element, $el_category->{html}];
3767        #redo B;  
3768        return; ## Go to the main phase.      $self->{application_cache_selection}->(undef);
3769    
3770        ## NOTE: Reprocess the token.
3771        !!!ack-later;
3772        return; ## Go to the "before head" insertion mode.
3773    
3774        ## ISSUE: There is an issue in the spec
3775    } # B    } # B
3776    
3777    die "$0: _tree_construction_root_element: This should never be reached";    die "$0: _tree_construction_root_element: This should never be reached";
# Line 2717  sub _reset_insertion_mode ($) { Line 3789  sub _reset_insertion_mode ($) {
3789            
3790      ## Step 3      ## Step 3
3791      S3: {      S3: {
       ## ISSUE: Oops! "If node is the first node in the stack of open  
       ## elements, then set last to true. If the context element of the  
       ## HTML fragment parsing algorithm is neither a td element nor a  
       ## th element, then set node to the context element. (fragment case)":  
       ## The second "if" is in the scope of the first "if"!?  
3792        if ($self->{open_elements}->[0]->[0] eq $node->[0]) {        if ($self->{open_elements}->[0]->[0] eq $node->[0]) {
3793          $last = 1;          $last = 1;
3794          if (defined $self->{inner_html_node}) {          if (defined $self->{inner_html_node}) {
3795            if ($self->{inner_html_node}->[1] eq 'td' or            !!!cp ('t28');
3796                $self->{inner_html_node}->[1] eq 'th') {            $node = $self->{inner_html_node};
3797              !!!cp ('t27');          } else {
3798              #            die "_reset_insertion_mode: t27";
           } else {  
             !!!cp ('t28');  
             $node = $self->{inner_html_node};  
           }  
3799          }          }
3800        }        }
3801              
3802        ## Step 4..13        ## Step 4..14
3803        my $new_mode = {        my $new_mode;
3804          if ($node->[1] & FOREIGN_EL) {
3805            !!!cp ('t28.1');
3806            ## NOTE: Strictly spaking, the line below only applies to MathML and
3807            ## SVG elements.  Currently the HTML syntax supports only MathML and
3808            ## SVG elements as foreigners.
3809            $new_mode = IN_BODY_IM | IN_FOREIGN_CONTENT_IM;
3810          } elsif ($node->[1] & TABLE_CELL_EL) {
3811            if ($last) {
3812              !!!cp ('t28.2');
3813              #
3814            } else {
3815              !!!cp ('t28.3');
3816              $new_mode = IN_CELL_IM;
3817            }
3818          } else {
3819            !!!cp ('t28.4');
3820            $new_mode = {
3821                        select => IN_SELECT_IM,                        select => IN_SELECT_IM,
3822                        td => IN_CELL_IM,                        ## NOTE: |option| and |optgroup| do not set
3823                        th => IN_CELL_IM,                        ## insertion mode to "in select" by themselves.
3824                        tr => IN_ROW_IM,                        tr => IN_ROW_IM,
3825                        tbody => IN_TABLE_BODY_IM,                        tbody => IN_TABLE_BODY_IM,
3826                        thead => IN_TABLE_BODY_IM,                        thead => IN_TABLE_BODY_IM,
# Line 2751  sub _reset_insertion_mode ($) { Line 3831  sub _reset_insertion_mode ($) {
3831                        head => IN_BODY_IM, # not in head!                        head => IN_BODY_IM, # not in head!
3832                        body => IN_BODY_IM,                        body => IN_BODY_IM,
3833                        frameset => IN_FRAMESET_IM,                        frameset => IN_FRAMESET_IM,
3834                       }->{$node->[1]};                       }->{$node->[0]->manakai_local_name};
3835          }
3836        $self->{insertion_mode} = $new_mode and return if defined $new_mode;        $self->{insertion_mode} = $new_mode and return if defined $new_mode;
3837                
3838        ## Step 14        ## Step 15
3839        if ($node->[1] eq 'html') {        if ($node->[1] & HTML_EL) {
3840          unless (defined $self->{head_element}) {          unless (defined $self->{head_element}) {
3841            !!!cp ('t29');            !!!cp ('t29');
3842            $self->{insertion_mode} = BEFORE_HEAD_IM;            $self->{insertion_mode} = BEFORE_HEAD_IM;
# Line 2769  sub _reset_insertion_mode ($) { Line 3850  sub _reset_insertion_mode ($) {
3850          !!!cp ('t31');          !!!cp ('t31');
3851        }        }
3852                
3853        ## Step 15        ## Step 16
3854        $self->{insertion_mode} = IN_BODY_IM and return if $last;        $self->{insertion_mode} = IN_BODY_IM and return if $last;
3855                
3856        ## Step 16        ## Step 17
3857        $i--;        $i--;
3858        $node = $self->{open_elements}->[$i];        $node = $self->{open_elements}->[$i];
3859                
3860        ## Step 17        ## Step 18
3861        redo S3;        redo S3;
3862      } # S3      } # S3
3863    
# Line 2880  sub _tree_construction_main ($) { Line 3961  sub _tree_construction_main ($) {
3961      !!!cp ('t39');      !!!cp ('t39');
3962    }; # $clear_up_to_marker    }; # $clear_up_to_marker
3963    
3964    my $parse_rcdata = sub ($$) {    my $insert;
3965      my ($content_model_flag, $insert) = @_;  
3966      my $parse_rcdata = sub ($) {
3967        my ($content_model_flag) = @_;
3968    
3969      ## Step 1      ## Step 1
3970      my $start_tag_name = $token->{tag_name};      my $start_tag_name = $token->{tag_name};
3971      my $el;      my $el;
3972      !!!create-element ($el, $start_tag_name, $token->{attributes});      !!!create-element ($el, $HTML_NS, $start_tag_name, $token->{attributes}, $token);
3973    
3974      ## Step 2      ## Step 2
3975      $insert->($el); # /context node/->append_child ($el)      $insert->($el);
3976    
3977      ## Step 3      ## Step 3
3978      $self->{content_model} = $content_model_flag; # CDATA or RCDATA      $self->{content_model} = $content_model_flag; # CDATA or RCDATA
# Line 2897  sub _tree_construction_main ($) { Line 3980  sub _tree_construction_main ($) {
3980    
3981      ## Step 4      ## Step 4
3982      my $text = '';      my $text = '';
3983        !!!nack ('t40.1');
3984      !!!next-token;      !!!next-token;
3985      while ($token->{type} == CHARACTER_TOKEN) { # or until stop tokenizing      while ($token->{type} == CHARACTER_TOKEN) { # or until stop tokenizing
3986        !!!cp ('t40');        !!!cp ('t40');
# Line 2919  sub _tree_construction_main ($) { Line 4003  sub _tree_construction_main ($) {
4003          $token->{tag_name} eq $start_tag_name) {          $token->{tag_name} eq $start_tag_name) {
4004        !!!cp ('t42');        !!!cp ('t42');
4005        ## Ignore the token        ## Ignore the token
     } elsif ($content_model_flag == CDATA_CONTENT_MODEL) {  
       !!!cp ('t43');  
       !!!parse-error (type => 'in CDATA:#'.$token->{type});  
     } elsif ($content_model_flag == RCDATA_CONTENT_MODEL) {  
       !!!cp ('t44');  
       !!!parse-error (type => 'in RCDATA:#'.$token->{type});  
4006      } else {      } else {
4007        die "$0: $content_model_flag in parse_rcdata";        ## NOTE: An end-of-file token.
4008          if ($content_model_flag == CDATA_CONTENT_MODEL) {
4009            !!!cp ('t43');
4010            !!!parse-error (type => 'in CDATA:#eof', token => $token);
4011          } elsif ($content_model_flag == RCDATA_CONTENT_MODEL) {
4012            !!!cp ('t44');
4013            !!!parse-error (type => 'in RCDATA:#eof', token => $token);
4014          } else {
4015            die "$0: $content_model_flag in parse_rcdata";
4016          }
4017      }      }
4018      !!!next-token;      !!!next-token;
4019    }; # $parse_rcdata    }; # $parse_rcdata
4020    
4021    my $script_start_tag = sub ($) {    my $script_start_tag = sub () {
     my $insert = $_[0];  
4022      my $script_el;      my $script_el;
4023      !!!create-element ($script_el, 'script', $token->{attributes});      !!!create-element ($script_el, $HTML_NS, 'script', $token->{attributes}, $token);
4024      ## TODO: mark as "parser-inserted"      ## TODO: mark as "parser-inserted"
4025    
4026      $self->{content_model} = CDATA_CONTENT_MODEL;      $self->{content_model} = CDATA_CONTENT_MODEL;
4027      delete $self->{escape}; # MUST      delete $self->{escape}; # MUST
4028            
4029      my $text = '';      my $text = '';
4030        !!!nack ('t45.1');
4031      !!!next-token;      !!!next-token;
4032      while ($token->{type} == CHARACTER_TOKEN) {      while ($token->{type} == CHARACTER_TOKEN) {
4033        !!!cp ('t45');        !!!cp ('t45');
# Line 2960  sub _tree_construction_main ($) { Line 4047  sub _tree_construction_main ($) {
4047        ## Ignore the token        ## Ignore the token
4048      } else {      } else {
4049        !!!cp ('t48');        !!!cp ('t48');
4050        !!!parse-error (type => 'in CDATA:#'.$token->{type});        !!!parse-error (type => 'in CDATA:#eof', token => $token);
4051        ## ISSUE: And ignore?        ## ISSUE: And ignore?
4052        ## TODO: mark as "already executed"        ## TODO: mark as "already executed"
4053      }      }
# Line 2983  sub _tree_construction_main ($) { Line 4070  sub _tree_construction_main ($) {
4070      !!!next-token;      !!!next-token;
4071    }; # $script_start_tag    }; # $script_start_tag
4072    
4073      ## NOTE: $open_tables->[-1]->[0] is the "current table" element node.
4074      ## NOTE: $open_tables->[-1]->[1] is the "tainted" flag.
4075      my $open_tables = [[$self->{open_elements}->[0]->[0]]];
4076    
4077    my $formatting_end_tag = sub {    my $formatting_end_tag = sub {
4078      my $tag_name = shift;      my $end_tag_token = shift;
4079        my $tag_name = $end_tag_token->{tag_name};
4080    
4081        ## NOTE: The adoption agency algorithm (AAA).
4082    
4083      FET: {      FET: {
4084        ## Step 1        ## Step 1
4085        my $formatting_element;        my $formatting_element;
4086        my $formatting_element_i_in_active;        my $formatting_element_i_in_active;
4087        AFE: for (reverse 0..$#$active_formatting_elements) {        AFE: for (reverse 0..$#$active_formatting_elements) {
4088          if ($active_formatting_elements->[$_]->[1] eq $tag_name) {          if ($active_formatting_elements->[$_]->[0] eq '#marker') {
4089              !!!cp ('t52');
4090              last AFE;
4091            } elsif ($active_formatting_elements->[$_]->[0]->manakai_local_name
4092                         eq $tag_name) {
4093            !!!cp ('t51');            !!!cp ('t51');
4094            $formatting_element = $active_formatting_elements->[$_];            $formatting_element = $active_formatting_elements->[$_];
4095            $formatting_element_i_in_active = $_;            $formatting_element_i_in_active = $_;
4096            last AFE;            last AFE;
         } elsif ($active_formatting_elements->[$_]->[0] eq '#marker') {  
           !!!cp ('t52');  
           last AFE;  
4097          }          }
4098        } # AFE        } # AFE
4099        unless (defined $formatting_element) {        unless (defined $formatting_element) {
4100          !!!cp ('t53');          !!!cp ('t53');
4101          !!!parse-error (type => 'unmatched end tag:'.$tag_name);          !!!parse-error (type => 'unmatched end tag', text => $tag_name, token => $end_tag_token);
4102          ## Ignore the token          ## Ignore the token
4103          !!!next-token;          !!!next-token;
4104          return;          return;
# Line 3020  sub _tree_construction_main ($) { Line 4115  sub _tree_construction_main ($) {
4115              last INSCOPE;              last INSCOPE;
4116            } else { # in open elements but not in scope            } else { # in open elements but not in scope
4117              !!!cp ('t55');              !!!cp ('t55');
4118              !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});              !!!parse-error (type => 'unmatched end tag',
4119                                text => $token->{tag_name},
4120                                token => $end_tag_token);
4121              ## Ignore the token              ## Ignore the token
4122              !!!next-token;              !!!next-token;
4123              return;              return;
4124            }            }
4125          } elsif ({          } elsif ($node->[1] & SCOPING_EL) {
                   table => 1, caption => 1, td => 1, th => 1,  
                   button => 1, marquee => 1, object => 1, html => 1,  
                  }->{$node->[1]}) {  
4126            !!!cp ('t56');            !!!cp ('t56');
4127            $in_scope = 0;            $in_scope = 0;
4128          }          }
4129        } # INSCOPE        } # INSCOPE
4130        unless (defined $formatting_element_i_in_open) {        unless (defined $formatting_element_i_in_open) {
4131          !!!cp ('t57');          !!!cp ('t57');
4132          !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});          !!!parse-error (type => 'unmatched end tag',
4133                            text => $token->{tag_name},
4134                            token => $end_tag_token);
4135          pop @$active_formatting_elements; # $formatting_element          pop @$active_formatting_elements; # $formatting_element
4136          !!!next-token; ## TODO: ok?          !!!next-token; ## TODO: ok?
4137          return;          return;
4138        }        }
4139        if (not $self->{open_elements}->[-1]->[0] eq $formatting_element->[0]) {        if (not $self->{open_elements}->[-1]->[0] eq $formatting_element->[0]) {
4140          !!!cp ('t58');          !!!cp ('t58');
4141          !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);          !!!parse-error (type => 'not closed',
4142                            text => $self->{open_elements}->[-1]->[0]
4143                                ->manakai_local_name,
4144                            token => $end_tag_token);
4145        }        }
4146                
4147        ## Step 2        ## Step 2
# Line 3050  sub _tree_construction_main ($) { Line 4149  sub _tree_construction_main ($) {
4149        my $furthest_block_i_in_open;        my $furthest_block_i_in_open;
4150        OE: for (reverse 0..$#{$self->{open_elements}}) {        OE: for (reverse 0..$#{$self->{open_elements}}) {
4151          my $node = $self->{open_elements}->[$_];          my $node = $self->{open_elements}->[$_];
4152          if (not $formatting_category->{$node->[1]} and          if (not ($node->[1] & FORMATTING_EL) and
4153              #not $phrasing_category->{$node->[1]} and              #not $phrasing_category->{$node->[1]} and
4154              ($special_category->{$node->[1]} or              ($node->[1] & SPECIAL_EL or
4155               $scoping_category->{$node->[1]})) {               $node->[1] & SCOPING_EL)) { ## Scoping is redundant, maybe
4156            !!!cp ('t59');            !!!cp ('t59');
4157            $furthest_block = $node;            $furthest_block = $node;
4158            $furthest_block_i_in_open = $_;            $furthest_block_i_in_open = $_;
# Line 3139  sub _tree_construction_main ($) { Line 4238  sub _tree_construction_main ($) {
4238        } # S7          } # S7  
4239                
4240        ## Step 8        ## Step 8
4241        $common_ancestor_node->[0]->append_child ($last_node->[0]);        if ($common_ancestor_node->[1] & TABLE_ROWS_EL) {
4242            my $foster_parent_element;
4243            my $next_sibling;
4244            OE: for (reverse 0..$#{$self->{open_elements}}) {
4245              if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {
4246                                 my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
4247                                 if (defined $parent and $parent->node_type == 1) {
4248                                   !!!cp ('t65.1');
4249                                   $foster_parent_element = $parent;
4250                                   $next_sibling = $self->{open_elements}->[$_]->[0];
4251                                 } else {
4252                                   !!!cp ('t65.2');
4253                                   $foster_parent_element
4254                                     = $self->{open_elements}->[$_ - 1]->[0];
4255                                 }
4256                                 last OE;
4257                               }
4258                             } # OE
4259                             $foster_parent_element = $self->{open_elements}->[0]->[0]
4260                               unless defined $foster_parent_element;
4261            $foster_parent_element->insert_before ($last_node->[0], $next_sibling);
4262            $open_tables->[-1]->[1] = 1; # tainted
4263          } else {
4264            !!!cp ('t65.3');
4265            $common_ancestor_node->[0]->append_child ($last_node->[0]);
4266          }
4267                
4268        ## Step 9        ## Step 9
4269        my $clone = [$formatting_element->[0]->clone_node (0),        my $clone = [$formatting_element->[0]->clone_node (0),
# Line 3185  sub _tree_construction_main ($) { Line 4309  sub _tree_construction_main ($) {
4309      } # FET      } # FET
4310    }; # $formatting_end_tag    }; # $formatting_end_tag
4311    
4312    my $insert_to_current = sub {    $insert = my $insert_to_current = sub {
4313      $self->{open_elements}->[-1]->[0]->append_child ($_[0]);      $self->{open_elements}->[-1]->[0]->append_child ($_[0]);
4314    }; # $insert_to_current    }; # $insert_to_current
4315    
4316    my $insert_to_foster = sub {    my $insert_to_foster = sub {
4317                         my $child = shift;      my $child = shift;
4318                         if ({      if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {
4319                              table => 1, tbody => 1, tfoot => 1,        # MUST
4320                              thead => 1, tr => 1,        my $foster_parent_element;
4321                             }->{$self->{open_elements}->[-1]->[1]}) {        my $next_sibling;
4322                           # MUST        OE: for (reverse 0..$#{$self->{open_elements}}) {
4323                           my $foster_parent_element;          if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {
                          my $next_sibling;  
                          OE: for (reverse 0..$#{$self->{open_elements}}) {  
                            if ($self->{open_elements}->[$_]->[1] eq 'table') {  
4324                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
4325                               if (defined $parent and $parent->node_type == 1) {                               if (defined $parent and $parent->node_type == 1) {
4326                                 !!!cp ('t70');                                 !!!cp ('t70');
# Line 3217  sub _tree_construction_main ($) { Line 4338  sub _tree_construction_main ($) {
4338                             unless defined $foster_parent_element;                             unless defined $foster_parent_element;
4339                           $foster_parent_element->insert_before                           $foster_parent_element->insert_before
4340                             ($child, $next_sibling);                             ($child, $next_sibling);
4341                         } else {        $open_tables->[-1]->[1] = 1; # tainted
4342                           !!!cp ('t72');      } else {
4343                           $self->{open_elements}->[-1]->[0]->append_child ($child);        !!!cp ('t72');
4344                         }        $self->{open_elements}->[-1]->[0]->append_child ($child);
4345        }
4346    }; # $insert_to_foster    }; # $insert_to_foster
4347    
4348    my $insert;    B: while (1) {
   
   B: {  
4349      if ($token->{type} == DOCTYPE_TOKEN) {      if ($token->{type} == DOCTYPE_TOKEN) {
4350        !!!cp ('t73');        !!!cp ('t73');
4351        !!!parse-error (type => 'DOCTYPE in the middle');        !!!parse-error (type => 'in html:#DOCTYPE', token => $token);
4352        ## Ignore the token        ## Ignore the token
4353        ## Stay in the phase        ## Stay in the phase
4354        !!!next-token;        !!!next-token;
4355        redo B;        next B;
     } elsif ($token->{type} == END_OF_FILE_TOKEN) {  
       if ($self->{insertion_mode} & AFTER_HTML_IMS) {  
         !!!cp ('t74');  
         #  
       } else {  
         ## Generate implied end tags  
         if ({  
              dd => 1, dt => 1, li => 1, p => 1, td => 1, th => 1, tr => 1,  
              tbody => 1, tfoot=> 1, thead => 1,  
             }->{$self->{open_elements}->[-1]->[1]}) {  
           !!!cp ('t75');  
           !!!back-token;  
           $token = {type => END_TAG_TOKEN, tag_name => $self->{open_elements}->[-1]->[1]};  
           redo B;  
         }  
           
         if (@{$self->{open_elements}} > 2 or  
             (@{$self->{open_elements}} == 2 and $self->{open_elements}->[1]->[1] ne 'body')) {  
           !!!cp ('t76');  
           !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
         } elsif (defined $self->{inner_html_node} and  
                  @{$self->{open_elements}} > 1 and  
                  $self->{open_elements}->[1]->[1] ne 'body') {  
 ## ISSUE: This case is never reached.  
           !!!cp ('t77');  
           !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
         } else {  
           !!!cp ('t78');  
         }  
   
         ## ISSUE: There is an issue in the spec.  
       }  
   
       ## Stop parsing  
       last B;  
4356      } elsif ($token->{type} == START_TAG_TOKEN and      } elsif ($token->{type} == START_TAG_TOKEN and
4357               $token->{tag_name} eq 'html') {               $token->{tag_name} eq 'html') {
4358        if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {        if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
4359          !!!cp ('t79');          !!!cp ('t79');
4360          ## Turn into the main phase          !!!parse-error (type => 'after html', text => 'html', token => $token);
         !!!parse-error (type => 'after html:html');  
4361          $self->{insertion_mode} = AFTER_BODY_IM;          $self->{insertion_mode} = AFTER_BODY_IM;
4362        } elsif ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {        } elsif ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {
4363          !!!cp ('t80');          !!!cp ('t80');
4364          ## Turn into the main phase          !!!parse-error (type => 'after html', text => 'html', token => $token);
         !!!parse-error (type => 'after html:html');  
4365          $self->{insertion_mode} = AFTER_FRAMESET_IM;          $self->{insertion_mode} = AFTER_FRAMESET_IM;
4366        } else {        } else {
4367          !!!cp ('t81');          !!!cp ('t81');
4368        }        }
4369    
4370  ## ISSUE: "aa<html>" is not a parse error.        !!!cp ('t82');
4371  ## ISSUE: "<html>" in fragment is not a parse error.        !!!parse-error (type => 'not first start tag', token => $token);
       unless ($token->{first_start_tag}) {  
         !!!cp ('t82');  
         !!!parse-error (type => 'not first start tag');  
       } else {  
         !!!cp ('t83');  
       }  
4372        my $top_el = $self->{open_elements}->[0]->[0];        my $top_el = $self->{open_elements}->[0]->[0];
4373        for my $attr_name (keys %{$token->{attributes}}) {        for my $attr_name (keys %{$token->{attributes}}) {
4374          unless ($top_el->has_attribute_ns (undef, $attr_name)) {          unless ($top_el->has_attribute_ns (undef, $attr_name)) {
# Line 3301  sub _tree_construction_main ($) { Line 4378  sub _tree_construction_main ($) {
4378               $token->{attributes}->{$attr_name}->{value});               $token->{attributes}->{$attr_name}->{value});
4379          }          }
4380        }        }
4381          !!!nack ('t84.1');
4382        !!!next-token;        !!!next-token;
4383        redo B;        next B;
4384      } elsif ($token->{type} == COMMENT_TOKEN) {      } elsif ($token->{type} == COMMENT_TOKEN) {
4385        my $comment = $self->{document}->create_comment ($token->{data});        my $comment = $self->{document}->create_comment ($token->{data});
4386        if ($self->{insertion_mode} & AFTER_HTML_IMS) {        if ($self->{insertion_mode} & AFTER_HTML_IMS) {
# Line 3316  sub _tree_construction_main ($) { Line 4394  sub _tree_construction_main ($) {
4394          $self->{open_elements}->[-1]->[0]->append_child ($comment);          $self->{open_elements}->[-1]->[0]->append_child ($comment);
4395        }        }
4396        !!!next-token;        !!!next-token;
4397        redo B;        next B;
4398      } elsif ($self->{insertion_mode} & HEAD_IMS) {      } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {
4399        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
4400          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          !!!cp ('t87.1');
4401            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);          $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
4402            !!!next-token;
4403            next B;
4404          } elsif ($token->{type} == START_TAG_TOKEN) {
4405            if ((not {mglyph => 1, malignmark => 1}->{$token->{tag_name}} and
4406                 $self->{open_elements}->[-1]->[1] & FOREIGN_FLOW_CONTENT_EL) or
4407                not ($self->{open_elements}->[-1]->[1] & FOREIGN_EL) or
4408                ($token->{tag_name} eq 'svg' and
4409                 $self->{open_elements}->[-1]->[1] & MML_AXML_EL)) {
4410              ## NOTE: "using the rules for secondary insertion mode"then"continue"
4411              !!!cp ('t87.2');
4412              #
4413            } elsif ({
4414                      b => 1, big => 1, blockquote => 1, body => 1, br => 1,
4415                      center => 1, code => 1, dd => 1, div => 1, dl => 1, dt => 1,
4416                      em => 1, embed => 1, font => 1, h1 => 1, h2 => 1, h3 => 1,
4417                      h4 => 1, h5 => 1, h6 => 1, head => 1, hr => 1, i => 1,
4418                      img => 1, li => 1, listing => 1, menu => 1, meta => 1,
4419                      nobr => 1, ol => 1, p => 1, pre => 1, ruby => 1, s => 1,
4420                      small => 1, span => 1, strong => 1, strike => 1, sub => 1,
4421                      sup => 1, table => 1, tt => 1, u => 1, ul => 1, var => 1,
4422                     }->{$token->{tag_name}}) {
4423              !!!cp ('t87.2');
4424              !!!parse-error (type => 'not closed',
4425                              text => $self->{open_elements}->[-1]->[0]
4426                                  ->manakai_local_name,
4427                              token => $token);
4428    
4429              pop @{$self->{open_elements}}
4430                  while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;
4431    
4432              $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
4433              ## Reprocess.
4434              next B;
4435            } else {
4436              my $nsuri = $self->{open_elements}->[-1]->[0]->namespace_uri;
4437              my $tag_name = $token->{tag_name};
4438              if ($nsuri eq $SVG_NS) {
4439                $tag_name = {
4440                   altglyph => 'altGlyph',
4441                   altglyphdef => 'altGlyphDef',
4442                   altglyphitem => 'altGlyphItem',
4443                   animatecolor => 'animateColor',
4444                   animatemotion => 'animateMotion',
4445                   animatetransform => 'animateTransform',
4446                   clippath => 'clipPath',
4447                   feblend => 'feBlend',
4448                   fecolormatrix => 'feColorMatrix',
4449                   fecomponenttransfer => 'feComponentTransfer',
4450                   fecomposite => 'feComposite',
4451                   feconvolvematrix => 'feConvolveMatrix',
4452                   fediffuselighting => 'feDiffuseLighting',
4453                   fedisplacementmap => 'feDisplacementMap',
4454                   fedistantlight => 'feDistantLight',
4455                   feflood => 'feFlood',
4456                   fefunca => 'feFuncA',
4457                   fefuncb => 'feFuncB',
4458                   fefuncg => 'feFuncG',
4459                   fefuncr => 'feFuncR',
4460                   fegaussianblur => 'feGaussianBlur',
4461                   feimage => 'feImage',
4462                   femerge => 'feMerge',
4463                   femergenode => 'feMergeNode',
4464                   femorphology => 'feMorphology',
4465                   feoffset => 'feOffset',
4466                   fepointlight => 'fePointLight',
4467                   fespecularlighting => 'feSpecularLighting',
4468                   fespotlight => 'feSpotLight',
4469                   fetile => 'feTile',
4470                   feturbulence => 'feTurbulence',
4471                   foreignobject => 'foreignObject',
4472                   glyphref => 'glyphRef',
4473                   lineargradient => 'linearGradient',
4474                   radialgradient => 'radialGradient',
4475                   #solidcolor => 'solidColor', ## NOTE: Commented in spec (SVG1.2)
4476                   textpath => 'textPath',  
4477                }->{$tag_name} || $tag_name;
4478              }
4479    
4480              ## "adjust SVG attributes" (SVG only) - done in insert-element-f
4481    
4482              ## "adjust foreign attributes" - done in insert-element-f
4483    
4484              !!!insert-element-f ($nsuri, $tag_name, $token->{attributes}, $token);
4485    
4486              if ($self->{self_closing}) {
4487                pop @{$self->{open_elements}};
4488                !!!ack ('t87.3');
4489              } else {
4490                !!!cp ('t87.4');
4491              }
4492    
4493              !!!next-token;
4494              next B;
4495            }
4496          } elsif ($token->{type} == END_TAG_TOKEN) {
4497            ## NOTE: "using the rules for secondary insertion mode" then "continue"
4498            !!!cp ('t87.5');
4499            #
4500          } elsif ($token->{type} == END_OF_FILE_TOKEN) {
4501            !!!cp ('t87.6');
4502            !!!parse-error (type => 'not closed',
4503                            text => $self->{open_elements}->[-1]->[0]
4504                                ->manakai_local_name,
4505                            token => $token);
4506    
4507            pop @{$self->{open_elements}}
4508                while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;
4509    
4510            $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
4511            ## Reprocess.
4512            next B;
4513          } else {
4514            die "$0: $token->{type}: Unknown token type";        
4515          }
4516        }
4517    
4518        if ($self->{insertion_mode} & HEAD_IMS) {
4519          if ($token->{type} == CHARACTER_TOKEN) {
4520            if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
4521              unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4522                !!!cp ('t88.2');
4523                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
4524                #
4525              } else {
4526                !!!cp ('t88.1');
4527                ## Ignore the token.
4528                #
4529              }
4530            unless (length $token->{data}) {            unless (length $token->{data}) {
4531              !!!cp ('t88');              !!!cp ('t88');
4532              !!!next-token;              !!!next-token;
4533              redo B;              next B;
4534            }            }
4535    ## TODO: set $token->{column} appropriately
4536          }          }
4537    
4538          if ($self->{insertion_mode} == BEFORE_HEAD_IM) {          if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4539            !!!cp ('t89');            !!!cp ('t89');
4540            ## As if <head>            ## As if <head>
4541            !!!create-element ($self->{head_element}, 'head');            !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4542            $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});            $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4543            push @{$self->{open_elements}}, [$self->{head_element}, 'head'];            push @{$self->{open_elements}},
4544                  [$self->{head_element}, $el_category->{head}];
4545    
4546            ## Reprocess in the "in head" insertion mode...            ## Reprocess in the "in head" insertion mode...
4547            pop @{$self->{open_elements}};            pop @{$self->{open_elements}};
# Line 3343  sub _tree_construction_main ($) { Line 4551  sub _tree_construction_main ($) {
4551            !!!cp ('t90');            !!!cp ('t90');
4552            ## As if </noscript>            ## As if </noscript>
4553            pop @{$self->{open_elements}};            pop @{$self->{open_elements}};
4554            !!!parse-error (type => 'in noscript:#character');            !!!parse-error (type => 'in noscript:#text', token => $token);
4555                        
4556            ## Reprocess in the "in head" insertion mode...            ## Reprocess in the "in head" insertion mode...
4557            ## As if </head>            ## As if </head>
# Line 3359  sub _tree_construction_main ($) { Line 4567  sub _tree_construction_main ($) {
4567            !!!cp ('t92');            !!!cp ('t92');
4568          }          }
4569    
4570              ## "after head" insertion mode          ## "after head" insertion mode
4571              ## As if <body>          ## As if <body>
4572              !!!insert-element ('body');          !!!insert-element ('body',, $token);
4573              $self->{insertion_mode} = IN_BODY_IM;          $self->{insertion_mode} = IN_BODY_IM;
4574              ## reprocess          ## reprocess
4575              redo B;          next B;
4576            } elsif ($token->{type} == START_TAG_TOKEN) {        } elsif ($token->{type} == START_TAG_TOKEN) {
4577              if ($token->{tag_name} eq 'head') {          if ($token->{tag_name} eq 'head') {
4578                if ($self->{insertion_mode} == BEFORE_HEAD_IM) {            if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4579                  !!!cp ('t93');              !!!cp ('t93');
4580                  !!!create-element ($self->{head_element}, $token->{tag_name}, $token->{attributes});              !!!create-element ($self->{head_element}, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);
4581                  $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});              $self->{open_elements}->[-1]->[0]->append_child
4582                  push @{$self->{open_elements}}, [$self->{head_element}, $token->{tag_name}];                  ($self->{head_element});
4583                  $self->{insertion_mode} = IN_HEAD_IM;              push @{$self->{open_elements}},
4584                  !!!next-token;                  [$self->{head_element}, $el_category->{head}];
4585                  redo B;              $self->{insertion_mode} = IN_HEAD_IM;
4586                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {              !!!nack ('t93.1');
4587                  !!!cp ('t94');              !!!next-token;
4588                  #              next B;
4589                } else {            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4590                  !!!cp ('t95');              !!!cp ('t93.2');
4591                  !!!parse-error (type => 'in head:head'); # or in head noscript              !!!parse-error (type => 'after head', text => 'head',
4592                  ## Ignore the token                              token => $token);
4593                  !!!next-token;              ## Ignore the token
4594                  redo B;              !!!nack ('t93.3');
4595                }              !!!next-token;
4596              } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {              next B;
4597                !!!cp ('t96');            } else {
4598                ## As if <head>              !!!cp ('t95');
4599                !!!create-element ($self->{head_element}, 'head');              !!!parse-error (type => 'in head:head',
4600                $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});                              token => $token); # or in head noscript
4601                push @{$self->{open_elements}}, [$self->{head_element}, 'head'];              ## Ignore the token
4602                !!!nack ('t95.1');
4603                !!!next-token;
4604                next B;
4605              }
4606            } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4607              !!!cp ('t96');
4608              ## As if <head>
4609              !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4610              $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4611              push @{$self->{open_elements}},
4612                  [$self->{head_element}, $el_category->{head}];
4613    
4614                $self->{insertion_mode} = IN_HEAD_IM;            $self->{insertion_mode} = IN_HEAD_IM;
4615                ## Reprocess in the "in head" insertion mode...            ## Reprocess in the "in head" insertion mode...
4616              } else {          } else {
4617                !!!cp ('t97');            !!!cp ('t97');
4618              }          }
4619    
4620              if ($token->{tag_name} eq 'base') {              if ($token->{tag_name} eq 'base') {
4621                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4622                  !!!cp ('t98');                  !!!cp ('t98');
4623                  ## As if </noscript>                  ## As if </noscript>
4624                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
4625                  !!!parse-error (type => 'in noscript:base');                  !!!parse-error (type => 'in noscript', text => 'base',
4626                                    token => $token);
4627                                
4628                  $self->{insertion_mode} = IN_HEAD_IM;                  $self->{insertion_mode} = IN_HEAD_IM;
4629                  ## Reprocess in the "in head" insertion mode...                  ## Reprocess in the "in head" insertion mode...
# Line 3414  sub _tree_construction_main ($) { Line 4634  sub _tree_construction_main ($) {
4634                ## NOTE: There is a "as if in head" code clone.                ## NOTE: There is a "as if in head" code clone.
4635                if ($self->{insertion_mode} == AFTER_HEAD_IM) {                if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4636                  !!!cp ('t100');                  !!!cp ('t100');
4637                  !!!parse-error (type => 'after head:'.$token->{tag_name});                  !!!parse-error (type => 'after head',
4638                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];                                  text => $token->{tag_name}, token => $token);
4639                    push @{$self->{open_elements}},
4640                        [$self->{head_element}, $el_category->{head}];
4641                } else {                } else {
4642                  !!!cp ('t101');                  !!!cp ('t101');
4643                }                }
4644                !!!insert-element ($token->{tag_name}, $token->{attributes});                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4645                pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.                pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
4646                pop @{$self->{open_elements}}                pop @{$self->{open_elements}} # <head>
4647                    if $self->{insertion_mode} == AFTER_HEAD_IM;                    if $self->{insertion_mode} == AFTER_HEAD_IM;
4648                  !!!nack ('t101.1');
4649                !!!next-token;                !!!next-token;
4650                redo B;                next B;
4651              } elsif ($token->{tag_name} eq 'link') {              } elsif ($token->{tag_name} eq 'link') {
4652                ## NOTE: There is a "as if in head" code clone.                ## NOTE: There is a "as if in head" code clone.
4653                if ($self->{insertion_mode} == AFTER_HEAD_IM) {                if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4654                  !!!cp ('t102');                  !!!cp ('t102');
4655                  !!!parse-error (type => 'after head:'.$token->{tag_name});                  !!!parse-error (type => 'after head',
4656                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];                                  text => $token->{tag_name}, token => $token);
4657                    push @{$self->{open_elements}},
4658                        [$self->{head_element}, $el_category->{head}];
4659                } else {                } else {
4660                  !!!cp ('t103');                  !!!cp ('t103');
4661                }                }
4662                !!!insert-element ($token->{tag_name}, $token->{attributes});                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4663                pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.                pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
4664                pop @{$self->{open_elements}}                pop @{$self->{open_elements}} # <head>
4665                    if $self->{insertion_mode} == AFTER_HEAD_IM;                    if $self->{insertion_mode} == AFTER_HEAD_IM;
4666                  !!!ack ('t103.1');
4667                !!!next-token;                !!!next-token;
4668                redo B;                next B;
4669              } elsif ($token->{tag_name} eq 'meta') {              } elsif ($token->{tag_name} eq 'meta') {
4670                ## NOTE: There is a "as if in head" code clone.                ## NOTE: There is a "as if in head" code clone.
4671                if ($self->{insertion_mode} == AFTER_HEAD_IM) {                if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4672                  !!!cp ('t104');                  !!!cp ('t104');
4673                  !!!parse-error (type => 'after head:'.$token->{tag_name});                  !!!parse-error (type => 'after head',
4674                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];                                  text => $token->{tag_name}, token => $token);
4675                    push @{$self->{open_elements}},
4676                        [$self->{head_element}, $el_category->{head}];
4677                } else {                } else {
4678                  !!!cp ('t105');                  !!!cp ('t105');
4679                }                }
4680                !!!insert-element ($token->{tag_name}, $token->{attributes});                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4681                my $meta_el = pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.                my $meta_el = pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
4682    
4683                unless ($self->{confident}) {                unless ($self->{confident}) {
4684                  if ($token->{attributes}->{charset}) { ## TODO: And if supported                  if ($token->{attributes}->{charset}) {
4685                    !!!cp ('t106');                    !!!cp ('t106');
4686                      ## NOTE: Whether the encoding is supported or not is handled
4687                      ## in the {change_encoding} callback.
4688                    $self->{change_encoding}                    $self->{change_encoding}
4689                        ->($self, $token->{attributes}->{charset}->{value});                        ->($self, $token->{attributes}->{charset}->{value},
4690                             $token);
4691                                        
4692                    $meta_el->[0]->get_attribute_node_ns (undef, 'charset')                    $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
4693                        ->set_user_data (manakai_has_reference =>                        ->set_user_data (manakai_has_reference =>
4694                                             $token->{attributes}->{charset}                                             $token->{attributes}->{charset}
4695                                                 ->{has_reference});                                                 ->{has_reference});
4696                  } elsif ($token->{attributes}->{content}) {                  } elsif ($token->{attributes}->{content}) {
                   ## ISSUE: Algorithm name in the spec was incorrect so that not linked to the definition.  
4697                    if ($token->{attributes}->{content}->{value}                    if ($token->{attributes}->{content}->{value}
4698                        =~ /\A[^;]*;[\x09-\x0D\x20]*[Cc][Hh][Aa][Rr][Ss][Ee][Tt]                        =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
4699                            [\x09-\x0D\x20]*=                            [\x09\x0A\x0C\x0D\x20]*=
4700                            [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|                            [\x09\x0A\x0C\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
4701                            ([^"'\x09-\x0D\x20][^\x09-\x0D\x20]*))/x) {                            ([^"'\x09\x0A\x0C\x0D\x20]
4702                               [^\x09\x0A\x0C\x0D\x20\x3B]*))/x) {
4703                      !!!cp ('t107');                      !!!cp ('t107');
4704                        ## NOTE: Whether the encoding is supported or not is handled
4705                        ## in the {change_encoding} callback.
4706                      $self->{change_encoding}                      $self->{change_encoding}
4707                          ->($self, defined $1 ? $1 : defined $2 ? $2 : $3);                          ->($self, defined $1 ? $1 : defined $2 ? $2 : $3,
4708                               $token);
4709                      $meta_el->[0]->get_attribute_node_ns (undef, 'content')                      $meta_el->[0]->get_attribute_node_ns (undef, 'content')
4710                          ->set_user_data (manakai_has_reference =>                          ->set_user_data (manakai_has_reference =>
4711                                               $token->{attributes}->{content}                                               $token->{attributes}->{content}
# Line 3497  sub _tree_construction_main ($) { Line 4731  sub _tree_construction_main ($) {
4731                  }                  }
4732                }                }
4733    
4734                pop @{$self->{open_elements}}                pop @{$self->{open_elements}} # <head>
4735                    if $self->{insertion_mode} == AFTER_HEAD_IM;                    if $self->{insertion_mode} == AFTER_HEAD_IM;
4736                  !!!ack ('t110.1');
4737                !!!next-token;                !!!next-token;
4738                redo B;                next B;
4739              } elsif ($token->{tag_name} eq 'title') {              } elsif ($token->{tag_name} eq 'title') {
4740                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4741                  !!!cp ('t111');                  !!!cp ('t111');
4742                  ## As if </noscript>                  ## As if </noscript>
4743                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
4744                  !!!parse-error (type => 'in noscript:title');                  !!!parse-error (type => 'in noscript', text => 'title',
4745                                    token => $token);
4746                                
4747                  $self->{insertion_mode} = IN_HEAD_IM;                  $self->{insertion_mode} = IN_HEAD_IM;
4748                  ## Reprocess in the "in head" insertion mode...                  ## Reprocess in the "in head" insertion mode...
4749                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4750                  !!!cp ('t112');                  !!!cp ('t112');
4751                  !!!parse-error (type => 'after head:'.$token->{tag_name});                  !!!parse-error (type => 'after head',
4752                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];                                  text => $token->{tag_name}, token => $token);
4753                    push @{$self->{open_elements}},
4754                        [$self->{head_element}, $el_category->{head}];
4755                } else {                } else {
4756                  !!!cp ('t113');                  !!!cp ('t113');
4757                }                }
# Line 3521  sub _tree_construction_main ($) { Line 4759  sub _tree_construction_main ($) {
4759                ## NOTE: There is a "as if in head" code clone.                ## NOTE: There is a "as if in head" code clone.
4760                my $parent = defined $self->{head_element} ? $self->{head_element}                my $parent = defined $self->{head_element} ? $self->{head_element}
4761                    : $self->{open_elements}->[-1]->[0];                    : $self->{open_elements}->[-1]->[0];
4762                $parse_rcdata->(RCDATA_CONTENT_MODEL,                $parse_rcdata->(RCDATA_CONTENT_MODEL);
4763                                sub { $parent->append_child ($_[0]) });                pop @{$self->{open_elements}} # <head>
               pop @{$self->{open_elements}}  
4764                    if $self->{insertion_mode} == AFTER_HEAD_IM;                    if $self->{insertion_mode} == AFTER_HEAD_IM;
4765                redo B;                next B;
4766              } elsif ($token->{tag_name} eq 'style') {              } elsif ($token->{tag_name} eq 'style' or
4767                         $token->{tag_name} eq 'noframes') {
4768                ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and                ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and
4769                ## insertion mode IN_HEAD_IM)                ## insertion mode IN_HEAD_IM)
4770                ## NOTE: There is a "as if in head" code clone.                ## NOTE: There is a "as if in head" code clone.
4771                if ($self->{insertion_mode} == AFTER_HEAD_IM) {                if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4772                  !!!cp ('t114');                  !!!cp ('t114');
4773                  !!!parse-error (type => 'after head:'.$token->{tag_name});                  !!!parse-error (type => 'after head',
4774                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];                                  text => $token->{tag_name}, token => $token);
4775                    push @{$self->{open_elements}},
4776                        [$self->{head_element}, $el_category->{head}];
4777                } else {                } else {
4778                  !!!cp ('t115');                  !!!cp ('t115');
4779                }                }
4780                $parse_rcdata->(CDATA_CONTENT_MODEL, $insert_to_current);                $parse_rcdata->(CDATA_CONTENT_MODEL);
4781                pop @{$self->{open_elements}}                pop @{$self->{open_elements}} # <head>
4782                    if $self->{insertion_mode} == AFTER_HEAD_IM;                    if $self->{insertion_mode} == AFTER_HEAD_IM;
4783                redo B;                next B;
4784              } elsif ($token->{tag_name} eq 'noscript') {              } elsif ($token->{tag_name} eq 'noscript') {
4785                if ($self->{insertion_mode} == IN_HEAD_IM) {                if ($self->{insertion_mode} == IN_HEAD_IM) {
4786                  !!!cp ('t116');                  !!!cp ('t116');
4787                  ## NOTE: and scripting is disalbed                  ## NOTE: and scripting is disalbed
4788                  !!!insert-element ($token->{tag_name}, $token->{attributes});                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4789                  $self->{insertion_mode} = IN_HEAD_NOSCRIPT_IM;                  $self->{insertion_mode} = IN_HEAD_NOSCRIPT_IM;
4790                    !!!nack ('t116.1');
4791                  !!!next-token;                  !!!next-token;
4792                  redo B;                  next B;
4793                } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {                } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4794                  !!!cp ('t117');                  !!!cp ('t117');
4795                  !!!parse-error (type => 'in noscript:noscript');                  !!!parse-error (type => 'in noscript', text => 'noscript',
4796                                    token => $token);
4797                  ## Ignore the token                  ## Ignore the token
4798                    !!!nack ('t117.1');
4799                  !!!next-token;                  !!!next-token;
4800                  redo B;                  next B;
4801                } else {                } else {
4802                  !!!cp ('t118');                  !!!cp ('t118');
4803                  #                  #
# Line 3564  sub _tree_construction_main ($) { Line 4807  sub _tree_construction_main ($) {
4807                  !!!cp ('t119');                  !!!cp ('t119');
4808                  ## As if </noscript>                  ## As if </noscript>
4809                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
4810                  !!!parse-error (type => 'in noscript:script');                  !!!parse-error (type => 'in noscript', text => 'script',
4811                                    token => $token);
4812                                
4813                  $self->{insertion_mode} = IN_HEAD_IM;                  $self->{insertion_mode} = IN_HEAD_IM;
4814                  ## Reprocess in the "in head" insertion mode...                  ## Reprocess in the "in head" insertion mode...
4815                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4816                  !!!cp ('t120');                  !!!cp ('t120');
4817                  !!!parse-error (type => 'after head:'.$token->{tag_name});                  !!!parse-error (type => 'after head',
4818                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];                                  text => $token->{tag_name}, token => $token);
4819                    push @{$self->{open_elements}},
4820                        [$self->{head_element}, $el_category->{head}];
4821                } else {                } else {
4822                  !!!cp ('t121');                  !!!cp ('t121');
4823                }                }
4824    
4825                ## NOTE: There is a "as if in head" code clone.                ## NOTE: There is a "as if in head" code clone.
4826                $script_start_tag->($insert_to_current);                $script_start_tag->();
4827                pop @{$self->{open_elements}}                pop @{$self->{open_elements}} # <head>
4828                    if $self->{insertion_mode} == AFTER_HEAD_IM;                    if $self->{insertion_mode} == AFTER_HEAD_IM;
4829                redo B;                next B;
4830              } elsif ($token->{tag_name} eq 'body' or              } elsif ($token->{tag_name} eq 'body' or
4831                       $token->{tag_name} eq 'frameset') {                       $token->{tag_name} eq 'frameset') {
4832                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4833                  !!!cp ('t122');                  !!!cp ('t122');
4834                  ## As if </noscript>                  ## As if </noscript>
4835                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
4836                  !!!parse-error (type => 'in noscript:'.$token->{tag_name});                  !!!parse-error (type => 'in noscript',
4837                                    text => $token->{tag_name}, token => $token);
4838                                    
4839                  ## Reprocess in the "in head" insertion mode...                  ## Reprocess in the "in head" insertion mode...
4840                  ## As if </head>                  ## As if </head>
# Line 3604  sub _tree_construction_main ($) { Line 4851  sub _tree_construction_main ($) {
4851                }                }
4852    
4853                ## "after head" insertion mode                ## "after head" insertion mode
4854                !!!insert-element ($token->{tag_name}, $token->{attributes});                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4855                if ($token->{tag_name} eq 'body') {                if ($token->{tag_name} eq 'body') {
4856                  !!!cp ('t126');                  !!!cp ('t126');
4857                  $self->{insertion_mode} = IN_BODY_IM;                  $self->{insertion_mode} = IN_BODY_IM;
# Line 3614  sub _tree_construction_main ($) { Line 4861  sub _tree_construction_main ($) {
4861                } else {                } else {
4862                  die "$0: tag name: $self->{tag_name}";                  die "$0: tag name: $self->{tag_name}";
4863                }                }
4864                  !!!nack ('t127.1');
4865                !!!next-token;                !!!next-token;
4866                redo B;                next B;
4867              } else {              } else {
4868                !!!cp ('t128');                !!!cp ('t128');
4869                #                #
# Line 3625  sub _tree_construction_main ($) { Line 4873  sub _tree_construction_main ($) {
4873                !!!cp ('t129');                !!!cp ('t129');
4874                ## As if </noscript>                ## As if </noscript>
4875                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
4876                !!!parse-error (type => 'in noscript:/'.$token->{tag_name});                !!!parse-error (type => 'in noscript:/',
4877                                  text => $token->{tag_name}, token => $token);
4878                                
4879                ## Reprocess in the "in head" insertion mode...                ## Reprocess in the "in head" insertion mode...
4880                ## As if </head>                ## As if </head>
# Line 3644  sub _tree_construction_main ($) { Line 4893  sub _tree_construction_main ($) {
4893    
4894              ## "after head" insertion mode              ## "after head" insertion mode
4895              ## As if <body>              ## As if <body>
4896              !!!insert-element ('body');              !!!insert-element ('body',, $token);
4897              $self->{insertion_mode} = IN_BODY_IM;              $self->{insertion_mode} = IN_BODY_IM;
4898              ## reprocess              ## reprocess
4899              redo B;              !!!ack-later;
4900                next B;
4901            } elsif ($token->{type} == END_TAG_TOKEN) {            } elsif ($token->{type} == END_TAG_TOKEN) {
4902              if ($token->{tag_name} eq 'head') {              if ($token->{tag_name} eq 'head') {
4903                if ($self->{insertion_mode} == BEFORE_HEAD_IM) {                if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4904                  !!!cp ('t132');                  !!!cp ('t132');
4905                  ## As if <head>                  ## As if <head>
4906                  !!!create-element ($self->{head_element}, 'head');                  !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4907                  $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});                  $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4908                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];                  push @{$self->{open_elements}},
4909                        [$self->{head_element}, $el_category->{head}];
4910    
4911                  ## Reprocess in the "in head" insertion mode...                  ## Reprocess in the "in head" insertion mode...
4912                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
4913                  $self->{insertion_mode} = AFTER_HEAD_IM;                  $self->{insertion_mode} = AFTER_HEAD_IM;
4914                  !!!next-token;                  !!!next-token;
4915                  redo B;                  next B;
4916                } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {                } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4917                  !!!cp ('t133');                  !!!cp ('t133');
4918                  ## As if </noscript>                  ## As if </noscript>
4919                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
4920                  !!!parse-error (type => 'in noscript:script');                  !!!parse-error (type => 'in noscript:/',
4921                                    text => 'head', token => $token);
4922                                    
4923                  ## Reprocess in the "in head" insertion mode...                  ## Reprocess in the "in head" insertion mode...
4924                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
4925                  $self->{insertion_mode} = AFTER_HEAD_IM;                  $self->{insertion_mode} = AFTER_HEAD_IM;
4926                  !!!next-token;                  !!!next-token;
4927                  redo B;                  next B;
4928                } elsif ($self->{insertion_mode} == IN_HEAD_IM) {                } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4929                  !!!cp ('t134');                  !!!cp ('t134');
4930                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
4931                  $self->{insertion_mode} = AFTER_HEAD_IM;                  $self->{insertion_mode} = AFTER_HEAD_IM;
4932                  !!!next-token;                  !!!next-token;
4933                  redo B;                  next B;
4934                  } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4935                    !!!cp ('t134.1');
4936                    !!!parse-error (type => 'unmatched end tag', text => 'head',
4937                                    token => $token);
4938                    ## Ignore the token
4939                    !!!next-token;
4940                    next B;
4941                } else {                } else {
4942                  !!!cp ('t135');                  die "$0: $self->{insertion_mode}: Unknown insertion mode";
                 #  
4943                }                }
4944              } elsif ($token->{tag_name} eq 'noscript') {              } elsif ($token->{tag_name} eq 'noscript') {
4945                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
# Line 3689  sub _tree_construction_main ($) { Line 4947  sub _tree_construction_main ($) {
4947                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
4948                  $self->{insertion_mode} = IN_HEAD_IM;                  $self->{insertion_mode} = IN_HEAD_IM;
4949                  !!!next-token;                  !!!next-token;
4950                  redo B;                  next B;
4951                } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {                } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM or
4952                           $self->{insertion_mode} == AFTER_HEAD_IM) {
4953                  !!!cp ('t137');                  !!!cp ('t137');
4954                  !!!parse-error (type => 'unmatched end tag:noscript');                  !!!parse-error (type => 'unmatched end tag',
4955                                    text => 'noscript', token => $token);
4956                  ## Ignore the token ## ISSUE: An issue in the spec.                  ## Ignore the token ## ISSUE: An issue in the spec.
4957                  !!!next-token;                  !!!next-token;
4958                  redo B;                  next B;
4959                } else {                } else {
4960                  !!!cp ('t138');                  !!!cp ('t138');
4961                  #                  #
# Line 3703  sub _tree_construction_main ($) { Line 4963  sub _tree_construction_main ($) {
4963              } elsif ({              } elsif ({
4964                        body => 1, html => 1,                        body => 1, html => 1,
4965                       }->{$token->{tag_name}}) {                       }->{$token->{tag_name}}) {
4966                if ($self->{insertion_mode} == BEFORE_HEAD_IM) {                if ($self->{insertion_mode} == BEFORE_HEAD_IM or
4967                  !!!cp ('t139');                    $self->{insertion_mode} == IN_HEAD_IM or
4968                  ## As if <head>                    $self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
                 !!!create-element ($self->{head_element}, 'head');  
                 $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});  
                 push @{$self->{open_elements}}, [$self->{head_element}, 'head'];  
   
                 $self->{insertion_mode} = IN_HEAD_IM;  
                 ## Reprocess in the "in head" insertion mode...  
               } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {  
4969                  !!!cp ('t140');                  !!!cp ('t140');
4970                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                  !!!parse-error (type => 'unmatched end tag',
4971                                    text => $token->{tag_name}, token => $token);
4972                  ## Ignore the token                  ## Ignore the token
4973                  !!!next-token;                  !!!next-token;
4974                  redo B;                  next B;
4975                  } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4976                    !!!cp ('t140.1');
4977                    !!!parse-error (type => 'unmatched end tag',
4978                                    text => $token->{tag_name}, token => $token);
4979                    ## Ignore the token
4980                    !!!next-token;
4981                    next B;
4982                } else {                } else {
4983                  !!!cp ('t141');                  die "$0: $self->{insertion_mode}: Unknown insertion mode";
4984                }                }
4985                              } elsif ($token->{tag_name} eq 'p') {
4986                #                !!!cp ('t142');
4987              } elsif ({                !!!parse-error (type => 'unmatched end tag',
4988                        p => 1, br => 1,                                text => $token->{tag_name}, token => $token);
4989                       }->{$token->{tag_name}}) {                ## Ignore the token
4990                  !!!next-token;
4991                  next B;
4992                } elsif ($token->{tag_name} eq 'br') {
4993                if ($self->{insertion_mode} == BEFORE_HEAD_IM) {                if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4994                  !!!cp ('t142');                  !!!cp ('t142.2');
4995                  ## As if <head>                  ## (before head) as if <head>, (in head) as if </head>
4996                  !!!create-element ($self->{head_element}, 'head');                  !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4997                  $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});                  $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4998                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];                  $self->{insertion_mode} = AFTER_HEAD_IM;
4999      
5000                    ## Reprocess in the "after head" insertion mode...
5001                  } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
5002                    !!!cp ('t143.2');
5003                    ## As if </head>
5004                    pop @{$self->{open_elements}};
5005                    $self->{insertion_mode} = AFTER_HEAD_IM;
5006      
5007                    ## Reprocess in the "after head" insertion mode...
5008                  } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5009                    !!!cp ('t143.3');
5010                    ## ISSUE: Two parse errors for <head><noscript></br>
5011                    !!!parse-error (type => 'unmatched end tag',
5012                                    text => 'br', token => $token);
5013                    ## As if </noscript>
5014                    pop @{$self->{open_elements}};
5015                  $self->{insertion_mode} = IN_HEAD_IM;                  $self->{insertion_mode} = IN_HEAD_IM;
5016    
5017                  ## Reprocess in the "in head" insertion mode...                  ## Reprocess in the "in head" insertion mode...
5018                } else {                  ## As if </head>
5019                  !!!cp ('t143');                  pop @{$self->{open_elements}};
5020                }                  $self->{insertion_mode} = AFTER_HEAD_IM;
5021    
5022                #                  ## Reprocess in the "after head" insertion mode...
5023              } else {                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
5024                if ($self->{insertion_mode} == AFTER_HEAD_IM) {                  !!!cp ('t143.4');
                 !!!cp ('t144');  
5025                  #                  #
5026                } else {                } else {
5027                  !!!cp ('t145');                  die "$0: $self->{insertion_mode}: Unknown insertion mode";
                 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
                 ## Ignore the token  
                 !!!next-token;  
                 redo B;  
5028                }                }
5029    
5030                  ## ISSUE: does not agree with IE7 - it doesn't ignore </br>.
5031                  !!!parse-error (type => 'unmatched end tag',
5032                                  text => 'br', token => $token);
5033                  ## Ignore the token
5034                  !!!next-token;
5035                  next B;
5036                } else {
5037                  !!!cp ('t145');
5038                  !!!parse-error (type => 'unmatched end tag',
5039                                  text => $token->{tag_name}, token => $token);
5040                  ## Ignore the token
5041                  !!!next-token;
5042                  next B;
5043              }              }
5044    
5045              if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {              if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5046                !!!cp ('t146');                !!!cp ('t146');
5047                ## As if </noscript>                ## As if </noscript>
5048                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
5049                !!!parse-error (type => 'in noscript:/'.$token->{tag_name});                !!!parse-error (type => 'in noscript:/',
5050                                  text => $token->{tag_name}, token => $token);
5051                                
5052                ## Reprocess in the "in head" insertion mode...                ## Reprocess in the "in head" insertion mode...
5053                ## As if </head>                ## As if </head>
# Line 3771  sub _tree_construction_main ($) { Line 5061  sub _tree_construction_main ($) {
5061    
5062                ## Reprocess in the "after head" insertion mode...                ## Reprocess in the "after head" insertion mode...
5063              } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {              } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {
5064    ## ISSUE: This case cannot be reached?
5065                !!!cp ('t148');                !!!cp ('t148');
5066                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                !!!parse-error (type => 'unmatched end tag',
5067                                  text => $token->{tag_name}, token => $token);
5068                ## Ignore the token ## ISSUE: An issue in the spec.                ## Ignore the token ## ISSUE: An issue in the spec.
5069                !!!next-token;                !!!next-token;
5070                redo B;                next B;
5071              } else {              } else {
5072                !!!cp ('t149');                !!!cp ('t149');
5073              }              }
5074    
5075              ## "after head" insertion mode              ## "after head" insertion mode
5076              ## As if <body>              ## As if <body>
5077              !!!insert-element ('body');              !!!insert-element ('body',, $token);
5078              $self->{insertion_mode} = IN_BODY_IM;              $self->{insertion_mode} = IN_BODY_IM;
5079              ## reprocess              ## reprocess
5080              redo B;              next B;
5081            } else {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
5082              die "$0: $token->{type}: Unknown token type";          if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
5083            }            !!!cp ('t149.1');
5084    
5085              ## NOTE: As if <head>
5086              !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
5087              $self->{open_elements}->[-1]->[0]->append_child
5088                  ($self->{head_element});
5089              #push @{$self->{open_elements}},
5090              #    [$self->{head_element}, $el_category->{head}];
5091              #$self->{insertion_mode} = IN_HEAD_IM;
5092              ## NOTE: Reprocess.
5093    
5094              ## NOTE: As if </head>
5095              #pop @{$self->{open_elements}};
5096              #$self->{insertion_mode} = IN_AFTER_HEAD_IM;
5097              ## NOTE: Reprocess.
5098              
5099              #
5100            } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
5101              !!!cp ('t149.2');
5102    
5103              ## NOTE: As if </head>
5104              pop @{$self->{open_elements}};
5105              #$self->{insertion_mode} = IN_AFTER_HEAD_IM;
5106              ## NOTE: Reprocess.
5107    
5108              #
5109            } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5110              !!!cp ('t149.3');
5111    
5112              !!!parse-error (type => 'in noscript:#eof', token => $token);
5113    
5114              ## As if </noscript>
5115              pop @{$self->{open_elements}};
5116              #$self->{insertion_mode} = IN_HEAD_IM;
5117              ## NOTE: Reprocess.
5118    
5119              ## NOTE: As if </head>
5120              pop @{$self->{open_elements}};
5121              #$self->{insertion_mode} = IN_AFTER_HEAD_IM;
5122              ## NOTE: Reprocess.
5123    
5124              #
5125            } else {
5126              !!!cp ('t149.4');
5127              #
5128            }
5129    
5130            ## NOTE: As if <body>
5131            !!!insert-element ('body',, $token);
5132            $self->{insertion_mode} = IN_BODY_IM;
5133            ## NOTE: Reprocess.
5134            next B;
5135          } else {
5136            die "$0: $token->{type}: Unknown token type";
5137          }
5138    
5139            ## ISSUE: An issue in the spec.            ## ISSUE: An issue in the spec.
5140      } elsif ($self->{insertion_mode} & BODY_IMS) {      } elsif ($self->{insertion_mode} & BODY_IMS) {
# Line 3800  sub _tree_construction_main ($) { Line 5146  sub _tree_construction_main ($) {
5146              $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});              $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
5147    
5148              !!!next-token;              !!!next-token;
5149              redo B;              next B;
5150            } elsif ($token->{type} == START_TAG_TOKEN) {            } elsif ($token->{type} == START_TAG_TOKEN) {
5151              if ({              if ({
5152                   caption => 1, col => 1, colgroup => 1, tbody => 1,                   caption => 1, col => 1, colgroup => 1, tbody => 1,
# Line 3808  sub _tree_construction_main ($) { Line 5154  sub _tree_construction_main ($) {
5154                  }->{$token->{tag_name}}) {                  }->{$token->{tag_name}}) {
5155                if ($self->{insertion_mode} == IN_CELL_IM) {                if ($self->{insertion_mode} == IN_CELL_IM) {
5156                  ## have an element in table scope                  ## have an element in table scope
5157                  my $tn;                  for (reverse 0..$#{$self->{open_elements}}) {
                 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
5158                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
5159                    if ($node->[1] eq 'td' or $node->[1] eq 'th') {                    if ($node->[1] & TABLE_CELL_EL) {
5160                      !!!cp ('t151');                      !!!cp ('t151');
5161                      $tn = $node->[1];  
5162                      last INSCOPE;                      ## Close the cell
5163                    } elsif ({                      !!!back-token; # <x>
5164                              table => 1, html => 1,                      $token = {type => END_TAG_TOKEN,
5165                             }->{$node->[1]}) {                                tag_name => $node->[0]->manakai_local_name,
5166                                  line => $token->{line},
5167                                  column => $token->{column}};
5168                        next B;
5169                      } elsif ($node->[1] & TABLE_SCOPING_EL) {
5170                      !!!cp ('t152');                      !!!cp ('t152');
5171                      last INSCOPE;                      ## ISSUE: This case can never be reached, maybe.
5172                    }                      last;
                 } # INSCOPE  
                   unless (defined $tn) {  
                     !!!cp ('t153');  
                     !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
                     ## Ignore the token  
                     !!!next-token;  
                     redo B;  
5173                    }                    }
5174                                    }
5175                  !!!cp ('t154');  
5176                  ## Close the cell                  !!!cp ('t153');
5177                  !!!back-token; # <?>                  !!!parse-error (type => 'start tag not allowed',
5178                  $token = {type => END_TAG_TOKEN, tag_name => $tn};                      text => $token->{tag_name}, token => $token);
5179                  redo B;                  ## Ignore the token
5180                    !!!nack ('t153.1');
5181                    !!!next-token;
5182                    next B;
5183                } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {                } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {
5184                  !!!parse-error (type => 'not closed:caption');                  !!!parse-error (type => 'not closed', text => 'caption',
5185                                    token => $token);
5186                                    
5187                  ## As if </caption>                  ## NOTE: As if </caption>.
5188                  ## have a table element in table scope                  ## have a table element in table scope
5189                  my $i;                  my $i;
5190                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: {
5191                    my $node = $self->{open_elements}->[$_];                    for (reverse 0..$#{$self->{open_elements}}) {
5192                    if ($node->[1] eq 'caption') {                      my $node = $self->{open_elements}->[$_];
5193                      !!!cp ('t155');                      if ($node->[1] & CAPTION_EL) {
5194                      $i = $_;                        !!!cp ('t155');
5195                      last INSCOPE;                        $i = $_;
5196                    } elsif ({                        last INSCOPE;
5197                              table => 1, html => 1,                      } elsif ($node->[1] & TABLE_SCOPING_EL) {
5198                             }->{$node->[1]}) {                        !!!cp ('t156');
5199                      !!!cp ('t156');                        last;
5200                      last INSCOPE;                      }
5201                    }                    }
5202    
5203                      !!!cp ('t157');
5204                      !!!parse-error (type => 'start tag not allowed',
5205                                      text => $token->{tag_name}, token => $token);
5206                      ## Ignore the token
5207                      !!!nack ('t157.1');
5208                      !!!next-token;
5209                      next B;
5210                  } # INSCOPE                  } # INSCOPE
                   unless (defined $i) {  
                     !!!cp ('t157');  
                     !!!parse-error (type => 'unmatched end tag:caption');  
                     ## Ignore the token  
                     !!!next-token;  
                     redo B;  
                   }  
5211                                    
5212                  ## generate implied end tags                  ## generate implied end tags
5213                  if ({                  while ($self->{open_elements}->[-1]->[1]
5214                       dd => 1, dt => 1, li => 1, p => 1,                             & END_TAG_OPTIONAL_EL) {
                      td => 1, th => 1, tr => 1,  
                      tbody => 1, tfoot=> 1, thead => 1,  
                     }->{$self->{open_elements}->[-1]->[1]}) {  
5215                    !!!cp ('t158');                    !!!cp ('t158');
5216                    !!!back-token; # <?>                    pop @{$self->{open_elements}};
                   $token = {type => END_TAG_TOKEN, tag_name => 'caption'};  
                   !!!back-token;  
                   $token = {type => END_TAG_TOKEN,  
                             tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
                   redo B;  
5217                  }                  }
5218    
5219                  if ($self->{open_elements}->[-1]->[1] ne 'caption') {                  unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
5220                    !!!cp ('t159');                    !!!cp ('t159');
5221                    !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                    !!!parse-error (type => 'not closed',
5222                                      text => $self->{open_elements}->[-1]->[0]
5223                                          ->manakai_local_name,
5224                                      token => $token);
5225                  } else {                  } else {
5226                    !!!cp ('t160');                    !!!cp ('t160');
5227                  }                  }
# Line 3891  sub _tree_construction_main ($) { Line 5233  sub _tree_construction_main ($) {
5233                  $self->{insertion_mode} = IN_TABLE_IM;                  $self->{insertion_mode} = IN_TABLE_IM;
5234                                    
5235                  ## reprocess                  ## reprocess
5236                  redo B;                  !!!ack-later;
5237                    next B;
5238                } else {                } else {
5239                  !!!cp ('t161');                  !!!cp ('t161');
5240                  #                  #
# Line 3907  sub _tree_construction_main ($) { Line 5250  sub _tree_construction_main ($) {
5250                  my $i;                  my $i;
5251                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5252                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
5253                    if ($node->[1] eq $token->{tag_name}) {                    if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
5254                      !!!cp ('t163');                      !!!cp ('t163');
5255                      $i = $_;                      $i = $_;
5256                      last INSCOPE;                      last INSCOPE;
5257                    } elsif ({                    } elsif ($node->[1] & TABLE_SCOPING_EL) {
                             table => 1, html => 1,  
                            }->{$node->[1]}) {  
5258                      !!!cp ('t164');                      !!!cp ('t164');
5259                      last INSCOPE;                      last INSCOPE;
5260                    }                    }
5261                  } # INSCOPE                  } # INSCOPE
5262                    unless (defined $i) {                    unless (defined $i) {
5263                      !!!cp ('t165');                      !!!cp ('t165');
5264                      !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                      !!!parse-error (type => 'unmatched end tag',
5265                                        text => $token->{tag_name},
5266                                        token => $token);
5267                      ## Ignore the token                      ## Ignore the token
5268                      !!!next-token;                      !!!next-token;
5269                      redo B;                      next B;
5270                    }                    }
5271                                    
5272                  ## generate implied end tags                  ## generate implied end tags
5273                  if ({                  while ($self->{open_elements}->[-1]->[1]
5274                       dd => 1, dt => 1, li => 1, p => 1,                             & END_TAG_OPTIONAL_EL) {
                      td => ($token->{tag_name} eq 'th'),  
                      th => ($token->{tag_name} eq 'td'),  
                      tr => 1,  
                      tbody => 1, tfoot=> 1, thead => 1,  
                     }->{$self->{open_elements}->[-1]->[1]}) {  
5275                    !!!cp ('t166');                    !!!cp ('t166');
5276                    !!!back-token;                    pop @{$self->{open_elements}};
                   $token = {type => END_TAG_TOKEN,  
                             tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
                   redo B;  
5277                  }                  }
5278                    
5279                  if ($self->{open_elements}->[-1]->[1] ne $token->{tag_name}) {                  if ($self->{open_elements}->[-1]->[0]->manakai_local_name
5280                            ne $token->{tag_name}) {
5281                    !!!cp ('t167');                    !!!cp ('t167');
5282                    !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                    !!!parse-error (type => 'not closed',
5283                                      text => $self->{open_elements}->[-1]->[0]
5284                                          ->manakai_local_name,
5285                                      token => $token);
5286                  } else {                  } else {
5287                    !!!cp ('t168');                    !!!cp ('t168');
5288                  }                  }
# Line 3955  sub _tree_construction_main ($) { Line 5294  sub _tree_construction_main ($) {
5294                  $self->{insertion_mode} = IN_ROW_IM;                  $self->{insertion_mode} = IN_ROW_IM;
5295                                    
5296                  !!!next-token;                  !!!next-token;
5297                  redo B;                  next B;
5298                } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {                } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {
5299                  !!!cp ('t169');                  !!!cp ('t169');
5300                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                  !!!parse-error (type => 'unmatched end tag',
5301                                    text => $token->{tag_name}, token => $token);
5302                  ## Ignore the token                  ## Ignore the token
5303                  !!!next-token;                  !!!next-token;
5304                  redo B;                  next B;
5305                } else {                } else {
5306                  !!!cp ('t170');                  !!!cp ('t170');
5307                  #                  #
# Line 3970  sub _tree_construction_main ($) { Line 5310  sub _tree_construction_main ($) {
5310                if ($self->{insertion_mode} == IN_CAPTION_IM) {                if ($self->{insertion_mode} == IN_CAPTION_IM) {
5311                  ## have a table element in table scope                  ## have a table element in table scope
5312                  my $i;                  my $i;
5313                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: {
5314                    my $node = $self->{open_elements}->[$_];                    for (reverse 0..$#{$self->{open_elements}}) {
5315                    if ($node->[1] eq $token->{tag_name}) {                      my $node = $self->{open_elements}->[$_];
5316                      !!!cp ('t171');                      if ($node->[1] & CAPTION_EL) {
5317                      $i = $_;                        !!!cp ('t171');
5318                      last INSCOPE;                        $i = $_;
5319                    } elsif ({                        last INSCOPE;
5320                              table => 1, html => 1,                      } elsif ($node->[1] & TABLE_SCOPING_EL) {
5321                             }->{$node->[1]}) {                        !!!cp ('t172');
5322                      !!!cp ('t172');                        last;
5323                      last INSCOPE;                      }
5324                    }                    }
5325    
5326                      !!!cp ('t173');
5327                      !!!parse-error (type => 'unmatched end tag',
5328                                      text => $token->{tag_name}, token => $token);
5329                      ## Ignore the token
5330                      !!!next-token;
5331                      next B;
5332                  } # INSCOPE                  } # INSCOPE
                   unless (defined $i) {  
                     !!!cp ('t173');  
                     !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
                     ## Ignore the token  
                     !!!next-token;  
                     redo B;  
                   }  
5333                                    
5334                  ## generate implied end tags                  ## generate implied end tags
5335                  if ({                  while ($self->{open_elements}->[-1]->[1]
5336                       dd => 1, dt => 1, li => 1, p => 1,                             & END_TAG_OPTIONAL_EL) {
                      td => 1, th => 1, tr => 1,  
                      tbody => 1, tfoot=> 1, thead => 1,  
                     }->{$self->{open_elements}->[-1]->[1]}) {  
5337                    !!!cp ('t174');                    !!!cp ('t174');
5338                    !!!back-token;                    pop @{$self->{open_elements}};
                   $token = {type => END_TAG_TOKEN,  
                             tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
                   redo B;  
5339                  }                  }
5340                                    
5341                  if ($self->{open_elements}->[-1]->[1] ne 'caption') {                  unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
5342                    !!!cp ('t175');                    !!!cp ('t175');
5343                    !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                    !!!parse-error (type => 'not closed',
5344                                      text => $self->{open_elements}->[-1]->[0]
5345                                          ->manakai_local_name,
5346                                      token => $token);
5347                  } else {                  } else {
5348                    !!!cp ('t176');                    !!!cp ('t176');
5349                  }                  }
# Line 4018  sub _tree_construction_main ($) { Line 5355  sub _tree_construction_main ($) {
5355                  $self->{insertion_mode} = IN_TABLE_IM;                  $self->{insertion_mode} = IN_TABLE_IM;
5356                                    
5357                  !!!next-token;                  !!!next-token;
5358                  redo B;                  next B;
5359                } elsif ($self->{insertion_mode} == IN_CELL_IM) {                } elsif ($self->{insertion_mode} == IN_CELL_IM) {
5360                  !!!cp ('t177');                  !!!cp ('t177');
5361                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                  !!!parse-error (type => 'unmatched end tag',
5362                                    text => $token->{tag_name}, token => $token);
5363                  ## Ignore the token                  ## Ignore the token
5364                  !!!next-token;                  !!!next-token;
5365                  redo B;                  next B;
5366                } else {                } else {
5367                  !!!cp ('t178');                  !!!cp ('t178');
5368                  #                  #
# Line 4037  sub _tree_construction_main ($) { Line 5375  sub _tree_construction_main ($) {
5375                ## have an element in table scope                ## have an element in table scope
5376                my $i;                my $i;
5377                my $tn;                my $tn;
5378                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: {
5379                  my $node = $self->{open_elements}->[$_];                  for (reverse 0..$#{$self->{open_elements}}) {
5380                  if ($node->[1] eq $token->{tag_name}) {                    my $node = $self->{open_elements}->[$_];
5381                    !!!cp ('t179');                    if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
5382                    $i = $_;                      !!!cp ('t179');
5383                    last INSCOPE;                      $i = $_;
5384                  } elsif ($node->[1] eq 'td' or $node->[1] eq 'th') {  
5385                    !!!cp ('t180');                      ## Close the cell
5386                    $tn = $node->[1];                      !!!back-token; # </x>
5387                    ## NOTE: There is exactly one |td| or |th| element                      $token = {type => END_TAG_TOKEN, tag_name => $tn,
5388                    ## in scope in the stack of open elements by definition.                                line => $token->{line},
5389                  } elsif ({                                column => $token->{column}};
5390                            table => 1, html => 1,                      next B;
5391                           }->{$node->[1]}) {                    } elsif ($node->[1] & TABLE_CELL_EL) {
5392                    !!!cp ('t181');                      !!!cp ('t180');
5393                    last INSCOPE;                      $tn = $node->[0]->manakai_local_name;
5394                        ## NOTE: There is exactly one |td| or |th| element
5395                        ## in scope in the stack of open elements by definition.
5396                      } elsif ($node->[1] & TABLE_SCOPING_EL) {
5397                        ## ISSUE: Can this be reached?
5398                        !!!cp ('t181');
5399                        last;
5400                      }
5401                  }                  }
5402                } # INSCOPE  
               unless (defined $i) {  
5403                  !!!cp ('t182');                  !!!cp ('t182');
5404                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                  !!!parse-error (type => 'unmatched end tag',
5405                        text => $token->{tag_name}, token => $token);
5406                  ## Ignore the token                  ## Ignore the token
5407                  !!!next-token;                  !!!next-token;
5408                  redo B;                  next B;
5409                } else {                } # INSCOPE
                 !!!cp ('t183');  
               }  
   
               ## Close the cell  
               !!!back-token; # </?>  
               $token = {type => END_TAG_TOKEN, tag_name => $tn};  
               redo B;  
5410              } elsif ($token->{tag_name} eq 'table' and              } elsif ($token->{tag_name} eq 'table' and
5411                       $self->{insertion_mode} == IN_CAPTION_IM) {                       $self->{insertion_mode} == IN_CAPTION_IM) {
5412                !!!parse-error (type => 'not closed:caption');                !!!parse-error (type => 'not closed', text => 'caption',
5413                                  token => $token);
5414    
5415                ## As if </caption>                ## As if </caption>
5416                ## have a table element in table scope                ## have a table element in table scope
5417                my $i;                my $i;
5418                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5419                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
5420                  if ($node->[1] eq 'caption') {                  if ($node->[1] & CAPTION_EL) {
5421                    !!!cp ('t184');                    !!!cp ('t184');
5422                    $i = $_;                    $i = $_;
5423                    last INSCOPE;                    last INSCOPE;
5424                  } elsif ({                  } elsif ($node->[1] & TABLE_SCOPING_EL) {
                           table => 1, html => 1,  
                          }->{$node->[1]}) {  
5425                    !!!cp ('t185');                    !!!cp ('t185');
5426                    last INSCOPE;                    last INSCOPE;
5427                  }                  }
5428                } # INSCOPE                } # INSCOPE
5429                unless (defined $i) {                unless (defined $i) {
5430                  !!!cp ('t186');                  !!!cp ('t186');
5431                  !!!parse-error (type => 'unmatched end tag:caption');                  !!!parse-error (type => 'unmatched end tag',
5432                                    text => 'caption', token => $token);
5433                  ## Ignore the token                  ## Ignore the token
5434                  !!!next-token;                  !!!next-token;
5435                  redo B;                  next B;
5436                }                }
5437                                
5438                ## generate implied end tags                ## generate implied end tags
5439                if ({                while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
                    dd => 1, dt => 1, li => 1, p => 1,  
                    td => 1, th => 1, tr => 1,  
                    tbody => 1, tfoot=> 1, thead => 1,  
                   }->{$self->{open_elements}->[-1]->[1]}) {  
5440                  !!!cp ('t187');                  !!!cp ('t187');
5441                  !!!back-token; # </table>                  pop @{$self->{open_elements}};
                 $token = {type => END_TAG_TOKEN, tag_name => 'caption'};  
                 !!!back-token;  
                 $token = {type => END_TAG_TOKEN,  
                           tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
                 redo B;  
5442                }                }
5443    
5444                if ($self->{open_elements}->[-1]->[1] ne 'caption') {                unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
5445                  !!!cp ('t188');                  !!!cp ('t188');
5446                  !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                  !!!parse-error (type => 'not closed',
5447                                    text => $self->{open_elements}->[-1]->[0]
5448                                        ->manakai_local_name,
5449                                    token => $token);
5450                } else {                } else {
5451                  !!!cp ('t189');                  !!!cp ('t189');
5452                }                }
# Line 4126  sub _tree_construction_main ($) { Line 5458  sub _tree_construction_main ($) {
5458                $self->{insertion_mode} = IN_TABLE_IM;                $self->{insertion_mode} = IN_TABLE_IM;
5459    
5460                ## reprocess                ## reprocess
5461                redo B;                next B;
5462              } elsif ({              } elsif ({
5463                        body => 1, col => 1, colgroup => 1, html => 1,                        body => 1, col => 1, colgroup => 1, html => 1,
5464                       }->{$token->{tag_name}}) {                       }->{$token->{tag_name}}) {
5465                if ($self->{insertion_mode} & BODY_TABLE_IMS) {                if ($self->{insertion_mode} & BODY_TABLE_IMS) {
5466                  !!!cp ('t190');                  !!!cp ('t190');
5467                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                  !!!parse-error (type => 'unmatched end tag',
5468                                    text => $token->{tag_name}, token => $token);
5469                  ## Ignore the token                  ## Ignore the token
5470                  !!!next-token;                  !!!next-token;
5471                  redo B;                  next B;
5472                } else {                } else {
5473                  !!!cp ('t191');                  !!!cp ('t191');
5474                  #                  #
# Line 4146  sub _tree_construction_main ($) { Line 5479  sub _tree_construction_main ($) {
5479                       }->{$token->{tag_name}} and                       }->{$token->{tag_name}} and
5480                       $self->{insertion_mode} == IN_CAPTION_IM) {                       $self->{insertion_mode} == IN_CAPTION_IM) {
5481                !!!cp ('t192');                !!!cp ('t192');
5482                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                !!!parse-error (type => 'unmatched end tag',
5483                                  text => $token->{tag_name}, token => $token);
5484                ## Ignore the token                ## Ignore the token
5485                !!!next-token;                !!!next-token;
5486                redo B;                next B;
5487              } else {              } else {
5488                !!!cp ('t193');                !!!cp ('t193');
5489                #                #
5490              }              }
5491          } elsif ($token->{type} == END_OF_FILE_TOKEN) {
5492            for my $entry (@{$self->{open_elements}}) {
5493              unless ($entry->[1] & ALL_END_TAG_OPTIONAL_EL) {
5494                !!!cp ('t75');
5495                !!!parse-error (type => 'in body:#eof', token => $token);
5496                last;
5497              }
5498            }
5499    
5500            ## Stop parsing.
5501            last B;
5502        } else {        } else {
5503          die "$0: $token->{type}: Unknown token type";          die "$0: $token->{type}: Unknown token type";
5504        }        }
# Line 4162  sub _tree_construction_main ($) { Line 5507  sub _tree_construction_main ($) {
5507        #        #
5508      } elsif ($self->{insertion_mode} & TABLE_IMS) {      } elsif ($self->{insertion_mode} & TABLE_IMS) {
5509        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
5510              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if (not $open_tables->[-1]->[1] and # tainted
5511                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);              $token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
5512              $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
5513                                
5514                unless (length $token->{data}) {            unless (length $token->{data}) {
5515                  !!!cp ('t194');              !!!cp ('t194');
5516                  !!!next-token;              !!!next-token;
5517                  redo B;              next B;
5518                } else {            } else {
5519                  !!!cp ('t195');              !!!cp ('t195');
5520                }            }
5521              }          }
5522    
5523              !!!parse-error (type => 'in table:#character');          !!!parse-error (type => 'in table:#text', token => $token);
5524    
5525              ## As if in body, but insert into foster parent element              ## As if in body, but insert into foster parent element
5526              ## ISSUE: Spec says that "whenever a node would be inserted              ## ISSUE: Spec says that "whenever a node would be inserted
# Line 4182  sub _tree_construction_main ($) { Line 5528  sub _tree_construction_main ($) {
5528              ## result in a new Text node.              ## result in a new Text node.
5529              $reconstruct_active_formatting_elements->($insert_to_foster);              $reconstruct_active_formatting_elements->($insert_to_foster);
5530                            
5531              if ({              if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {
                  table => 1, tbody => 1, tfoot => 1,  
                  thead => 1, tr => 1,  
                 }->{$self->{open_elements}->[-1]->[1]}) {  
5532                # MUST                # MUST
5533                my $foster_parent_element;                my $foster_parent_element;
5534                my $next_sibling;                my $next_sibling;
5535                my $prev_sibling;                my $prev_sibling;
5536                OE: for (reverse 0..$#{$self->{open_elements}}) {                OE: for (reverse 0..$#{$self->{open_elements}}) {
5537                  if ($self->{open_elements}->[$_]->[1] eq 'table') {                  if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {
5538                    my $parent = $self->{open_elements}->[$_]->[0]->parent_node;                    my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
5539                    if (defined $parent and $parent->node_type == 1) {                    if (defined $parent and $parent->node_type == 1) {
5540                      !!!cp ('t196');                      !!!cp ('t196');
# Line 4219  sub _tree_construction_main ($) { Line 5562  sub _tree_construction_main ($) {
5562                    ($self->{document}->create_text_node ($token->{data}),                    ($self->{document}->create_text_node ($token->{data}),
5563                     $next_sibling);                     $next_sibling);
5564                }                }
5565              } else {            $open_tables->[-1]->[1] = 1; # tainted
5566                !!!cp ('t200');          } else {
5567                $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});            !!!cp ('t200');
5568              }            $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
5569            }
5570                            
5571              !!!next-token;          !!!next-token;
5572              redo B;          next B;
5573        } elsif ($token->{type} == START_TAG_TOKEN) {        } elsif ($token->{type} == START_TAG_TOKEN) {
5574              if ({          if ({
5575                   tr => ($self->{insertion_mode} != IN_ROW_IM),               tr => ($self->{insertion_mode} != IN_ROW_IM),
5576                   th => 1, td => 1,               th => 1, td => 1,
5577                  }->{$token->{tag_name}}) {              }->{$token->{tag_name}}) {
5578                if ($self->{insertion_mode} == IN_TABLE_IM) {            if ($self->{insertion_mode} == IN_TABLE_IM) {
5579                  ## Clear back to table context              ## Clear back to table context
5580                  while ($self->{open_elements}->[-1]->[1] ne 'table' and              while (not ($self->{open_elements}->[-1]->[1]
5581                         $self->{open_elements}->[-1]->[1] ne 'html') {                              & TABLE_SCOPING_EL)) {
5582                    !!!cp ('t201');                !!!cp ('t201');
5583                    !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                pop @{$self->{open_elements}};
5584                    pop @{$self->{open_elements}};              }
5585                  }              
5586                                !!!insert-element ('tbody',, $token);
5587                  !!!insert-element ('tbody');              $self->{insertion_mode} = IN_TABLE_BODY_IM;
5588                  $self->{insertion_mode} = IN_TABLE_BODY_IM;              ## reprocess in the "in table body" insertion mode...
5589                  ## reprocess in the "in table body" insertion mode...            }
5590                }            
5591              if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {
5592                if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {              unless ($token->{tag_name} eq 'tr') {
5593                  unless ($token->{tag_name} eq 'tr') {                !!!cp ('t202');
5594                    !!!cp ('t202');                !!!parse-error (type => 'missing start tag:tr', token => $token);
5595                    !!!parse-error (type => 'missing start tag:tr');              }
                 }  
5596                                    
5597                  ## Clear back to table body context              ## Clear back to table body context
5598                  while (not {              while (not ($self->{open_elements}->[-1]->[1]
5599                    tbody => 1, tfoot => 1, thead => 1, html => 1,                              & TABLE_ROWS_SCOPING_EL)) {
5600                  }->{$self->{open_elements}->[-1]->[1]}) {                !!!cp ('t203');
5601                    !!!cp ('t203');                ## ISSUE: Can this case be reached?
5602                    !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                pop @{$self->{open_elements}};
5603                    pop @{$self->{open_elements}};              }
                 }  
5604                                    
5605                  $self->{insertion_mode} = IN_ROW_IM;                  $self->{insertion_mode} = IN_ROW_IM;
5606                  if ($token->{tag_name} eq 'tr') {                  if ($token->{tag_name} eq 'tr') {
5607                    !!!cp ('t204');                    !!!cp ('t204');
5608                    !!!insert-element ($token->{tag_name}, $token->{attributes});                    !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5609                      !!!nack ('t204');
5610                    !!!next-token;                    !!!next-token;
5611                    redo B;                    next B;
5612                  } else {                  } else {
5613                    !!!cp ('t205');                    !!!cp ('t205');
5614                    !!!insert-element ('tr');                    !!!insert-element ('tr',, $token);
5615                    ## reprocess in the "in row" insertion mode                    ## reprocess in the "in row" insertion mode
5616                  }                  }
5617                } else {                } else {
# Line 4276  sub _tree_construction_main ($) { Line 5619  sub _tree_construction_main ($) {
5619                }                }
5620    
5621                ## Clear back to table row context                ## Clear back to table row context
5622                while (not {                while (not ($self->{open_elements}->[-1]->[1]
5623                  tr => 1, html => 1,                                & TABLE_ROW_SCOPING_EL)) {
               }->{$self->{open_elements}->[-1]->[1]}) {  
5624                  !!!cp ('t207');                  !!!cp ('t207');
                 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
5625                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
5626                }                }
5627                                
5628                !!!insert-element ($token->{tag_name}, $token->{attributes});                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5629                $self->{insertion_mode} = IN_CELL_IM;                $self->{insertion_mode} = IN_CELL_IM;
5630    
5631                push @$active_formatting_elements, ['#marker', ''];                push @$active_formatting_elements, ['#marker', ''];
5632                                
5633                  !!!nack ('t207.1');
5634                !!!next-token;                !!!next-token;
5635                redo B;                next B;
5636              } elsif ({              } elsif ({
5637                        caption => 1, col => 1, colgroup => 1,                        caption => 1, col => 1, colgroup => 1,
5638                        tbody => 1, tfoot => 1, thead => 1,                        tbody => 1, tfoot => 1, thead => 1,
# Line 4302  sub _tree_construction_main ($) { Line 5644  sub _tree_construction_main ($) {
5644                  my $i;                  my $i;
5645                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5646                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
5647                    if ($node->[1] eq 'tr') {                    if ($node->[1] & TABLE_ROW_EL) {
5648                      !!!cp ('t208');                      !!!cp ('t208');
5649                      $i = $_;                      $i = $_;
5650                      last INSCOPE;                      last INSCOPE;
5651                    } elsif ({                    } elsif ($node->[1] & TABLE_SCOPING_EL) {
                             table => 1, html => 1,  
                            }->{$node->[1]}) {  
5652                      !!!cp ('t209');                      !!!cp ('t209');
5653                      last INSCOPE;                      last INSCOPE;
5654                    }                    }
5655                  } # INSCOPE                  } # INSCOPE
5656                  unless (defined $i) {                  unless (defined $i) {
5657                   !!!cp ('t210');                    !!!cp ('t210');
5658                   !!!parse-error (type => 'unmacthed end tag:'.$token->{tag_name});  ## TODO: This type is wrong.
5659                      !!!parse-error (type => 'unmacthed end tag',
5660                                      text => $token->{tag_name}, token => $token);
5661                    ## Ignore the token                    ## Ignore the token
5662                      !!!nack ('t210.1');
5663                    !!!next-token;                    !!!next-token;
5664                    redo B;                    next B;
5665                  }                  }
5666                                    
5667                  ## Clear back to table row context                  ## Clear back to table row context
5668                  while (not {                  while (not ($self->{open_elements}->[-1]->[1]
5669                    tr => 1, html => 1,                                  & TABLE_ROW_SCOPING_EL)) {
                 }->{$self->{open_elements}->[-1]->[1]}) {  
5670                    !!!cp ('t211');                    !!!cp ('t211');
5671                    !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                    ## ISSUE: Can this case be reached?
5672                    pop @{$self->{open_elements}};                    pop @{$self->{open_elements}};
5673                  }                  }
5674                                    
# Line 4335  sub _tree_construction_main ($) { Line 5677  sub _tree_construction_main ($) {
5677                  if ($token->{tag_name} eq 'tr') {                  if ($token->{tag_name} eq 'tr') {
5678                    !!!cp ('t212');                    !!!cp ('t212');
5679                    ## reprocess                    ## reprocess
5680                    redo B;                    !!!ack-later;
5681                      next B;
5682                  } else {                  } else {
5683                    !!!cp ('t213');                    !!!cp ('t213');
5684                    ## reprocess in the "in table body" insertion mode...                    ## reprocess in the "in table body" insertion mode...
# Line 4347  sub _tree_construction_main ($) { Line 5690  sub _tree_construction_main ($) {
5690                  my $i;                  my $i;
5691                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5692                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
5693                    if ({                    if ($node->[1] & TABLE_ROW_GROUP_EL) {
                        tbody => 1, thead => 1, tfoot => 1,  
                       }->{$node->[1]}) {  
5694                      !!!cp ('t214');                      !!!cp ('t214');
5695                      $i = $_;                      $i = $_;
5696                      last INSCOPE;                      last INSCOPE;
5697                    } elsif ({                    } elsif ($node->[1] & TABLE_SCOPING_EL) {
                             table => 1, html => 1,  
                            }->{$node->[1]}) {  
5698                      !!!cp ('t215');                      !!!cp ('t215');
5699                      last INSCOPE;                      last INSCOPE;
5700                    }                    }
5701                  } # INSCOPE                  } # INSCOPE
5702                  unless (defined $i) {                  unless (defined $i) {
5703                    !!!cp ('t216');                    !!!cp ('t216');
5704                    !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  ## TODO: This erorr type is wrong.
5705                      !!!parse-error (type => 'unmatched end tag',
5706                                      text => $token->{tag_name}, token => $token);
5707                    ## Ignore the token                    ## Ignore the token
5708                      !!!nack ('t216.1');
5709                    !!!next-token;                    !!!next-token;
5710                    redo B;                    next B;
5711                  }                  }
5712    
5713                  ## Clear back to table body context                  ## Clear back to table body context
5714                  while (not {                  while (not ($self->{open_elements}->[-1]->[1]
5715                    tbody => 1, tfoot => 1, thead => 1, html => 1,                                  & TABLE_ROWS_SCOPING_EL)) {
                 }->{$self->{open_elements}->[-1]->[1]}) {  
5716                    !!!cp ('t217');                    !!!cp ('t217');
5717                    !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                    ## ISSUE: Can this state be reached?
5718                    pop @{$self->{open_elements}};                    pop @{$self->{open_elements}};
5719                  }                  }
5720                                    
# Line 4393  sub _tree_construction_main ($) { Line 5734  sub _tree_construction_main ($) {
5734    
5735                if ($token->{tag_name} eq 'col') {                if ($token->{tag_name} eq 'col') {
5736                  ## Clear back to table context                  ## Clear back to table context
5737                  while ($self->{open_elements}->[-1]->[1] ne 'table' and                  while (not ($self->{open_elements}->[-1]->[1]
5738                         $self->{open_elements}->[-1]->[1] ne 'html') {                                  & TABLE_SCOPING_EL)) {
5739                    !!!cp ('t219');                    !!!cp ('t219');
5740                    !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                    ## ISSUE: Can this state be reached?
5741                    pop @{$self->{open_elements}};                    pop @{$self->{open_elements}};
5742                  }                  }
5743                                    
5744                  !!!insert-element ('colgroup');                  !!!insert-element ('colgroup',, $token);
5745                  $self->{insertion_mode} = IN_COLUMN_GROUP_IM;                  $self->{insertion_mode} = IN_COLUMN_GROUP_IM;
5746                  ## reprocess                  ## reprocess
5747                  redo B;                  !!!ack-later;
5748                    next B;
5749                } elsif ({                } elsif ({
5750                          caption => 1,                          caption => 1,
5751                          colgroup => 1,                          colgroup => 1,
5752                          tbody => 1, tfoot => 1, thead => 1,                          tbody => 1, tfoot => 1, thead => 1,
5753                         }->{$token->{tag_name}}) {                         }->{$token->{tag_name}}) {
5754                  ## Clear back to table context                  ## Clear back to table context
5755                  while ($self->{open_elements}->[-1]->[1] ne 'table' and                  while (not ($self->{open_elements}->[-1]->[1]
5756                         $self->{open_elements}->[-1]->[1] ne 'html') {                                  & TABLE_SCOPING_EL)) {
5757                    !!!cp ('t220');                    !!!cp ('t220');
5758                    !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                    ## ISSUE: Can this state be reached?
5759                    pop @{$self->{open_elements}};                    pop @{$self->{open_elements}};
5760                  }                  }
5761                                    
5762                  push @$active_formatting_elements, ['#marker', '']                  push @$active_formatting_elements, ['#marker', '']
5763                      if $token->{tag_name} eq 'caption';                      if $token->{tag_name} eq 'caption';
5764                                    
5765                  !!!insert-element ($token->{tag_name}, $token->{attributes});                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5766                  $self->{insertion_mode} = {                  $self->{insertion_mode} = {
5767                                             caption => IN_CAPTION_IM,                                             caption => IN_CAPTION_IM,
5768                                             colgroup => IN_COLUMN_GROUP_IM,                                             colgroup => IN_COLUMN_GROUP_IM,
# Line 4429  sub _tree_construction_main ($) { Line 5771  sub _tree_construction_main ($) {
5771                                             thead => IN_TABLE_BODY_IM,                                             thead => IN_TABLE_BODY_IM,
5772                                            }->{$token->{tag_name}};                                            }->{$token->{tag_name}};
5773                  !!!next-token;                  !!!next-token;
5774                  redo B;                  !!!nack ('t220.1');
5775                    next B;
5776                } else {                } else {
5777                  die "$0: in table: <>: $token->{tag_name}";                  die "$0: in table: <>: $token->{tag_name}";
5778                }                }
5779              } elsif ($token->{tag_name} eq 'table') {              } elsif ($token->{tag_name} eq 'table') {
5780                !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                !!!parse-error (type => 'not closed',
5781                                  text => $self->{open_elements}->[-1]->[0]
5782                                      ->manakai_local_name,
5783                                  token => $token);
5784    
5785                ## As if </table>                ## As if </table>
5786                ## have a table element in table scope                ## have a table element in table scope
5787                my $i;                my $i;
5788                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5789                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
5790                  if ($node->[1] eq 'table') {                  if ($node->[1] & TABLE_EL) {
5791                    !!!cp ('t221');                    !!!cp ('t221');
5792                    $i = $_;                    $i = $_;
5793                    last INSCOPE;                    last INSCOPE;
5794                  } elsif ({                  } elsif ($node->[1] & TABLE_SCOPING_EL) {
                           table => 1, html => 1,  
                          }->{$node->[1]}) {  
5795                    !!!cp ('t222');                    !!!cp ('t222');
5796                    last INSCOPE;                    last INSCOPE;
5797                  }                  }
5798                } # INSCOPE                } # INSCOPE
5799                unless (defined $i) {                unless (defined $i) {
5800                  !!!cp ('t223');                  !!!cp ('t223');
5801                  !!!parse-error (type => 'unmatched end tag:table');  ## TODO: The following is wrong, maybe.
5802                    !!!parse-error (type => 'unmatched end tag', text => 'table',
5803                                    token => $token);
5804                  ## Ignore tokens </table><table>                  ## Ignore tokens </table><table>
5805                    !!!nack ('t223.1');
5806                  !!!next-token;                  !!!next-token;
5807                  redo B;                  next B;
5808                }                }
5809                                
5810    ## TODO: Followings are removed from the latest spec.
5811                ## generate implied end tags                ## generate implied end tags
5812                if ({                while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
                    dd => 1, dt => 1, li => 1, p => 1,  
                    td => 1, th => 1, tr => 1,  
                    tbody => 1, tfoot=> 1, thead => 1,  
                   }->{$self->{open_elements}->[-1]->[1]}) {  
5813                  !!!cp ('t224');                  !!!cp ('t224');
5814                  !!!back-token; # <table>                  pop @{$self->{open_elements}};
                 $token = {type => END_TAG_TOKEN, tag_name => 'table'};  
                 !!!back-token;  
                 $token = {type => END_TAG_TOKEN,  
                           tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
                 redo B;  
5815                }                }
5816    
5817                if ($self->{open_elements}->[-1]->[1] ne 'table') {                unless ($self->{open_elements}->[-1]->[1] & TABLE_EL) {
5818                  !!!cp ('t225');                  !!!cp ('t225');
5819                  !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                  ## NOTE: |<table><tr><table>|
5820                    !!!parse-error (type => 'not closed',
5821                                    text => $self->{open_elements}->[-1]->[0]
5822                                        ->manakai_local_name,
5823                                    token => $token);
5824                } else {                } else {
5825                  !!!cp ('t226');                  !!!cp ('t226');
5826                }                }
5827    
5828                splice @{$self->{open_elements}}, $i;                splice @{$self->{open_elements}}, $i;
5829                  pop @{$open_tables};
5830    
5831                $self->_reset_insertion_mode;                $self->_reset_insertion_mode;
5832    
5833                ## reprocess            ## reprocess
5834                redo B;            !!!ack-later;
5835              next B;
5836            } elsif ($token->{tag_name} eq 'style') {
5837              if (not $open_tables->[-1]->[1]) { # tainted
5838                !!!cp ('t227.8');
5839                ## NOTE: This is a "as if in head" code clone.
5840                $parse_rcdata->(CDATA_CONTENT_MODEL);
5841                next B;
5842              } else {
5843                !!!cp ('t227.7');
5844                #
5845              }
5846            } elsif ($token->{tag_name} eq 'script') {
5847              if (not $open_tables->[-1]->[1]) { # tainted
5848                !!!cp ('t227.6');
5849                ## NOTE: This is a "as if in head" code clone.
5850                $script_start_tag->();
5851                next B;
5852              } else {
5853                !!!cp ('t227.5');
5854                #
5855              }
5856            } elsif ($token->{tag_name} eq 'input') {
5857              if (not $open_tables->[-1]->[1]) { # tainted
5858                if ($token->{attributes}->{type}) { ## TODO: case
5859                  my $type = lc $token->{attributes}->{type}->{value};
5860                  if ($type eq 'hidden') {
5861                    !!!cp ('t227.3');
5862                    !!!parse-error (type => 'in table',
5863                                    text => $token->{tag_name}, token => $token);
5864    
5865                    !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5866    
5867                    ## TODO: form element pointer
5868    
5869                    pop @{$self->{open_elements}};
5870    
5871                    !!!next-token;
5872                    !!!ack ('t227.2.1');
5873                    next B;
5874                  } else {
5875                    !!!cp ('t227.2');
5876                    #
5877                  }
5878                } else {
5879                  !!!cp ('t227.1');
5880                  #
5881                }
5882              } else {
5883                !!!cp ('t227.4');
5884                #
5885              }
5886          } else {          } else {
5887            !!!cp ('t227');            !!!cp ('t227');
           !!!parse-error (type => 'in table:'.$token->{tag_name});  
   
           $insert = $insert_to_foster;  
5888            #            #
5889          }          }
5890    
5891            !!!parse-error (type => 'in table', text => $token->{tag_name},
5892                            token => $token);
5893    
5894            $insert = $insert_to_foster;
5895            #
5896        } elsif ($token->{type} == END_TAG_TOKEN) {        } elsif ($token->{type} == END_TAG_TOKEN) {
5897              if ($token->{tag_name} eq 'tr' and              if ($token->{tag_name} eq 'tr' and
5898                  $self->{insertion_mode} == IN_ROW_IM) {                  $self->{insertion_mode} == IN_ROW_IM) {
# Line 4502  sub _tree_construction_main ($) { Line 5900  sub _tree_construction_main ($) {
5900                my $i;                my $i;
5901                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5902                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
5903                  if ($node->[1] eq $token->{tag_name}) {                  if ($node->[1] & TABLE_ROW_EL) {
5904                    !!!cp ('t228');                    !!!cp ('t228');
5905                    $i = $_;                    $i = $_;
5906                    last INSCOPE;                    last INSCOPE;
5907                  } elsif ({                  } elsif ($node->[1] & TABLE_SCOPING_EL) {
                           table => 1, html => 1,  
                          }->{$node->[1]}) {  
5908                    !!!cp ('t229');                    !!!cp ('t229');
5909                    last INSCOPE;                    last INSCOPE;
5910                  }                  }
5911                } # INSCOPE                } # INSCOPE
5912                unless (defined $i) {                unless (defined $i) {
5913                  !!!cp ('t230');                  !!!cp ('t230');
5914                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                  !!!parse-error (type => 'unmatched end tag',
5915                                    text => $token->{tag_name}, token => $token);
5916                  ## Ignore the token                  ## Ignore the token
5917                    !!!nack ('t230.1');
5918                  !!!next-token;                  !!!next-token;
5919                  redo B;                  next B;
5920                } else {                } else {
5921                  !!!cp ('t232');                  !!!cp ('t232');
5922                }                }
5923    
5924                ## Clear back to table row context                ## Clear back to table row context
5925                while (not {                while (not ($self->{open_elements}->[-1]->[1]
5926                  tr => 1, html => 1,                                & TABLE_ROW_SCOPING_EL)) {
               }->{$self->{open_elements}->[-1]->[1]}) {  
5927                  !!!cp ('t231');                  !!!cp ('t231');
5928                  !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  ## ISSUE: Can this state be reached?
5929                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
5930                }                }
5931    
5932                pop @{$self->{open_elements}}; # tr                pop @{$self->{open_elements}}; # tr
5933                $self->{insertion_mode} = IN_TABLE_BODY_IM;                $self->{insertion_mode} = IN_TABLE_BODY_IM;
5934                !!!next-token;                !!!next-token;
5935                redo B;                !!!nack ('t231.1');
5936                  next B;
5937              } elsif ($token->{tag_name} eq 'table') {              } elsif ($token->{tag_name} eq 'table') {
5938                if ($self->{insertion_mode} == IN_ROW_IM) {                if ($self->{insertion_mode} == IN_ROW_IM) {
5939                  ## As if </tr>                  ## As if </tr>
# Line 4543  sub _tree_construction_main ($) { Line 5941  sub _tree_construction_main ($) {
5941                  my $i;                  my $i;
5942                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5943                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
5944                    if ($node->[1] eq 'tr') {                    if ($node->[1] & TABLE_ROW_EL) {
5945                      !!!cp ('t233');                      !!!cp ('t233');
5946                      $i = $_;                      $i = $_;
5947                      last INSCOPE;                      last INSCOPE;
5948                    } elsif ({                    } elsif ($node->[1] & TABLE_SCOPING_EL) {
                             table => 1, html => 1,  
                            }->{$node->[1]}) {  
5949                      !!!cp ('t234');                      !!!cp ('t234');
5950                      last INSCOPE;                      last INSCOPE;
5951                    }                    }
5952                  } # INSCOPE                  } # INSCOPE
5953                  unless (defined $i) {                  unless (defined $i) {
5954                    !!!cp ('t235');                    !!!cp ('t235');
5955                    !!!parse-error (type => 'unmatched end tag:'.$token->{type});  ## TODO: The following is wrong.
5956                      !!!parse-error (type => 'unmatched end tag',
5957                                      text => $token->{type}, token => $token);
5958                    ## Ignore the token                    ## Ignore the token
5959                      !!!nack ('t236.1');
5960                    !!!next-token;                    !!!next-token;
5961                    redo B;                    next B;
5962                  }                  }
5963                                    
5964                  ## Clear back to table row context                  ## Clear back to table row context
5965                  while (not {                  while (not ($self->{open_elements}->[-1]->[1]
5966                    tr => 1, html => 1,                                  & TABLE_ROW_SCOPING_EL)) {
                 }->{$self->{open_elements}->[-1]->[1]}) {  
5967                    !!!cp ('t236');                    !!!cp ('t236');
5968                    !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  ## ISSUE: Can this state be reached?
5969                    pop @{$self->{open_elements}};                    pop @{$self->{open_elements}};
5970                  }                  }
5971                                    
# Line 4581  sub _tree_construction_main ($) { Line 5979  sub _tree_construction_main ($) {
5979                  my $i;                  my $i;
5980                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5981                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
5982                    if ({                    if ($node->[1] & TABLE_ROW_GROUP_EL) {
                        tbody => 1, thead => 1, tfoot => 1,  
                       }->{$node->[1]}) {  
5983                      !!!cp ('t237');                      !!!cp ('t237');
5984                      $i = $_;                      $i = $_;
5985                      last INSCOPE;                      last INSCOPE;
5986                    } elsif ({                    } elsif ($node->[1] & TABLE_SCOPING_EL) {
                             table => 1, html => 1,  
                            }->{$node->[1]}) {  
5987                      !!!cp ('t238');                      !!!cp ('t238');
5988                      last INSCOPE;                      last INSCOPE;
5989                    }                    }
5990                  } # INSCOPE                  } # INSCOPE
5991                  unless (defined $i) {                  unless (defined $i) {
5992                    !!!cp ('t239');                    !!!cp ('t239');
5993                    !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                    !!!parse-error (type => 'unmatched end tag',
5994                                      text => $token->{tag_name}, token => $token);
5995                    ## Ignore the token                    ## Ignore the token
5996                      !!!nack ('t239.1');
5997                    !!!next-token;                    !!!next-token;
5998                    redo B;                    next B;
5999                  }                  }
6000                                    
6001                  ## Clear back to table body context                  ## Clear back to table body context
6002                  while (not {                  while (not ($self->{open_elements}->[-1]->[1]
6003                    tbody => 1, tfoot => 1, thead => 1, html => 1,                                  & TABLE_ROWS_SCOPING_EL)) {
                 }->{$self->{open_elements}->[-1]->[1]}) {  
6004                    !!!cp ('t240');                    !!!cp ('t240');
                   !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
6005                    pop @{$self->{open_elements}};                    pop @{$self->{open_elements}};
6006                  }                  }
6007                                    
# Line 4623  sub _tree_construction_main ($) { Line 6017  sub _tree_construction_main ($) {
6017                  ## reprocess in the "in table" insertion mode...                  ## reprocess in the "in table" insertion mode...
6018                }                }
6019    
6020                  ## NOTE: </table> in the "in table" insertion mode.
6021                  ## When you edit the code fragment below, please ensure that
6022                  ## the code for <table> in the "in table" insertion mode
6023                  ## is synced with it.
6024    
6025                ## have a table element in table scope                ## have a table element in table scope
6026                my $i;                my $i;
6027                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6028                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
6029                  if ($node->[1] eq $token->{tag_name}) {                  if ($node->[1] & TABLE_EL) {
6030                    !!!cp ('t241');                    !!!cp ('t241');
6031                    $i = $_;                    $i = $_;
6032                    last INSCOPE;                    last INSCOPE;
6033                  } elsif ({                  } elsif ($node->[1] & TABLE_SCOPING_EL) {
                           table => 1, html => 1,  
                          }->{$node->[1]}) {  
6034                    !!!cp ('t242');                    !!!cp ('t242');
6035                    last INSCOPE;                    last INSCOPE;
6036                  }                  }
6037                } # INSCOPE                } # INSCOPE
6038                unless (defined $i) {                unless (defined $i) {
6039                  !!!cp ('t243');                  !!!cp ('t243');
6040                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                  !!!parse-error (type => 'unmatched end tag',
6041                                    text => $token->{tag_name}, token => $token);
6042                  ## Ignore the token                  ## Ignore the token
6043                    !!!nack ('t243.1');
6044                  !!!next-token;                  !!!next-token;
6045                  redo B;                  next B;
               }  
   
               ## generate implied end tags  
               if ({  
                    dd => 1, dt => 1, li => 1, p => 1,  
                    td => 1, th => 1, tr => 1,  
                    tbody => 1, tfoot=> 1, thead => 1,  
                   }->{$self->{open_elements}->[-1]->[1]}) {  
                 !!!cp ('t244');  
                 !!!back-token;  
                 $token = {type => END_TAG_TOKEN,  
                           tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
                 redo B;  
               }  
                 
               if ($self->{open_elements}->[-1]->[1] ne 'table') {  
                 !!!cp ('t245');  
                 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
               } else {  
                 !!!cp ('t246');  
6046                }                }
6047                                    
6048                splice @{$self->{open_elements}}, $i;                splice @{$self->{open_elements}}, $i;
6049                  pop @{$open_tables};
6050                                
6051                $self->_reset_insertion_mode;                $self->_reset_insertion_mode;
6052                                
6053                !!!next-token;                !!!next-token;
6054                redo B;                next B;
6055              } elsif ({              } elsif ({
6056                        tbody => 1, tfoot => 1, thead => 1,                        tbody => 1, tfoot => 1, thead => 1,
6057                       }->{$token->{tag_name}} and                       }->{$token->{tag_name}} and
# Line 4681  sub _tree_construction_main ($) { Line 6061  sub _tree_construction_main ($) {
6061                  my $i;                  my $i;
6062                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6063                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
6064                    if ($node->[1] eq $token->{tag_name}) {                    if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
6065                      !!!cp ('t247');                      !!!cp ('t247');
6066                      $i = $_;                      $i = $_;
6067                      last INSCOPE;                      last INSCOPE;
6068                    } elsif ({                    } elsif ($node->[1] & TABLE_SCOPING_EL) {
                             table => 1, html => 1,  
                            }->{$node->[1]}) {  
6069                      !!!cp ('t248');                      !!!cp ('t248');
6070                      last INSCOPE;                      last INSCOPE;
6071                    }                    }
6072                  } # INSCOPE                  } # INSCOPE
6073                    unless (defined $i) {                    unless (defined $i) {
6074                      !!!cp ('t249');                      !!!cp ('t249');
6075                      !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                      !!!parse-error (type => 'unmatched end tag',
6076                                        text => $token->{tag_name}, token => $token);
6077                      ## Ignore the token                      ## Ignore the token
6078                        !!!nack ('t249.1');
6079                      !!!next-token;                      !!!next-token;
6080                      redo B;                      next B;
6081                    }                    }
6082                                    
6083                  ## As if </tr>                  ## As if </tr>
# Line 4705  sub _tree_construction_main ($) { Line 6085  sub _tree_construction_main ($) {
6085                  my $i;                  my $i;
6086                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6087                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
6088                    if ($node->[1] eq 'tr') {                    if ($node->[1] & TABLE_ROW_EL) {
6089                      !!!cp ('t250');                      !!!cp ('t250');
6090                      $i = $_;                      $i = $_;
6091                      last INSCOPE;                      last INSCOPE;
6092                    } elsif ({                    } elsif ($node->[1] & TABLE_SCOPING_EL) {
                             table => 1, html => 1,  
                            }->{$node->[1]}) {  
6093                      !!!cp ('t251');                      !!!cp ('t251');
6094                      last INSCOPE;                      last INSCOPE;
6095                    }                    }
6096                  } # INSCOPE                  } # INSCOPE
6097                    unless (defined $i) {                    unless (defined $i) {
6098                      !!!cp ('t252');                      !!!cp ('t252');
6099                      !!!parse-error (type => 'unmatched end tag:tr');                      !!!parse-error (type => 'unmatched end tag',
6100                                        text => 'tr', token => $token);
6101                      ## Ignore the token                      ## Ignore the token
6102                        !!!nack ('t252.1');
6103                      !!!next-token;                      !!!next-token;
6104                      redo B;                      next B;
6105                    }                    }
6106                                    
6107                  ## Clear back to table row context                  ## Clear back to table row context
6108                  while (not {                  while (not ($self->{open_elements}->[-1]->[1]
6109                    tr => 1, html => 1,                                  & TABLE_ROW_SCOPING_EL)) {
                 }->{$self->{open_elements}->[-1]->[1]}) {  
6110                    !!!cp ('t253');                    !!!cp ('t253');
6111                    !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  ## ISSUE: Can this case be reached?
6112                    pop @{$self->{open_elements}};                    pop @{$self->{open_elements}};
6113                  }                  }
6114                                    
# Line 4742  sub _tree_construction_main ($) { Line 6121  sub _tree_construction_main ($) {
6121                my $i;                my $i;
6122                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6123                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
6124                  if ($node->[1] eq $token->{tag_name}) {                  if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
6125                    !!!cp ('t254');                    !!!cp ('t254');
6126                    $i = $_;                    $i = $_;
6127                    last INSCOPE;                    last INSCOPE;
6128                  } elsif ({                  } elsif ($node->[1] & TABLE_SCOPING_EL) {
                           table => 1, html => 1,  
                          }->{$node->[1]}) {  
6129                    !!!cp ('t255');                    !!!cp ('t255');
6130                    last INSCOPE;                    last INSCOPE;
6131                  }                  }
6132                } # INSCOPE                } # INSCOPE
6133                unless (defined $i) {                unless (defined $i) {
6134                  !!!cp ('t256');                  !!!cp ('t256');
6135                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                  !!!parse-error (type => 'unmatched end tag',
6136                                    text => $token->{tag_name}, token => $token);
6137                  ## Ignore the token                  ## Ignore the token
6138                    !!!nack ('t256.1');
6139                  !!!next-token;                  !!!next-token;
6140                  redo B;                  next B;
6141                }                }
6142    
6143                ## Clear back to table body context                ## Clear back to table body context
6144                while (not {                while (not ($self->{open_elements}->[-1]->[1]
6145                  tbody => 1, tfoot => 1, thead => 1, html => 1,                                & TABLE_ROWS_SCOPING_EL)) {
               }->{$self->{open_elements}->[-1]->[1]}) {  
6146                  !!!cp ('t257');                  !!!cp ('t257');
6147                  !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  ## ISSUE: Can this case be reached?
6148                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
6149                }                }
6150    
6151                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
6152                $self->{insertion_mode} = IN_TABLE_IM;                $self->{insertion_mode} = IN_TABLE_IM;
6153                  !!!nack ('t257.1');
6154                !!!next-token;                !!!next-token;
6155                redo B;                next B;
6156              } elsif ({              } elsif ({
6157                        body => 1, caption => 1, col => 1, colgroup => 1,                        body => 1, caption => 1, col => 1, colgroup => 1,
6158                        html => 1, td => 1, th => 1,                        html => 1, td => 1, th => 1,
6159                        tr => 1, # $self->{insertion_mode} == IN_ROW_IM                        tr => 1, # $self->{insertion_mode} == IN_ROW_IM
6160                        tbody => 1, tfoot => 1, thead => 1, # $self->{insertion_mode} == IN_TABLE_IM                        tbody => 1, tfoot => 1, thead => 1, # $self->{insertion_mode} == IN_TABLE_IM
6161                       }->{$token->{tag_name}}) {                       }->{$token->{tag_name}}) {
6162                !!!cp ('t258');            !!!cp ('t258');
6163                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});            !!!parse-error (type => 'unmatched end tag',
6164                ## Ignore the token                            text => $token->{tag_name}, token => $token);
6165                !!!next-token;            ## Ignore the token
6166                redo B;            !!!nack ('t258.1');
6167               !!!next-token;
6168              next B;
6169          } else {          } else {
6170            !!!cp ('t259');            !!!cp ('t259');
6171            !!!parse-error (type => 'in table:/'.$token->{tag_name});            !!!parse-error (type => 'in table:/',
6172                              text => $token->{tag_name}, token => $token);
6173    
6174            $insert = $insert_to_foster;            $insert = $insert_to_foster;
6175            #            #
6176          }          }
6177          } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6178            unless ($self->{open_elements}->[-1]->[1] & HTML_EL and
6179                    @{$self->{open_elements}} == 1) { # redundant, maybe
6180              !!!parse-error (type => 'in body:#eof', token => $token);
6181              !!!cp ('t259.1');
6182              #
6183            } else {
6184              !!!cp ('t259.2');
6185              #
6186            }
6187    
6188            ## Stop parsing
6189            last B;
6190        } else {        } else {
6191          die "$0: $token->{type}: Unknown token type";          die "$0: $token->{type}: Unknown token type";
6192        }        }
6193      } elsif ($self->{insertion_mode} == IN_COLUMN_GROUP_IM) {      } elsif ($self->{insertion_mode} == IN_COLUMN_GROUP_IM) {
6194            if ($token->{type} == CHARACTER_TOKEN) {            if ($token->{type} == CHARACTER_TOKEN) {
6195              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {              if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6196                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6197                unless (length $token->{data}) {                unless (length $token->{data}) {
6198                  !!!cp ('t260');                  !!!cp ('t260');
6199                  !!!next-token;                  !!!next-token;
6200                  redo B;                  next B;
6201                }                }
6202              }              }
6203                            
# Line 4811  sub _tree_construction_main ($) { Line 6206  sub _tree_construction_main ($) {
6206            } elsif ($token->{type} == START_TAG_TOKEN) {            } elsif ($token->{type} == START_TAG_TOKEN) {
6207              if ($token->{tag_name} eq 'col') {              if ($token->{tag_name} eq 'col') {
6208                !!!cp ('t262');                !!!cp ('t262');
6209                !!!insert-element ($token->{tag_name}, $token->{attributes});                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6210                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
6211                  !!!ack ('t262.1');
6212                !!!next-token;                !!!next-token;
6213                redo B;                next B;
6214              } else {              } else {
6215                !!!cp ('t263');                !!!cp ('t263');
6216                #                #
6217              }              }
6218            } elsif ($token->{type} == END_TAG_TOKEN) {            } elsif ($token->{type} == END_TAG_TOKEN) {
6219              if ($token->{tag_name} eq 'colgroup') {              if ($token->{tag_name} eq 'colgroup') {
6220                if ($self->{open_elements}->[-1]->[1] eq 'html') {                if ($self->{open_elements}->[-1]->[1] & HTML_EL) {
6221                  !!!cp ('t264');                  !!!cp ('t264');
6222                  !!!parse-error (type => 'unmatched end tag:colgroup');                  !!!parse-error (type => 'unmatched end tag',
6223                                    text => 'colgroup', token => $token);
6224                  ## Ignore the token                  ## Ignore the token
6225                  !!!next-token;                  !!!next-token;
6226                  redo B;                  next B;
6227                } else {                } else {
6228                  !!!cp ('t265');                  !!!cp ('t265');
6229                  pop @{$self->{open_elements}}; # colgroup                  pop @{$self->{open_elements}}; # colgroup
6230                  $self->{insertion_mode} = IN_TABLE_IM;                  $self->{insertion_mode} = IN_TABLE_IM;
6231                  !!!next-token;                  !!!next-token;
6232                  redo B;                              next B;            
6233                }                }
6234              } elsif ($token->{tag_name} eq 'col') {              } elsif ($token->{tag_name} eq 'col') {
6235                !!!cp ('t266');                !!!cp ('t266');
6236                !!!parse-error (type => 'unmatched end tag:col');                !!!parse-error (type => 'unmatched end tag',
6237                                  text => 'col', token => $token);
6238                ## Ignore the token                ## Ignore the token
6239                !!!next-token;                !!!next-token;
6240                redo B;                next B;
6241              } else {              } else {
6242                !!!cp ('t267');                !!!cp ('t267');
6243                #                #
6244              }              }
6245            } else {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6246              !!!cp ('t268');          if ($self->{open_elements}->[-1]->[1] & HTML_EL and
6247              #              @{$self->{open_elements}} == 1) { # redundant, maybe
6248            }            !!!cp ('t270.2');
6249              ## Stop parsing.
6250              last B;
6251            } else {
6252              ## NOTE: As if </colgroup>.
6253              !!!cp ('t270.1');
6254              pop @{$self->{open_elements}}; # colgroup
6255              $self->{insertion_mode} = IN_TABLE_IM;
6256              ## Reprocess.
6257              next B;
6258            }
6259          } else {
6260            die "$0: $token->{type}: Unknown token type";
6261          }
6262    
6263            ## As if </colgroup>            ## As if </colgroup>
6264            if ($self->{open_elements}->[-1]->[1] eq 'html') {            if ($self->{open_elements}->[-1]->[1] & HTML_EL) {
6265              !!!cp ('t269');              !!!cp ('t269');
6266              !!!parse-error (type => 'unmatched end tag:colgroup');  ## TODO: Wrong error type?
6267                !!!parse-error (type => 'unmatched end tag',
6268                                text => 'colgroup', token => $token);
6269              ## Ignore the token              ## Ignore the token
6270                !!!nack ('t269.1');
6271              !!!next-token;              !!!next-token;
6272              redo B;              next B;
6273            } else {            } else {
6274              !!!cp ('t270');              !!!cp ('t270');
6275              pop @{$self->{open_elements}}; # colgroup              pop @{$self->{open_elements}}; # colgroup
6276              $self->{insertion_mode} = IN_TABLE_IM;              $self->{insertion_mode} = IN_TABLE_IM;
6277                !!!ack-later;
6278              ## reprocess              ## reprocess
6279              redo B;              next B;
6280            }            }
6281      } elsif ($self->{insertion_mode} == IN_SELECT_IM) {      } elsif ($self->{insertion_mode} & SELECT_IMS) {
6282        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
6283          !!!cp ('t271');          !!!cp ('t271');
6284          $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});          $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
6285          !!!next-token;          !!!next-token;
6286          redo B;          next B;
6287        } elsif ($token->{type} == START_TAG_TOKEN) {        } elsif ($token->{type} == START_TAG_TOKEN) {
6288              if ($token->{tag_name} eq 'option') {          if ($token->{tag_name} eq 'option') {
6289                if ($self->{open_elements}->[-1]->[1] eq 'option') {            if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {
6290                  !!!cp ('t272');              !!!cp ('t272');
6291                  ## As if </option>              ## As if </option>
6292                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
6293                } else {            } else {
6294                  !!!cp ('t273');              !!!cp ('t273');
6295                }            }
6296    
6297                !!!insert-element ($token->{tag_name}, $token->{attributes});            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6298                !!!next-token;            !!!nack ('t273.1');
6299                redo B;            !!!next-token;
6300              } elsif ($token->{tag_name} eq 'optgroup') {            next B;
6301                if ($self->{open_elements}->[-1]->[1] eq 'option') {          } elsif ($token->{tag_name} eq 'optgroup') {
6302                  !!!cp ('t274');            if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {
6303                  ## As if </option>              !!!cp ('t274');
6304                  pop @{$self->{open_elements}};              ## As if </option>
6305                } else {              pop @{$self->{open_elements}};
6306                  !!!cp ('t275');            } else {
6307                }              !!!cp ('t275');
6308              }
6309    
6310                if ($self->{open_elements}->[-1]->[1] eq 'optgroup') {            if ($self->{open_elements}->[-1]->[1] & OPTGROUP_EL) {
6311                  !!!cp ('t276');              !!!cp ('t276');
6312                  ## As if </optgroup>              ## As if </optgroup>
6313                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
6314                } else {            } else {
6315                  !!!cp ('t277');              !!!cp ('t277');
6316                }            }
6317    
6318                !!!insert-element ($token->{tag_name}, $token->{attributes});            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6319                !!!next-token;            !!!nack ('t277.1');
6320                redo B;            !!!next-token;
6321              } elsif ($token->{tag_name} eq 'select') {            next B;
6322                !!!parse-error (type => 'not closed:select');          } elsif ({
6323                ## As if </select> instead                     select => 1, input => 1, textarea => 1,
6324                ## have an element in table scope                   }->{$token->{tag_name}} or
6325                my $i;                   ($self->{insertion_mode} == IN_SELECT_IN_TABLE_IM and
6326                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                    {
6327                  my $node = $self->{open_elements}->[$_];                     caption => 1, table => 1,
6328                  if ($node->[1] eq $token->{tag_name}) {                     tbody => 1, tfoot => 1, thead => 1,
6329                    !!!cp ('t278');                     tr => 1, td => 1, th => 1,
6330                    $i = $_;                    }->{$token->{tag_name}})) {
6331                    last INSCOPE;            ## TODO: The type below is not good - <select> is replaced by </select>
6332                  } elsif ({            !!!parse-error (type => 'not closed', text => 'select',
6333                            table => 1, html => 1,                            token => $token);
6334                           }->{$node->[1]}) {            ## NOTE: As if the token were </select> (<select> case) or
6335                    !!!cp ('t279');            ## as if there were </select> (otherwise).
6336                    last INSCOPE;            ## have an element in table scope
6337                  }            my $i;
6338                } # INSCOPE            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6339                unless (defined $i) {              my $node = $self->{open_elements}->[$_];
6340                  !!!cp ('t280');              if ($node->[1] & SELECT_EL) {
6341                  !!!parse-error (type => 'unmatched end tag:select');                !!!cp ('t278');
6342                  ## Ignore the token                $i = $_;
6343                  !!!next-token;                last INSCOPE;
6344                  redo B;              } elsif ($node->[1] & TABLE_SCOPING_EL) {
6345                }                !!!cp ('t279');
6346                  last INSCOPE;
6347                }
6348              } # INSCOPE
6349              unless (defined $i) {
6350                !!!cp ('t280');
6351                !!!parse-error (type => 'unmatched end tag',
6352                                text => 'select', token => $token);
6353                ## Ignore the token
6354                !!!nack ('t280.1');
6355                !!!next-token;
6356                next B;
6357              }
6358                                
6359                !!!cp ('t281');            !!!cp ('t281');
6360                splice @{$self->{open_elements}}, $i;            splice @{$self->{open_elements}}, $i;
6361    
6362                $self->_reset_insertion_mode;            $self->_reset_insertion_mode;
6363    
6364                !!!next-token;            if ($token->{tag_name} eq 'select') {
6365                redo B;              !!!nack ('t281.2');
6366                !!!next-token;
6367                next B;
6368              } else {
6369                !!!cp ('t281.1');
6370                !!!ack-later;
6371                ## Reprocess the token.
6372                next B;
6373              }
6374          } else {          } else {
6375            !!!cp ('t282');            !!!cp ('t282');
6376            !!!parse-error (type => 'in select:'.$token->{tag_name});            !!!parse-error (type => 'in select',
6377                              text => $token->{tag_name}, token => $token);
6378            ## Ignore the token            ## Ignore the token
6379              !!!nack ('t282.1');
6380            !!!next-token;            !!!next-token;
6381            redo B;            next B;
6382          }          }
6383        } elsif ($token->{type} == END_TAG_TOKEN) {        } elsif ($token->{type} == END_TAG_TOKEN) {
6384              if ($token->{tag_name} eq 'optgroup') {          if ($token->{tag_name} eq 'optgroup') {
6385                if ($self->{open_elements}->[-1]->[1] eq 'option' and            if ($self->{open_elements}->[-1]->[1] & OPTION_EL and
6386                    $self->{open_elements}->[-2]->[1] eq 'optgroup') {                $self->{open_elements}->[-2]->[1] & OPTGROUP_EL) {
6387                  !!!cp ('t283');              !!!cp ('t283');
6388                  ## As if </option>              ## As if </option>
6389                  splice @{$self->{open_elements}}, -2;              splice @{$self->{open_elements}}, -2;
6390                } elsif ($self->{open_elements}->[-1]->[1] eq 'optgroup') {            } elsif ($self->{open_elements}->[-1]->[1] & OPTGROUP_EL) {
6391                  !!!cp ('t284');              !!!cp ('t284');
6392                  pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
6393                } else {            } else {
6394                  !!!cp ('t285');              !!!cp ('t285');
6395                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});              !!!parse-error (type => 'unmatched end tag',
6396                  ## Ignore the token                              text => $token->{tag_name}, token => $token);
6397                }              ## Ignore the token
6398                !!!next-token;            }
6399                redo B;            !!!nack ('t285.1');
6400              } elsif ($token->{tag_name} eq 'option') {            !!!next-token;
6401                if ($self->{open_elements}->[-1]->[1] eq 'option') {            next B;
6402                  !!!cp ('t286');          } elsif ($token->{tag_name} eq 'option') {
6403                  pop @{$self->{open_elements}};            if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {
6404                } else {              !!!cp ('t286');
6405                  !!!cp ('t287');              pop @{$self->{open_elements}};
6406                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});            } else {
6407                  ## Ignore the token              !!!cp ('t287');
6408                }              !!!parse-error (type => 'unmatched end tag',
6409                !!!next-token;                              text => $token->{tag_name}, token => $token);
6410                redo B;              ## Ignore the token
6411              } elsif ($token->{tag_name} eq 'select') {            }
6412                ## have an element in table scope            !!!nack ('t287.1');
6413                my $i;            !!!next-token;
6414                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {            next B;
6415                  my $node = $self->{open_elements}->[$_];          } elsif ($token->{tag_name} eq 'select') {
6416                  if ($node->[1] eq $token->{tag_name}) {            ## have an element in table scope
6417                    !!!cp ('t288');            my $i;
6418                    $i = $_;            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6419                    last INSCOPE;              my $node = $self->{open_elements}->[$_];
6420                  } elsif ({              if ($node->[1] & SELECT_EL) {
6421                            table => 1, html => 1,                !!!cp ('t288');
6422                           }->{$node->[1]}) {                $i = $_;
6423                    !!!cp ('t289');                last INSCOPE;
6424                    last INSCOPE;              } elsif ($node->[1] & TABLE_SCOPING_EL) {
6425                  }                !!!cp ('t289');
6426                } # INSCOPE                last INSCOPE;
6427                unless (defined $i) {              }
6428                  !!!cp ('t290');            } # INSCOPE
6429                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});            unless (defined $i) {
6430                  ## Ignore the token              !!!cp ('t290');
6431                  !!!next-token;              !!!parse-error (type => 'unmatched end tag',
6432                  redo B;                              text => $token->{tag_name}, token => $token);
6433                }              ## Ignore the token
6434                !!!nack ('t290.1');
6435                !!!next-token;
6436                next B;
6437              }
6438                                
6439                !!!cp ('t291');            !!!cp ('t291');
6440                splice @{$self->{open_elements}}, $i;            splice @{$self->{open_elements}}, $i;
6441    
6442                $self->_reset_insertion_mode;            $self->_reset_insertion_mode;
6443    
6444                !!!next-token;            !!!nack ('t291.1');
6445                redo B;            !!!next-token;
6446              } elsif ({            next B;
6447                        caption => 1, table => 1, tbody => 1,          } elsif ($self->{insertion_mode} == IN_SELECT_IN_TABLE_IM and
6448                        tfoot => 1, thead => 1, tr => 1, td => 1, th => 1,                   {
6449                       }->{$token->{tag_name}}) {                    caption => 1, table => 1, tbody => 1,
6450                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                    tfoot => 1, thead => 1, tr => 1, td => 1, th => 1,
6451                     }->{$token->{tag_name}}) {
6452    ## TODO: The following is wrong?
6453              !!!parse-error (type => 'unmatched end tag',
6454                              text => $token->{tag_name}, token => $token);
6455                                
6456                ## have an element in table scope            ## have an element in table scope
6457                my $i;            my $i;
6458                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6459                  my $node = $self->{open_elements}->[$_];              my $node = $self->{open_elements}->[$_];
6460                  if ($node->[1] eq $token->{tag_name}) {              if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
6461                    !!!cp ('t292');                !!!cp ('t292');
6462                    $i = $_;                $i = $_;
6463                    last INSCOPE;                last INSCOPE;
6464                  } elsif ({              } elsif ($node->[1] & TABLE_SCOPING_EL) {
6465                            table => 1, html => 1,                !!!cp ('t293');
6466                           }->{$node->[1]}) {                last INSCOPE;
6467                    !!!cp ('t293');              }
6468                    last INSCOPE;            } # INSCOPE
6469                  }            unless (defined $i) {
6470                } # INSCOPE              !!!cp ('t294');
6471                unless (defined $i) {              ## Ignore the token
6472                  !!!cp ('t294');              !!!nack ('t294.1');
6473                  ## Ignore the token              !!!next-token;
6474                  !!!next-token;              next B;
6475                  redo B;            }
               }  
6476                                
6477                ## As if </select>            ## As if </select>
6478                ## have an element in table scope            ## have an element in table scope
6479                undef $i;            undef $i;
6480                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6481                  my $node = $self->{open_elements}->[$_];              my $node = $self->{open_elements}->[$_];
6482                  if ($node->[1] eq 'select') {              if ($node->[1] & SELECT_EL) {
6483                    !!!cp ('t295');                !!!cp ('t295');
6484                    $i = $_;                $i = $_;
6485                    last INSCOPE;                last INSCOPE;
6486                  } elsif ({              } elsif ($node->[1] & TABLE_SCOPING_EL) {
6487                            table => 1, html => 1,  ## ISSUE: Can this state be reached?
6488                           }->{$node->[1]}) {                !!!cp ('t296');
6489                    !!!cp ('t296');                last INSCOPE;
6490                    last INSCOPE;              }
6491                  }            } # INSCOPE
6492                } # INSCOPE            unless (defined $i) {
6493                unless (defined $i) {              !!!cp ('t297');
6494                  !!!cp ('t297');  ## TODO: The following error type is correct?
6495                  !!!parse-error (type => 'unmatched end tag:select');              !!!parse-error (type => 'unmatched end tag',
6496                  ## Ignore the </select> token                              text => 'select', token => $token);
6497                  !!!next-token; ## TODO: ok?              ## Ignore the </select> token
6498                  redo B;              !!!nack ('t297.1');
6499                }              !!!next-token; ## TODO: ok?
6500                next B;
6501              }
6502                                
6503                !!!cp ('t298');            !!!cp ('t298');
6504                splice @{$self->{open_elements}}, $i;            splice @{$self->{open_elements}}, $i;
6505    
6506                $self->_reset_insertion_mode;            $self->_reset_insertion_mode;
6507    
6508                ## reprocess            !!!ack-later;
6509                redo B;            ## reprocess
6510              next B;
6511          } else {          } else {
6512            !!!cp ('t299');            !!!cp ('t299');
6513            !!!parse-error (type => 'in select:/'.$token->{tag_name});            !!!parse-error (type => 'in select:/',
6514                              text => $token->{tag_name}, token => $token);
6515            ## Ignore the token            ## Ignore the token
6516              !!!nack ('t299.3');
6517            !!!next-token;            !!!next-token;
6518            redo B;            next B;
6519          }          }
6520          } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6521            unless ($self->{open_elements}->[-1]->[1] & HTML_EL and
6522                    @{$self->{open_elements}} == 1) { # redundant, maybe
6523              !!!cp ('t299.1');
6524              !!!parse-error (type => 'in body:#eof', token => $token);
6525            } else {
6526              !!!cp ('t299.2');
6527            }
6528    
6529            ## Stop parsing.
6530            last B;
6531        } else {        } else {
6532          die "$0: $token->{type}: Unknown token type";          die "$0: $token->{type}: Unknown token type";
6533        }        }
6534      } elsif ($self->{insertion_mode} & BODY_AFTER_IMS) {      } elsif ($self->{insertion_mode} & BODY_AFTER_IMS) {
6535        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
6536          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6537            my $data = $1;            my $data = $1;
6538            ## As if in body            ## As if in body
6539            $reconstruct_active_formatting_elements->($insert_to_current);            $reconstruct_active_formatting_elements->($insert_to_current);
# Line 5082  sub _tree_construction_main ($) { Line 6543  sub _tree_construction_main ($) {
6543            unless (length $token->{data}) {            unless (length $token->{data}) {
6544              !!!cp ('t300');              !!!cp ('t300');
6545              !!!next-token;              !!!next-token;
6546              redo B;              next B;
6547            }            }
6548          }          }
6549                    
6550          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6551            !!!cp ('t301');            !!!cp ('t301');
6552            !!!parse-error (type => 'after html:#character');            !!!parse-error (type => 'after html:#text', token => $token);
6553              #
           ## Reprocess in the "main" phase, "after body" insertion mode...  
6554          } else {          } else {
6555            !!!cp ('t302');            !!!cp ('t302');
6556              ## "after body" insertion mode
6557              !!!parse-error (type => 'after body:#text', token => $token);
6558              #
6559          }          }
           
         ## "after body" insertion mode  
         !!!parse-error (type => 'after body:#character');  
6560    
6561          $self->{insertion_mode} = IN_BODY_IM;          $self->{insertion_mode} = IN_BODY_IM;
6562          ## reprocess          ## reprocess
6563          redo B;          next B;
6564        } elsif ($token->{type} == START_TAG_TOKEN) {        } elsif ($token->{type} == START_TAG_TOKEN) {
6565          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6566            !!!cp ('t303');            !!!cp ('t303');
6567            !!!parse-error (type => 'after html:'.$token->{tag_name});            !!!parse-error (type => 'after html',
6568                                        text => $token->{tag_name}, token => $token);
6569            ## Reprocess in the "main" phase, "after body" insertion mode...            #
6570          } else {          } else {
6571            !!!cp ('t304');            !!!cp ('t304');
6572              ## "after body" insertion mode
6573              !!!parse-error (type => 'after body',
6574                              text => $token->{tag_name}, token => $token);
6575              #
6576          }          }
6577    
         ## "after body" insertion mode  
         !!!parse-error (type => 'after body:'.$token->{tag_name});  
   
6578          $self->{insertion_mode} = IN_BODY_IM;          $self->{insertion_mode} = IN_BODY_IM;
6579            !!!ack-later;
6580          ## reprocess          ## reprocess
6581          redo B;          next B;
6582        } elsif ($token->{type} == END_TAG_TOKEN) {        } elsif ($token->{type} == END_TAG_TOKEN) {
6583          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6584            !!!cp ('t305');            !!!cp ('t305');
6585            !!!parse-error (type => 'after html:/'.$token->{tag_name});            !!!parse-error (type => 'after html:/',
6586                              text => $token->{tag_name}, token => $token);
6587                        
6588            $self->{insertion_mode} = AFTER_BODY_IM;            $self->{insertion_mode} = IN_BODY_IM;
6589            ## Reprocess in the "main" phase, "after body" insertion mode...            ## Reprocess.
6590              next B;
6591          } else {          } else {
6592            !!!cp ('t306');            !!!cp ('t306');
6593          }          }
# Line 5132  sub _tree_construction_main ($) { Line 6596  sub _tree_construction_main ($) {
6596          if ($token->{tag_name} eq 'html') {          if ($token->{tag_name} eq 'html') {
6597            if (defined $self->{inner_html_node}) {            if (defined $self->{inner_html_node}) {
6598              !!!cp ('t307');              !!!cp ('t307');
6599              !!!parse-error (type => 'unmatched end tag:html');              !!!parse-error (type => 'unmatched end tag',
6600                                text => 'html', token => $token);
6601              ## Ignore the token              ## Ignore the token
6602              !!!next-token;              !!!next-token;
6603              redo B;              next B;
6604            } else {            } else {
6605              !!!cp ('t308');              !!!cp ('t308');
6606              $self->{insertion_mode} = AFTER_HTML_BODY_IM;              $self->{insertion_mode} = AFTER_HTML_BODY_IM;
6607              !!!next-token;              !!!next-token;
6608              redo B;              next B;
6609            }            }
6610          } else {          } else {
6611            !!!cp ('t309');            !!!cp ('t309');
6612            !!!parse-error (type => 'after body:/'.$token->{tag_name});            !!!parse-error (type => 'after body:/',
6613                              text => $token->{tag_name}, token => $token);
6614    
6615            $self->{insertion_mode} = IN_BODY_IM;            $self->{insertion_mode} = IN_BODY_IM;
6616            ## reprocess            ## reprocess
6617            redo B;            next B;
6618          }          }
6619          } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6620            !!!cp ('t309.2');
6621            ## Stop parsing
6622            last B;
6623        } else {        } else {
6624          die "$0: $token->{type}: Unknown token type";          die "$0: $token->{type}: Unknown token type";
6625        }        }
6626      } elsif ($self->{insertion_mode} & FRAME_IMS) {      } elsif ($self->{insertion_mode} & FRAME_IMS) {
6627        if ($token->{type} == CHARACTER_TOKEN) {        if ($token->{type} == CHARACTER_TOKEN) {
6628          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6629            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6630                        
6631            unless (length $token->{data}) {            unless (length $token->{data}) {
6632              !!!cp ('t310');              !!!cp ('t310');
6633              !!!next-token;              !!!next-token;
6634              redo B;              next B;
6635            }            }
6636          }          }
6637                    
6638          if ($token->{data} =~ s/^[^\x09\x0A\x0B\x0C\x20]+//) {          if ($token->{data} =~ s/^[^\x09\x0A\x0C\x20]+//) {
6639            if ($self->{insertion_mode} == IN_FRAMESET_IM) {            if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6640              !!!cp ('t311');              !!!cp ('t311');
6641              !!!parse-error (type => 'in frameset:#character');              !!!parse-error (type => 'in frameset:#text', token => $token);
6642            } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {            } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6643              !!!cp ('t312');              !!!cp ('t312');
6644              !!!parse-error (type => 'after frameset:#character');              !!!parse-error (type => 'after frameset:#text', token => $token);
6645            } else { # "after html frameset"            } else { # "after after frameset"
6646              !!!cp ('t313');              !!!cp ('t313');
6647              !!!parse-error (type => 'after html:#character');              !!!parse-error (type => 'after html:#text', token => $token);
   
             $self->{insertion_mode} = AFTER_FRAMESET_IM;  
             ## Reprocess in the "main" phase, "after frameset"...  
             !!!parse-error (type => 'after frameset:#character');  
6648            }            }
6649                        
6650            ## Ignore the token.            ## Ignore the token.
# Line 5189  sub _tree_construction_main ($) { Line 6655  sub _tree_construction_main ($) {
6655              !!!cp ('t315');              !!!cp ('t315');
6656              !!!next-token;              !!!next-token;
6657            }            }
6658            redo B;            next B;
6659          }          }
6660                    
6661          die qq[$0: Character "$token->{data}"];          die qq[$0: Character "$token->{data}"];
6662        } elsif ($token->{type} == START_TAG_TOKEN) {        } elsif ($token->{type} == START_TAG_TOKEN) {
         if ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {  
           !!!cp ('t316');  
           !!!parse-error (type => 'after html:'.$token->{tag_name});  
   
           $self->{insertion_mode} = AFTER_FRAMESET_IM;  
           ## Process in the "main" phase, "after frameset" insertion mode...  
         } else {  
           !!!cp ('t317');  
         }  
   
6663          if ($token->{tag_name} eq 'frameset' and          if ($token->{tag_name} eq 'frameset' and
6664              $self->{insertion_mode} == IN_FRAMESET_IM) {              $self->{insertion_mode} == IN_FRAMESET_IM) {
6665            !!!cp ('t318');            !!!cp ('t318');
6666            !!!insert-element ($token->{tag_name}, $token->{attributes});            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6667              !!!nack ('t318.1');
6668            !!!next-token;            !!!next-token;
6669            redo B;            next B;
6670          } elsif ($token->{tag_name} eq 'frame' and          } elsif ($token->{tag_name} eq 'frame' and
6671                   $self->{insertion_mode} == IN_FRAMESET_IM) {                   $self->{insertion_mode} == IN_FRAMESET_IM) {
6672            !!!cp ('t319');            !!!cp ('t319');
6673            !!!insert-element ($token->{tag_name}, $token->{attributes});            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6674            pop @{$self->{open_elements}};            pop @{$self->{open_elements}};
6675              !!!ack ('t319.1');
6676            !!!next-token;            !!!next-token;
6677            redo B;            next B;
6678          } elsif ($token->{tag_name} eq 'noframes') {          } elsif ($token->{tag_name} eq 'noframes') {
6679            !!!cp ('t320');            !!!cp ('t320');
6680            ## NOTE: As if in body.            ## NOTE: As if in head.
6681            $parse_rcdata->(CDATA_CONTENT_MODEL, $insert_to_current);            $parse_rcdata->(CDATA_CONTENT_MODEL);
6682            redo B;            next B;
6683    
6684              ## NOTE: |<!DOCTYPE HTML><frameset></frameset></html><noframes></noframes>|
6685              ## has no parse error.
6686          } else {          } else {
6687            if ($self->{insertion_mode} == IN_FRAMESET_IM) {            if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6688              !!!cp ('t321');              !!!cp ('t321');
6689              !!!parse-error (type => 'in frameset:'.$token->{tag_name});              !!!parse-error (type => 'in frameset',
6690            } else {                              text => $token->{tag_name}, token => $token);
6691              } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6692              !!!cp ('t322');              !!!cp ('t322');
6693              !!!parse-error (type => 'after frameset:'.$token->{tag_name});              !!!parse-error (type => 'after frameset',
6694                                text => $token->{tag_name}, token => $token);
6695              } else { # "after after frameset"
6696                !!!cp ('t322.2');
6697                !!!parse-error (type => 'after after frameset',
6698                                text => $token->{tag_name}, token => $token);
6699            }            }
6700            ## Ignore the token            ## Ignore the token
6701              !!!nack ('t322.1');
6702            !!!next-token;            !!!next-token;
6703            redo B;            next B;
6704          }          }
6705        } elsif ($token->{type} == END_TAG_TOKEN) {        } elsif ($token->{type} == END_TAG_TOKEN) {
         if ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {  
           !!!cp ('t323');  
           !!!parse-error (type => 'after html:/'.$token->{tag_name});  
   
           $self->{insertion_mode} = AFTER_FRAMESET_IM;  
           ## Process in the "main" phase, "after frameset" insertion mode...  
         } else {  
           !!!cp ('t324');  
         }  
   
6706          if ($token->{tag_name} eq 'frameset' and          if ($token->{tag_name} eq 'frameset' and
6707              $self->{insertion_mode} == IN_FRAMESET_IM) {              $self->{insertion_mode} == IN_FRAMESET_IM) {
6708            if ($self->{open_elements}->[-1]->[1] eq 'html' and            if ($self->{open_elements}->[-1]->[1] & HTML_EL and
6709                @{$self->{open_elements}} == 1) {                @{$self->{open_elements}} == 1) {
6710              !!!cp ('t325');              !!!cp ('t325');
6711              !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});              !!!parse-error (type => 'unmatched end tag',
6712                                text => $token->{tag_name}, token => $token);
6713              ## Ignore the token              ## Ignore the token
6714              !!!next-token;              !!!next-token;
6715            } else {            } else {
# Line 5260  sub _tree_construction_main ($) { Line 6719  sub _tree_construction_main ($) {
6719            }            }
6720    
6721            if (not defined $self->{inner_html_node} and            if (not defined $self->{inner_html_node} and
6722                $self->{open_elements}->[-1]->[1] ne 'frameset') {                not ($self->{open_elements}->[-1]->[1] & FRAMESET_EL)) {
6723              !!!cp ('t327');              !!!cp ('t327');
6724              $self->{insertion_mode} = AFTER_FRAMESET_IM;              $self->{insertion_mode} = AFTER_FRAMESET_IM;
6725            } else {            } else {
6726              !!!cp ('t328');              !!!cp ('t328');
6727            }            }
6728            redo B;            next B;
6729          } elsif ($token->{tag_name} eq 'html' and          } elsif ($token->{tag_name} eq 'html' and
6730                   $self->{insertion_mode} == AFTER_FRAMESET_IM) {                   $self->{insertion_mode} == AFTER_FRAMESET_IM) {
6731            !!!cp ('t329');            !!!cp ('t329');
6732            $self->{insertion_mode} = AFTER_HTML_FRAMESET_IM;            $self->{insertion_mode} = AFTER_HTML_FRAMESET_IM;
6733            !!!next-token;            !!!next-token;
6734            redo B;            next B;
6735          } else {          } else {
6736            if ($self->{insertion_mode} == IN_FRAMESET_IM) {            if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6737              !!!cp ('t330');              !!!cp ('t330');
6738              !!!parse-error (type => 'in frameset:/'.$token->{tag_name});              !!!parse-error (type => 'in frameset:/',
6739            } else {                              text => $token->{tag_name}, token => $token);
6740              } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6741                !!!cp ('t330.1');
6742                !!!parse-error (type => 'after frameset:/',
6743                                text => $token->{tag_name}, token => $token);
6744              } else { # "after after html"
6745              !!!cp ('t331');              !!!cp ('t331');
6746              !!!parse-error (type => 'after frameset:/'.$token->{tag_name});              !!!parse-error (type => 'after after frameset:/',
6747                                text => $token->{tag_name}, token => $token);
6748            }            }
6749            ## Ignore the token            ## Ignore the token
6750            !!!next-token;            !!!next-token;
6751            redo B;            next B;
6752          }          }
6753          } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6754            unless ($self->{open_elements}->[-1]->[1] & HTML_EL and
6755                    @{$self->{open_elements}} == 1) { # redundant, maybe
6756              !!!cp ('t331.1');
6757              !!!parse-error (type => 'in body:#eof', token => $token);
6758            } else {
6759              !!!cp ('t331.2');
6760            }
6761            
6762            ## Stop parsing
6763            last B;
6764        } else {        } else {
6765          die "$0: $token->{type}: Unknown token type";          die "$0: $token->{type}: Unknown token type";
6766        }        }
# Line 5299  sub _tree_construction_main ($) { Line 6775  sub _tree_construction_main ($) {
6775        if ($token->{tag_name} eq 'script') {        if ($token->{tag_name} eq 'script') {
6776          !!!cp ('t332');          !!!cp ('t332');
6777          ## NOTE: This is an "as if in head" code clone          ## NOTE: This is an "as if in head" code clone
6778          $script_start_tag->($insert);          $script_start_tag->();
6779          redo B;          next B;
6780        } elsif ($token->{tag_name} eq 'style') {        } elsif ($token->{tag_name} eq 'style') {
6781          !!!cp ('t333');          !!!cp ('t333');
6782          ## NOTE: This is an "as if in head" code clone          ## NOTE: This is an "as if in head" code clone
6783          $parse_rcdata->(CDATA_CONTENT_MODEL, $insert);          $parse_rcdata->(CDATA_CONTENT_MODEL);
6784          redo B;          next B;
6785        } elsif ({        } elsif ({
6786                  base => 1, link => 1,                  base => 1, link => 1,
6787                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
6788          !!!cp ('t334');          !!!cp ('t334');
6789          ## NOTE: This is an "as if in head" code clone, only "-t" differs          ## NOTE: This is an "as if in head" code clone, only "-t" differs
6790          !!!insert-element-t ($token->{tag_name}, $token->{attributes});          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6791          pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.          pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
6792            !!!ack ('t334.1');
6793          !!!next-token;          !!!next-token;
6794          redo B;          next B;
6795        } elsif ($token->{tag_name} eq 'meta') {        } elsif ($token->{tag_name} eq 'meta') {
6796          ## NOTE: This is an "as if in head" code clone, only "-t" differs          ## NOTE: This is an "as if in head" code clone, only "-t" differs
6797          !!!insert-element-t ($token->{tag_name}, $token->{attributes});          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6798          my $meta_el = pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.          my $meta_el = pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
6799    
6800          unless ($self->{confident}) {          unless ($self->{confident}) {
6801            if ($token->{attributes}->{charset}) { ## TODO: And if supported            if ($token->{attributes}->{charset}) {
6802              !!!cp ('t335');              !!!cp ('t335');
6803                ## NOTE: Whether the encoding is supported or not is handled
6804                ## in the {change_encoding} callback.
6805              $self->{change_encoding}              $self->{change_encoding}
6806                  ->($self, $token->{attributes}->{charset}->{value});                  ->($self, $token->{attributes}->{charset}->{value}, $token);
6807                            
6808              $meta_el->[0]->get_attribute_node_ns (undef, 'charset')              $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
6809                  ->set_user_data (manakai_has_reference =>                  ->set_user_data (manakai_has_reference =>
6810                                       $token->{attributes}->{charset}                                       $token->{attributes}->{charset}
6811                                           ->{has_reference});                                           ->{has_reference});
6812            } elsif ($token->{attributes}->{content}) {            } elsif ($token->{attributes}->{content}) {
             ## ISSUE: Algorithm name in the spec was incorrect so that not linked to the definition.  
6813              if ($token->{attributes}->{content}->{value}              if ($token->{attributes}->{content}->{value}
6814                  =~ /\A[^;]*;[\x09-\x0D\x20]*[Cc][Hh][Aa][Rr][Ss][Ee][Tt]                  =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
6815                      [\x09-\x0D\x20]*=                      [\x09\x0A\x0C\x0D\x20]*=
6816                      [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|                      [\x09\x0A\x0C\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
6817                      ([^"'\x09-\x0D\x20][^\x09-\x0D\x20]*))/x) {                      ([^"'\x09\x0A\x0C\x0D\x20][^\x09\x0A\x0C\x0D\x20\x3B]*))
6818                       /x) {
6819                !!!cp ('t336');                !!!cp ('t336');
6820                  ## NOTE: Whether the encoding is supported or not is handled
6821                  ## in the {change_encoding} callback.
6822                $self->{change_encoding}                $self->{change_encoding}
6823                    ->($self, defined $1 ? $1 : defined $2 ? $2 : $3);                    ->($self, defined $1 ? $1 : defined $2 ? $2 : $3, $token);
6824                $meta_el->[0]->get_attribute_node_ns (undef, 'content')                $meta_el->[0]->get_attribute_node_ns (undef, 'content')
6825                    ->set_user_data (manakai_has_reference =>                    ->set_user_data (manakai_has_reference =>
6826                                         $token->{attributes}->{content}                                         $token->{attributes}->{content}
# Line 5363  sub _tree_construction_main ($) { Line 6844  sub _tree_construction_main ($) {
6844            }            }
6845          }          }
6846    
6847            !!!ack ('t338.1');
6848          !!!next-token;          !!!next-token;
6849          redo B;          next B;
6850        } elsif ($token->{tag_name} eq 'title') {        } elsif ($token->{tag_name} eq 'title') {
6851          !!!cp ('t341');          !!!cp ('t341');
         !!!parse-error (type => 'in body:title');  
6852          ## NOTE: This is an "as if in head" code clone          ## NOTE: This is an "as if in head" code clone
6853          $parse_rcdata->(RCDATA_CONTENT_MODEL, sub {          $parse_rcdata->(RCDATA_CONTENT_MODEL);
6854            if (defined $self->{head_element}) {          next B;
             !!!cp ('t339');  
             $self->{head_element}->append_child ($_[0]);  
           } else {  
             !!!cp ('t340');  
             $insert->($_[0]);  
           }  
         });  
         redo B;  
6855        } elsif ($token->{tag_name} eq 'body') {        } elsif ($token->{tag_name} eq 'body') {
6856          !!!parse-error (type => 'in body:body');          !!!parse-error (type => 'in body', text => 'body', token => $token);
6857                                
6858          if (@{$self->{open_elements}} == 1 or          if (@{$self->{open_elements}} == 1 or
6859              $self->{open_elements}->[1]->[1] ne 'body') {              not ($self->{open_elements}->[1]->[1] & BODY_EL)) {
6860            !!!cp ('t342');            !!!cp ('t342');
6861            ## Ignore the token            ## Ignore the token
6862          } else {          } else {
# Line 5397  sub _tree_construction_main ($) { Line 6870  sub _tree_construction_main ($) {
6870              }              }
6871            }            }
6872          }          }
6873            !!!nack ('t343.1');
6874          !!!next-token;          !!!next-token;
6875          redo B;          next B;
6876        } elsif ({        } elsif ({
6877                  address => 1, blockquote => 1, center => 1, dir => 1,                  address => 1, blockquote => 1, center => 1, dir => 1,
6878                  div => 1, dl => 1, fieldset => 1, listing => 1,                  div => 1, dl => 1, fieldset => 1,
6879                    h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
6880                  menu => 1, ol => 1, p => 1, ul => 1,                  menu => 1, ol => 1, p => 1, ul => 1,
6881                  pre => 1,                  pre => 1, listing => 1,
6882                    form => 1,
6883                    table => 1,
6884                    hr => 1,
6885                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
6886            if ($token->{tag_name} eq 'form' and defined $self->{form_element}) {
6887              !!!cp ('t350');
6888              !!!parse-error (type => 'in form:form', token => $token);
6889              ## Ignore the token
6890              !!!nack ('t350.1');
6891              !!!next-token;
6892              next B;
6893            }
6894    
6895          ## has a p element in scope          ## has a p element in scope
6896          INSCOPE: for (reverse @{$self->{open_elements}}) {          INSCOPE: for (reverse @{$self->{open_elements}}) {
6897            if ($_->[1] eq 'p') {            if ($_->[1] & P_EL) {
6898              !!!cp ('t344');              !!!cp ('t344');
6899              !!!back-token;              !!!back-token; # <form>
6900              $token = {type => END_TAG_TOKEN, tag_name => 'p'};              $token = {type => END_TAG_TOKEN, tag_name => 'p',
6901              redo B;                        line => $token->{line}, column => $token->{column}};
6902            } elsif ({              next B;
6903                      table => 1, caption => 1, td => 1, th => 1,            } elsif ($_->[1] & SCOPING_EL) {
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$_->[1]}) {  
6904              !!!cp ('t345');              !!!cp ('t345');
6905              last INSCOPE;              last INSCOPE;
6906            }            }
6907          } # INSCOPE          } # INSCOPE
6908                        
6909          !!!insert-element-t ($token->{tag_name}, $token->{attributes});          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6910          if ($token->{tag_name} eq 'pre') {          if ($token->{tag_name} eq 'pre' or $token->{tag_name} eq 'listing') {
6911              !!!nack ('t346.1');
6912            !!!next-token;            !!!next-token;
6913            if ($token->{type} == CHARACTER_TOKEN) {            if ($token->{type} == CHARACTER_TOKEN) {
6914              $token->{data} =~ s/^\x0A//;              $token->{data} =~ s/^\x0A//;
# Line 5435  sub _tree_construction_main ($) { Line 6921  sub _tree_construction_main ($) {
6921            } else {            } else {
6922              !!!cp ('t348');              !!!cp ('t348');
6923            }            }
6924          } else {          } elsif ($token->{tag_name} eq 'form') {
6925            !!!cp ('t347');            !!!cp ('t347.1');
6926              $self->{form_element} = $self->{open_elements}->[-1]->[0];
6927    
6928              !!!nack ('t347.2');
6929            !!!next-token;            !!!next-token;
6930          }          } elsif ($token->{tag_name} eq 'table') {
6931          redo B;            !!!cp ('t382');
6932        } elsif ($token->{tag_name} eq 'form') {            push @{$open_tables}, [$self->{open_elements}->[-1]->[0]];
6933          if (defined $self->{form_element}) {            
6934            !!!cp ('t350');            $self->{insertion_mode} = IN_TABLE_IM;
6935            !!!parse-error (type => 'in form:form');  
6936            ## Ignore the token            !!!nack ('t382.1');
6937              !!!next-token;
6938            } elsif ($token->{tag_name} eq 'hr') {
6939              !!!cp ('t386');
6940              pop @{$self->{open_elements}};
6941            
6942              !!!nack ('t386.1');
6943            !!!next-token;            !!!next-token;
           redo B;  
6944          } else {          } else {
6945            ## has a p element in scope            !!!nack ('t347.1');
           INSCOPE: for (reverse @{$self->{open_elements}}) {  
             if ($_->[1] eq 'p') {  
               !!!cp ('t351');  
               !!!back-token;  
               $token = {type => END_TAG_TOKEN, tag_name => 'p'};  
               redo B;  
             } elsif ({  
                       table => 1, caption => 1, td => 1, th => 1,  
                       button => 1, marquee => 1, object => 1, html => 1,  
                      }->{$_->[1]}) {  
               !!!cp ('t352');  
               last INSCOPE;  
             }  
           } # INSCOPE  
               
           !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
           $self->{form_element} = $self->{open_elements}->[-1]->[0];  
6946            !!!next-token;            !!!next-token;
           redo B;  
6947          }          }
6948        } elsif ($token->{tag_name} eq 'li') {          next B;
6949          } elsif ({li => 1, dt => 1, dd => 1}->{$token->{tag_name}}) {
6950          ## has a p element in scope          ## has a p element in scope
6951          INSCOPE: for (reverse @{$self->{open_elements}}) {          INSCOPE: for (reverse @{$self->{open_elements}}) {
6952            if ($_->[1] eq 'p') {            if ($_->[1] & P_EL) {
6953              !!!cp ('t353');              !!!cp ('t353');
6954              !!!back-token;              !!!back-token; # <x>
6955              $token = {type => END_TAG_TOKEN, tag_name => 'p'};              $token = {type => END_TAG_TOKEN, tag_name => 'p',
6956              redo B;                        line => $token->{line}, column => $token->{column}};
6957            } elsif ({              next B;
6958                      table => 1, caption => 1, td => 1, th => 1,            } elsif ($_->[1] & SCOPING_EL) {
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$_->[1]}) {  
6959              !!!cp ('t354');              !!!cp ('t354');
6960              last INSCOPE;              last INSCOPE;
6961            }            }
# Line 5489  sub _tree_construction_main ($) { Line 6964  sub _tree_construction_main ($) {
6964          ## Step 1          ## Step 1
6965          my $i = -1;          my $i = -1;
6966          my $node = $self->{open_elements}->[$i];          my $node = $self->{open_elements}->[$i];
6967            my $li_or_dtdd = {li => {li => 1},
6968                              dt => {dt => 1, dd => 1},
6969                              dd => {dt => 1, dd => 1}}->{$token->{tag_name}};
6970          LI: {          LI: {
6971            ## Step 2            ## Step 2
6972            if ($node->[1] eq 'li') {            if ($li_or_dtdd->{$node->[0]->manakai_local_name}) {
6973              if ($i != -1) {              if ($i != -1) {
6974                !!!cp ('t355');                !!!cp ('t355');
6975                !!!parse-error (type => 'end tag missing:'.                !!!parse-error (type => 'not closed',
6976                                $self->{open_elements}->[-1]->[1]);                                text => $self->{open_elements}->[-1]->[0]
6977                                      ->manakai_local_name,
6978                                  token => $token);
6979              } else {              } else {
6980                !!!cp ('t356');                !!!cp ('t356');
6981              }              }
# Line 5506  sub _tree_construction_main ($) { Line 6986  sub _tree_construction_main ($) {
6986            }            }
6987                        
6988            ## Step 3            ## Step 3
6989            if (not $formatting_category->{$node->[1]} and            if (not ($node->[1] & FORMATTING_EL) and
6990                #not $phrasing_category->{$node->[1]} and                #not $phrasing_category->{$node->[1]} and
6991                ($special_category->{$node->[1]} or                ($node->[1] & SPECIAL_EL or
6992                 $scoping_category->{$node->[1]}) and                 $node->[1] & SCOPING_EL) and
6993                $node->[1] ne 'address' and $node->[1] ne 'div') {                not ($node->[1] & ADDRESS_EL) and
6994                  not ($node->[1] & DIV_EL)) {
6995              !!!cp ('t358');              !!!cp ('t358');
6996              last LI;              last LI;
6997            }            }
# Line 5522  sub _tree_construction_main ($) { Line 7003  sub _tree_construction_main ($) {
7003            redo LI;            redo LI;
7004          } # LI          } # LI
7005                        
7006          !!!insert-element-t ($token->{tag_name}, $token->{attributes});          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7007            !!!nack ('t359.1');
7008          !!!next-token;          !!!next-token;
7009          redo B;          next B;
       } elsif ($token->{tag_name} eq 'dd' or $token->{tag_name} eq 'dt') {  
         ## has a p element in scope  
         INSCOPE: for (reverse @{$self->{open_elements}}) {  
           if ($_->[1] eq 'p') {  
             !!!cp ('t360');  
             !!!back-token;  
             $token = {type => END_TAG_TOKEN, tag_name => 'p'};  
             redo B;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$_->[1]}) {  
             !!!cp ('t361');  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         ## Step 1  
         my $i = -1;  
         my $node = $self->{open_elements}->[$i];  
         LI: {  
           ## Step 2  
           if ($node->[1] eq 'dt' or $node->[1] eq 'dd') {  
             if ($i != -1) {  
               !!!cp ('t362');  
               !!!parse-error (type => 'end tag missing:'.  
                               $self->{open_elements}->[-1]->[1]);  
             } else {  
               !!!cp ('t363');  
             }  
             splice @{$self->{open_elements}}, $i;  
             last LI;  
           } else {  
             !!!cp ('t364');  
           }  
             
           ## Step 3  
           if (not $formatting_category->{$node->[1]} and  
               #not $phrasing_category->{$node->[1]} and  
               ($special_category->{$node->[1]} or  
                $scoping_category->{$node->[1]}) and  
               $node->[1] ne 'address' and $node->[1] ne 'div') {  
             !!!cp ('t365');  
             last LI;  
           }  
             
           !!!cp ('t366');  
           ## Step 4  
           $i--;  
           $node = $self->{open_elements}->[$i];  
           redo LI;  
         } # LI  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         !!!next-token;  
         redo B;  
7010        } elsif ($token->{tag_name} eq 'plaintext') {        } elsif ($token->{tag_name} eq 'plaintext') {
7011          ## has a p element in scope          ## has a p element in scope
7012          INSCOPE: for (reverse @{$self->{open_elements}}) {          INSCOPE: for (reverse @{$self->{open_elements}}) {
7013            if ($_->[1] eq 'p') {            if ($_->[1] & P_EL) {
7014              !!!cp ('t367');              !!!cp ('t367');
7015              !!!back-token;              !!!back-token; # <plaintext>
7016              $token = {type => END_TAG_TOKEN, tag_name => 'p'};              $token = {type => END_TAG_TOKEN, tag_name => 'p',
7017              redo B;                        line => $token->{line}, column => $token->{column}};
7018            } elsif ({              next B;
7019                      table => 1, caption => 1, td => 1, th => 1,            } elsif ($_->[1] & SCOPING_EL) {
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$_->[1]}) {  
7020              !!!cp ('t368');              !!!cp ('t368');
7021              last INSCOPE;              last INSCOPE;
7022            }            }
7023          } # INSCOPE          } # INSCOPE
7024                        
7025          !!!insert-element-t ($token->{tag_name}, $token->{attributes});          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7026                        
7027          $self->{content_model} = PLAINTEXT_CONTENT_MODEL;          $self->{content_model} = PLAINTEXT_CONTENT_MODEL;
7028                        
7029            !!!nack ('t368.1');
7030          !!!next-token;          !!!next-token;
7031          redo B;          next B;
       } elsif ({  
                 h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,  
                }->{$token->{tag_name}}) {  
         ## has a p element in scope  
         INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
           my $node = $self->{open_elements}->[$_];  
           if ($node->[1] eq 'p') {  
             !!!cp ('t369');  
             !!!back-token;  
             $token = {type => END_TAG_TOKEN, tag_name => 'p'};  
             redo B;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
             !!!cp ('t370');  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         ## NOTE: See <http://html5.org/tools/web-apps-tracker?from=925&to=926>  
         ## has an element in scope  
         #my $i;  
         #INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
         #  my $node = $self->{open_elements}->[$_];  
         #  if ({  
         #       h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,  
         #      }->{$node->[1]}) {  
         #    $i = $_;  
         #    last INSCOPE;  
         #  } elsif ({  
         #            table => 1, caption => 1, td => 1, th => 1,  
         #            button => 1, marquee => 1, object => 1, html => 1,  
         #           }->{$node->[1]}) {  
         #    last INSCOPE;  
         #  }  
         #} # INSCOPE  
         #    
         #if (defined $i) {  
         #  !!! parse-error (type => 'in hn:hn');  
         #  splice @{$self->{open_elements}}, $i;  
         #}  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
             
         !!!next-token;  
         redo B;  
7032        } elsif ($token->{tag_name} eq 'a') {        } elsif ($token->{tag_name} eq 'a') {
7033          AFE: for my $i (reverse 0..$#$active_formatting_elements) {          AFE: for my $i (reverse 0..$#$active_formatting_elements) {
7034            my $node = $active_formatting_elements->[$i];            my $node = $active_formatting_elements->[$i];
7035            if ($node->[1] eq 'a') {            if ($node->[1] & A_EL) {
7036              !!!cp ('t371');              !!!cp ('t371');
7037              !!!parse-error (type => 'in a:a');              !!!parse-error (type => 'in a:a', token => $token);
7038                            
7039              !!!back-token;              !!!back-token; # <a>
7040              $token = {type => END_TAG_TOKEN, tag_name => 'a'};              $token = {type => END_TAG_TOKEN, tag_name => 'a',
7041              $formatting_end_tag->($token->{tag_name});                        line => $token->{line}, column => $token->{column}};
7042                $formatting_end_tag->($token);
7043                            
7044              AFE2: for (reverse 0..$#$active_formatting_elements) {              AFE2: for (reverse 0..$#$active_formatting_elements) {
7045                if ($active_formatting_elements->[$_]->[0] eq $node->[0]) {                if ($active_formatting_elements->[$_]->[0] eq $node->[0]) {
# Line 5685  sub _tree_construction_main ($) { Line 7064  sub _tree_construction_main ($) {
7064                        
7065          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
7066    
7067          !!!insert-element-t ($token->{tag_name}, $token->{attributes});          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7068          push @$active_formatting_elements, $self->{open_elements}->[-1];          push @$active_formatting_elements, $self->{open_elements}->[-1];
7069    
7070            !!!nack ('t374.1');
7071          !!!next-token;          !!!next-token;
7072          redo B;          next B;
       } elsif ({  
                 b => 1, big => 1, em => 1, font => 1, i => 1,  
                 s => 1, small => 1, strile => 1,  
                 strong => 1, tt => 1, u => 1,  
                }->{$token->{tag_name}}) {  
         !!!cp ('t375');  
         $reconstruct_active_formatting_elements->($insert_to_current);  
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         push @$active_formatting_elements, $self->{open_elements}->[-1];  
           
         !!!next-token;  
         redo B;  
7073        } elsif ($token->{tag_name} eq 'nobr') {        } elsif ($token->{tag_name} eq 'nobr') {
7074          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
7075    
7076          ## has a |nobr| element in scope          ## has a |nobr| element in scope
7077          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7078            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
7079            if ($node->[1] eq 'nobr') {            if ($node->[1] & NOBR_EL) {
7080              !!!cp ('t376');              !!!cp ('t376');
7081              !!!parse-error (type => 'in nobr:nobr');              !!!parse-error (type => 'in nobr:nobr', token => $token);
7082              !!!back-token;              !!!back-token; # <nobr>
7083              $token = {type => END_TAG_TOKEN, tag_name => 'nobr'};              $token = {type => END_TAG_TOKEN, tag_name => 'nobr',
7084              redo B;                        line => $token->{line}, column => $token->{column}};
7085            } elsif ({              next B;
7086                      table => 1, caption => 1, td => 1, th => 1,            } elsif ($node->[1] & SCOPING_EL) {
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
7087              !!!cp ('t377');              !!!cp ('t377');
7088              last INSCOPE;              last INSCOPE;
7089            }            }
7090          } # INSCOPE          } # INSCOPE
7091                    
7092          !!!insert-element-t ($token->{tag_name}, $token->{attributes});          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7093          push @$active_formatting_elements, $self->{open_elements}->[-1];          push @$active_formatting_elements, $self->{open_elements}->[-1];
7094                    
7095            !!!nack ('t377.1');
7096          !!!next-token;          !!!next-token;
7097          redo B;          next B;
7098        } elsif ($token->{tag_name} eq 'button') {        } elsif ($token->{tag_name} eq 'button') {
7099          ## has a button element in scope          ## has a button element in scope
7100          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7101            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
7102            if ($node->[1] eq 'button') {            if ($node->[1] & BUTTON_EL) {
7103              !!!cp ('t378');              !!!cp ('t378');
7104              !!!parse-error (type => 'in button:button');              !!!parse-error (type => 'in button:button', token => $token);
7105              !!!back-token;              !!!back-token; # <button>
7106              $token = {type => END_TAG_TOKEN, tag_name => 'button'};              $token = {type => END_TAG_TOKEN, tag_name => 'button',
7107              redo B;                        line => $token->{line}, column => $token->{column}};
7108            } elsif ({              next B;
7109                      table => 1, caption => 1, td => 1, th => 1,            } elsif ($node->[1] & SCOPING_EL) {
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
7110              !!!cp ('t379');              !!!cp ('t379');
7111              last INSCOPE;              last INSCOPE;
7112            }            }
# Line 5750  sub _tree_construction_main ($) { Line 7114  sub _tree_construction_main ($) {
7114                        
7115          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
7116                        
7117          !!!insert-element-t ($token->{tag_name}, $token->{attributes});          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7118          push @$active_formatting_elements, ['#marker', ''];  
7119            ## TODO: associate with $self->{form_element} if defined
7120    
         !!!next-token;  
         redo B;  
       } elsif ($token->{tag_name} eq 'marquee' or  
                $token->{tag_name} eq 'object') {  
         !!!cp ('t380');  
         $reconstruct_active_formatting_elements->($insert_to_current);  
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
7121          push @$active_formatting_elements, ['#marker', ''];          push @$active_formatting_elements, ['#marker', ''];
7122            
7123          !!!next-token;          !!!nack ('t379.1');
         redo B;  
       } elsif ($token->{tag_name} eq 'xmp') {  
         !!!cp ('t381');  
         $reconstruct_active_formatting_elements->($insert_to_current);  
         $parse_rcdata->(CDATA_CONTENT_MODEL, $insert);  
         redo B;  
       } elsif ($token->{tag_name} eq 'table') {  
         ## has a p element in scope  
         INSCOPE: for (reverse @{$self->{open_elements}}) {  
           if ($_->[1] eq 'p') {  
             !!!cp ('t382');  
             !!!back-token;  
             $token = {type => END_TAG_TOKEN, tag_name => 'p'};  
             redo B;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$_->[1]}) {  
             !!!cp ('t383');  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
             
         $self->{insertion_mode} = IN_TABLE_IM;  
             
7124          !!!next-token;          !!!next-token;
7125          redo B;          next B;
7126        } elsif ({        } elsif ({
7127                  area => 1, basefont => 1, bgsound => 1, br => 1,                  xmp => 1,
7128                  embed => 1, img => 1, param => 1, spacer => 1, wbr => 1,                  iframe => 1,
7129                  image => 1,                  noembed => 1,
7130                    noframes => 1, ## NOTE: This is an "as if in head" code clone.
7131                    noscript => 0, ## TODO: 1 if scripting is enabled
7132                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
7133          if ($token->{tag_name} eq 'image') {          if ($token->{tag_name} eq 'xmp') {
7134            !!!cp ('t384');            !!!cp ('t381');
7135            !!!parse-error (type => 'image');            $reconstruct_active_formatting_elements->($insert_to_current);
           $token->{tag_name} = 'img';  
7136          } else {          } else {
7137            !!!cp ('t385');            !!!cp ('t399');
7138          }          }
7139            ## NOTE: There is an "as if in body" code clone.
7140          ## NOTE: There is an "as if <br>" code clone.          $parse_rcdata->(CDATA_CONTENT_MODEL);
7141          $reconstruct_active_formatting_elements->($insert_to_current);          next B;
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         pop @{$self->{open_elements}};  
           
         !!!next-token;  
         redo B;  
       } elsif ($token->{tag_name} eq 'hr') {  
         ## has a p element in scope  
         INSCOPE: for (reverse @{$self->{open_elements}}) {  
           if ($_->[1] eq 'p') {  
             !!!cp ('t386');  
             !!!back-token;  
             $token = {type => END_TAG_TOKEN, tag_name => 'p'};  
             redo B;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$_->[1]}) {  
             !!!cp ('t387');  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         pop @{$self->{open_elements}};  
             
         !!!next-token;  
         redo B;  
       } elsif ($token->{tag_name} eq 'input') {  
         !!!cp ('t388');  
         $reconstruct_active_formatting_elements->($insert_to_current);  
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         ## TODO: associate with $self->{form_element} if defined  
         pop @{$self->{open_elements}};  
           
         !!!next-token;  
         redo B;  
7142        } elsif ($token->{tag_name} eq 'isindex') {        } elsif ($token->{tag_name} eq 'isindex') {
7143          !!!parse-error (type => 'isindex');          !!!parse-error (type => 'isindex', token => $token);
7144                    
7145          if (defined $self->{form_element}) {          if (defined $self->{form_element}) {
7146            !!!cp ('t389');            !!!cp ('t389');
7147            ## Ignore the token            ## Ignore the token
7148              !!!nack ('t389'); ## NOTE: Not acknowledged.
7149            !!!next-token;            !!!next-token;
7150            redo B;            next B;
7151          } else {          } else {
7152              !!!ack ('t391.1');
7153    
7154            my $at = $token->{attributes};            my $at = $token->{attributes};
7155            my $form_attrs;            my $form_attrs;
7156            $form_attrs->{action} = $at->{action} if $at->{action};            $form_attrs->{action} = $at->{action} if $at->{action};
# Line 5864  sub _tree_construction_main ($) { Line 7160  sub _tree_construction_main ($) {
7160            delete $at->{prompt};            delete $at->{prompt};
7161            my @tokens = (            my @tokens = (
7162                          {type => START_TAG_TOKEN, tag_name => 'form',                          {type => START_TAG_TOKEN, tag_name => 'form',
7163                           attributes => $form_attrs},                           attributes => $form_attrs,
7164                          {type => START_TAG_TOKEN, tag_name => 'hr'},                           line => $token->{line}, column => $token->{column}},
7165                          {type => START_TAG_TOKEN, tag_name => 'p'},                          {type => START_TAG_TOKEN, tag_name => 'hr',
7166                          {type => START_TAG_TOKEN, tag_name => 'label'},                           line => $token->{line}, column => $token->{column}},
7167                            {type => START_TAG_TOKEN, tag_name => 'p',
7168                             line => $token->{line}, column => $token->{column}},
7169                            {type => START_TAG_TOKEN, tag_name => 'label',
7170                             line => $token->{line}, column => $token->{column}},
7171                         );                         );
7172            if ($prompt_attr) {            if ($prompt_attr) {
7173              !!!cp ('t390');              !!!cp ('t390');
7174              push @tokens, {type => CHARACTER_TOKEN, data => $prompt_attr->{value}};              push @tokens, {type => CHARACTER_TOKEN, data => $prompt_attr->{value},
7175                               #line => $token->{line}, column => $token->{column},
7176                              };
7177            } else {            } else {
7178              !!!cp ('t391');              !!!cp ('t391');
7179              push @tokens, {type => CHARACTER_TOKEN,              push @tokens, {type => CHARACTER_TOKEN,
7180                             data => 'This is a searchable index. Insert your search keywords here: '}; # SHOULD                             data => 'This is a searchable index. Insert your search keywords here: ',
7181                               #line => $token->{line}, column => $token->{column},
7182                              }; # SHOULD
7183              ## TODO: make this configurable              ## TODO: make this configurable
7184            }            }
7185            push @tokens,            push @tokens,
7186                          {type => START_TAG_TOKEN, tag_name => 'input', attributes => $at},                          {type => START_TAG_TOKEN, tag_name => 'input', attributes => $at,
7187                             line => $token->{line}, column => $token->{column}},
7188                          #{type => CHARACTER_TOKEN, data => ''}, # SHOULD                          #{type => CHARACTER_TOKEN, data => ''}, # SHOULD
7189                          {type => END_TAG_TOKEN, tag_name => 'label'},                          {type => END_TAG_TOKEN, tag_name => 'label',
7190                          {type => END_TAG_TOKEN, tag_name => 'p'},                           line => $token->{line}, column => $token->{column}},
7191                          {type => START_TAG_TOKEN, tag_name => 'hr'},                          {type => END_TAG_TOKEN, tag_name => 'p',
7192                          {type => END_TAG_TOKEN, tag_name => 'form'};                           line => $token->{line}, column => $token->{column}},
7193            $token = shift @tokens;                          {type => START_TAG_TOKEN, tag_name => 'hr',
7194                             line => $token->{line}, column => $token->{column}},
7195                            {type => END_TAG_TOKEN, tag_name => 'form',
7196                             line => $token->{line}, column => $token->{column}};
7197            !!!back-token (@tokens);            !!!back-token (@tokens);
7198            redo B;            !!!next-token;
7199              next B;
7200          }          }
7201        } elsif ($token->{tag_name} eq 'textarea') {        } elsif ($token->{tag_name} eq 'textarea') {
7202          my $tag_name = $token->{tag_name};          my $tag_name = $token->{tag_name};
7203          my $el;          my $el;
7204          !!!create-element ($el, $token->{tag_name}, $token->{attributes});          !!!create-element ($el, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);
7205                    
7206          ## TODO: $self->{form_element} if defined          ## TODO: $self->{form_element} if defined
7207          $self->{content_model} = RCDATA_CONTENT_MODEL;          $self->{content_model} = RCDATA_CONTENT_MODEL;
# Line 5901  sub _tree_construction_main ($) { Line 7210  sub _tree_construction_main ($) {
7210          $insert->($el);          $insert->($el);
7211                    
7212          my $text = '';          my $text = '';
7213            !!!nack ('t392.1');
7214          !!!next-token;          !!!next-token;
7215          if ($token->{type} == CHARACTER_TOKEN) {          if ($token->{type} == CHARACTER_TOKEN) {
7216            $token->{data} =~ s/^\x0A//;            $token->{data} =~ s/^\x0A//;
# Line 5931  sub _tree_construction_main ($) { Line 7241  sub _tree_construction_main ($) {
7241            ## Ignore the token            ## Ignore the token
7242          } else {          } else {
7243            !!!cp ('t398');            !!!cp ('t398');
7244            !!!parse-error (type => 'in RCDATA:#'.$token->{type});            !!!parse-error (type => 'in RCDATA:#eof', token => $token);
7245          }          }
7246          !!!next-token;          !!!next-token;
7247            next B;
7248          } elsif ($token->{tag_name} eq 'rt' or
7249                   $token->{tag_name} eq 'rp') {
7250            ## has a |ruby| element in scope
7251            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7252              my $node = $self->{open_elements}->[$_];
7253              if ($node->[1] & RUBY_EL) {
7254                !!!cp ('t398.1');
7255                ## generate implied end tags
7256                while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7257                  !!!cp ('t398.2');
7258                  pop @{$self->{open_elements}};
7259                }
7260                unless ($self->{open_elements}->[-1]->[1] & RUBY_EL) {
7261                  !!!cp ('t398.3');
7262                  !!!parse-error (type => 'not closed',
7263                                  text => $self->{open_elements}->[-1]->[0]
7264                                      ->manakai_local_name,
7265                                  token => $token);
7266                  pop @{$self->{open_elements}}
7267                      while not $self->{open_elements}->[-1]->[1] & RUBY_EL;
7268                }
7269                last INSCOPE;
7270              } elsif ($node->[1] & SCOPING_EL) {
7271                !!!cp ('t398.4');
7272                last INSCOPE;
7273              }
7274            } # INSCOPE
7275    
7276            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7277    
7278            !!!nack ('t398.5');
7279            !!!next-token;
7280          redo B;          redo B;
7281        } elsif ({        } elsif ($token->{tag_name} eq 'math' or
7282                  iframe => 1,                 $token->{tag_name} eq 'svg') {
                 noembed => 1,  
                 noframes => 1,  
                 noscript => 0, ## TODO: 1 if scripting is enabled  
                }->{$token->{tag_name}}) {  
         !!!cp ('t399');  
         ## NOTE: There is an "as if in body" code clone.  
         $parse_rcdata->(CDATA_CONTENT_MODEL, $insert);  
         redo B;  
       } elsif ($token->{tag_name} eq 'select') {  
         !!!cp ('t400');  
7283          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
7284    
7285            ## "Adjust MathML attributes" ('math' only) - done in insert-element-f
7286    
7287            ## "adjust SVG attributes" ('svg' only) - done in insert-element-f
7288    
7289            ## "adjust foreign attributes" - done in insert-element-f
7290                    
7291          !!!insert-element-t ($token->{tag_name}, $token->{attributes});          !!!insert-element-f ($token->{tag_name} eq 'math' ? $MML_NS : $SVG_NS, $token->{tag_name}, $token->{attributes}, $token);
7292                    
7293          $self->{insertion_mode} = IN_SELECT_IM;          if ($self->{self_closing}) {
7294              pop @{$self->{open_elements}};
7295              !!!ack ('t398.1');
7296            } else {
7297              !!!cp ('t398.2');
7298              $self->{insertion_mode} |= IN_FOREIGN_CONTENT_IM;
7299              ## NOTE: |<body><math><mi><svg>| -> "in foreign content" insertion
7300              ## mode, "in body" (not "in foreign content") secondary insertion
7301              ## mode, maybe.
7302            }
7303    
7304          !!!next-token;          !!!next-token;
7305          redo B;          next B;
7306        } elsif ({        } elsif ({
7307                  caption => 1, col => 1, colgroup => 1, frame => 1,                  caption => 1, col => 1, colgroup => 1, frame => 1,
7308                  frameset => 1, head => 1, option => 1, optgroup => 1,                  frameset => 1, head => 1, option => 1, optgroup => 1,
# Line 5961  sub _tree_construction_main ($) { Line 7310  sub _tree_construction_main ($) {
7310                  thead => 1, tr => 1,                  thead => 1, tr => 1,
7311                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
7312          !!!cp ('t401');          !!!cp ('t401');
7313          !!!parse-error (type => 'in body:'.$token->{tag_name});          !!!parse-error (type => 'in body',
7314                            text => $token->{tag_name}, token => $token);
7315          ## Ignore the token          ## Ignore the token
7316            !!!nack ('t401.1'); ## NOTE: |<col/>| or |<frame/>| here is an error.
7317          !!!next-token;          !!!next-token;
7318          redo B;          next B;
7319                    
7320          ## ISSUE: An issue on HTML5 new elements in the spec.          ## ISSUE: An issue on HTML5 new elements in the spec.
7321        } else {        } else {
7322          !!!cp ('t402');          if ($token->{tag_name} eq 'image') {
7323              !!!cp ('t384');
7324              !!!parse-error (type => 'image', token => $token);
7325              $token->{tag_name} = 'img';
7326            } else {
7327              !!!cp ('t385');
7328            }
7329    
7330            ## NOTE: There is an "as if <br>" code clone.
7331          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
7332                    
7333          !!!insert-element-t ($token->{tag_name}, $token->{attributes});          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7334    
7335            if ({
7336                 applet => 1, marquee => 1, object => 1,
7337                }->{$token->{tag_name}}) {
7338              !!!cp ('t380');
7339              push @$active_formatting_elements, ['#marker', ''];
7340              !!!nack ('t380.1');
7341            } elsif ({
7342                      b => 1, big => 1, em => 1, font => 1, i => 1,
7343                      s => 1, small => 1, strile => 1,
7344                      strong => 1, tt => 1, u => 1,
7345                     }->{$token->{tag_name}}) {
7346              !!!cp ('t375');
7347              push @$active_formatting_elements, $self->{open_elements}->[-1];
7348              !!!nack ('t375.1');
7349            } elsif ($token->{tag_name} eq 'input') {
7350              !!!cp ('t388');
7351              ## TODO: associate with $self->{form_element} if defined
7352              pop @{$self->{open_elements}};
7353              !!!ack ('t388.2');
7354            } elsif ({
7355                      area => 1, basefont => 1, bgsound => 1, br => 1,
7356                      embed => 1, img => 1, param => 1, spacer => 1, wbr => 1,
7357                      #image => 1,
7358                     }->{$token->{tag_name}}) {
7359              !!!cp ('t388.1');
7360              pop @{$self->{open_elements}};
7361              !!!ack ('t388.3');
7362            } elsif ($token->{tag_name} eq 'select') {
7363              ## TODO: associate with $self->{form_element} if defined
7364            
7365              if ($self->{insertion_mode} & TABLE_IMS or
7366                  $self->{insertion_mode} & BODY_TABLE_IMS or
7367                  $self->{insertion_mode} == IN_COLUMN_GROUP_IM) {
7368                !!!cp ('t400.1');
7369                $self->{insertion_mode} = IN_SELECT_IN_TABLE_IM;
7370              } else {
7371                !!!cp ('t400.2');
7372                $self->{insertion_mode} = IN_SELECT_IM;
7373              }
7374              !!!nack ('t400.3');
7375            } else {
7376              !!!nack ('t402');
7377            }
7378                    
7379          !!!next-token;          !!!next-token;
7380          redo B;          next B;
7381        }        }
7382      } elsif ($token->{type} == END_TAG_TOKEN) {      } elsif ($token->{type} == END_TAG_TOKEN) {
7383        if ($token->{tag_name} eq 'body') {        if ($token->{tag_name} eq 'body') {
7384          if (@{$self->{open_elements}} > 1 and          ## has a |body| element in scope
7385              $self->{open_elements}->[1]->[1] eq 'body') {          my $i;
7386            for (@{$self->{open_elements}}) {          INSCOPE: {
7387              unless ({            for (reverse @{$self->{open_elements}}) {
7388                         dd => 1, dt => 1, li => 1, p => 1, td => 1,              if ($_->[1] & BODY_EL) {
7389                         th => 1, tr => 1, body => 1, html => 1,                !!!cp ('t405');
7390                       tbody => 1, tfoot => 1, thead => 1,                $i = $_;
7391                      }->{$_->[1]}) {                last INSCOPE;
7392                !!!cp ('t403');              } elsif ($_->[1] & SCOPING_EL) {
7393                !!!parse-error (type => 'not closed:'.$_->[1]);                !!!cp ('t405.1');
7394              } else {                last;
               !!!cp ('t404');  
7395              }              }
7396            }            }
7397    
7398            $self->{insertion_mode} = AFTER_BODY_IM;            !!!parse-error (type => 'start tag not allowed',
7399                              text => $token->{tag_name}, token => $token);
7400              ## NOTE: Ignore the token.
7401            !!!next-token;            !!!next-token;
7402            redo B;            next B;
7403          } else {          } # INSCOPE
7404            !!!cp ('t405');  
7405            !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});          for (@{$self->{open_elements}}) {
7406            ## Ignore the token            unless ($_->[1] & ALL_END_TAG_OPTIONAL_EL) {
7407            !!!next-token;              !!!cp ('t403');
7408            redo B;              !!!parse-error (type => 'not closed',
7409                                text => $_->[0]->manakai_local_name,
7410                                token => $token);
7411                last;
7412              } else {
7413                !!!cp ('t404');
7414              }
7415          }          }
7416    
7417            $self->{insertion_mode} = AFTER_BODY_IM;
7418            !!!next-token;
7419            next B;
7420        } elsif ($token->{tag_name} eq 'html') {        } elsif ($token->{tag_name} eq 'html') {
7421          if (@{$self->{open_elements}} > 1 and $self->{open_elements}->[1]->[1] eq 'body') {          ## TODO: Update this code.  It seems that the code below is not
7422            ## up-to-date, though it has same effect as speced.
7423            if (@{$self->{open_elements}} > 1 and
7424                $self->{open_elements}->[1]->[1] & BODY_EL) {
7425            ## ISSUE: There is an issue in the spec.            ## ISSUE: There is an issue in the spec.
7426            if ($self->{open_elements}->[-1]->[1] ne 'body') {            unless ($self->{open_elements}->[-1]->[1] & BODY_EL) {
7427              !!!cp ('t406');              !!!cp ('t406');
7428              !!!parse-error (type => 'not closed:'.$self->{open_elements}->[1]->[1]);              !!!parse-error (type => 'not closed',
7429                                text => $self->{open_elements}->[1]->[0]
7430                                    ->manakai_local_name,
7431                                token => $token);
7432            } else {            } else {
7433              !!!cp ('t407');              !!!cp ('t407');
7434            }            }
7435            $self->{insertion_mode} = AFTER_BODY_IM;            $self->{insertion_mode} = AFTER_BODY_IM;
7436            ## reprocess            ## reprocess
7437            redo B;            next B;
7438          } else {          } else {
7439            !!!cp ('t408');            !!!cp ('t408');
7440            !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});            !!!parse-error (type => 'unmatched end tag',
7441                              text => $token->{tag_name}, token => $token);
7442            ## Ignore the token            ## Ignore the token
7443            !!!next-token;            !!!next-token;
7444            redo B;            next B;
7445          }          }
7446        } elsif ({        } elsif ({
7447                  address => 1, blockquote => 1, center => 1, dir => 1,                  address => 1, blockquote => 1, center => 1, dir => 1,
7448                  div => 1, dl => 1, fieldset => 1, listing => 1,                  div => 1, dl => 1, fieldset => 1, listing => 1,
7449                  menu => 1, ol => 1, pre => 1, ul => 1,                  menu => 1, ol => 1, pre => 1, ul => 1,
                 p => 1,  
7450                  dd => 1, dt => 1, li => 1,                  dd => 1, dt => 1, li => 1,
7451                  button => 1, marquee => 1, object => 1,                  applet => 1, button => 1, marquee => 1, object => 1,
7452                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
7453          ## has an element in scope          ## has an element in scope
7454          my $i;          my $i;
7455          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7456            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
7457            if ($node->[1] eq $token->{tag_name}) {            if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
             ## generate implied end tags  
             if ({  
                  dd => ($token->{tag_name} ne 'dd'),  
                  dt => ($token->{tag_name} ne 'dt'),  
                  li => ($token->{tag_name} ne 'li'),  
                  p => ($token->{tag_name} ne 'p'),  
                  td => 1, th => 1, tr => 1,  
                  tbody => 1, tfoot=> 1, thead => 1,  
                 }->{$self->{open_elements}->[-1]->[1]}) {  
               !!!cp ('t409');  
               !!!back-token;  
               $token = {type => END_TAG_TOKEN,  
                         tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
               redo B;  
             }  
               
7458              !!!cp ('t410');              !!!cp ('t410');
7459              $i = $_;              $i = $_;
7460              last INSCOPE unless $token->{tag_name} eq 'p';              last INSCOPE;
7461            } elsif ({            } elsif ($node->[1] & SCOPING_EL) {
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
7462              !!!cp ('t411');              !!!cp ('t411');
7463              last INSCOPE;              last INSCOPE;
7464            }            }
7465          } # INSCOPE          } # INSCOPE
7466            
7467          if ($self->{open_elements}->[-1]->[1] ne $token->{tag_name}) {          unless (defined $i) { # has an element in scope
7468            if (defined $i) {            !!!cp ('t413');
7469              !!!parse-error (type => 'unmatched end tag',
7470                              text => $token->{tag_name}, token => $token);
7471              ## NOTE: Ignore the token.
7472            } else {
7473              ## Step 1. generate implied end tags
7474              while ({
7475                      ## END_TAG_OPTIONAL_EL
7476                      dd => ($token->{tag_name} ne 'dd'),
7477                      dt => ($token->{tag_name} ne 'dt'),
7478                      li => ($token->{tag_name} ne 'li'),
7479                      p => 1,
7480                      rt => 1,
7481                      rp => 1,
7482                     }->{$self->{open_elements}->[-1]->[0]->manakai_local_name}) {
7483                !!!cp ('t409');
7484                pop @{$self->{open_elements}};
7485              }
7486    
7487              ## Step 2.
7488              if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7489                      ne $token->{tag_name}) {
7490              !!!cp ('t412');              !!!cp ('t412');
7491              !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);              !!!parse-error (type => 'not closed',
7492                                text => $self->{open_elements}->[-1]->[0]
7493                                    ->manakai_local_name,
7494                                token => $token);
7495            } else {            } else {
7496              !!!cp ('t413');              !!!cp ('t414');
             !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
7497            }            }
7498          }  
7499                      ## Step 3.
         if (defined $i) {  
           !!!cp ('t414');  
7500            splice @{$self->{open_elements}}, $i;            splice @{$self->{open_elements}}, $i;
7501          } elsif ($token->{tag_name} eq 'p') {  
7502            !!!cp ('t415');            ## Step 4.
7503            ## As if <p>, then reprocess the current token            $clear_up_to_marker->()
7504            my $el;                if {
7505            !!!create-element ($el, 'p');                  applet => 1, button => 1, marquee => 1, object => 1,
7506            $insert->($el);                }->{$token->{tag_name}};
         } else {  
           !!!cp ('t416');  
7507          }          }
         $clear_up_to_marker->()  
           if {  
             button => 1, marquee => 1, object => 1,  
           }->{$token->{tag_name}};  
7508          !!!next-token;          !!!next-token;
7509          redo B;          next B;
7510        } elsif ($token->{tag_name} eq 'form') {        } elsif ($token->{tag_name} eq 'form') {
7511            undef $self->{form_element};
7512    
7513          ## has an element in scope          ## has an element in scope
7514            my $i;
7515          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7516            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
7517            if ($node->[1] eq $token->{tag_name}) {            if ($node->[1] & FORM_EL) {
             ## generate implied end tags  
             if ({  
                  dd => 1, dt => 1, li => 1, p => 1,  
                  td => 1, th => 1, tr => 1,  
                  tbody => 1, tfoot=> 1, thead => 1,  
                 }->{$self->{open_elements}->[-1]->[1]}) {  
               !!!cp ('t417');  
               !!!back-token;  
               $token = {type => END_TAG_TOKEN,  
                         tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
               redo B;  
             }  
               
7518              !!!cp ('t418');              !!!cp ('t418');
7519                $i = $_;
7520              last INSCOPE;              last INSCOPE;
7521            } elsif ({            } elsif ($node->[1] & SCOPING_EL) {
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
7522              !!!cp ('t419');              !!!cp ('t419');
7523              last INSCOPE;              last INSCOPE;
7524            }            }
7525          } # INSCOPE          } # INSCOPE
7526            
7527          if ($self->{open_elements}->[-1]->[1] eq $token->{tag_name}) {          unless (defined $i) { # has an element in scope
           !!!cp ('t420');  
           pop @{$self->{open_elements}};  
         } else {  
7528            !!!cp ('t421');            !!!cp ('t421');
7529            !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});            !!!parse-error (type => 'unmatched end tag',
7530                              text => $token->{tag_name}, token => $token);
7531              ## NOTE: Ignore the token.
7532            } else {
7533              ## Step 1. generate implied end tags
7534              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7535                !!!cp ('t417');
7536                pop @{$self->{open_elements}};
7537              }
7538              
7539              ## Step 2.
7540              if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7541                      ne $token->{tag_name}) {
7542                !!!cp ('t417.1');
7543                !!!parse-error (type => 'not closed',
7544                                text => $self->{open_elements}->[-1]->[0]
7545                                    ->manakai_local_name,
7546                                token => $token);
7547              } else {
7548                !!!cp ('t420');
7549              }  
7550              
7551              ## Step 3.
7552              splice @{$self->{open_elements}}, $i;
7553          }          }
7554    
         undef $self->{form_element};  
7555          !!!next-token;          !!!next-token;
7556          redo B;          next B;
7557        } elsif ({        } elsif ({
7558                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
7559                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
# Line 6138  sub _tree_construction_main ($) { Line 7561  sub _tree_construction_main ($) {
7561          my $i;          my $i;
7562          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7563            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
7564            if ({            if ($node->[1] & HEADING_EL) {
                h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,  
               }->{$node->[1]}) {  
             ## generate implied end tags  
             if ({  
                  dd => 1, dt => 1, li => 1, p => 1,  
                  td => 1, th => 1, tr => 1,  
                  tbody => 1, tfoot=> 1, thead => 1,  
                 }->{$self->{open_elements}->[-1]->[1]}) {  
               !!!cp ('t422');  
               !!!back-token;  
               $token = {type => END_TAG_TOKEN,  
                         tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
               redo B;  
             }  
   
7565              !!!cp ('t423');              !!!cp ('t423');
7566              $i = $_;              $i = $_;
7567              last INSCOPE;              last INSCOPE;
7568            } elsif ({            } elsif ($node->[1] & SCOPING_EL) {
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
7569              !!!cp ('t424');              !!!cp ('t424');
7570              last INSCOPE;              last INSCOPE;
7571            }            }
7572          } # INSCOPE          } # INSCOPE
7573    
7574            unless (defined $i) { # has an element in scope
7575              !!!cp ('t425.1');
7576              !!!parse-error (type => 'unmatched end tag',
7577                              text => $token->{tag_name}, token => $token);
7578              ## NOTE: Ignore the token.
7579            } else {
7580              ## Step 1. generate implied end tags
7581              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7582                !!!cp ('t422');
7583                pop @{$self->{open_elements}};
7584              }
7585              
7586              ## Step 2.
7587              if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7588                      ne $token->{tag_name}) {
7589                !!!cp ('t425');
7590                !!!parse-error (type => 'unmatched end tag',
7591                                text => $token->{tag_name}, token => $token);
7592              } else {
7593                !!!cp ('t426');
7594              }
7595    
7596              ## Step 3.
7597              splice @{$self->{open_elements}}, $i;
7598            }
7599                    
7600          if ($self->{open_elements}->[-1]->[1] ne $token->{tag_name}) {          !!!next-token;
7601            !!!cp ('t425');          next B;
7602            !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});        } elsif ($token->{tag_name} eq 'p') {
7603            ## has an element in scope
7604            my $i;
7605            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7606              my $node = $self->{open_elements}->[$_];
7607              if ($node->[1] & P_EL) {
7608                !!!cp ('t410.1');
7609                $i = $_;
7610                last INSCOPE;
7611              } elsif ($node->[1] & SCOPING_EL) {
7612                !!!cp ('t411.1');
7613                last INSCOPE;
7614              }
7615            } # INSCOPE
7616    
7617            if (defined $i) {
7618              if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7619                      ne $token->{tag_name}) {
7620                !!!cp ('t412.1');
7621                !!!parse-error (type => 'not closed',
7622                                text => $self->{open_elements}->[-1]->[0]
7623                                    ->manakai_local_name,
7624                                token => $token);
7625              } else {
7626                !!!cp ('t414.1');
7627              }
7628    
7629              splice @{$self->{open_elements}}, $i;
7630          } else {          } else {
7631            !!!cp ('t426');            !!!cp ('t413.1');
7632              !!!parse-error (type => 'unmatched end tag',
7633                              text => $token->{tag_name}, token => $token);
7634    
7635              !!!cp ('t415.1');
7636              ## As if <p>, then reprocess the current token
7637              my $el;
7638              !!!create-element ($el, $HTML_NS, 'p',, $token);
7639              $insert->($el);
7640              ## NOTE: Not inserted into |$self->{open_elements}|.
7641          }          }
7642            
         splice @{$self->{open_elements}}, $i if defined $i;  
7643          !!!next-token;          !!!next-token;
7644          redo B;          next B;
7645        } elsif ({        } elsif ({
7646                  a => 1,                  a => 1,
7647                  b => 1, big => 1, em => 1, font => 1, i => 1,                  b => 1, big => 1, em => 1, font => 1, i => 1,
# Line 6183  sub _tree_construction_main ($) { Line 7649  sub _tree_construction_main ($) {
7649                  strong => 1, tt => 1, u => 1,                  strong => 1, tt => 1, u => 1,
7650                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
7651          !!!cp ('t427');          !!!cp ('t427');
7652          $formatting_end_tag->($token->{tag_name});          $formatting_end_tag->($token);
7653          redo B;          next B;
7654        } elsif ($token->{tag_name} eq 'br') {        } elsif ($token->{tag_name} eq 'br') {
7655          !!!cp ('t428');          !!!cp ('t428');
7656          !!!parse-error (type => 'unmatched end tag:br');          !!!parse-error (type => 'unmatched end tag',
7657                            text => 'br', token => $token);
7658    
7659          ## As if <br>          ## As if <br>
7660          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
7661                    
7662          my $el;          my $el;
7663          !!!create-element ($el, 'br');          !!!create-element ($el, $HTML_NS, 'br',, $token);
7664          $insert->($el);          $insert->($el);
7665                    
7666          ## Ignore the token.          ## Ignore the token.
7667          !!!next-token;          !!!next-token;
7668          redo B;          next B;
7669        } elsif ({        } elsif ({
7670                  caption => 1, col => 1, colgroup => 1, frame => 1,                  caption => 1, col => 1, colgroup => 1, frame => 1,
7671                  frameset => 1, head => 1, option => 1, optgroup => 1,                  frameset => 1, head => 1, option => 1, optgroup => 1,
# Line 6212  sub _tree_construction_main ($) { Line 7679  sub _tree_construction_main ($) {
7679                  noscript => 0, ## TODO: if scripting is enabled                  noscript => 0, ## TODO: if scripting is enabled
7680                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
7681          !!!cp ('t429');          !!!cp ('t429');
7682          !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});          !!!parse-error (type => 'unmatched end tag',
7683                            text => $token->{tag_name}, token => $token);
7684          ## Ignore the token          ## Ignore the token
7685          !!!next-token;          !!!next-token;
7686          redo B;          next B;
7687                    
7688          ## ISSUE: Issue on HTML5 new elements in spec          ## ISSUE: Issue on HTML5 new elements in spec
7689                    
# Line 6226  sub _tree_construction_main ($) { Line 7694  sub _tree_construction_main ($) {
7694    
7695          ## Step 2          ## Step 2
7696          S2: {          S2: {
7697            if ($node->[1] eq $token->{tag_name}) {            if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
7698              ## Step 1              ## Step 1
7699              ## generate implied end tags              ## generate implied end tags
7700              if ({              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
                  dd => 1, dt => 1, li => 1, p => 1,  
                  td => 1, th => 1, tr => 1,  
                  tbody => 1, tfoot => 1, thead => 1,  
                 }->{$self->{open_elements}->[-1]->[1]}) {  
7701                !!!cp ('t430');                !!!cp ('t430');
7702                !!!back-token;                ## NOTE: |<ruby><rt></ruby>|.
7703                $token = {type => END_TAG_TOKEN,                ## ISSUE: <ruby><rt></rt> will also take this code path,
7704                          tag_name => $self->{open_elements}->[-1]->[1]}; # MUST                ## which seems wrong.
7705                redo B;                pop @{$self->{open_elements}};
7706                  $node_i++;
7707              }              }
7708                    
7709              ## Step 2              ## Step 2
7710              if ($token->{tag_name} ne $self->{open_elements}->[-1]->[1]) {              if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7711                        ne $token->{tag_name}) {
7712                !!!cp ('t431');                !!!cp ('t431');
7713                ## NOTE: <x><y></x>                ## NOTE: <x><y></x>
7714                !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                !!!parse-error (type => 'not closed',
7715                                  text => $self->{open_elements}->[-1]->[0]
7716                                      ->manakai_local_name,
7717                                  token => $token);
7718              } else {              } else {
7719                !!!cp ('t432');                !!!cp ('t432');
7720              }              }
7721                            
7722              ## Step 3              ## Step 3
7723              splice @{$self->{open_elements}}, $node_i;              splice @{$self->{open_elements}}, $node_i if $node_i < 0;
7724    
7725              !!!next-token;              !!!next-token;
7726              last S2;              last S2;
7727            } else {            } else {
7728              ## Step 3              ## Step 3
7729              if (not $formatting_category->{$node->[1]} and              if (not ($node->[1] & FORMATTING_EL) and
7730                  #not $phrasing_category->{$node->[1]} and                  #not $phrasing_category->{$node->[1]} and
7731                  ($special_category->{$node->[1]} or                  ($node->[1] & SPECIAL_EL or
7732                   $scoping_category->{$node->[1]})) {                   $node->[1] & SCOPING_EL)) {
7733                !!!cp ('t433');                !!!cp ('t433');
7734                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                !!!parse-error (type => 'unmatched end tag',
7735                                  text => $token->{tag_name}, token => $token);
7736                ## Ignore the token                ## Ignore the token
7737                !!!next-token;                !!!next-token;
7738                last S2;                last S2;
# Line 6278  sub _tree_construction_main ($) { Line 7748  sub _tree_construction_main ($) {
7748            ## Step 5;            ## Step 5;
7749            redo S2;            redo S2;
7750          } # S2          } # S2
7751          redo B;          next B;
7752        }        }
7753      }      }
7754      redo B;      next B;
7755      } continue { # B
7756        if ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {
7757          ## NOTE: The code below is executed in cases where it does not have
7758          ## to be, but it it is harmless even in those cases.
7759          ## has an element in scope
7760          INSCOPE: {
7761            for (reverse 0..$#{$self->{open_elements}}) {
7762              my $node = $self->{open_elements}->[$_];
7763              if ($node->[1] & FOREIGN_EL) {
7764                last INSCOPE;
7765              } elsif ($node->[1] & SCOPING_EL) {
7766                last;
7767              }
7768            }
7769            
7770            ## NOTE: No foreign element in scope.
7771            $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
7772          } # INSCOPE
7773        }
7774    } # B    } # B
7775    
   ## NOTE: The "trailing end" phase in HTML5 is split into  
   ## two insertion modes: "after html body" and "after html frameset".  
   ## NOTE: States in the main stage is preserved while  
   ## the parser stays in the trailing end phase. # MUST  
   
7776    ## Stop parsing # MUST    ## Stop parsing # MUST
7777        
7778    ## TODO: script stuffs    ## TODO: script stuffs
7779  } # _tree_construct_main  } # _tree_construct_main
7780    
7781  sub set_inner_html ($$$) {  sub set_inner_html ($$$$;$) {
7782    my $class = shift;    my $class = shift;
7783    my $node = shift;    my $node = shift;
7784    my $s = \$_[0];    #my $s = \$_[0];
7785    my $onerror = $_[1];    my $onerror = $_[1];
7786      my $get_wrapper = $_[2] || sub ($) { return $_[0] };
7787    
7788    ## ISSUE: Should {confident} be true?    ## ISSUE: Should {confident} be true?
7789    
# Line 6317  sub set_inner_html ($$$) { Line 7802  sub set_inner_html ($$$) {
7802      }      }
7803    
7804      ## Step 3, 4, 5 # MUST      ## Step 3, 4, 5 # MUST
7805      $class->parse_string ($$s => $node, $onerror);      $class->parse_char_string ($_[0] => $node, $onerror, $get_wrapper);
7806    } elsif ($nt == 1) {    } elsif ($nt == 1) {
7807      ## TODO: If non-html element      ## TODO: If non-html element
7808    
7809      ## NOTE: Most of this code is copied from |parse_string|      ## NOTE: Most of this code is copied from |parse_string|
7810    
7811    ## TODO: Support for $get_wrapper
7812    
7813      ## Step 1 # MUST      ## Step 1 # MUST
7814      my $this_doc = $node->owner_document;      my $this_doc = $node->owner_document;
7815      my $doc = $this_doc->implementation->create_document;      my $doc = $this_doc->implementation->create_document;
# Line 6330  sub set_inner_html ($$$) { Line 7817  sub set_inner_html ($$$) {
7817      my $p = $class->new;      my $p = $class->new;
7818      $p->{document} = $doc;      $p->{document} = $doc;
7819    
7820      ## Step 9 # MUST      ## Step 8 # MUST
7821      my $i = 0;      my $i = 0;
7822      my $line = 1;      $p->{line_prev} = $p->{line} = 1;
7823      my $column = 0;      $p->{column_prev} = $p->{column} = 0;
7824      $p->{set_next_char} = sub {      require Whatpm::Charset::DecodeHandle;
7825        my $input = Whatpm::Charset::DecodeHandle::CharString->new (\($_[0]));
7826        $input = $get_wrapper->($input);
7827        $p->{set_nc} = sub {
7828        my $self = shift;        my $self = shift;
7829    
7830        pop @{$self->{prev_char}};        my $char = '';
7831        unshift @{$self->{prev_char}}, $self->{next_char};        if (defined $self->{next_nc}) {
7832            $char = $self->{next_nc};
7833            delete $self->{next_nc};
7834            $self->{nc} = ord $char;
7835          } else {
7836            $self->{char_buffer} = '';
7837            $self->{char_buffer_pos} = 0;
7838            
7839            my $count = $input->manakai_read_until
7840                ($self->{char_buffer}, qr/[^\x00\x0A\x0D]/,
7841                 $self->{char_buffer_pos});
7842            if ($count) {
7843              $self->{line_prev} = $self->{line};
7844              $self->{column_prev} = $self->{column};
7845              $self->{column}++;
7846              $self->{nc}
7847                  = ord substr ($self->{char_buffer},
7848                                $self->{char_buffer_pos}++, 1);
7849              return;
7850            }
7851            
7852            if ($input->read ($char, 1)) {
7853              $self->{nc} = ord $char;
7854            } else {
7855              $self->{nc} = -1;
7856              return;
7857            }
7858          }
7859    
7860        $self->{next_char} = -1 and return if $i >= length $$s;        ($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column});
7861        $self->{next_char} = ord substr $$s, $i++, 1;        $p->{column}++;
7862        $column++;  
7863          if ($self->{nc} == 0x000A) { # LF
7864        if ($self->{next_char} == 0x000A) { # LF          $p->{line}++;
7865          $line++;          $p->{column} = 0;
         $column = 0;  
7866          !!!cp ('i1');          !!!cp ('i1');
7867        } elsif ($self->{next_char} == 0x000D) { # CR        } elsif ($self->{nc} == 0x000D) { # CR
7868          $i++ if substr ($$s, $i, 1) eq "\x0A";  ## TODO: support for abort/streaming
7869          $self->{next_char} = 0x000A; # LF # MUST          my $next = '';
7870          $line++;          if ($input->read ($next, 1) and $next ne "\x0A") {
7871          $column = 0;            $self->{next_nc} = $next;
7872            }
7873            $self->{nc} = 0x000A; # LF # MUST
7874            $p->{line}++;
7875            $p->{column} = 0;
7876          !!!cp ('i2');          !!!cp ('i2');
7877        } elsif ($self->{next_char} > 0x10FFFF) {        } elsif ($self->{nc} == 0x0000) { # NULL
         $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST  
         !!!cp ('i3');  
       } elsif ($self->{next_char} == 0x0000) { # NULL  
7878          !!!cp ('i4');          !!!cp ('i4');
7879          !!!parse-error (type => 'NULL');          !!!parse-error (type => 'NULL');
7880          $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST          $self->{nc} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
7881        }        }
7882      };      };
7883      $p->{prev_char} = [-1, -1, -1];  
7884      $p->{next_char} = -1;      $p->{read_until} = sub {
7885              #my ($scalar, $specials_range, $offset) = @_;
7886          return 0 if defined $p->{next_nc};
7887    
7888          my $pattern = qr/[^$_[1]\x00\x0A\x0D]/;
7889          my $offset = $_[2] || 0;
7890          
7891          if ($p->{char_buffer_pos} < length $p->{char_buffer}) {
7892            pos ($p->{char_buffer}) = $p->{char_buffer_pos};
7893            if ($p->{char_buffer} =~ /\G(?>$pattern)+/) {
7894              substr ($_[0], $offset)
7895                  = substr ($p->{char_buffer}, $-[0], $+[0] - $-[0]);
7896              my $count = $+[0] - $-[0];
7897              if ($count) {
7898                $p->{column} += $count;
7899                $p->{char_buffer_pos} += $count;
7900                $p->{line_prev} = $p->{line};
7901                $p->{column_prev} = $p->{column} - 1;
7902                $p->{nc} = -1;
7903              }
7904              return $count;
7905            } else {
7906              return 0;
7907            }
7908          } else {
7909            my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]);
7910            if ($count) {
7911              $p->{column} += $count;
7912              $p->{column_prev} += $count;
7913              $p->{nc} = -1;
7914            }
7915            return $count;
7916          }
7917        }; # $p->{read_until}
7918    
7919      my $ponerror = $onerror || sub {      my $ponerror = $onerror || sub {
7920        my (%opt) = @_;        my (%opt) = @_;
7921        warn "Parse error ($opt{type}) at line $opt{line} column $opt{column}\n";        my $line = $opt{line};
7922          my $column = $opt{column};
7923          if (defined $opt{token} and defined $opt{token}->{line}) {
7924            $line = $opt{token}->{line};
7925            $column = $opt{token}->{column};
7926          }
7927          warn "Parse error ($opt{type}) at line $line column $column\n";
7928      };      };
7929      $p->{parse_error} = sub {      $p->{parse_error} = sub {
7930        $ponerror->(@_, line => $line, column => $column);        $ponerror->(line => $p->{line}, column => $p->{column}, @_);
7931      };      };
7932            
7933        my $char_onerror = sub {
7934          my (undef, $type, %opt) = @_;
7935          $ponerror->(layer => 'encode',
7936                      line => $p->{line}, column => $p->{column} + 1,
7937                      %opt, type => $type);
7938        }; # $char_onerror
7939        $input->onerror ($char_onerror);
7940    
7941      $p->_initialize_tokenizer;      $p->_initialize_tokenizer;
7942      $p->_initialize_tree_constructor;      $p->_initialize_tree_constructor;
7943    
# Line 6395  sub set_inner_html ($$$) { Line 7959  sub set_inner_html ($$$) {
7959          unless defined $p->{content_model};          unless defined $p->{content_model};
7960          ## ISSUE: What is "the name of the element"? local name?          ## ISSUE: What is "the name of the element"? local name?
7961    
7962      $p->{inner_html_node} = [$node, $node_ln];      $p->{inner_html_node} = [$node, $el_category->{$node_ln}];
7963          ## TODO: Foreign element OK?
7964    
7965      ## Step 4      ## Step 3
7966      my $root = $doc->create_element_ns      my $root = $doc->create_element_ns
7967        ('http://www.w3.org/1999/xhtml', [undef, 'html']);        ('http://www.w3.org/1999/xhtml', [undef, 'html']);
7968    
7969      ## Step 5 # MUST      ## Step 4 # MUST
7970      $doc->append_child ($root);      $doc->append_child ($root);
7971    
7972      ## Step 6 # MUST      ## Step 5 # MUST
7973      push @{$p->{open_elements}}, [$root, 'html'];      push @{$p->{open_elements}}, [$root, $el_category->{html}];
7974    
7975      undef $p->{head_element};      undef $p->{head_element};
7976    
7977      ## Step 7 # MUST      ## Step 6 # MUST
7978      $p->_reset_insertion_mode;      $p->_reset_insertion_mode;
7979    
7980      ## Step 8 # MUST      ## Step 7 # MUST
7981      my $anode = $node;      my $anode = $node;
7982      AN: while (defined $anode) {      AN: while (defined $anode) {
7983        if ($anode->node_type == 1) {        if ($anode->node_type == 1) {
# Line 6428  sub set_inner_html ($$$) { Line 7993  sub set_inner_html ($$$) {
7993        $anode = $anode->parent_node;        $anode = $anode->parent_node;
7994      } # AN      } # AN
7995            
7996      ## Step 3 # MUST      ## Step 9 # MUST
     ## Step 10 # MUST  
7997      {      {
7998        my $self = $p;        my $self = $p;
7999        !!!next-token;        !!!next-token;
8000      }      }
8001      $p->_tree_construction_main;      $p->_tree_construction_main;
8002    
8003      ## Step 11 # MUST      ## Step 10 # MUST
8004      my @cn = @{$node->child_nodes};      my @cn = @{$node->child_nodes};
8005      for (@cn) {      for (@cn) {
8006        $node->remove_child ($_);        $node->remove_child ($_);
8007      }      }
8008      ## ISSUE: mutation events? read-only?      ## ISSUE: mutation events? read-only?
8009    
8010      ## Step 12 # MUST      ## Step 11 # MUST
8011      @cn = @{$root->child_nodes};      @cn = @{$root->child_nodes};
8012      for (@cn) {      for (@cn) {
8013        $this_doc->adopt_node ($_);        $this_doc->adopt_node ($_);
# Line 6452  sub set_inner_html ($$$) { Line 8016  sub set_inner_html ($$$) {
8016      ## ISSUE: mutation events?      ## ISSUE: mutation events?
8017    
8018      $p->_terminate_tree_constructor;      $p->_terminate_tree_constructor;
8019    
8020        delete $p->{parse_error}; # delete loop
8021    } else {    } else {
8022      die "$0: |set_inner_html| is not defined for node of type $nt";      die "$0: |set_inner_html| is not defined for node of type $nt";
8023    }    }

Legend:
Removed from v.1.81  
changed lines
  Added in v.1.192

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24