/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.44 by wakaba, Sat Jul 21 07:34:32 2007 UTC revision 1.202 by wakaba, Sat Oct 4 14:31:28 2008 UTC
# Line 1  Line 1 
1  package Whatpm::HTML;  package Whatpm::HTML;
2  use strict;  use strict;
3  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};
4    use Error qw(:try);
5    
6    ## NOTE: This module don't check all HTML5 parse errors; character
7    ## encoding related parse errors are expected to be handled by relevant
8    ## modules.
9    ## Parse errors for control characters that are not allowed in HTML5
10    ## documents, for surrogate code points, and for noncharacter code
11    ## points, as well as U+FFFD substitions for characters whose code points
12    ## is higher than U+10FFFF may be detected by combining the parser with
13    ## the checker implemented by Whatpm::Charset::UnicodeChecker (for its
14    ## usage example, see |t/HTML-tree.t| in the Whatpm package or the
15    ## WebHACC::Language::HTML module in the WebHACC package).
16    
17  ## ISSUE:  ## ISSUE:
18  ## var doc = implementation.createDocument (null, null, null);  ## var doc = implementation.createDocument (null, null, null);
19  ## doc.write ('');  ## doc.write ('');
20  ## alert (doc.compatMode);  ## alert (doc.compatMode);
21    
22  ## ISSUE: HTML5 revision 967 says that the encoding layer MUST NOT  require IO::Handle;
23  ## strip BOM and the HTML layer MUST ignore it.  Whether we can do it  
24  ## is not yet clear.  my $HTML_NS = q<http://www.w3.org/1999/xhtml>;
25  ## "{U+FEFF}..." in UTF-16BE/UTF-16LE is three or four characters?  my $MML_NS = q<http://www.w3.org/1998/Math/MathML>;
26  ## "{U+FEFF}..." in GB18030?  my $SVG_NS = q<http://www.w3.org/2000/svg>;
27    my $XLINK_NS = q<http://www.w3.org/1999/xlink>;
28  my $permitted_slash_tag_name = {  my $XML_NS = q<http://www.w3.org/XML/1998/namespace>;
29    base => 1,  my $XMLNS_NS = q<http://www.w3.org/2000/xmlns/>;
30    link => 1,  
31    meta => 1,  sub A_EL () { 0b1 }
32    hr => 1,  sub ADDRESS_EL () { 0b10 }
33    br => 1,  sub BODY_EL () { 0b100 }
34    img=> 1,  sub BUTTON_EL () { 0b1000 }
35    embed => 1,  sub CAPTION_EL () { 0b10000 }
36    param => 1,  sub DD_EL () { 0b100000 }
37    area => 1,  sub DIV_EL () { 0b1000000 }
38    col => 1,  sub DT_EL () { 0b10000000 }
39    input => 1,  sub FORM_EL () { 0b100000000 }
40    sub FORMATTING_EL () { 0b1000000000 }
41    sub FRAMESET_EL () { 0b10000000000 }
42    sub HEADING_EL () { 0b100000000000 }
43    sub HTML_EL () { 0b1000000000000 }
44    sub LI_EL () { 0b10000000000000 }
45    sub NOBR_EL () { 0b100000000000000 }
46    sub OPTION_EL () { 0b1000000000000000 }
47    sub OPTGROUP_EL () { 0b10000000000000000 }
48    sub P_EL () { 0b100000000000000000 }
49    sub SELECT_EL () { 0b1000000000000000000 }
50    sub TABLE_EL () { 0b10000000000000000000 }
51    sub TABLE_CELL_EL () { 0b100000000000000000000 }
52    sub TABLE_ROW_EL () { 0b1000000000000000000000 }
53    sub TABLE_ROW_GROUP_EL () { 0b10000000000000000000000 }
54    sub MISC_SCOPING_EL () { 0b100000000000000000000000 }
55    sub MISC_SPECIAL_EL () { 0b1000000000000000000000000 }
56    sub FOREIGN_EL () { 0b10000000000000000000000000 }
57    sub FOREIGN_FLOW_CONTENT_EL () { 0b100000000000000000000000000 }
58    sub MML_AXML_EL () { 0b1000000000000000000000000000 }
59    sub RUBY_EL () { 0b10000000000000000000000000000 }
60    sub RUBY_COMPONENT_EL () { 0b100000000000000000000000000000 }
61    
62    sub TABLE_ROWS_EL () {
63      TABLE_EL |
64      TABLE_ROW_EL |
65      TABLE_ROW_GROUP_EL
66    }
67    
68    ## NOTE: Used in "generate implied end tags" algorithm.
69    ## NOTE: There is a code where a modified version of
70    ## END_TAG_OPTIONAL_EL is used in "generate implied end tags"
71    ## implementation (search for the algorithm name).
72    sub END_TAG_OPTIONAL_EL () {
73      DD_EL |
74      DT_EL |
75      LI_EL |
76      OPTION_EL |
77      OPTGROUP_EL |
78      P_EL |
79      RUBY_COMPONENT_EL
80    }
81    
82    ## NOTE: Used in </body> and EOF algorithms.
83    sub ALL_END_TAG_OPTIONAL_EL () {
84      DD_EL |
85      DT_EL |
86      LI_EL |
87      P_EL |
88    
89      ## ISSUE: option, optgroup, rt, rp?
90    
91      BODY_EL |
92      HTML_EL |
93      TABLE_CELL_EL |
94      TABLE_ROW_EL |
95      TABLE_ROW_GROUP_EL
96    }
97    
98    sub SCOPING_EL () {
99      BUTTON_EL |
100      CAPTION_EL |
101      HTML_EL |
102      TABLE_EL |
103      TABLE_CELL_EL |
104      MISC_SCOPING_EL
105    }
106    
107    sub TABLE_SCOPING_EL () {
108      HTML_EL |
109      TABLE_EL
110    }
111    
112    sub TABLE_ROWS_SCOPING_EL () {
113      HTML_EL |
114      TABLE_ROW_GROUP_EL
115    }
116    
117    sub TABLE_ROW_SCOPING_EL () {
118      HTML_EL |
119      TABLE_ROW_EL
120    }
121    
122    sub SPECIAL_EL () {
123      ADDRESS_EL |
124      BODY_EL |
125      DIV_EL |
126    
127      DD_EL |
128      DT_EL |
129      LI_EL |
130      P_EL |
131    
132      FORM_EL |
133      FRAMESET_EL |
134      HEADING_EL |
135      SELECT_EL |
136      TABLE_ROW_EL |
137      TABLE_ROW_GROUP_EL |
138      MISC_SPECIAL_EL
139    }
140    
141    my $el_category = {
142      a => A_EL | FORMATTING_EL,
143      address => ADDRESS_EL,
144      applet => MISC_SCOPING_EL,
145      area => MISC_SPECIAL_EL,
146      article => MISC_SPECIAL_EL,
147      aside => MISC_SPECIAL_EL,
148      b => FORMATTING_EL,
149      base => MISC_SPECIAL_EL,
150      basefont => MISC_SPECIAL_EL,
151      bgsound => MISC_SPECIAL_EL,
152      big => FORMATTING_EL,
153      blockquote => MISC_SPECIAL_EL,
154      body => BODY_EL,
155      br => MISC_SPECIAL_EL,
156      button => BUTTON_EL,
157      caption => CAPTION_EL,
158      center => MISC_SPECIAL_EL,
159      col => MISC_SPECIAL_EL,
160      colgroup => MISC_SPECIAL_EL,
161      command => MISC_SPECIAL_EL,
162      datagrid => MISC_SPECIAL_EL,
163      dd => DD_EL,
164      details => MISC_SPECIAL_EL,
165      dialog => MISC_SPECIAL_EL,
166      dir => MISC_SPECIAL_EL,
167      div => DIV_EL,
168      dl => MISC_SPECIAL_EL,
169      dt => DT_EL,
170      em => FORMATTING_EL,
171      embed => MISC_SPECIAL_EL,
172      eventsource => MISC_SPECIAL_EL,
173      fieldset => MISC_SPECIAL_EL,
174      figure => MISC_SPECIAL_EL,
175      font => FORMATTING_EL,
176      footer => MISC_SPECIAL_EL,
177      form => FORM_EL,
178      frame => MISC_SPECIAL_EL,
179      frameset => FRAMESET_EL,
180      h1 => HEADING_EL,
181      h2 => HEADING_EL,
182      h3 => HEADING_EL,
183      h4 => HEADING_EL,
184      h5 => HEADING_EL,
185      h6 => HEADING_EL,
186      head => MISC_SPECIAL_EL,
187      header => MISC_SPECIAL_EL,
188      hr => MISC_SPECIAL_EL,
189      html => HTML_EL,
190      i => FORMATTING_EL,
191      iframe => MISC_SPECIAL_EL,
192      img => MISC_SPECIAL_EL,
193      #image => MISC_SPECIAL_EL, ## NOTE: Commented out in the spec.
194      input => MISC_SPECIAL_EL,
195      isindex => MISC_SPECIAL_EL,
196      li => LI_EL,
197      link => MISC_SPECIAL_EL,
198      listing => MISC_SPECIAL_EL,
199      marquee => MISC_SCOPING_EL,
200      menu => MISC_SPECIAL_EL,
201      meta => MISC_SPECIAL_EL,
202      nav => MISC_SPECIAL_EL,
203      nobr => NOBR_EL | FORMATTING_EL,
204      noembed => MISC_SPECIAL_EL,
205      noframes => MISC_SPECIAL_EL,
206      noscript => MISC_SPECIAL_EL,
207      object => MISC_SCOPING_EL,
208      ol => MISC_SPECIAL_EL,
209      optgroup => OPTGROUP_EL,
210      option => OPTION_EL,
211      p => P_EL,
212      param => MISC_SPECIAL_EL,
213      plaintext => MISC_SPECIAL_EL,
214      pre => MISC_SPECIAL_EL,
215      rp => RUBY_COMPONENT_EL,
216      rt => RUBY_COMPONENT_EL,
217      ruby => RUBY_EL,
218      s => FORMATTING_EL,
219      script => MISC_SPECIAL_EL,
220      select => SELECT_EL,
221      section => MISC_SPECIAL_EL,
222      small => FORMATTING_EL,
223      spacer => MISC_SPECIAL_EL,
224      strike => FORMATTING_EL,
225      strong => FORMATTING_EL,
226      style => MISC_SPECIAL_EL,
227      table => TABLE_EL,
228      tbody => TABLE_ROW_GROUP_EL,
229      td => TABLE_CELL_EL,
230      textarea => MISC_SPECIAL_EL,
231      tfoot => TABLE_ROW_GROUP_EL,
232      th => TABLE_CELL_EL,
233      thead => TABLE_ROW_GROUP_EL,
234      title => MISC_SPECIAL_EL,
235      tr => TABLE_ROW_EL,
236      tt => FORMATTING_EL,
237      u => FORMATTING_EL,
238      ul => MISC_SPECIAL_EL,
239      wbr => MISC_SPECIAL_EL,
240    };
241    
242    my $el_category_f = {
243      $MML_NS => {
244        'annotation-xml' => MML_AXML_EL,
245        mi => FOREIGN_FLOW_CONTENT_EL,
246        mo => FOREIGN_FLOW_CONTENT_EL,
247        mn => FOREIGN_FLOW_CONTENT_EL,
248        ms => FOREIGN_FLOW_CONTENT_EL,
249        mtext => FOREIGN_FLOW_CONTENT_EL,
250      },
251      $SVG_NS => {
252        foreignObject => FOREIGN_FLOW_CONTENT_EL | MISC_SCOPING_EL,
253        desc => FOREIGN_FLOW_CONTENT_EL,
254        title => FOREIGN_FLOW_CONTENT_EL,
255      },
256      ## NOTE: In addition, FOREIGN_EL is set to non-HTML elements.
257    };
258    
259    my $svg_attr_name = {
260      attributename => 'attributeName',
261      attributetype => 'attributeType',
262      basefrequency => 'baseFrequency',
263      baseprofile => 'baseProfile',
264      calcmode => 'calcMode',
265      clippathunits => 'clipPathUnits',
266      contentscripttype => 'contentScriptType',
267      contentstyletype => 'contentStyleType',
268      diffuseconstant => 'diffuseConstant',
269      edgemode => 'edgeMode',
270      externalresourcesrequired => 'externalResourcesRequired',
271      filterres => 'filterRes',
272      filterunits => 'filterUnits',
273      glyphref => 'glyphRef',
274      gradienttransform => 'gradientTransform',
275      gradientunits => 'gradientUnits',
276      kernelmatrix => 'kernelMatrix',
277      kernelunitlength => 'kernelUnitLength',
278      keypoints => 'keyPoints',
279      keysplines => 'keySplines',
280      keytimes => 'keyTimes',
281      lengthadjust => 'lengthAdjust',
282      limitingconeangle => 'limitingConeAngle',
283      markerheight => 'markerHeight',
284      markerunits => 'markerUnits',
285      markerwidth => 'markerWidth',
286      maskcontentunits => 'maskContentUnits',
287      maskunits => 'maskUnits',
288      numoctaves => 'numOctaves',
289      pathlength => 'pathLength',
290      patterncontentunits => 'patternContentUnits',
291      patterntransform => 'patternTransform',
292      patternunits => 'patternUnits',
293      pointsatx => 'pointsAtX',
294      pointsaty => 'pointsAtY',
295      pointsatz => 'pointsAtZ',
296      preservealpha => 'preserveAlpha',
297      preserveaspectratio => 'preserveAspectRatio',
298      primitiveunits => 'primitiveUnits',
299      refx => 'refX',
300      refy => 'refY',
301      repeatcount => 'repeatCount',
302      repeatdur => 'repeatDur',
303      requiredextensions => 'requiredExtensions',
304      requiredfeatures => 'requiredFeatures',
305      specularconstant => 'specularConstant',
306      specularexponent => 'specularExponent',
307      spreadmethod => 'spreadMethod',
308      startoffset => 'startOffset',
309      stddeviation => 'stdDeviation',
310      stitchtiles => 'stitchTiles',
311      surfacescale => 'surfaceScale',
312      systemlanguage => 'systemLanguage',
313      tablevalues => 'tableValues',
314      targetx => 'targetX',
315      targety => 'targetY',
316      textlength => 'textLength',
317      viewbox => 'viewBox',
318      viewtarget => 'viewTarget',
319      xchannelselector => 'xChannelSelector',
320      ychannelselector => 'yChannelSelector',
321      zoomandpan => 'zoomAndPan',
322  };  };
323    
324  my $c1_entity_char = {  my $foreign_attr_xname = {
325      'xlink:actuate' => [$XLINK_NS, ['xlink', 'actuate']],
326      'xlink:arcrole' => [$XLINK_NS, ['xlink', 'arcrole']],
327      'xlink:href' => [$XLINK_NS, ['xlink', 'href']],
328      'xlink:role' => [$XLINK_NS, ['xlink', 'role']],
329      'xlink:show' => [$XLINK_NS, ['xlink', 'show']],
330      'xlink:title' => [$XLINK_NS, ['xlink', 'title']],
331      'xlink:type' => [$XLINK_NS, ['xlink', 'type']],
332      'xml:base' => [$XML_NS, ['xml', 'base']],
333      'xml:lang' => [$XML_NS, ['xml', 'lang']],
334      'xml:space' => [$XML_NS, ['xml', 'space']],
335      'xmlns' => [$XMLNS_NS, [undef, 'xmlns']],
336      'xmlns:xlink' => [$XMLNS_NS, ['xmlns', 'xlink']],
337    };
338    
339    ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.
340    
341    my $charref_map = {
342      0x0D => 0x000A,
343    0x80 => 0x20AC,    0x80 => 0x20AC,
344    0x81 => 0xFFFD,    0x81 => 0xFFFD,
345    0x82 => 0x201A,    0x82 => 0x201A,
# Line 60  my $c1_entity_char = { Line 372  my $c1_entity_char = {
372    0x9D => 0xFFFD,    0x9D => 0xFFFD,
373    0x9E => 0x017E,    0x9E => 0x017E,
374    0x9F => 0x0178,    0x9F => 0x0178,
375  }; # $c1_entity_char  }; # $charref_map
376    $charref_map->{$_} = 0xFFFD
377        for 0x0000..0x0008, 0x000B, 0x000E..0x001F, 0x007F,
378            0xD800..0xDFFF, 0xFDD0..0xFDDF, ## ISSUE: 0xFDEF
379            0xFFFE, 0xFFFF, 0x1FFFE, 0x1FFFF, 0x2FFFE, 0x2FFFF, 0x3FFFE, 0x3FFFF,
380            0x4FFFE, 0x4FFFF, 0x5FFFE, 0x5FFFF, 0x6FFFE, 0x6FFFF, 0x7FFFE,
381            0x7FFFF, 0x8FFFE, 0x8FFFF, 0x9FFFE, 0x9FFFF, 0xAFFFE, 0xAFFFF,
382            0xBFFFE, 0xBFFFF, 0xCFFFE, 0xCFFFF, 0xDFFFE, 0xDFFFF, 0xEFFFE,
383            0xEFFFF, 0xFFFFE, 0xFFFFF, 0x10FFFE, 0x10FFFF;
384    
385  my $special_category = {  ## TODO: Invoke the reset algorithm when a resettable element is
386    address => 1, area => 1, base => 1, basefont => 1, bgsound => 1,  ## created (cf. HTML5 revision 2259).
387    blockquote => 1, body => 1, br => 1, center => 1, col => 1, colgroup => 1,  
388    dd => 1, dir => 1, div => 1, dl => 1, dt => 1, embed => 1, fieldset => 1,  sub parse_byte_string ($$$$;$) {
389    form => 1, frame => 1, frameset => 1, h1 => 1, h2 => 1, h3 => 1,    my $self = shift;
390    h4 => 1, h5 => 1, h6 => 1, head => 1, hr => 1, iframe => 1, image => 1,    my $charset_name = shift;
391    img => 1, input => 1, isindex => 1, li => 1, link => 1, listing => 1,    open my $input, '<', ref $_[0] ? $_[0] : \($_[0]);
392    menu => 1, meta => 1, noembed => 1, noframes => 1, noscript => 1,    return $self->parse_byte_stream ($charset_name, $input, @_[1..$#_]);
393    ol => 1, optgroup => 1, option => 1, p => 1, param => 1, plaintext => 1,  } # parse_byte_string
394    pre => 1, script => 1, select => 1, spacer => 1, style => 1, tbody => 1,  
395    textarea => 1, tfoot => 1, thead => 1, title => 1, tr => 1, ul => 1, wbr => 1,  sub parse_byte_stream ($$$$;$$) {
396  };    # my ($self, $charset_name, $byte_stream, $doc, $onerror, $get_wrapper) = @_;
397  my $scoping_category = {    my $self = ref $_[0] ? shift : shift->new;
398    button => 1, caption => 1, html => 1, marquee => 1, object => 1,    my $charset_name = shift;
399    table => 1, td => 1, th => 1,    my $byte_stream = $_[0];
400  };  
401  my $formatting_category = {    my $onerror = $_[2] || sub {
402    a => 1, b => 1, big => 1, em => 1, font => 1, i => 1, nobr => 1,      my (%opt) = @_;
403    s => 1, small => 1, strile => 1, strong => 1, tt => 1, u => 1,      warn "Parse error ($opt{type})\n";
404  };    };
405  # $phrasing_category: all other elements    $self->{parse_error} = $onerror; # updated later by parse_char_string
406    
407      my $get_wrapper = $_[3] || sub ($) {
408        return $_[0]; # $_[0] = byte stream handle, returned = arg to char handle
409      };
410    
411      ## HTML5 encoding sniffing algorithm
412      require Message::Charset::Info;
413      my $charset;
414      my $buffer;
415      my ($char_stream, $e_status);
416    
417      SNIFFING: {
418        ## NOTE: By setting |allow_fallback| option true when the
419        ## |get_decode_handle| method is invoked, we ignore what the HTML5
420        ## spec requires, i.e. unsupported encoding should be ignored.
421          ## TODO: We should not do this unless the parser is invoked
422          ## in the conformance checking mode, in which this behavior
423          ## would be useful.
424    
425        ## Step 1
426        if (defined $charset_name) {
427          $charset = Message::Charset::Info->get_by_html_name ($charset_name);
428              ## TODO: Is this ok?  Transfer protocol's parameter should be
429              ## interpreted in its semantics?
430    
431          ($char_stream, $e_status) = $charset->get_decode_handle
432              ($byte_stream, allow_error_reporting => 1,
433               allow_fallback => 1);
434          if ($char_stream) {
435            $self->{confident} = 1;
436            last SNIFFING;
437          } else {
438            !!!parse-error (type => 'charset:not supported',
439                            layer => 'encode',
440                            line => 1, column => 1,
441                            value => $charset_name,
442                            level => $self->{level}->{uncertain});
443          }
444        }
445    
446        ## Step 2
447        my $byte_buffer = '';
448        for (1..1024) {
449          my $char = $byte_stream->getc;
450          last unless defined $char;
451          $byte_buffer .= $char;
452        } ## TODO: timeout
453    
454        ## Step 3
455        if ($byte_buffer =~ /^\xFE\xFF/) {
456          $charset = Message::Charset::Info->get_by_html_name ('utf-16be');
457          ($char_stream, $e_status) = $charset->get_decode_handle
458              ($byte_stream, allow_error_reporting => 1,
459               allow_fallback => 1, byte_buffer => \$byte_buffer);
460          $self->{confident} = 1;
461          last SNIFFING;
462        } elsif ($byte_buffer =~ /^\xFF\xFE/) {
463          $charset = Message::Charset::Info->get_by_html_name ('utf-16le');
464          ($char_stream, $e_status) = $charset->get_decode_handle
465              ($byte_stream, allow_error_reporting => 1,
466               allow_fallback => 1, byte_buffer => \$byte_buffer);
467          $self->{confident} = 1;
468          last SNIFFING;
469        } elsif ($byte_buffer =~ /^\xEF\xBB\xBF/) {
470          $charset = Message::Charset::Info->get_by_html_name ('utf-8');
471          ($char_stream, $e_status) = $charset->get_decode_handle
472              ($byte_stream, allow_error_reporting => 1,
473               allow_fallback => 1, byte_buffer => \$byte_buffer);
474          $self->{confident} = 1;
475          last SNIFFING;
476        }
477    
478        ## Step 4
479        ## TODO: <meta charset>
480    
481        ## Step 5
482        ## TODO: from history
483    
484        ## Step 6
485        require Whatpm::Charset::UniversalCharDet;
486        $charset_name = Whatpm::Charset::UniversalCharDet->detect_byte_string
487            ($byte_buffer);
488        if (defined $charset_name) {
489          $charset = Message::Charset::Info->get_by_html_name ($charset_name);
490    
491          require Whatpm::Charset::DecodeHandle;
492          $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new
493              ($byte_stream);
494          ($char_stream, $e_status) = $charset->get_decode_handle
495              ($buffer, allow_error_reporting => 1,
496               allow_fallback => 1, byte_buffer => \$byte_buffer);
497          if ($char_stream) {
498            $buffer->{buffer} = $byte_buffer;
499            !!!parse-error (type => 'sniffing:chardet',
500                            text => $charset_name,
501                            level => $self->{level}->{info},
502                            layer => 'encode',
503                            line => 1, column => 1);
504            $self->{confident} = 0;
505            last SNIFFING;
506          }
507        }
508    
509        ## Step 7: default
510        ## TODO: Make this configurable.
511        $charset = Message::Charset::Info->get_by_html_name ('windows-1252');
512            ## NOTE: We choose |windows-1252| here, since |utf-8| should be
513            ## detectable in the step 6.
514        require Whatpm::Charset::DecodeHandle;
515        $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new
516            ($byte_stream);
517        ($char_stream, $e_status)
518            = $charset->get_decode_handle ($buffer,
519                                           allow_error_reporting => 1,
520                                           allow_fallback => 1,
521                                           byte_buffer => \$byte_buffer);
522        $buffer->{buffer} = $byte_buffer;
523        !!!parse-error (type => 'sniffing:default',
524                        text => 'windows-1252',
525                        level => $self->{level}->{info},
526                        line => 1, column => 1,
527                        layer => 'encode');
528        $self->{confident} = 0;
529      } # SNIFFING
530    
531      if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
532        $self->{input_encoding} = $charset->get_iana_name; ## TODO: Should we set actual charset decoder's encoding name?
533        !!!parse-error (type => 'chardecode:fallback',
534                        #text => $self->{input_encoding},
535                        level => $self->{level}->{uncertain},
536                        line => 1, column => 1,
537                        layer => 'encode');
538      } elsif (not ($e_status &
539                    Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) {
540        $self->{input_encoding} = $charset->get_iana_name;
541        !!!parse-error (type => 'chardecode:no error',
542                        text => $self->{input_encoding},
543                        level => $self->{level}->{uncertain},
544                        line => 1, column => 1,
545                        layer => 'encode');
546      } else {
547        $self->{input_encoding} = $charset->get_iana_name;
548      }
549    
550      $self->{change_encoding} = sub {
551        my $self = shift;
552        $charset_name = shift;
553        my $token = shift;
554    
555        $charset = Message::Charset::Info->get_by_html_name ($charset_name);
556        ($char_stream, $e_status) = $charset->get_decode_handle
557            ($byte_stream, allow_error_reporting => 1, allow_fallback => 1,
558             byte_buffer => \ $buffer->{buffer});
559        
560        if ($char_stream) { # if supported
561          ## "Change the encoding" algorithm:
562    
563          ## Step 1    
564          if ($charset->{category} &
565              Message::Charset::Info::CHARSET_CATEGORY_UTF16 ()) {
566            $charset = Message::Charset::Info->get_by_html_name ('utf-8');
567            ($char_stream, $e_status) = $charset->get_decode_handle
568                ($byte_stream,
569                 byte_buffer => \ $buffer->{buffer});
570          }
571          $charset_name = $charset->get_iana_name;
572          
573          ## Step 2
574          if (defined $self->{input_encoding} and
575              $self->{input_encoding} eq $charset_name) {
576            !!!parse-error (type => 'charset label:matching',
577                            text => $charset_name,
578                            level => $self->{level}->{info});
579            $self->{confident} = 1;
580            return;
581          }
582    
583          !!!parse-error (type => 'charset label detected',
584                          text => $self->{input_encoding},
585                          value => $charset_name,
586                          level => $self->{level}->{warn},
587                          token => $token);
588          
589          ## Step 3
590          # if (can) {
591            ## change the encoding on the fly.
592            #$self->{confident} = 1;
593            #return;
594          # }
595          
596          ## Step 4
597          throw Whatpm::HTML::RestartParser ();
598        }
599      }; # $self->{change_encoding}
600    
601      my $char_onerror = sub {
602        my (undef, $type, %opt) = @_;
603        !!!parse-error (layer => 'encode',
604                        line => $self->{line}, column => $self->{column} + 1,
605                        %opt, type => $type);
606        if ($opt{octets}) {
607          ${$opt{octets}} = "\x{FFFD}"; # relacement character
608        }
609      };
610    
611      my $wrapped_char_stream = $get_wrapper->($char_stream);
612      $wrapped_char_stream->onerror ($char_onerror);
613    
614      my @args = ($_[1], $_[2]); # $doc, $onerror - $get_wrapper = undef;
615      my $return;
616      try {
617        $return = $self->parse_char_stream ($wrapped_char_stream, @args);  
618      } catch Whatpm::HTML::RestartParser with {
619        ## NOTE: Invoked after {change_encoding}.
620    
621        if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
622          $self->{input_encoding} = $charset->get_iana_name; ## TODO: Should we set actual charset decoder's encoding name?
623          !!!parse-error (type => 'chardecode:fallback',
624                          level => $self->{level}->{uncertain},
625                          #text => $self->{input_encoding},
626                          line => 1, column => 1,
627                          layer => 'encode');
628        } elsif (not ($e_status &
629                      Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) {
630          $self->{input_encoding} = $charset->get_iana_name;
631          !!!parse-error (type => 'chardecode:no error',
632                          text => $self->{input_encoding},
633                          level => $self->{level}->{uncertain},
634                          line => 1, column => 1,
635                          layer => 'encode');
636        } else {
637          $self->{input_encoding} = $charset->get_iana_name;
638        }
639        $self->{confident} = 1;
640    
641        $wrapped_char_stream = $get_wrapper->($char_stream);
642        $wrapped_char_stream->onerror ($char_onerror);
643    
644        $return = $self->parse_char_stream ($wrapped_char_stream, @args);
645      };
646      return $return;
647    } # parse_byte_stream
648    
649    ## NOTE: HTML5 spec says that the encoding layer MUST NOT strip BOM
650    ## and the HTML layer MUST ignore it.  However, we does strip BOM in
651    ## the encoding layer and the HTML layer does not ignore any U+FEFF,
652    ## because the core part of our HTML parser expects a string of character,
653    ## not a string of bytes or code units or anything which might contain a BOM.
654    ## Therefore, any parser interface that accepts a string of bytes,
655    ## such as |parse_byte_string| in this module, must ensure that it does
656    ## strip the BOM and never strip any ZWNBSP.
657    
658  sub parse_string ($$$;$) {  sub parse_char_string ($$$;$$) {
659    my $self = shift->new;    #my ($self, $s, $doc, $onerror, $get_wrapper) = @_;
660    my $s = \$_[0];    my $self = shift;
661      my $s = ref $_[0] ? $_[0] : \($_[0]);
662      require Whatpm::Charset::DecodeHandle;
663      my $input = Whatpm::Charset::DecodeHandle::CharString->new ($s);
664      return $self->parse_char_stream ($input, @_[1..$#_]);
665    } # parse_char_string
666    *parse_string = \&parse_char_string; ## NOTE: Alias for backward compatibility.
667    
668    sub parse_char_stream ($$$;$$) {
669      my $self = ref $_[0] ? shift : shift->new;
670      my $input = $_[0];
671    $self->{document} = $_[1];    $self->{document} = $_[1];
672      @{$self->{document}->child_nodes} = ();
673    
674    ## NOTE: |set_inner_html| copies most of this method's code    ## NOTE: |set_inner_html| copies most of this method's code
675    
676    my $i = 0;    $self->{confident} = 1 unless exists $self->{confident};
677    my $line = 1;    $self->{document}->input_encoding ($self->{input_encoding})
678    my $column = 0;        if defined $self->{input_encoding};
679    $self->{set_next_input_character} = sub {  ## TODO: |{input_encoding}| is needless?
680    
681      $self->{line_prev} = $self->{line} = 1;
682      $self->{column_prev} = -1;
683      $self->{column} = 0;
684      $self->{set_nc} = sub {
685      my $self = shift;      my $self = shift;
686    
687      pop @{$self->{prev_input_character}};      my $char = '';
688      unshift @{$self->{prev_input_character}}, $self->{next_input_character};      if (defined $self->{next_nc}) {
689          $char = $self->{next_nc};
690          delete $self->{next_nc};
691          $self->{nc} = ord $char;
692        } else {
693          $self->{char_buffer} = '';
694          $self->{char_buffer_pos} = 0;
695    
696      $self->{next_input_character} = -1 and return if $i >= length $$s;        my $count = $input->manakai_read_until
697      $self->{next_input_character} = ord substr $$s, $i++, 1;           ($self->{char_buffer}, qr/[^\x00\x0A\x0D]/, $self->{char_buffer_pos});
698      $column++;        if ($count) {
699            $self->{line_prev} = $self->{line};
700            $self->{column_prev} = $self->{column};
701            $self->{column}++;
702            $self->{nc}
703                = ord substr ($self->{char_buffer}, $self->{char_buffer_pos}++, 1);
704            return;
705          }
706    
707          if ($input->read ($char, 1)) {
708            $self->{nc} = ord $char;
709          } else {
710            $self->{nc} = -1;
711            return;
712          }
713        }
714    
715        ($self->{line_prev}, $self->{column_prev})
716            = ($self->{line}, $self->{column});
717        $self->{column}++;
718            
719      if ($self->{next_input_character} == 0x000A) { # LF      if ($self->{nc} == 0x000A) { # LF
720        $line++;        !!!cp ('j1');
721        $column = 0;        $self->{line}++;
722      } elsif ($self->{next_input_character} == 0x000D) { # CR        $self->{column} = 0;
723        $i++ if substr ($$s, $i, 1) eq "\x0A";      } elsif ($self->{nc} == 0x000D) { # CR
724        $self->{next_input_character} = 0x000A; # LF # MUST        !!!cp ('j2');
725        $line++;  ## TODO: support for abort/streaming
726        $column = 0;        my $next = '';
727      } elsif ($self->{next_input_character} > 0x10FFFF) {        if ($input->read ($next, 1) and $next ne "\x0A") {
728        $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST          $self->{next_nc} = $next;
729      } elsif ($self->{next_input_character} == 0x0000) { # NULL        }
730          $self->{nc} = 0x000A; # LF # MUST
731          $self->{line}++;
732          $self->{column} = 0;
733        } elsif ($self->{nc} == 0x0000) { # NULL
734          !!!cp ('j4');
735        !!!parse-error (type => 'NULL');        !!!parse-error (type => 'NULL');
736        $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST        $self->{nc} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
737      }      }
738    };    };
739    $self->{prev_input_character} = [-1, -1, -1];  
740    $self->{next_input_character} = -1;    $self->{read_until} = sub {
741        #my ($scalar, $specials_range, $offset) = @_;
742        return 0 if defined $self->{next_nc};
743    
744        my $pattern = qr/[^$_[1]\x00\x0A\x0D]/;
745        my $offset = $_[2] || 0;
746    
747        if ($self->{char_buffer_pos} < length $self->{char_buffer}) {
748          pos ($self->{char_buffer}) = $self->{char_buffer_pos};
749          if ($self->{char_buffer} =~ /\G(?>$pattern)+/) {
750            substr ($_[0], $offset)
751                = substr ($self->{char_buffer}, $-[0], $+[0] - $-[0]);
752            my $count = $+[0] - $-[0];
753            if ($count) {
754              $self->{column} += $count;
755              $self->{char_buffer_pos} += $count;
756              $self->{line_prev} = $self->{line};
757              $self->{column_prev} = $self->{column} - 1;
758              $self->{nc} = -1;
759            }
760            return $count;
761          } else {
762            return 0;
763          }
764        } else {
765          my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]);
766          if ($count) {
767            $self->{column} += $count;
768            $self->{line_prev} = $self->{line};
769            $self->{column_prev} = $self->{column} - 1;
770            $self->{nc} = -1;
771          }
772          return $count;
773        }
774      }; # $self->{read_until}
775    
776    my $onerror = $_[2] || sub {    my $onerror = $_[2] || sub {
777      my (%opt) = @_;      my (%opt) = @_;
778      warn "Parse error ($opt{type}) at line $opt{line} column $opt{column}\n";      my $line = $opt{token} ? $opt{token}->{line} : $opt{line};
779        my $column = $opt{token} ? $opt{token}->{column} : $opt{column};
780        warn "Parse error ($opt{type}) at line $line column $column\n";
781    };    };
782    $self->{parse_error} = sub {    $self->{parse_error} = sub {
783      $onerror->(@_, line => $line, column => $column);      $onerror->(line => $self->{line}, column => $self->{column}, @_);
784    };    };
785    
786      my $char_onerror = sub {
787        my (undef, $type, %opt) = @_;
788        !!!parse-error (layer => 'encode',
789                        line => $self->{line}, column => $self->{column} + 1,
790                        %opt, type => $type);
791      }; # $char_onerror
792    
793      if ($_[3]) {
794        $input = $_[3]->($input);
795        $input->onerror ($char_onerror);
796      } else {
797        $input->onerror ($char_onerror) unless defined $input->onerror;
798      }
799    
800    $self->_initialize_tokenizer;    $self->_initialize_tokenizer;
801    $self->_initialize_tree_constructor;    $self->_initialize_tree_constructor;
802    $self->_construct_tree;    $self->_construct_tree;
803    $self->_terminate_tree_constructor;    $self->_terminate_tree_constructor;
804    
805      delete $self->{parse_error}; # remove loop
806    
807    return $self->{document};    return $self->{document};
808  } # parse_string  } # parse_char_stream
809    
810  sub new ($) {  sub new ($) {
811    my $class = shift;    my $class = shift;
812    my $self = bless {}, $class;    my $self = bless {
813    $self->{set_next_input_character} = sub {      level => {must => 'm',
814      $self->{next_input_character} = -1;                should => 's',
815                  warn => 'w',
816                  info => 'i',
817                  uncertain => 'u'},
818      }, $class;
819      $self->{set_nc} = sub {
820        $self->{nc} = -1;
821    };    };
822    $self->{parse_error} = sub {    $self->{parse_error} = sub {
823      #      #
824    };    };
825      $self->{change_encoding} = sub {
826        # if ($_[0] is a supported encoding) {
827        #   run "change the encoding" algorithm;
828        #   throw Whatpm::HTML::RestartParser (charset => $new_encoding);
829        # }
830      };
831      $self->{application_cache_selection} = sub {
832        #
833      };
834    return $self;    return $self;
835  } # new  } # new
836    
# Line 159  sub CDATA_CONTENT_MODEL () { CM_LIMITED_ Line 843  sub CDATA_CONTENT_MODEL () { CM_LIMITED_
843  sub RCDATA_CONTENT_MODEL () { CM_ENTITY | CM_LIMITED_MARKUP }  sub RCDATA_CONTENT_MODEL () { CM_ENTITY | CM_LIMITED_MARKUP }
844  sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP }  sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP }
845    
846    sub DATA_STATE () { 0 }
847    #sub ENTITY_DATA_STATE () { 1 }
848    sub TAG_OPEN_STATE () { 2 }
849    sub CLOSE_TAG_OPEN_STATE () { 3 }
850    sub TAG_NAME_STATE () { 4 }
851    sub BEFORE_ATTRIBUTE_NAME_STATE () { 5 }
852    sub ATTRIBUTE_NAME_STATE () { 6 }
853    sub AFTER_ATTRIBUTE_NAME_STATE () { 7 }
854    sub BEFORE_ATTRIBUTE_VALUE_STATE () { 8 }
855    sub ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE () { 9 }
856    sub ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE () { 10 }
857    sub ATTRIBUTE_VALUE_UNQUOTED_STATE () { 11 }
858    #sub ENTITY_IN_ATTRIBUTE_VALUE_STATE () { 12 }
859    sub MARKUP_DECLARATION_OPEN_STATE () { 13 }
860    sub COMMENT_START_STATE () { 14 }
861    sub COMMENT_START_DASH_STATE () { 15 }
862    sub COMMENT_STATE () { 16 }
863    sub COMMENT_END_STATE () { 17 }
864    sub COMMENT_END_DASH_STATE () { 18 }
865    sub BOGUS_COMMENT_STATE () { 19 }
866    sub DOCTYPE_STATE () { 20 }
867    sub BEFORE_DOCTYPE_NAME_STATE () { 21 }
868    sub DOCTYPE_NAME_STATE () { 22 }
869    sub AFTER_DOCTYPE_NAME_STATE () { 23 }
870    sub BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE () { 24 }
871    sub DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE () { 25 }
872    sub DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE () { 26 }
873    sub AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE () { 27 }
874    sub BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 28 }
875    sub DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE () { 29 }
876    sub DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE () { 30 }
877    sub AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 31 }
878    sub BOGUS_DOCTYPE_STATE () { 32 }
879    sub AFTER_ATTRIBUTE_VALUE_QUOTED_STATE () { 33 }
880    sub SELF_CLOSING_START_TAG_STATE () { 34 }
881    sub CDATA_SECTION_STATE () { 35 }
882    sub MD_HYPHEN_STATE () { 36 } # "markup declaration open state" in the spec
883    sub MD_DOCTYPE_STATE () { 37 } # "markup declaration open state" in the spec
884    sub MD_CDATA_STATE () { 38 } # "markup declaration open state" in the spec
885    sub CDATA_RCDATA_CLOSE_TAG_STATE () { 39 } # "close tag open state" in the spec
886    sub CDATA_SECTION_MSE1_STATE () { 40 } # "CDATA section state" in the spec
887    sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec
888    sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec
889    sub SYSTEM_STATE () { 43 } # "after DOCTYPE name state" in the spec
890    ## NOTE: "Entity data state", "entity in attribute value state", and
891    ## "consume a character reference" algorithm are jointly implemented
892    ## using the following six states:
893    sub ENTITY_STATE () { 44 }
894    sub ENTITY_HASH_STATE () { 45 }
895    sub NCR_NUM_STATE () { 46 }
896    sub HEXREF_X_STATE () { 47 }
897    sub HEXREF_HEX_STATE () { 48 }
898    sub ENTITY_NAME_STATE () { 49 }
899    sub PCDATA_STATE () { 50 } # "data state" in the spec
900    
901    sub DOCTYPE_TOKEN () { 1 }
902    sub COMMENT_TOKEN () { 2 }
903    sub START_TAG_TOKEN () { 3 }
904    sub END_TAG_TOKEN () { 4 }
905    sub END_OF_FILE_TOKEN () { 5 }
906    sub CHARACTER_TOKEN () { 6 }
907    
908    sub AFTER_HTML_IMS () { 0b100 }
909    sub HEAD_IMS ()       { 0b1000 }
910    sub BODY_IMS ()       { 0b10000 }
911    sub BODY_TABLE_IMS () { 0b100000 }
912    sub TABLE_IMS ()      { 0b1000000 }
913    sub ROW_IMS ()        { 0b10000000 }
914    sub BODY_AFTER_IMS () { 0b100000000 }
915    sub FRAME_IMS ()      { 0b1000000000 }
916    sub SELECT_IMS ()     { 0b10000000000 }
917    sub IN_FOREIGN_CONTENT_IM () { 0b100000000000 }
918        ## NOTE: "in foreign content" insertion mode is special; it is combined
919        ## with the secondary insertion mode.  In this parser, they are stored
920        ## together in the bit-or'ed form.
921    
922    ## NOTE: "initial" and "before html" insertion modes have no constants.
923    
924    ## NOTE: "after after body" insertion mode.
925    sub AFTER_HTML_BODY_IM () { AFTER_HTML_IMS | BODY_AFTER_IMS }
926    
927    ## NOTE: "after after frameset" insertion mode.
928    sub AFTER_HTML_FRAMESET_IM () { AFTER_HTML_IMS | FRAME_IMS }
929    
930    sub IN_HEAD_IM () { HEAD_IMS | 0b00 }
931    sub IN_HEAD_NOSCRIPT_IM () { HEAD_IMS | 0b01 }
932    sub AFTER_HEAD_IM () { HEAD_IMS | 0b10 }
933    sub BEFORE_HEAD_IM () { HEAD_IMS | 0b11 }
934    sub IN_BODY_IM () { BODY_IMS }
935    sub IN_CELL_IM () { BODY_IMS | BODY_TABLE_IMS | 0b01 }
936    sub IN_CAPTION_IM () { BODY_IMS | BODY_TABLE_IMS | 0b10 }
937    sub IN_ROW_IM () { TABLE_IMS | ROW_IMS | 0b01 }
938    sub IN_TABLE_BODY_IM () { TABLE_IMS | ROW_IMS | 0b10 }
939    sub IN_TABLE_IM () { TABLE_IMS }
940    sub AFTER_BODY_IM () { BODY_AFTER_IMS }
941    sub IN_FRAMESET_IM () { FRAME_IMS | 0b01 }
942    sub AFTER_FRAMESET_IM () { FRAME_IMS | 0b10 }
943    sub IN_SELECT_IM () { SELECT_IMS | 0b01 }
944    sub IN_SELECT_IN_TABLE_IM () { SELECT_IMS | 0b10 }
945    sub IN_COLUMN_GROUP_IM () { 0b10 }
946    
947  ## Implementations MUST act as if state machine in the spec  ## Implementations MUST act as if state machine in the spec
948    
949  sub _initialize_tokenizer ($) {  sub _initialize_tokenizer ($) {
950    my $self = shift;    my $self = shift;
951    $self->{state} = 'data'; # MUST    $self->{state} = DATA_STATE; # MUST
952      #$self->{s_kwd}; # state keyword - initialized when used
953      #$self->{entity__value}; # initialized when used
954      #$self->{entity__match}; # initialized when used
955    $self->{content_model} = PCDATA_CONTENT_MODEL; # be    $self->{content_model} = PCDATA_CONTENT_MODEL; # be
956    undef $self->{current_token}; # start tag, end tag, comment, or DOCTYPE    undef $self->{ct}; # current token
957    undef $self->{current_attribute};    undef $self->{ca}; # current attribute
958    undef $self->{last_emitted_start_tag_name};    undef $self->{last_stag_name}; # last emitted start tag name
959    undef $self->{last_attribute_value_state};    #$self->{prev_state}; # initialized when used
960    $self->{char} = [];    delete $self->{self_closing};
961    # $self->{next_input_character}    $self->{char_buffer} = '';
962      $self->{char_buffer_pos} = 0;
963      $self->{nc} = -1; # next input character
964      #$self->{next_nc}
965    !!!next-input-character;    !!!next-input-character;
966    $self->{token} = [];    $self->{token} = [];
967    # $self->{escape}    # $self->{escape}
968  } # _initialize_tokenizer  } # _initialize_tokenizer
969    
970  ## A token has:  ## A token has:
971  ##   ->{type} eq 'DOCTYPE', 'start tag', 'end tag', 'comment',  ##   ->{type} == DOCTYPE_TOKEN, START_TAG_TOKEN, END_TAG_TOKEN, COMMENT_TOKEN,
972  ##       'character', or 'end-of-file'  ##       CHARACTER_TOKEN, or END_OF_FILE_TOKEN
973  ##   ->{name} (DOCTYPE, start tag (tag name), end tag (tag name))  ##   ->{name} (DOCTYPE_TOKEN)
974  ##   ->{public_identifier} (DOCTYPE)  ##   ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)
975  ##   ->{system_identifier} (DOCTYPE)  ##   ->{pubid} (DOCTYPE_TOKEN)
976  ##   ->{correct} == 1 or 0 (DOCTYPE)  ##   ->{sysid} (DOCTYPE_TOKEN)
977  ##   ->{attributes} isa HASH (start tag, end tag)  ##   ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag
978  ##   ->{data} (comment, character)  ##   ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)
979    ##        ->{name}
980    ##        ->{value}
981    ##        ->{has_reference} == 1 or 0
982    ##   ->{data} (COMMENT_TOKEN, CHARACTER_TOKEN)
983    ## NOTE: The "self-closing flag" is hold as |$self->{self_closing}|.
984    ##     |->{self_closing}| is used to save the value of |$self->{self_closing}|
985    ##     while the token is pushed back to the stack.
986    
987  ## Emitted token MUST immediately be handled by the tree construction state.  ## Emitted token MUST immediately be handled by the tree construction state.
988    
# Line 194  sub _initialize_tokenizer ($) { Line 992  sub _initialize_tokenizer ($) {
992  ## has completed loading.  If one has, then it MUST be executed  ## has completed loading.  If one has, then it MUST be executed
993  ## and removed from the list.  ## and removed from the list.
994    
995    ## TODO: Polytheistic slash SHOULD NOT be used. (Applied only to atheists.)
996    ## (This requirement was dropped from HTML5 spec, unfortunately.)
997    
998    my $is_space = {
999      0x0009 => 1, # CHARACTER TABULATION (HT)
1000      0x000A => 1, # LINE FEED (LF)
1001      #0x000B => 0, # LINE TABULATION (VT)
1002      0x000C => 1, # FORM FEED (FF)
1003      #0x000D => 1, # CARRIAGE RETURN (CR)
1004      0x0020 => 1, # SPACE (SP)
1005    };
1006    
1007  sub _get_next_token ($) {  sub _get_next_token ($) {
1008    my $self = shift;    my $self = shift;
1009    
1010      if ($self->{self_closing}) {
1011        !!!parse-error (type => 'nestc', token => $self->{ct});
1012        ## NOTE: The |self_closing| flag is only set by start tag token.
1013        ## In addition, when a start tag token is emitted, it is always set to
1014        ## |ct|.
1015        delete $self->{self_closing};
1016      }
1017    
1018    if (@{$self->{token}}) {    if (@{$self->{token}}) {
1019        $self->{self_closing} = $self->{token}->[0]->{self_closing};
1020      return shift @{$self->{token}};      return shift @{$self->{token}};
1021    }    }
1022    
1023    A: {    A: {
1024      if ($self->{state} eq 'data') {      if ($self->{state} == PCDATA_STATE) {
1025        if ($self->{next_input_character} == 0x0026) { # &        ## NOTE: Same as |DATA_STATE|, but only for |PCDATA| content model.
1026          if ($self->{content_model} & CM_ENTITY) { # PCDATA | RCDATA  
1027            $self->{state} = 'entity data';        if ($self->{nc} == 0x0026) { # &
1028            !!!cp (0.1);
1029            ## NOTE: In the spec, the tokenizer is switched to the
1030            ## "entity data state".  In this implementation, the tokenizer
1031            ## is switched to the |ENTITY_STATE|, which is an implementation
1032            ## of the "consume a character reference" algorithm.
1033            $self->{entity_add} = -1;
1034            $self->{prev_state} = DATA_STATE;
1035            $self->{state} = ENTITY_STATE;
1036            !!!next-input-character;
1037            redo A;
1038          } elsif ($self->{nc} == 0x003C) { # <
1039            !!!cp (0.2);
1040            $self->{state} = TAG_OPEN_STATE;
1041            !!!next-input-character;
1042            redo A;
1043          } elsif ($self->{nc} == -1) {
1044            !!!cp (0.3);
1045            !!!emit ({type => END_OF_FILE_TOKEN,
1046                      line => $self->{line}, column => $self->{column}});
1047            last A; ## TODO: ok?
1048          } else {
1049            !!!cp (0.4);
1050            #
1051          }
1052    
1053          # Anything else
1054          my $token = {type => CHARACTER_TOKEN,
1055                       data => chr $self->{nc},
1056                       line => $self->{line}, column => $self->{column},
1057                      };
1058          $self->{read_until}->($token->{data}, q[<&], length $token->{data});
1059    
1060          ## Stay in the state.
1061          !!!next-input-character;
1062          !!!emit ($token);
1063          redo A;
1064        } elsif ($self->{state} == DATA_STATE) {
1065          $self->{s_kwd} = '' unless defined $self->{s_kwd};
1066          if ($self->{nc} == 0x0026) { # &
1067            $self->{s_kwd} = '';
1068            if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA
1069                not $self->{escape}) {
1070              !!!cp (1);
1071              ## NOTE: In the spec, the tokenizer is switched to the
1072              ## "entity data state".  In this implementation, the tokenizer
1073              ## is switched to the |ENTITY_STATE|, which is an implementation
1074              ## of the "consume a character reference" algorithm.
1075              $self->{entity_add} = -1;
1076              $self->{prev_state} = DATA_STATE;
1077              $self->{state} = ENTITY_STATE;
1078            !!!next-input-character;            !!!next-input-character;
1079            redo A;            redo A;
1080          } else {          } else {
1081              !!!cp (2);
1082            #            #
1083          }          }
1084        } elsif ($self->{next_input_character} == 0x002D) { # -        } elsif ($self->{nc} == 0x002D) { # -
1085          if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA          if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1086            unless ($self->{escape}) {            $self->{s_kwd} .= '-';
1087              if ($self->{prev_input_character}->[0] == 0x002D and # -            
1088                  $self->{prev_input_character}->[1] == 0x0021 and # !            if ($self->{s_kwd} eq '<!--') {
1089                  $self->{prev_input_character}->[2] == 0x003C) { # <              !!!cp (3);
1090                $self->{escape} = 1;              $self->{escape} = 1; # unless $self->{escape};
1091              }              $self->{s_kwd} = '--';
1092                #
1093              } elsif ($self->{s_kwd} eq '---') {
1094                !!!cp (4);
1095                $self->{s_kwd} = '--';
1096                #
1097              } else {
1098                !!!cp (5);
1099                #
1100            }            }
1101          }          }
1102                    
1103          #          #
1104        } elsif ($self->{next_input_character} == 0x003C) { # <        } elsif ($self->{nc} == 0x0021) { # !
1105            if (length $self->{s_kwd}) {
1106              !!!cp (5.1);
1107              $self->{s_kwd} .= '!';
1108              #
1109            } else {
1110              !!!cp (5.2);
1111              #$self->{s_kwd} = '';
1112              #
1113            }
1114            #
1115          } elsif ($self->{nc} == 0x003C) { # <
1116          if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA          if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA
1117              (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA              (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA
1118               not $self->{escape})) {               not $self->{escape})) {
1119            $self->{state} = 'tag open';            !!!cp (6);
1120              $self->{state} = TAG_OPEN_STATE;
1121            !!!next-input-character;            !!!next-input-character;
1122            redo A;            redo A;
1123          } else {          } else {
1124              !!!cp (7);
1125              $self->{s_kwd} = '';
1126            #            #
1127          }          }
1128        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1129          if ($self->{escape} and          if ($self->{escape} and
1130              ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA              ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA
1131            if ($self->{prev_input_character}->[0] == 0x002D and # -            if ($self->{s_kwd} eq '--') {
1132                $self->{prev_input_character}->[1] == 0x002D) { # -              !!!cp (8);
1133              delete $self->{escape};              delete $self->{escape};
1134              } else {
1135                !!!cp (9);
1136            }            }
1137            } else {
1138              !!!cp (10);
1139          }          }
1140                    
1141            $self->{s_kwd} = '';
1142          #          #
1143        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1144          !!!emit ({type => 'end-of-file'});          !!!cp (11);
1145            $self->{s_kwd} = '';
1146            !!!emit ({type => END_OF_FILE_TOKEN,
1147                      line => $self->{line}, column => $self->{column}});
1148          last A; ## TODO: ok?          last A; ## TODO: ok?
1149          } else {
1150            !!!cp (12);
1151            $self->{s_kwd} = '';
1152            #
1153        }        }
       # Anything else  
       my $token = {type => 'character',  
                    data => chr $self->{next_input_character}};  
       ## Stay in the data state  
       !!!next-input-character;  
1154    
1155        !!!emit ($token);        # Anything else
1156          my $token = {type => CHARACTER_TOKEN,
1157        redo A;                     data => chr $self->{nc},
1158      } elsif ($self->{state} eq 'entity data') {                     line => $self->{line}, column => $self->{column},
1159        ## (cannot happen in CDATA state)                    };
1160                if ($self->{read_until}->($token->{data}, q[-!<>&],
1161        my $token = $self->_tokenize_attempt_to_consume_an_entity (0);                                  length $token->{data})) {
1162            $self->{s_kwd} = '';
1163        $self->{state} = 'data';        }
1164        # next-input-character is already done  
1165          ## Stay in the data state.
1166        unless (defined $token) {        if ($self->{content_model} == PCDATA_CONTENT_MODEL) {
1167          !!!emit ({type => 'character', data => '&'});          !!!cp (13);
1168            $self->{state} = PCDATA_STATE;
1169        } else {        } else {
1170          !!!emit ($token);          !!!cp (14);
1171            ## Stay in the state.
1172        }        }
1173          !!!next-input-character;
1174          !!!emit ($token);
1175        redo A;        redo A;
1176      } elsif ($self->{state} eq 'tag open') {      } elsif ($self->{state} == TAG_OPEN_STATE) {
1177        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1178          if ($self->{next_input_character} == 0x002F) { # /          if ($self->{nc} == 0x002F) { # /
1179              !!!cp (15);
1180            !!!next-input-character;            !!!next-input-character;
1181            $self->{state} = 'close tag open';            $self->{state} = CLOSE_TAG_OPEN_STATE;
1182            redo A;            redo A;
1183            } elsif ($self->{nc} == 0x0021) { # !
1184              !!!cp (15.1);
1185              $self->{s_kwd} = '<' unless $self->{escape};
1186              #
1187          } else {          } else {
1188            ## reconsume            !!!cp (16);
1189            $self->{state} = 'data';            #
   
           !!!emit ({type => 'character', data => '<'});  
   
           redo A;  
1190          }          }
1191    
1192            ## reconsume
1193            $self->{state} = DATA_STATE;
1194            !!!emit ({type => CHARACTER_TOKEN, data => '<',
1195                      line => $self->{line_prev},
1196                      column => $self->{column_prev},
1197                     });
1198            redo A;
1199        } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA        } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA
1200          if ($self->{next_input_character} == 0x0021) { # !          if ($self->{nc} == 0x0021) { # !
1201            $self->{state} = 'markup declaration open';            !!!cp (17);
1202              $self->{state} = MARKUP_DECLARATION_OPEN_STATE;
1203            !!!next-input-character;            !!!next-input-character;
1204            redo A;            redo A;
1205          } elsif ($self->{next_input_character} == 0x002F) { # /          } elsif ($self->{nc} == 0x002F) { # /
1206            $self->{state} = 'close tag open';            !!!cp (18);
1207              $self->{state} = CLOSE_TAG_OPEN_STATE;
1208            !!!next-input-character;            !!!next-input-character;
1209            redo A;            redo A;
1210          } elsif (0x0041 <= $self->{next_input_character} and          } elsif (0x0041 <= $self->{nc} and
1211                   $self->{next_input_character} <= 0x005A) { # A..Z                   $self->{nc} <= 0x005A) { # A..Z
1212            $self->{current_token}            !!!cp (19);
1213              = {type => 'start tag',            $self->{ct}
1214                 tag_name => chr ($self->{next_input_character} + 0x0020)};              = {type => START_TAG_TOKEN,
1215            $self->{state} = 'tag name';                 tag_name => chr ($self->{nc} + 0x0020),
1216                   line => $self->{line_prev},
1217                   column => $self->{column_prev}};
1218              $self->{state} = TAG_NAME_STATE;
1219            !!!next-input-character;            !!!next-input-character;
1220            redo A;            redo A;
1221          } elsif (0x0061 <= $self->{next_input_character} and          } elsif (0x0061 <= $self->{nc} and
1222                   $self->{next_input_character} <= 0x007A) { # a..z                   $self->{nc} <= 0x007A) { # a..z
1223            $self->{current_token} = {type => 'start tag',            !!!cp (20);
1224                              tag_name => chr ($self->{next_input_character})};            $self->{ct} = {type => START_TAG_TOKEN,
1225            $self->{state} = 'tag name';                                      tag_name => chr ($self->{nc}),
1226                                        line => $self->{line_prev},
1227                                        column => $self->{column_prev}};
1228              $self->{state} = TAG_NAME_STATE;
1229            !!!next-input-character;            !!!next-input-character;
1230            redo A;            redo A;
1231          } elsif ($self->{next_input_character} == 0x003E) { # >          } elsif ($self->{nc} == 0x003E) { # >
1232            !!!parse-error (type => 'empty start tag');            !!!cp (21);
1233            $self->{state} = 'data';            !!!parse-error (type => 'empty start tag',
1234                              line => $self->{line_prev},
1235                              column => $self->{column_prev});
1236              $self->{state} = DATA_STATE;
1237            !!!next-input-character;            !!!next-input-character;
1238    
1239            !!!emit ({type => 'character', data => '<>'});            !!!emit ({type => CHARACTER_TOKEN, data => '<>',
1240                        line => $self->{line_prev},
1241                        column => $self->{column_prev},
1242                       });
1243    
1244            redo A;            redo A;
1245          } elsif ($self->{next_input_character} == 0x003F) { # ?          } elsif ($self->{nc} == 0x003F) { # ?
1246            !!!parse-error (type => 'pio');            !!!cp (22);
1247            $self->{state} = 'bogus comment';            !!!parse-error (type => 'pio',
1248            ## $self->{next_input_character} is intentionally left as is                            line => $self->{line_prev},
1249                              column => $self->{column_prev});
1250              $self->{state} = BOGUS_COMMENT_STATE;
1251              $self->{ct} = {type => COMMENT_TOKEN, data => '',
1252                                        line => $self->{line_prev},
1253                                        column => $self->{column_prev},
1254                                       };
1255              ## $self->{nc} is intentionally left as is
1256            redo A;            redo A;
1257          } else {          } else {
1258            !!!parse-error (type => 'bare stago');            !!!cp (23);
1259            $self->{state} = 'data';            !!!parse-error (type => 'bare stago',
1260                              line => $self->{line_prev},
1261                              column => $self->{column_prev});
1262              $self->{state} = DATA_STATE;
1263            ## reconsume            ## reconsume
1264    
1265            !!!emit ({type => 'character', data => '<'});            !!!emit ({type => CHARACTER_TOKEN, data => '<',
1266                        line => $self->{line_prev},
1267                        column => $self->{column_prev},
1268                       });
1269    
1270            redo A;            redo A;
1271          }          }
1272        } else {        } else {
1273          die "$0: $self->{content_model} in tag open";          die "$0: $self->{content_model} in tag open";
1274        }        }
1275      } elsif ($self->{state} eq 'close tag open') {      } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {
1276        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA        ## NOTE: The "close tag open state" in the spec is implemented as
1277          if (defined $self->{last_emitted_start_tag_name}) {        ## |CLOSE_TAG_OPEN_STATE| and |CDATA_RCDATA_CLOSE_TAG_STATE|.
           ## NOTE: <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>  
           my @next_char;  
           TAGNAME: for (my $i = 0; $i < length $self->{last_emitted_start_tag_name}; $i++) {  
             push @next_char, $self->{next_input_character};  
             my $c = ord substr ($self->{last_emitted_start_tag_name}, $i, 1);  
             my $C = 0x0061 <= $c && $c <= 0x007A ? $c - 0x0020 : $c;  
             if ($self->{next_input_character} == $c or $self->{next_input_character} == $C) {  
               !!!next-input-character;  
               next TAGNAME;  
             } else {  
               $self->{next_input_character} = shift @next_char; # reconsume  
               !!!back-next-input-character (@next_char);  
               $self->{state} = 'data';  
1278    
1279                !!!emit ({type => 'character', data => '</'});        my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"
1280            if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1281                redo A;          if (defined $self->{last_stag_name}) {
1282              }            $self->{state} = CDATA_RCDATA_CLOSE_TAG_STATE;
1283            }            $self->{s_kwd} = '';
1284            push @next_char, $self->{next_input_character};            ## Reconsume.
1285                    redo A;
           unless ($self->{next_input_character} == 0x0009 or # HT  
                   $self->{next_input_character} == 0x000A or # LF  
                   $self->{next_input_character} == 0x000B or # VT  
                   $self->{next_input_character} == 0x000C or # FF  
                   $self->{next_input_character} == 0x0020 or # SP  
                   $self->{next_input_character} == 0x003E or # >  
                   $self->{next_input_character} == 0x002F or # /  
                   $self->{next_input_character} == -1) {  
             $self->{next_input_character} = shift @next_char; # reconsume  
             !!!back-next-input-character (@next_char);  
             $self->{state} = 'data';  
             !!!emit ({type => 'character', data => '</'});  
             redo A;  
           } else {  
             $self->{next_input_character} = shift @next_char;  
             !!!back-next-input-character (@next_char);  
             # and consume...  
           }  
1286          } else {          } else {
1287            ## No start tag token has ever been emitted            ## No start tag token has ever been emitted
1288            # next-input-character is already done            ## NOTE: See <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>.
1289            $self->{state} = 'data';            !!!cp (28);
1290            !!!emit ({type => 'character', data => '</'});            $self->{state} = DATA_STATE;
1291              ## Reconsume.
1292              !!!emit ({type => CHARACTER_TOKEN, data => '</',
1293                        line => $l, column => $c,
1294                       });
1295            redo A;            redo A;
1296          }          }
1297        }        }
1298          
1299        if (0x0041 <= $self->{next_input_character} and        if (0x0041 <= $self->{nc} and
1300            $self->{next_input_character} <= 0x005A) { # A..Z            $self->{nc} <= 0x005A) { # A..Z
1301          $self->{current_token} = {type => 'end tag',          !!!cp (29);
1302                            tag_name => chr ($self->{next_input_character} + 0x0020)};          $self->{ct}
1303          $self->{state} = 'tag name';              = {type => END_TAG_TOKEN,
1304          !!!next-input-character;                 tag_name => chr ($self->{nc} + 0x0020),
1305          redo A;                 line => $l, column => $c};
1306        } elsif (0x0061 <= $self->{next_input_character} and          $self->{state} = TAG_NAME_STATE;
1307                 $self->{next_input_character} <= 0x007A) { # a..z          !!!next-input-character;
1308          $self->{current_token} = {type => 'end tag',          redo A;
1309                            tag_name => chr ($self->{next_input_character})};        } elsif (0x0061 <= $self->{nc} and
1310          $self->{state} = 'tag name';                 $self->{nc} <= 0x007A) { # a..z
1311          !!!next-input-character;          !!!cp (30);
1312          redo A;          $self->{ct} = {type => END_TAG_TOKEN,
1313        } elsif ($self->{next_input_character} == 0x003E) { # >                                    tag_name => chr ($self->{nc}),
1314          !!!parse-error (type => 'empty end tag');                                    line => $l, column => $c};
1315          $self->{state} = 'data';          $self->{state} = TAG_NAME_STATE;
1316            !!!next-input-character;
1317            redo A;
1318          } elsif ($self->{nc} == 0x003E) { # >
1319            !!!cp (31);
1320            !!!parse-error (type => 'empty end tag',
1321                            line => $self->{line_prev}, ## "<" in "</>"
1322                            column => $self->{column_prev} - 1);
1323            $self->{state} = DATA_STATE;
1324          !!!next-input-character;          !!!next-input-character;
1325          redo A;          redo A;
1326        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1327            !!!cp (32);
1328          !!!parse-error (type => 'bare etago');          !!!parse-error (type => 'bare etago');
1329          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1330          # reconsume          # reconsume
1331    
1332          !!!emit ({type => 'character', data => '</'});          !!!emit ({type => CHARACTER_TOKEN, data => '</',
1333                      line => $l, column => $c,
1334                     });
1335    
1336          redo A;          redo A;
1337        } else {        } else {
1338            !!!cp (33);
1339          !!!parse-error (type => 'bogus end tag');          !!!parse-error (type => 'bogus end tag');
1340          $self->{state} = 'bogus comment';          $self->{state} = BOGUS_COMMENT_STATE;
1341          ## $self->{next_input_character} is intentionally left as is          $self->{ct} = {type => COMMENT_TOKEN, data => '',
1342          redo A;                                    line => $self->{line_prev}, # "<" of "</"
1343                                      column => $self->{column_prev} - 1,
1344                                     };
1345            ## NOTE: $self->{nc} is intentionally left as is.
1346            ## Although the "anything else" case of the spec not explicitly
1347            ## states that the next input character is to be reconsumed,
1348            ## it will be included to the |data| of the comment token
1349            ## generated from the bogus end tag, as defined in the
1350            ## "bogus comment state" entry.
1351            redo A;
1352          }
1353        } elsif ($self->{state} == CDATA_RCDATA_CLOSE_TAG_STATE) {
1354          my $ch = substr $self->{last_stag_name}, length $self->{s_kwd}, 1;
1355          if (length $ch) {
1356            my $CH = $ch;
1357            $ch =~ tr/a-z/A-Z/;
1358            my $nch = chr $self->{nc};
1359            if ($nch eq $ch or $nch eq $CH) {
1360              !!!cp (24);
1361              ## Stay in the state.
1362              $self->{s_kwd} .= $nch;
1363              !!!next-input-character;
1364              redo A;
1365            } else {
1366              !!!cp (25);
1367              $self->{state} = DATA_STATE;
1368              ## Reconsume.
1369              !!!emit ({type => CHARACTER_TOKEN,
1370                        data => '</' . $self->{s_kwd},
1371                        line => $self->{line_prev},
1372                        column => $self->{column_prev} - 1 - length $self->{s_kwd},
1373                       });
1374              redo A;
1375            }
1376          } else { # after "<{tag-name}"
1377            unless ($is_space->{$self->{nc}} or
1378                    {
1379                     0x003E => 1, # >
1380                     0x002F => 1, # /
1381                     -1 => 1, # EOF
1382                    }->{$self->{nc}}) {
1383              !!!cp (26);
1384              ## Reconsume.
1385              $self->{state} = DATA_STATE;
1386              !!!emit ({type => CHARACTER_TOKEN,
1387                        data => '</' . $self->{s_kwd},
1388                        line => $self->{line_prev},
1389                        column => $self->{column_prev} - 1 - length $self->{s_kwd},
1390                       });
1391              redo A;
1392            } else {
1393              !!!cp (27);
1394              $self->{ct}
1395                  = {type => END_TAG_TOKEN,
1396                     tag_name => $self->{last_stag_name},
1397                     line => $self->{line_prev},
1398                     column => $self->{column_prev} - 1 - length $self->{s_kwd}};
1399              $self->{state} = TAG_NAME_STATE;
1400              ## Reconsume.
1401              redo A;
1402            }
1403        }        }
1404      } elsif ($self->{state} eq 'tag name') {      } elsif ($self->{state} == TAG_NAME_STATE) {
1405        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
1406            $self->{next_input_character} == 0x000A or # LF          !!!cp (34);
1407            $self->{next_input_character} == 0x000B or # VT          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1408            $self->{next_input_character} == 0x000C or # FF          !!!next-input-character;
1409            $self->{next_input_character} == 0x0020) { # SP          redo A;
1410          $self->{state} = 'before attribute name';        } elsif ($self->{nc} == 0x003E) { # >
1411          !!!next-input-character;          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1412          redo A;            !!!cp (35);
1413        } elsif ($self->{next_input_character} == 0x003E) { # >            $self->{last_stag_name} = $self->{ct}->{tag_name};
1414          if ($self->{current_token}->{type} eq 'start tag') {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
           $self->{current_token}->{first_start_tag}  
               = not defined $self->{last_emitted_start_tag_name};  
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} eq 'end tag') {  
1415            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1416            if ($self->{current_token}->{attributes}) {            #if ($self->{ct}->{attributes}) {
1417              !!!parse-error (type => 'end tag attribute');            #  ## NOTE: This should never be reached.
1418            }            #  !!! cp (36);
1419              #  !!! parse-error (type => 'end tag attribute');
1420              #} else {
1421                !!!cp (37);
1422              #}
1423          } else {          } else {
1424            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1425          }          }
1426          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1427          !!!next-input-character;          !!!next-input-character;
1428    
1429          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1430    
1431          redo A;          redo A;
1432        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{nc} and
1433                 $self->{next_input_character} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1434          $self->{current_token}->{tag_name} .= chr ($self->{next_input_character} + 0x0020);          !!!cp (38);
1435            $self->{ct}->{tag_name} .= chr ($self->{nc} + 0x0020);
1436            # start tag or end tag            # start tag or end tag
1437          ## Stay in this state          ## Stay in this state
1438          !!!next-input-character;          !!!next-input-character;
1439          redo A;          redo A;
1440        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1441          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1442          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1443            $self->{current_token}->{first_start_tag}            !!!cp (39);
1444                = not defined $self->{last_emitted_start_tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1445            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
         } elsif ($self->{current_token}->{type} eq 'end tag') {  
1446            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1447            if ($self->{current_token}->{attributes}) {            #if ($self->{ct}->{attributes}) {
1448              !!!parse-error (type => 'end tag attribute');            #  ## NOTE: This state should never be reached.
1449            }            #  !!! cp (40);
1450              #  !!! parse-error (type => 'end tag attribute');
1451              #} else {
1452                !!!cp (41);
1453              #}
1454          } else {          } else {
1455            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1456          }          }
1457          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1458          # reconsume          # reconsume
1459    
1460          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1461    
1462          redo A;          redo A;
1463        } elsif ($self->{next_input_character} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1464            !!!cp (42);
1465            $self->{state} = SELF_CLOSING_START_TAG_STATE;
1466          !!!next-input-character;          !!!next-input-character;
         if ($self->{next_input_character} == 0x003E and # >  
             $self->{current_token}->{type} eq 'start tag' and  
             $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {  
           # permitted slash  
           #  
         } else {  
           !!!parse-error (type => 'nestc');  
         }  
         $self->{state} = 'before attribute name';  
         # next-input-character is already done  
1467          redo A;          redo A;
1468        } else {        } else {
1469          $self->{current_token}->{tag_name} .= chr $self->{next_input_character};          !!!cp (44);
1470            $self->{ct}->{tag_name} .= chr $self->{nc};
1471            # start tag or end tag            # start tag or end tag
1472          ## Stay in the state          ## Stay in the state
1473          !!!next-input-character;          !!!next-input-character;
1474          redo A;          redo A;
1475        }        }
1476      } elsif ($self->{state} eq 'before attribute name') {      } elsif ($self->{state} == BEFORE_ATTRIBUTE_NAME_STATE) {
1477        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
1478            $self->{next_input_character} == 0x000A or # LF          !!!cp (45);
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
1479          ## Stay in the state          ## Stay in the state
1480          !!!next-input-character;          !!!next-input-character;
1481          redo A;          redo A;
1482        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1483          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1484            $self->{current_token}->{first_start_tag}            !!!cp (46);
1485                = not defined $self->{last_emitted_start_tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1486            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
         } elsif ($self->{current_token}->{type} eq 'end tag') {  
1487            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1488            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1489                !!!cp (47);
1490              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1491              } else {
1492                !!!cp (48);
1493            }            }
1494          } else {          } else {
1495            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1496          }          }
1497          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1498          !!!next-input-character;          !!!next-input-character;
1499    
1500          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1501    
1502          redo A;          redo A;
1503        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{nc} and
1504                 $self->{next_input_character} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1505          $self->{current_attribute} = {name => chr ($self->{next_input_character} + 0x0020),          !!!cp (49);
1506                                value => ''};          $self->{ca}
1507          $self->{state} = 'attribute name';              = {name => chr ($self->{nc} + 0x0020),
1508                   value => '',
1509                   line => $self->{line}, column => $self->{column}};
1510            $self->{state} = ATTRIBUTE_NAME_STATE;
1511          !!!next-input-character;          !!!next-input-character;
1512          redo A;          redo A;
1513        } elsif ($self->{next_input_character} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1514            !!!cp (50);
1515            $self->{state} = SELF_CLOSING_START_TAG_STATE;
1516          !!!next-input-character;          !!!next-input-character;
         if ($self->{next_input_character} == 0x003E and # >  
             $self->{current_token}->{type} eq 'start tag' and  
             $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {  
           # permitted slash  
           #  
         } else {  
           !!!parse-error (type => 'nestc');  
         }  
         ## Stay in the state  
         # next-input-character is already done  
1517          redo A;          redo A;
1518        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1519          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1520          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1521            $self->{current_token}->{first_start_tag}            !!!cp (52);
1522                = not defined $self->{last_emitted_start_tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1523            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
         } elsif ($self->{current_token}->{type} eq 'end tag') {  
1524            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1525            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1526                !!!cp (53);
1527              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1528              } else {
1529                !!!cp (54);
1530            }            }
1531          } else {          } else {
1532            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1533          }          }
1534          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1535          # reconsume          # reconsume
1536    
1537          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1538    
1539          redo A;          redo A;
1540        } else {        } else {
1541          $self->{current_attribute} = {name => chr ($self->{next_input_character}),          if ({
1542                                value => ''};               0x0022 => 1, # "
1543          $self->{state} = 'attribute name';               0x0027 => 1, # '
1544                 0x003D => 1, # =
1545                }->{$self->{nc}}) {
1546              !!!cp (55);
1547              !!!parse-error (type => 'bad attribute name');
1548            } else {
1549              !!!cp (56);
1550            }
1551            $self->{ca}
1552                = {name => chr ($self->{nc}),
1553                   value => '',
1554                   line => $self->{line}, column => $self->{column}};
1555            $self->{state} = ATTRIBUTE_NAME_STATE;
1556          !!!next-input-character;          !!!next-input-character;
1557          redo A;          redo A;
1558        }        }
1559      } elsif ($self->{state} eq 'attribute name') {      } elsif ($self->{state} == ATTRIBUTE_NAME_STATE) {
1560        my $before_leave = sub {        my $before_leave = sub {
1561          if (exists $self->{current_token}->{attributes} # start tag or end tag          if (exists $self->{ct}->{attributes} # start tag or end tag
1562              ->{$self->{current_attribute}->{name}}) { # MUST              ->{$self->{ca}->{name}}) { # MUST
1563            !!!parse-error (type => 'duplicate attribute:'.$self->{current_attribute}->{name});            !!!cp (57);
1564            ## Discard $self->{current_attribute} # MUST            !!!parse-error (type => 'duplicate attribute', text => $self->{ca}->{name}, line => $self->{ca}->{line}, column => $self->{ca}->{column});
1565          } else {            ## Discard $self->{ca} # MUST
1566            $self->{current_token}->{attributes}->{$self->{current_attribute}->{name}}          } else {
1567              = $self->{current_attribute};            !!!cp (58);
1568              $self->{ct}->{attributes}->{$self->{ca}->{name}}
1569                = $self->{ca};
1570          }          }
1571        }; # $before_leave        }; # $before_leave
1572    
1573        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
1574            $self->{next_input_character} == 0x000A or # LF          !!!cp (59);
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
1575          $before_leave->();          $before_leave->();
1576          $self->{state} = 'after attribute name';          $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;
1577          !!!next-input-character;          !!!next-input-character;
1578          redo A;          redo A;
1579        } elsif ($self->{next_input_character} == 0x003D) { # =        } elsif ($self->{nc} == 0x003D) { # =
1580            !!!cp (60);
1581          $before_leave->();          $before_leave->();
1582          $self->{state} = 'before attribute value';          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1583          !!!next-input-character;          !!!next-input-character;
1584          redo A;          redo A;
1585        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1586          $before_leave->();          $before_leave->();
1587          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1588            $self->{current_token}->{first_start_tag}            !!!cp (61);
1589                = not defined $self->{last_emitted_start_tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1590            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1591          } elsif ($self->{current_token}->{type} eq 'end tag') {            !!!cp (62);
1592            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1593            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1594              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1595            }            }
1596          } else {          } else {
1597            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1598          }          }
1599          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1600          !!!next-input-character;          !!!next-input-character;
1601    
1602          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1603    
1604          redo A;          redo A;
1605        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{nc} and
1606                 $self->{next_input_character} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1607          $self->{current_attribute}->{name} .= chr ($self->{next_input_character} + 0x0020);          !!!cp (63);
1608            $self->{ca}->{name} .= chr ($self->{nc} + 0x0020);
1609          ## Stay in the state          ## Stay in the state
1610          !!!next-input-character;          !!!next-input-character;
1611          redo A;          redo A;
1612        } elsif ($self->{next_input_character} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1613            !!!cp (64);
1614          $before_leave->();          $before_leave->();
1615            $self->{state} = SELF_CLOSING_START_TAG_STATE;
1616          !!!next-input-character;          !!!next-input-character;
         if ($self->{next_input_character} == 0x003E and # >  
             $self->{current_token}->{type} eq 'start tag' and  
             $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {  
           # permitted slash  
           #  
         } else {  
           !!!parse-error (type => 'nestc');  
         }  
         $self->{state} = 'before attribute name';  
         # next-input-character is already done  
1617          redo A;          redo A;
1618        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1619          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1620          $before_leave->();          $before_leave->();
1621          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1622            $self->{current_token}->{first_start_tag}            !!!cp (66);
1623                = not defined $self->{last_emitted_start_tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1624            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
         } elsif ($self->{current_token}->{type} eq 'end tag') {  
1625            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1626            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1627                !!!cp (67);
1628              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1629              } else {
1630                ## NOTE: This state should never be reached.
1631                !!!cp (68);
1632            }            }
1633          } else {          } else {
1634            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1635          }          }
1636          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1637          # reconsume          # reconsume
1638    
1639          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1640    
1641          redo A;          redo A;
1642        } else {        } else {
1643          $self->{current_attribute}->{name} .= chr ($self->{next_input_character});          if ($self->{nc} == 0x0022 or # "
1644                $self->{nc} == 0x0027) { # '
1645              !!!cp (69);
1646              !!!parse-error (type => 'bad attribute name');
1647            } else {
1648              !!!cp (70);
1649            }
1650            $self->{ca}->{name} .= chr ($self->{nc});
1651          ## Stay in the state          ## Stay in the state
1652          !!!next-input-character;          !!!next-input-character;
1653          redo A;          redo A;
1654        }        }
1655      } elsif ($self->{state} eq 'after attribute name') {      } elsif ($self->{state} == AFTER_ATTRIBUTE_NAME_STATE) {
1656        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
1657            $self->{next_input_character} == 0x000A or # LF          !!!cp (71);
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
1658          ## Stay in the state          ## Stay in the state
1659          !!!next-input-character;          !!!next-input-character;
1660          redo A;          redo A;
1661        } elsif ($self->{next_input_character} == 0x003D) { # =        } elsif ($self->{nc} == 0x003D) { # =
1662          $self->{state} = 'before attribute value';          !!!cp (72);
1663            $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1664          !!!next-input-character;          !!!next-input-character;
1665          redo A;          redo A;
1666        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1667          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1668            $self->{current_token}->{first_start_tag}            !!!cp (73);
1669                = not defined $self->{last_emitted_start_tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1670            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
         } elsif ($self->{current_token}->{type} eq 'end tag') {  
1671            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1672            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1673                !!!cp (74);
1674              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1675              } else {
1676                ## NOTE: This state should never be reached.
1677                !!!cp (75);
1678            }            }
1679          } else {          } else {
1680            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1681          }          }
1682          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1683          !!!next-input-character;          !!!next-input-character;
1684    
1685          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1686    
1687          redo A;          redo A;
1688        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{nc} and
1689                 $self->{next_input_character} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1690          $self->{current_attribute} = {name => chr ($self->{next_input_character} + 0x0020),          !!!cp (76);
1691                                value => ''};          $self->{ca}
1692          $self->{state} = 'attribute name';              = {name => chr ($self->{nc} + 0x0020),
1693                   value => '',
1694                   line => $self->{line}, column => $self->{column}};
1695            $self->{state} = ATTRIBUTE_NAME_STATE;
1696          !!!next-input-character;          !!!next-input-character;
1697          redo A;          redo A;
1698        } elsif ($self->{next_input_character} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1699            !!!cp (77);
1700            $self->{state} = SELF_CLOSING_START_TAG_STATE;
1701          !!!next-input-character;          !!!next-input-character;
         if ($self->{next_input_character} == 0x003E and # >  
             $self->{current_token}->{type} eq 'start tag' and  
             $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {  
           # permitted slash  
           #  
         } else {  
           !!!parse-error (type => 'nestc');  
           ## TODO: Different error type for <aa / bb> than <aa/>  
         }  
         $self->{state} = 'before attribute name';  
         # next-input-character is already done  
1702          redo A;          redo A;
1703        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1704          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1705          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1706            $self->{current_token}->{first_start_tag}            !!!cp (79);
1707                = not defined $self->{last_emitted_start_tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1708            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
         } elsif ($self->{current_token}->{type} eq 'end tag') {  
1709            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1710            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1711                !!!cp (80);
1712              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1713              } else {
1714                ## NOTE: This state should never be reached.
1715                !!!cp (81);
1716            }            }
1717          } else {          } else {
1718            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1719          }          }
1720          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1721          # reconsume          # reconsume
1722    
1723          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1724    
1725          redo A;          redo A;
1726        } else {        } else {
1727          $self->{current_attribute} = {name => chr ($self->{next_input_character}),          if ($self->{nc} == 0x0022 or # "
1728                                value => ''};              $self->{nc} == 0x0027) { # '
1729          $self->{state} = 'attribute name';            !!!cp (78);
1730              !!!parse-error (type => 'bad attribute name');
1731            } else {
1732              !!!cp (82);
1733            }
1734            $self->{ca}
1735                = {name => chr ($self->{nc}),
1736                   value => '',
1737                   line => $self->{line}, column => $self->{column}};
1738            $self->{state} = ATTRIBUTE_NAME_STATE;
1739          !!!next-input-character;          !!!next-input-character;
1740          redo A;                  redo A;        
1741        }        }
1742      } elsif ($self->{state} eq 'before attribute value') {      } elsif ($self->{state} == BEFORE_ATTRIBUTE_VALUE_STATE) {
1743        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
1744            $self->{next_input_character} == 0x000A or # LF          !!!cp (83);
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP        
1745          ## Stay in the state          ## Stay in the state
1746          !!!next-input-character;          !!!next-input-character;
1747          redo A;          redo A;
1748        } elsif ($self->{next_input_character} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
1749          $self->{state} = 'attribute value (double-quoted)';          !!!cp (84);
1750            $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;
1751          !!!next-input-character;          !!!next-input-character;
1752          redo A;          redo A;
1753        } elsif ($self->{next_input_character} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1754          $self->{state} = 'attribute value (unquoted)';          !!!cp (85);
1755            $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1756          ## reconsume          ## reconsume
1757          redo A;          redo A;
1758        } elsif ($self->{next_input_character} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
1759          $self->{state} = 'attribute value (single-quoted)';          !!!cp (86);
1760            $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;
1761          !!!next-input-character;          !!!next-input-character;
1762          redo A;          redo A;
1763        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1764          if ($self->{current_token}->{type} eq 'start tag') {          !!!parse-error (type => 'empty unquoted attribute value');
1765            $self->{current_token}->{first_start_tag}          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1766                = not defined $self->{last_emitted_start_tag_name};            !!!cp (87);
1767            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1768          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1769            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1770            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1771                !!!cp (88);
1772              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1773              } else {
1774                ## NOTE: This state should never be reached.
1775                !!!cp (89);
1776            }            }
1777          } else {          } else {
1778            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1779          }          }
1780          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1781          !!!next-input-character;          !!!next-input-character;
1782    
1783          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1784    
1785          redo A;          redo A;
1786        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1787          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1788          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1789            $self->{current_token}->{first_start_tag}            !!!cp (90);
1790                = not defined $self->{last_emitted_start_tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1791            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
         } elsif ($self->{current_token}->{type} eq 'end tag') {  
1792            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1793            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1794                !!!cp (91);
1795              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1796              } else {
1797                ## NOTE: This state should never be reached.
1798                !!!cp (92);
1799            }            }
1800          } else {          } else {
1801            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1802          }          }
1803          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1804          ## reconsume          ## reconsume
1805    
1806          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1807    
1808          redo A;          redo A;
1809        } else {        } else {
1810          $self->{current_attribute}->{value} .= chr ($self->{next_input_character});          if ($self->{nc} == 0x003D) { # =
1811          $self->{state} = 'attribute value (unquoted)';            !!!cp (93);
1812              !!!parse-error (type => 'bad attribute value');
1813            } else {
1814              !!!cp (94);
1815            }
1816            $self->{ca}->{value} .= chr ($self->{nc});
1817            $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1818          !!!next-input-character;          !!!next-input-character;
1819          redo A;          redo A;
1820        }        }
1821      } elsif ($self->{state} eq 'attribute value (double-quoted)') {      } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {
1822        if ($self->{next_input_character} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
1823          $self->{state} = 'before attribute name';          !!!cp (95);
1824            $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1825          !!!next-input-character;          !!!next-input-character;
1826          redo A;          redo A;
1827        } elsif ($self->{next_input_character} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1828          $self->{last_attribute_value_state} = 'attribute value (double-quoted)';          !!!cp (96);
1829          $self->{state} = 'entity in attribute value';          ## NOTE: In the spec, the tokenizer is switched to the
1830            ## "entity in attribute value state".  In this implementation, the
1831            ## tokenizer is switched to the |ENTITY_STATE|, which is an
1832            ## implementation of the "consume a character reference" algorithm.
1833            $self->{prev_state} = $self->{state};
1834            $self->{entity_add} = 0x0022; # "
1835            $self->{state} = ENTITY_STATE;
1836          !!!next-input-character;          !!!next-input-character;
1837          redo A;          redo A;
1838        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1839          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
1840          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1841            $self->{current_token}->{first_start_tag}            !!!cp (97);
1842                = not defined $self->{last_emitted_start_tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1843            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
         } elsif ($self->{current_token}->{type} eq 'end tag') {  
1844            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1845            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1846                !!!cp (98);
1847              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1848              } else {
1849                ## NOTE: This state should never be reached.
1850                !!!cp (99);
1851            }            }
1852          } else {          } else {
1853            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1854          }          }
1855          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1856          ## reconsume          ## reconsume
1857    
1858          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1859    
1860          redo A;          redo A;
1861        } else {        } else {
1862          $self->{current_attribute}->{value} .= chr ($self->{next_input_character});          !!!cp (100);
1863            $self->{ca}->{value} .= chr ($self->{nc});
1864            $self->{read_until}->($self->{ca}->{value},
1865                                  q["&],
1866                                  length $self->{ca}->{value});
1867    
1868          ## Stay in the state          ## Stay in the state
1869          !!!next-input-character;          !!!next-input-character;
1870          redo A;          redo A;
1871        }        }
1872      } elsif ($self->{state} eq 'attribute value (single-quoted)') {      } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {
1873        if ($self->{next_input_character} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
1874          $self->{state} = 'before attribute name';          !!!cp (101);
1875          !!!next-input-character;          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1876          redo A;          !!!next-input-character;
1877        } elsif ($self->{next_input_character} == 0x0026) { # &          redo A;
1878          $self->{last_attribute_value_state} = 'attribute value (single-quoted)';        } elsif ($self->{nc} == 0x0026) { # &
1879          $self->{state} = 'entity in attribute value';          !!!cp (102);
1880            ## NOTE: In the spec, the tokenizer is switched to the
1881            ## "entity in attribute value state".  In this implementation, the
1882            ## tokenizer is switched to the |ENTITY_STATE|, which is an
1883            ## implementation of the "consume a character reference" algorithm.
1884            $self->{entity_add} = 0x0027; # '
1885            $self->{prev_state} = $self->{state};
1886            $self->{state} = ENTITY_STATE;
1887          !!!next-input-character;          !!!next-input-character;
1888          redo A;          redo A;
1889        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1890          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
1891          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1892            $self->{current_token}->{first_start_tag}            !!!cp (103);
1893                = not defined $self->{last_emitted_start_tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1894            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
         } elsif ($self->{current_token}->{type} eq 'end tag') {  
1895            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1896            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1897                !!!cp (104);
1898              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1899              } else {
1900                ## NOTE: This state should never be reached.
1901                !!!cp (105);
1902            }            }
1903          } else {          } else {
1904            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1905          }          }
1906          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1907          ## reconsume          ## reconsume
1908    
1909          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1910    
1911          redo A;          redo A;
1912        } else {        } else {
1913          $self->{current_attribute}->{value} .= chr ($self->{next_input_character});          !!!cp (106);
1914            $self->{ca}->{value} .= chr ($self->{nc});
1915            $self->{read_until}->($self->{ca}->{value},
1916                                  q['&],
1917                                  length $self->{ca}->{value});
1918    
1919          ## Stay in the state          ## Stay in the state
1920          !!!next-input-character;          !!!next-input-character;
1921          redo A;          redo A;
1922        }        }
1923      } elsif ($self->{state} eq 'attribute value (unquoted)') {      } elsif ($self->{state} == ATTRIBUTE_VALUE_UNQUOTED_STATE) {
1924        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
1925            $self->{next_input_character} == 0x000A or # LF          !!!cp (107);
1926            $self->{next_input_character} == 0x000B or # HT          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1927            $self->{next_input_character} == 0x000C or # FF          !!!next-input-character;
1928            $self->{next_input_character} == 0x0020) { # SP          redo A;
1929          $self->{state} = 'before attribute name';        } elsif ($self->{nc} == 0x0026) { # &
1930          !!!next-input-character;          !!!cp (108);
1931          redo A;          ## NOTE: In the spec, the tokenizer is switched to the
1932        } elsif ($self->{next_input_character} == 0x0026) { # &          ## "entity in attribute value state".  In this implementation, the
1933          $self->{last_attribute_value_state} = 'attribute value (unquoted)';          ## tokenizer is switched to the |ENTITY_STATE|, which is an
1934          $self->{state} = 'entity in attribute value';          ## implementation of the "consume a character reference" algorithm.
1935          !!!next-input-character;          $self->{entity_add} = -1;
1936          redo A;          $self->{prev_state} = $self->{state};
1937        } elsif ($self->{next_input_character} == 0x003E) { # >          $self->{state} = ENTITY_STATE;
1938          if ($self->{current_token}->{type} eq 'start tag') {          !!!next-input-character;
1939            $self->{current_token}->{first_start_tag}          redo A;
1940                = not defined $self->{last_emitted_start_tag_name};        } elsif ($self->{nc} == 0x003E) { # >
1941            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1942          } elsif ($self->{current_token}->{type} eq 'end tag') {            !!!cp (109);
1943              $self->{last_stag_name} = $self->{ct}->{tag_name};
1944            } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1945            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1946            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1947                !!!cp (110);
1948              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1949              } else {
1950                ## NOTE: This state should never be reached.
1951                !!!cp (111);
1952            }            }
1953          } else {          } else {
1954            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1955          }          }
1956          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1957          !!!next-input-character;          !!!next-input-character;
1958    
1959          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1960    
1961          redo A;          redo A;
1962        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1963          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1964          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1965            $self->{current_token}->{first_start_tag}            !!!cp (112);
1966                = not defined $self->{last_emitted_start_tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1967            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
         } elsif ($self->{current_token}->{type} eq 'end tag') {  
1968            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1969            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1970                !!!cp (113);
1971              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1972              } else {
1973                ## NOTE: This state should never be reached.
1974                !!!cp (114);
1975            }            }
1976          } else {          } else {
1977            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1978          }          }
1979          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1980          ## reconsume          ## reconsume
1981    
1982          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1983    
1984          redo A;          redo A;
1985        } else {        } else {
1986          $self->{current_attribute}->{value} .= chr ($self->{next_input_character});          if ({
1987                 0x0022 => 1, # "
1988                 0x0027 => 1, # '
1989                 0x003D => 1, # =
1990                }->{$self->{nc}}) {
1991              !!!cp (115);
1992              !!!parse-error (type => 'bad attribute value');
1993            } else {
1994              !!!cp (116);
1995            }
1996            $self->{ca}->{value} .= chr ($self->{nc});
1997            $self->{read_until}->($self->{ca}->{value},
1998                                  q["'=& >],
1999                                  length $self->{ca}->{value});
2000    
2001          ## Stay in the state          ## Stay in the state
2002          !!!next-input-character;          !!!next-input-character;
2003          redo A;          redo A;
2004        }        }
2005      } elsif ($self->{state} eq 'entity in attribute value') {      } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {
2006        my $token = $self->_tokenize_attempt_to_consume_an_entity (1);        if ($is_space->{$self->{nc}}) {
2007            !!!cp (118);
2008            $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
2009            !!!next-input-character;
2010            redo A;
2011          } elsif ($self->{nc} == 0x003E) { # >
2012            if ($self->{ct}->{type} == START_TAG_TOKEN) {
2013              !!!cp (119);
2014              $self->{last_stag_name} = $self->{ct}->{tag_name};
2015            } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2016              $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
2017              if ($self->{ct}->{attributes}) {
2018                !!!cp (120);
2019                !!!parse-error (type => 'end tag attribute');
2020              } else {
2021                ## NOTE: This state should never be reached.
2022                !!!cp (121);
2023              }
2024            } else {
2025              die "$0: $self->{ct}->{type}: Unknown token type";
2026            }
2027            $self->{state} = DATA_STATE;
2028            !!!next-input-character;
2029    
2030            !!!emit ($self->{ct}); # start tag or end tag
2031    
2032        unless (defined $token) {          redo A;
2033          $self->{current_attribute}->{value} .= '&';        } elsif ($self->{nc} == 0x002F) { # /
2034            !!!cp (122);
2035            $self->{state} = SELF_CLOSING_START_TAG_STATE;
2036            !!!next-input-character;
2037            redo A;
2038          } elsif ($self->{nc} == -1) {
2039            !!!parse-error (type => 'unclosed tag');
2040            if ($self->{ct}->{type} == START_TAG_TOKEN) {
2041              !!!cp (122.3);
2042              $self->{last_stag_name} = $self->{ct}->{tag_name};
2043            } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2044              if ($self->{ct}->{attributes}) {
2045                !!!cp (122.1);
2046                !!!parse-error (type => 'end tag attribute');
2047              } else {
2048                ## NOTE: This state should never be reached.
2049                !!!cp (122.2);
2050              }
2051            } else {
2052              die "$0: $self->{ct}->{type}: Unknown token type";
2053            }
2054            $self->{state} = DATA_STATE;
2055            ## Reconsume.
2056            !!!emit ($self->{ct}); # start tag or end tag
2057            redo A;
2058        } else {        } else {
2059          $self->{current_attribute}->{value} .= $token->{data};          !!!cp ('124.1');
2060          ## ISSUE: spec says "append the returned character token to the current attribute's value"          !!!parse-error (type => 'no space between attributes');
2061            $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
2062            ## reconsume
2063            redo A;
2064        }        }
2065        } elsif ($self->{state} == SELF_CLOSING_START_TAG_STATE) {
2066          if ($self->{nc} == 0x003E) { # >
2067            if ($self->{ct}->{type} == END_TAG_TOKEN) {
2068              !!!cp ('124.2');
2069              !!!parse-error (type => 'nestc', token => $self->{ct});
2070              ## TODO: Different type than slash in start tag
2071              $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
2072              if ($self->{ct}->{attributes}) {
2073                !!!cp ('124.4');
2074                !!!parse-error (type => 'end tag attribute');
2075              } else {
2076                !!!cp ('124.5');
2077              }
2078              ## TODO: Test |<title></title/>|
2079            } else {
2080              !!!cp ('124.3');
2081              $self->{self_closing} = 1;
2082            }
2083    
2084        $self->{state} = $self->{last_attribute_value_state};          $self->{state} = DATA_STATE;
2085        # next-input-character is already done          !!!next-input-character;
       redo A;  
     } elsif ($self->{state} eq 'bogus comment') {  
       ## (only happen if PCDATA state)  
         
       my $token = {type => 'comment', data => ''};  
   
       BC: {  
         if ($self->{next_input_character} == 0x003E) { # >  
           $self->{state} = 'data';  
           !!!next-input-character;  
   
           !!!emit ($token);  
   
           redo A;  
         } elsif ($self->{next_input_character} == -1) {  
           $self->{state} = 'data';  
           ## reconsume  
2086    
2087            !!!emit ($token);          !!!emit ($self->{ct}); # start tag or end tag
2088    
2089            redo A;          redo A;
2090          } elsif ($self->{nc} == -1) {
2091            !!!parse-error (type => 'unclosed tag');
2092            if ($self->{ct}->{type} == START_TAG_TOKEN) {
2093              !!!cp (124.7);
2094              $self->{last_stag_name} = $self->{ct}->{tag_name};
2095            } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2096              if ($self->{ct}->{attributes}) {
2097                !!!cp (124.5);
2098                !!!parse-error (type => 'end tag attribute');
2099              } else {
2100                ## NOTE: This state should never be reached.
2101                !!!cp (124.6);
2102              }
2103          } else {          } else {
2104            $token->{data} .= chr ($self->{next_input_character});            die "$0: $self->{ct}->{type}: Unknown token type";
           !!!next-input-character;  
           redo BC;  
2105          }          }
2106        } # BC          $self->{state} = DATA_STATE;
2107      } elsif ($self->{state} eq 'markup declaration open') {          ## Reconsume.
2108            !!!emit ($self->{ct}); # start tag or end tag
2109            redo A;
2110          } else {
2111            !!!cp ('124.4');
2112            !!!parse-error (type => 'nestc');
2113            ## TODO: This error type is wrong.
2114            $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
2115            ## Reconsume.
2116            redo A;
2117          }
2118        } elsif ($self->{state} == BOGUS_COMMENT_STATE) {
2119        ## (only happen if PCDATA state)        ## (only happen if PCDATA state)
2120    
2121        my @next_char;        ## NOTE: Unlike spec's "bogus comment state", this implementation
2122        push @next_char, $self->{next_input_character};        ## consumes characters one-by-one basis.
2123                
2124        if ($self->{next_input_character} == 0x002D) { # -        if ($self->{nc} == 0x003E) { # >
2125            !!!cp (124);
2126            $self->{state} = DATA_STATE;
2127          !!!next-input-character;          !!!next-input-character;
2128          push @next_char, $self->{next_input_character};  
2129          if ($self->{next_input_character} == 0x002D) { # -          !!!emit ($self->{ct}); # comment
2130            $self->{current_token} = {type => 'comment', data => ''};          redo A;
2131            $self->{state} = 'comment start';        } elsif ($self->{nc} == -1) {
2132            !!!next-input-character;          !!!cp (125);
2133            redo A;          $self->{state} = DATA_STATE;
2134          }          ## reconsume
2135        } elsif ($self->{next_input_character} == 0x0044 or # D  
2136                 $self->{next_input_character} == 0x0064) { # d          !!!emit ($self->{ct}); # comment
2137            redo A;
2138          } else {
2139            !!!cp (126);
2140            $self->{ct}->{data} .= chr ($self->{nc}); # comment
2141            $self->{read_until}->($self->{ct}->{data},
2142                                  q[>],
2143                                  length $self->{ct}->{data});
2144    
2145            ## Stay in the state.
2146          !!!next-input-character;          !!!next-input-character;
2147          push @next_char, $self->{next_input_character};          redo A;
2148          if ($self->{next_input_character} == 0x004F or # O        }
2149              $self->{next_input_character} == 0x006F) { # o      } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {
2150            !!!next-input-character;        ## (only happen if PCDATA state)
2151            push @next_char, $self->{next_input_character};        
2152            if ($self->{next_input_character} == 0x0043 or # C        if ($self->{nc} == 0x002D) { # -
2153                $self->{next_input_character} == 0x0063) { # c          !!!cp (133);
2154              !!!next-input-character;          $self->{state} = MD_HYPHEN_STATE;
2155              push @next_char, $self->{next_input_character};          !!!next-input-character;
2156              if ($self->{next_input_character} == 0x0054 or # T          redo A;
2157                  $self->{next_input_character} == 0x0074) { # t        } elsif ($self->{nc} == 0x0044 or # D
2158                !!!next-input-character;                 $self->{nc} == 0x0064) { # d
2159                push @next_char, $self->{next_input_character};          ## ASCII case-insensitive.
2160                if ($self->{next_input_character} == 0x0059 or # Y          !!!cp (130);
2161                    $self->{next_input_character} == 0x0079) { # y          $self->{state} = MD_DOCTYPE_STATE;
2162                  !!!next-input-character;          $self->{s_kwd} = chr $self->{nc};
2163                  push @next_char, $self->{next_input_character};          !!!next-input-character;
2164                  if ($self->{next_input_character} == 0x0050 or # P          redo A;
2165                      $self->{next_input_character} == 0x0070) { # p        } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and
2166                    !!!next-input-character;                 $self->{open_elements}->[-1]->[1] & FOREIGN_EL and
2167                    push @next_char, $self->{next_input_character};                 $self->{nc} == 0x005B) { # [
2168                    if ($self->{next_input_character} == 0x0045 or # E          !!!cp (135.4);                
2169                        $self->{next_input_character} == 0x0065) { # e          $self->{state} = MD_CDATA_STATE;
2170                      ## ISSUE: What a stupid code this is!          $self->{s_kwd} = '[';
2171                      $self->{state} = 'DOCTYPE';          !!!next-input-character;
2172                      !!!next-input-character;          redo A;
2173                      redo A;        } else {
2174                    }          !!!cp (136);
                 }  
               }  
             }  
           }  
         }  
2175        }        }
2176    
2177        !!!parse-error (type => 'bogus comment');        !!!parse-error (type => 'bogus comment',
2178        $self->{next_input_character} = shift @next_char;                        line => $self->{line_prev},
2179        !!!back-next-input-character (@next_char);                        column => $self->{column_prev} - 1);
2180        $self->{state} = 'bogus comment';        ## Reconsume.
2181          $self->{state} = BOGUS_COMMENT_STATE;
2182          $self->{ct} = {type => COMMENT_TOKEN, data => '',
2183                                    line => $self->{line_prev},
2184                                    column => $self->{column_prev} - 1,
2185                                   };
2186        redo A;        redo A;
2187              } elsif ($self->{state} == MD_HYPHEN_STATE) {
2188        ## ISSUE: typos in spec: chacacters, is is a parse error        if ($self->{nc} == 0x002D) { # -
2189        ## ISSUE: spec is somewhat unclear on "is the first character that will be in the comment"; what is "that will be in the comment" is what the algorithm defines, isn't it?          !!!cp (127);
2190      } elsif ($self->{state} eq 'comment start') {          $self->{ct} = {type => COMMENT_TOKEN, data => '',
2191        if ($self->{next_input_character} == 0x002D) { # -                                    line => $self->{line_prev},
2192          $self->{state} = 'comment start dash';                                    column => $self->{column_prev} - 2,
2193                                     };
2194            $self->{state} = COMMENT_START_STATE;
2195            !!!next-input-character;
2196            redo A;
2197          } else {
2198            !!!cp (128);
2199            !!!parse-error (type => 'bogus comment',
2200                            line => $self->{line_prev},
2201                            column => $self->{column_prev} - 2);
2202            $self->{state} = BOGUS_COMMENT_STATE;
2203            ## Reconsume.
2204            $self->{ct} = {type => COMMENT_TOKEN,
2205                                      data => '-',
2206                                      line => $self->{line_prev},
2207                                      column => $self->{column_prev} - 2,
2208                                     };
2209            redo A;
2210          }
2211        } elsif ($self->{state} == MD_DOCTYPE_STATE) {
2212          ## ASCII case-insensitive.
2213          if ($self->{nc} == [
2214                undef,
2215                0x004F, # O
2216                0x0043, # C
2217                0x0054, # T
2218                0x0059, # Y
2219                0x0050, # P
2220              ]->[length $self->{s_kwd}] or
2221              $self->{nc} == [
2222                undef,
2223                0x006F, # o
2224                0x0063, # c
2225                0x0074, # t
2226                0x0079, # y
2227                0x0070, # p
2228              ]->[length $self->{s_kwd}]) {
2229            !!!cp (131);
2230            ## Stay in the state.
2231            $self->{s_kwd} .= chr $self->{nc};
2232            !!!next-input-character;
2233            redo A;
2234          } elsif ((length $self->{s_kwd}) == 6 and
2235                   ($self->{nc} == 0x0045 or # E
2236                    $self->{nc} == 0x0065)) { # e
2237            !!!cp (129);
2238            $self->{state} = DOCTYPE_STATE;
2239            $self->{ct} = {type => DOCTYPE_TOKEN,
2240                                      quirks => 1,
2241                                      line => $self->{line_prev},
2242                                      column => $self->{column_prev} - 7,
2243                                     };
2244            !!!next-input-character;
2245            redo A;
2246          } else {
2247            !!!cp (132);        
2248            !!!parse-error (type => 'bogus comment',
2249                            line => $self->{line_prev},
2250                            column => $self->{column_prev} - 1 - length $self->{s_kwd});
2251            $self->{state} = BOGUS_COMMENT_STATE;
2252            ## Reconsume.
2253            $self->{ct} = {type => COMMENT_TOKEN,
2254                                      data => $self->{s_kwd},
2255                                      line => $self->{line_prev},
2256                                      column => $self->{column_prev} - 1 - length $self->{s_kwd},
2257                                     };
2258            redo A;
2259          }
2260        } elsif ($self->{state} == MD_CDATA_STATE) {
2261          if ($self->{nc} == {
2262                '[' => 0x0043, # C
2263                '[C' => 0x0044, # D
2264                '[CD' => 0x0041, # A
2265                '[CDA' => 0x0054, # T
2266                '[CDAT' => 0x0041, # A
2267              }->{$self->{s_kwd}}) {
2268            !!!cp (135.1);
2269            ## Stay in the state.
2270            $self->{s_kwd} .= chr $self->{nc};
2271            !!!next-input-character;
2272            redo A;
2273          } elsif ($self->{s_kwd} eq '[CDATA' and
2274                   $self->{nc} == 0x005B) { # [
2275            !!!cp (135.2);
2276            $self->{ct} = {type => CHARACTER_TOKEN,
2277                                      data => '',
2278                                      line => $self->{line_prev},
2279                                      column => $self->{column_prev} - 7};
2280            $self->{state} = CDATA_SECTION_STATE;
2281            !!!next-input-character;
2282            redo A;
2283          } else {
2284            !!!cp (135.3);
2285            !!!parse-error (type => 'bogus comment',
2286                            line => $self->{line_prev},
2287                            column => $self->{column_prev} - 1 - length $self->{s_kwd});
2288            $self->{state} = BOGUS_COMMENT_STATE;
2289            ## Reconsume.
2290            $self->{ct} = {type => COMMENT_TOKEN,
2291                                      data => $self->{s_kwd},
2292                                      line => $self->{line_prev},
2293                                      column => $self->{column_prev} - 1 - length $self->{s_kwd},
2294                                     };
2295            redo A;
2296          }
2297        } elsif ($self->{state} == COMMENT_START_STATE) {
2298          if ($self->{nc} == 0x002D) { # -
2299            !!!cp (137);
2300            $self->{state} = COMMENT_START_DASH_STATE;
2301          !!!next-input-character;          !!!next-input-character;
2302          redo A;          redo A;
2303        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2304            !!!cp (138);
2305          !!!parse-error (type => 'bogus comment');          !!!parse-error (type => 'bogus comment');
2306          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2307          !!!next-input-character;          !!!next-input-character;
2308    
2309          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2310    
2311          redo A;          redo A;
2312        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2313            !!!cp (139);
2314          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2315          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2316          ## reconsume          ## reconsume
2317    
2318          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2319    
2320          redo A;          redo A;
2321        } else {        } else {
2322          $self->{current_token}->{data} # comment          !!!cp (140);
2323              .= chr ($self->{next_input_character});          $self->{ct}->{data} # comment
2324          $self->{state} = 'comment';              .= chr ($self->{nc});
2325            $self->{state} = COMMENT_STATE;
2326          !!!next-input-character;          !!!next-input-character;
2327          redo A;          redo A;
2328        }        }
2329      } elsif ($self->{state} eq 'comment start dash') {      } elsif ($self->{state} == COMMENT_START_DASH_STATE) {
2330        if ($self->{next_input_character} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2331          $self->{state} = 'comment end';          !!!cp (141);
2332            $self->{state} = COMMENT_END_STATE;
2333          !!!next-input-character;          !!!next-input-character;
2334          redo A;          redo A;
2335        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2336            !!!cp (142);
2337          !!!parse-error (type => 'bogus comment');          !!!parse-error (type => 'bogus comment');
2338          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2339          !!!next-input-character;          !!!next-input-character;
2340    
2341          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2342    
2343          redo A;          redo A;
2344        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2345            !!!cp (143);
2346          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2347          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2348          ## reconsume          ## reconsume
2349    
2350          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2351    
2352          redo A;          redo A;
2353        } else {        } else {
2354          $self->{current_token}->{data} # comment          !!!cp (144);
2355              .= '-' . chr ($self->{next_input_character});          $self->{ct}->{data} # comment
2356          $self->{state} = 'comment';              .= '-' . chr ($self->{nc});
2357            $self->{state} = COMMENT_STATE;
2358          !!!next-input-character;          !!!next-input-character;
2359          redo A;          redo A;
2360        }        }
2361      } elsif ($self->{state} eq 'comment') {      } elsif ($self->{state} == COMMENT_STATE) {
2362        if ($self->{next_input_character} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2363          $self->{state} = 'comment end dash';          !!!cp (145);
2364            $self->{state} = COMMENT_END_DASH_STATE;
2365          !!!next-input-character;          !!!next-input-character;
2366          redo A;          redo A;
2367        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2368            !!!cp (146);
2369          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2370          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2371          ## reconsume          ## reconsume
2372    
2373          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2374    
2375          redo A;          redo A;
2376        } else {        } else {
2377          $self->{current_token}->{data} .= chr ($self->{next_input_character}); # comment          !!!cp (147);
2378            $self->{ct}->{data} .= chr ($self->{nc}); # comment
2379            $self->{read_until}->($self->{ct}->{data},
2380                                  q[-],
2381                                  length $self->{ct}->{data});
2382    
2383          ## Stay in the state          ## Stay in the state
2384          !!!next-input-character;          !!!next-input-character;
2385          redo A;          redo A;
2386        }        }
2387      } elsif ($self->{state} eq 'comment end dash') {      } elsif ($self->{state} == COMMENT_END_DASH_STATE) {
2388        if ($self->{next_input_character} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2389          $self->{state} = 'comment end';          !!!cp (148);
2390            $self->{state} = COMMENT_END_STATE;
2391          !!!next-input-character;          !!!next-input-character;
2392          redo A;          redo A;
2393        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2394            !!!cp (149);
2395          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2396          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2397          ## reconsume          ## reconsume
2398    
2399          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2400    
2401          redo A;          redo A;
2402        } else {        } else {
2403          $self->{current_token}->{data} .= '-' . chr ($self->{next_input_character}); # comment          !!!cp (150);
2404          $self->{state} = 'comment';          $self->{ct}->{data} .= '-' . chr ($self->{nc}); # comment
2405            $self->{state} = COMMENT_STATE;
2406          !!!next-input-character;          !!!next-input-character;
2407          redo A;          redo A;
2408        }        }
2409      } elsif ($self->{state} eq 'comment end') {      } elsif ($self->{state} == COMMENT_END_STATE) {
2410        if ($self->{next_input_character} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2411          $self->{state} = 'data';          !!!cp (151);
2412            $self->{state} = DATA_STATE;
2413          !!!next-input-character;          !!!next-input-character;
2414    
2415          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2416    
2417          redo A;          redo A;
2418        } elsif ($self->{next_input_character} == 0x002D) { # -        } elsif ($self->{nc} == 0x002D) { # -
2419          !!!parse-error (type => 'dash in comment');          !!!cp (152);
2420          $self->{current_token}->{data} .= '-'; # comment          !!!parse-error (type => 'dash in comment',
2421                            line => $self->{line_prev},
2422                            column => $self->{column_prev});
2423            $self->{ct}->{data} .= '-'; # comment
2424          ## Stay in the state          ## Stay in the state
2425          !!!next-input-character;          !!!next-input-character;
2426          redo A;          redo A;
2427        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2428            !!!cp (153);
2429          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2430          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2431          ## reconsume          ## reconsume
2432    
2433          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2434    
2435          redo A;          redo A;
2436        } else {        } else {
2437          !!!parse-error (type => 'dash in comment');          !!!cp (154);
2438          $self->{current_token}->{data} .= '--' . chr ($self->{next_input_character}); # comment          !!!parse-error (type => 'dash in comment',
2439          $self->{state} = 'comment';                          line => $self->{line_prev},
2440                            column => $self->{column_prev});
2441            $self->{ct}->{data} .= '--' . chr ($self->{nc}); # comment
2442            $self->{state} = COMMENT_STATE;
2443          !!!next-input-character;          !!!next-input-character;
2444          redo A;          redo A;
2445        }        }
2446      } elsif ($self->{state} eq 'DOCTYPE') {      } elsif ($self->{state} == DOCTYPE_STATE) {
2447        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
2448            $self->{next_input_character} == 0x000A or # LF          !!!cp (155);
2449            $self->{next_input_character} == 0x000B or # VT          $self->{state} = BEFORE_DOCTYPE_NAME_STATE;
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
         $self->{state} = 'before DOCTYPE name';  
2450          !!!next-input-character;          !!!next-input-character;
2451          redo A;          redo A;
2452        } else {        } else {
2453            !!!cp (156);
2454          !!!parse-error (type => 'no space before DOCTYPE name');          !!!parse-error (type => 'no space before DOCTYPE name');
2455          $self->{state} = 'before DOCTYPE name';          $self->{state} = BEFORE_DOCTYPE_NAME_STATE;
2456          ## reconsume          ## reconsume
2457          redo A;          redo A;
2458        }        }
2459      } elsif ($self->{state} eq 'before DOCTYPE name') {      } elsif ($self->{state} == BEFORE_DOCTYPE_NAME_STATE) {
2460        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
2461            $self->{next_input_character} == 0x000A or # LF          !!!cp (157);
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
2462          ## Stay in the state          ## Stay in the state
2463          !!!next-input-character;          !!!next-input-character;
2464          redo A;          redo A;
2465        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2466            !!!cp (158);
2467          !!!parse-error (type => 'no DOCTYPE name');          !!!parse-error (type => 'no DOCTYPE name');
2468          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2469          !!!next-input-character;          !!!next-input-character;
2470    
2471          !!!emit ({type => 'DOCTYPE'}); # incorrect          !!!emit ($self->{ct}); # DOCTYPE (quirks)
2472    
2473          redo A;          redo A;
2474        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2475            !!!cp (159);
2476          !!!parse-error (type => 'no DOCTYPE name');          !!!parse-error (type => 'no DOCTYPE name');
2477          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2478          ## reconsume          ## reconsume
2479    
2480          !!!emit ({type => 'DOCTYPE'}); # incorrect          !!!emit ($self->{ct}); # DOCTYPE (quirks)
2481    
2482          redo A;          redo A;
2483        } else {        } else {
2484          $self->{current_token}          !!!cp (160);
2485              = {type => 'DOCTYPE',          $self->{ct}->{name} = chr $self->{nc};
2486                 name => chr ($self->{next_input_character}),          delete $self->{ct}->{quirks};
2487                 correct => 1};          $self->{state} = DOCTYPE_NAME_STATE;
 ## ISSUE: "Set the token's name name to the" in the spec  
         $self->{state} = 'DOCTYPE name';  
2488          !!!next-input-character;          !!!next-input-character;
2489          redo A;          redo A;
2490        }        }
2491      } elsif ($self->{state} eq 'DOCTYPE name') {      } elsif ($self->{state} == DOCTYPE_NAME_STATE) {
2492  ## ISSUE: Redundant "First," in the spec.  ## ISSUE: Redundant "First," in the spec.
2493        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
2494            $self->{next_input_character} == 0x000A or # LF          !!!cp (161);
2495            $self->{next_input_character} == 0x000B or # VT          $self->{state} = AFTER_DOCTYPE_NAME_STATE;
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
         $self->{state} = 'after DOCTYPE name';  
2496          !!!next-input-character;          !!!next-input-character;
2497          redo A;          redo A;
2498        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2499          $self->{state} = 'data';          !!!cp (162);
2500            $self->{state} = DATA_STATE;
2501          !!!next-input-character;          !!!next-input-character;
2502    
2503          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2504    
2505          redo A;          redo A;
2506        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2507            !!!cp (163);
2508          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2509          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2510          ## reconsume          ## reconsume
2511    
2512          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2513          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2514    
2515          redo A;          redo A;
2516        } else {        } else {
2517          $self->{current_token}->{name}          !!!cp (164);
2518            .= chr ($self->{next_input_character}); # DOCTYPE          $self->{ct}->{name}
2519              .= chr ($self->{nc}); # DOCTYPE
2520          ## Stay in the state          ## Stay in the state
2521          !!!next-input-character;          !!!next-input-character;
2522          redo A;          redo A;
2523        }        }
2524      } elsif ($self->{state} eq 'after DOCTYPE name') {      } elsif ($self->{state} == AFTER_DOCTYPE_NAME_STATE) {
2525        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
2526            $self->{next_input_character} == 0x000A or # LF          !!!cp (165);
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
2527          ## Stay in the state          ## Stay in the state
2528          !!!next-input-character;          !!!next-input-character;
2529          redo A;          redo A;
2530        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2531          $self->{state} = 'data';          !!!cp (166);
2532            $self->{state} = DATA_STATE;
2533          !!!next-input-character;          !!!next-input-character;
2534    
2535          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2536    
2537          redo A;          redo A;
2538        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2539            !!!cp (167);
2540          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2541          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2542          ## reconsume          ## reconsume
2543    
2544          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2545          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2546    
2547          redo A;          redo A;
2548        } elsif ($self->{next_input_character} == 0x0050 or # P        } elsif ($self->{nc} == 0x0050 or # P
2549                 $self->{next_input_character} == 0x0070) { # p                 $self->{nc} == 0x0070) { # p
2550            $self->{state} = PUBLIC_STATE;
2551            $self->{s_kwd} = chr $self->{nc};
2552          !!!next-input-character;          !!!next-input-character;
2553          if ($self->{next_input_character} == 0x0055 or # U          redo A;
2554              $self->{next_input_character} == 0x0075) { # u        } elsif ($self->{nc} == 0x0053 or # S
2555            !!!next-input-character;                 $self->{nc} == 0x0073) { # s
2556            if ($self->{next_input_character} == 0x0042 or # B          $self->{state} = SYSTEM_STATE;
2557                $self->{next_input_character} == 0x0062) { # b          $self->{s_kwd} = chr $self->{nc};
             !!!next-input-character;  
             if ($self->{next_input_character} == 0x004C or # L  
                 $self->{next_input_character} == 0x006C) { # l  
               !!!next-input-character;  
               if ($self->{next_input_character} == 0x0049 or # I  
                   $self->{next_input_character} == 0x0069) { # i  
                 !!!next-input-character;  
                 if ($self->{next_input_character} == 0x0043 or # C  
                     $self->{next_input_character} == 0x0063) { # c  
                   $self->{state} = 'before DOCTYPE public identifier';  
                   !!!next-input-character;  
                   redo A;  
                 }  
               }  
             }  
           }  
         }  
   
         #  
       } elsif ($self->{next_input_character} == 0x0053 or # S  
                $self->{next_input_character} == 0x0073) { # s  
2558          !!!next-input-character;          !!!next-input-character;
2559          if ($self->{next_input_character} == 0x0059 or # Y          redo A;
             $self->{next_input_character} == 0x0079) { # y  
           !!!next-input-character;  
           if ($self->{next_input_character} == 0x0053 or # S  
               $self->{next_input_character} == 0x0073) { # s  
             !!!next-input-character;  
             if ($self->{next_input_character} == 0x0054 or # T  
                 $self->{next_input_character} == 0x0074) { # t  
               !!!next-input-character;  
               if ($self->{next_input_character} == 0x0045 or # E  
                   $self->{next_input_character} == 0x0065) { # e  
                 !!!next-input-character;  
                 if ($self->{next_input_character} == 0x004D or # M  
                     $self->{next_input_character} == 0x006D) { # m  
                   $self->{state} = 'before DOCTYPE system identifier';  
                   !!!next-input-character;  
                   redo A;  
                 }  
               }  
             }  
           }  
         }  
   
         #  
2560        } else {        } else {
2561            !!!cp (180);
2562            !!!parse-error (type => 'string after DOCTYPE name');
2563            $self->{ct}->{quirks} = 1;
2564    
2565            $self->{state} = BOGUS_DOCTYPE_STATE;
2566          !!!next-input-character;          !!!next-input-character;
2567          #          redo A;
2568        }        }
2569        } elsif ($self->{state} == PUBLIC_STATE) {
2570          ## ASCII case-insensitive
2571          if ($self->{nc} == [
2572                undef,
2573                0x0055, # U
2574                0x0042, # B
2575                0x004C, # L
2576                0x0049, # I
2577              ]->[length $self->{s_kwd}] or
2578              $self->{nc} == [
2579                undef,
2580                0x0075, # u
2581                0x0062, # b
2582                0x006C, # l
2583                0x0069, # i
2584              ]->[length $self->{s_kwd}]) {
2585            !!!cp (175);
2586            ## Stay in the state.
2587            $self->{s_kwd} .= chr $self->{nc};
2588            !!!next-input-character;
2589            redo A;
2590          } elsif ((length $self->{s_kwd}) == 5 and
2591                   ($self->{nc} == 0x0043 or # C
2592                    $self->{nc} == 0x0063)) { # c
2593            !!!cp (168);
2594            $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2595            !!!next-input-character;
2596            redo A;
2597          } else {
2598            !!!cp (169);
2599            !!!parse-error (type => 'string after DOCTYPE name',
2600                            line => $self->{line_prev},
2601                            column => $self->{column_prev} + 1 - length $self->{s_kwd});
2602            $self->{ct}->{quirks} = 1;
2603    
2604        !!!parse-error (type => 'string after DOCTYPE name');          $self->{state} = BOGUS_DOCTYPE_STATE;
2605        $self->{state} = 'bogus DOCTYPE';          ## Reconsume.
2606        # next-input-character is already done          redo A;
2607        redo A;        }
2608      } elsif ($self->{state} eq 'before DOCTYPE public identifier') {      } elsif ($self->{state} == SYSTEM_STATE) {
2609        if ({        ## ASCII case-insensitive
2610              0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,        if ($self->{nc} == [
2611              #0x000D => 1, # HT, LF, VT, FF, SP, CR              undef,
2612            }->{$self->{next_input_character}}) {              0x0059, # Y
2613                0x0053, # S
2614                0x0054, # T
2615                0x0045, # E
2616              ]->[length $self->{s_kwd}] or
2617              $self->{nc} == [
2618                undef,
2619                0x0079, # y
2620                0x0073, # s
2621                0x0074, # t
2622                0x0065, # e
2623              ]->[length $self->{s_kwd}]) {
2624            !!!cp (170);
2625            ## Stay in the state.
2626            $self->{s_kwd} .= chr $self->{nc};
2627            !!!next-input-character;
2628            redo A;
2629          } elsif ((length $self->{s_kwd}) == 5 and
2630                   ($self->{nc} == 0x004D or # M
2631                    $self->{nc} == 0x006D)) { # m
2632            !!!cp (171);
2633            $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2634            !!!next-input-character;
2635            redo A;
2636          } else {
2637            !!!cp (172);
2638            !!!parse-error (type => 'string after DOCTYPE name',
2639                            line => $self->{line_prev},
2640                            column => $self->{column_prev} + 1 - length $self->{s_kwd});
2641            $self->{ct}->{quirks} = 1;
2642    
2643            $self->{state} = BOGUS_DOCTYPE_STATE;
2644            ## Reconsume.
2645            redo A;
2646          }
2647        } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2648          if ($is_space->{$self->{nc}}) {
2649            !!!cp (181);
2650          ## Stay in the state          ## Stay in the state
2651          !!!next-input-character;          !!!next-input-character;
2652          redo A;          redo A;
2653        } elsif ($self->{next_input_character} eq 0x0022) { # "        } elsif ($self->{nc} eq 0x0022) { # "
2654          $self->{current_token}->{public_identifier} = ''; # DOCTYPE          !!!cp (182);
2655          $self->{state} = 'DOCTYPE public identifier (double-quoted)';          $self->{ct}->{pubid} = ''; # DOCTYPE
2656            $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;
2657          !!!next-input-character;          !!!next-input-character;
2658          redo A;          redo A;
2659        } elsif ($self->{next_input_character} eq 0x0027) { # '        } elsif ($self->{nc} eq 0x0027) { # '
2660          $self->{current_token}->{public_identifier} = ''; # DOCTYPE          !!!cp (183);
2661          $self->{state} = 'DOCTYPE public identifier (single-quoted)';          $self->{ct}->{pubid} = ''; # DOCTYPE
2662            $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;
2663          !!!next-input-character;          !!!next-input-character;
2664          redo A;          redo A;
2665        } elsif ($self->{next_input_character} eq 0x003E) { # >        } elsif ($self->{nc} eq 0x003E) { # >
2666            !!!cp (184);
2667          !!!parse-error (type => 'no PUBLIC literal');          !!!parse-error (type => 'no PUBLIC literal');
2668    
2669          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2670          !!!next-input-character;          !!!next-input-character;
2671    
2672          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2673          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2674    
2675          redo A;          redo A;
2676        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2677            !!!cp (185);
2678          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2679    
2680          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2681          ## reconsume          ## reconsume
2682    
2683          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2684          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2685    
2686          redo A;          redo A;
2687        } else {        } else {
2688            !!!cp (186);
2689          !!!parse-error (type => 'string after PUBLIC');          !!!parse-error (type => 'string after PUBLIC');
2690          $self->{state} = 'bogus DOCTYPE';          $self->{ct}->{quirks} = 1;
2691    
2692            $self->{state} = BOGUS_DOCTYPE_STATE;
2693          !!!next-input-character;          !!!next-input-character;
2694          redo A;          redo A;
2695        }        }
2696      } elsif ($self->{state} eq 'DOCTYPE public identifier (double-quoted)') {      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2697        if ($self->{next_input_character} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
2698          $self->{state} = 'after DOCTYPE public identifier';          !!!cp (187);
2699            $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2700          !!!next-input-character;          !!!next-input-character;
2701          redo A;          redo A;
2702        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == 0x003E) { # >
2703            !!!cp (188);
2704          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2705    
2706          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2707            !!!next-input-character;
2708    
2709            $self->{ct}->{quirks} = 1;
2710            !!!emit ($self->{ct}); # DOCTYPE
2711    
2712            redo A;
2713          } elsif ($self->{nc} == -1) {
2714            !!!cp (189);
2715            !!!parse-error (type => 'unclosed PUBLIC literal');
2716    
2717            $self->{state} = DATA_STATE;
2718          ## reconsume          ## reconsume
2719    
2720          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2721          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2722    
2723          redo A;          redo A;
2724        } else {        } else {
2725          $self->{current_token}->{public_identifier} # DOCTYPE          !!!cp (190);
2726              .= chr $self->{next_input_character};          $self->{ct}->{pubid} # DOCTYPE
2727                .= chr $self->{nc};
2728            $self->{read_until}->($self->{ct}->{pubid}, q[">],
2729                                  length $self->{ct}->{pubid});
2730    
2731          ## Stay in the state          ## Stay in the state
2732          !!!next-input-character;          !!!next-input-character;
2733          redo A;          redo A;
2734        }        }
2735      } elsif ($self->{state} eq 'DOCTYPE public identifier (single-quoted)') {      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {
2736        if ($self->{next_input_character} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
2737          $self->{state} = 'after DOCTYPE public identifier';          !!!cp (191);
2738            $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2739          !!!next-input-character;          !!!next-input-character;
2740          redo A;          redo A;
2741        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == 0x003E) { # >
2742            !!!cp (192);
2743          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2744    
2745          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2746            !!!next-input-character;
2747    
2748            $self->{ct}->{quirks} = 1;
2749            !!!emit ($self->{ct}); # DOCTYPE
2750    
2751            redo A;
2752          } elsif ($self->{nc} == -1) {
2753            !!!cp (193);
2754            !!!parse-error (type => 'unclosed PUBLIC literal');
2755    
2756            $self->{state} = DATA_STATE;
2757          ## reconsume          ## reconsume
2758    
2759          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2760          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2761    
2762          redo A;          redo A;
2763        } else {        } else {
2764          $self->{current_token}->{public_identifier} # DOCTYPE          !!!cp (194);
2765              .= chr $self->{next_input_character};          $self->{ct}->{pubid} # DOCTYPE
2766                .= chr $self->{nc};
2767            $self->{read_until}->($self->{ct}->{pubid}, q['>],
2768                                  length $self->{ct}->{pubid});
2769    
2770          ## Stay in the state          ## Stay in the state
2771          !!!next-input-character;          !!!next-input-character;
2772          redo A;          redo A;
2773        }        }
2774      } elsif ($self->{state} eq 'after DOCTYPE public identifier') {      } elsif ($self->{state} == AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2775        if ({        if ($is_space->{$self->{nc}}) {
2776              0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,          !!!cp (195);
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_input_character}}) {  
2777          ## Stay in the state          ## Stay in the state
2778          !!!next-input-character;          !!!next-input-character;
2779          redo A;          redo A;
2780        } elsif ($self->{next_input_character} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
2781          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          !!!cp (196);
2782          $self->{state} = 'DOCTYPE system identifier (double-quoted)';          $self->{ct}->{sysid} = ''; # DOCTYPE
2783            $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2784          !!!next-input-character;          !!!next-input-character;
2785          redo A;          redo A;
2786        } elsif ($self->{next_input_character} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
2787          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          !!!cp (197);
2788          $self->{state} = 'DOCTYPE system identifier (single-quoted)';          $self->{ct}->{sysid} = ''; # DOCTYPE
2789            $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2790          !!!next-input-character;          !!!next-input-character;
2791          redo A;          redo A;
2792        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2793          $self->{state} = 'data';          !!!cp (198);
2794            $self->{state} = DATA_STATE;
2795          !!!next-input-character;          !!!next-input-character;
2796    
2797          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2798    
2799          redo A;          redo A;
2800        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2801            !!!cp (199);
2802          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2803    
2804          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2805          ## reconsume          ## reconsume
2806    
2807          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2808          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2809    
2810          redo A;          redo A;
2811        } else {        } else {
2812            !!!cp (200);
2813          !!!parse-error (type => 'string after PUBLIC literal');          !!!parse-error (type => 'string after PUBLIC literal');
2814          $self->{state} = 'bogus DOCTYPE';          $self->{ct}->{quirks} = 1;
2815    
2816            $self->{state} = BOGUS_DOCTYPE_STATE;
2817          !!!next-input-character;          !!!next-input-character;
2818          redo A;          redo A;
2819        }        }
2820      } elsif ($self->{state} eq 'before DOCTYPE system identifier') {      } elsif ($self->{state} == BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2821        if ({        if ($is_space->{$self->{nc}}) {
2822              0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,          !!!cp (201);
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_input_character}}) {  
2823          ## Stay in the state          ## Stay in the state
2824          !!!next-input-character;          !!!next-input-character;
2825          redo A;          redo A;
2826        } elsif ($self->{next_input_character} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
2827          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          !!!cp (202);
2828          $self->{state} = 'DOCTYPE system identifier (double-quoted)';          $self->{ct}->{sysid} = ''; # DOCTYPE
2829            $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2830          !!!next-input-character;          !!!next-input-character;
2831          redo A;          redo A;
2832        } elsif ($self->{next_input_character} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
2833          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          !!!cp (203);
2834          $self->{state} = 'DOCTYPE system identifier (single-quoted)';          $self->{ct}->{sysid} = ''; # DOCTYPE
2835            $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2836          !!!next-input-character;          !!!next-input-character;
2837          redo A;          redo A;
2838        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2839            !!!cp (204);
2840          !!!parse-error (type => 'no SYSTEM literal');          !!!parse-error (type => 'no SYSTEM literal');
2841          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2842          !!!next-input-character;          !!!next-input-character;
2843    
2844          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2845          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2846    
2847          redo A;          redo A;
2848        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2849            !!!cp (205);
2850          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2851    
2852          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2853          ## reconsume          ## reconsume
2854    
2855          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2856          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2857    
2858          redo A;          redo A;
2859        } else {        } else {
2860            !!!cp (206);
2861          !!!parse-error (type => 'string after SYSTEM');          !!!parse-error (type => 'string after SYSTEM');
2862          $self->{state} = 'bogus DOCTYPE';          $self->{ct}->{quirks} = 1;
2863    
2864            $self->{state} = BOGUS_DOCTYPE_STATE;
2865          !!!next-input-character;          !!!next-input-character;
2866          redo A;          redo A;
2867        }        }
2868      } elsif ($self->{state} eq 'DOCTYPE system identifier (double-quoted)') {      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2869        if ($self->{next_input_character} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
2870          $self->{state} = 'after DOCTYPE system identifier';          !!!cp (207);
2871            $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2872            !!!next-input-character;
2873            redo A;
2874          } elsif ($self->{nc} == 0x003E) { # >
2875            !!!cp (208);
2876            !!!parse-error (type => 'unclosed SYSTEM literal');
2877    
2878            $self->{state} = DATA_STATE;
2879          !!!next-input-character;          !!!next-input-character;
2880    
2881            $self->{ct}->{quirks} = 1;
2882            !!!emit ($self->{ct}); # DOCTYPE
2883    
2884          redo A;          redo A;
2885        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2886            !!!cp (209);
2887          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2888    
2889          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2890          ## reconsume          ## reconsume
2891    
2892          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2893          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2894    
2895          redo A;          redo A;
2896        } else {        } else {
2897          $self->{current_token}->{system_identifier} # DOCTYPE          !!!cp (210);
2898              .= chr $self->{next_input_character};          $self->{ct}->{sysid} # DOCTYPE
2899                .= chr $self->{nc};
2900            $self->{read_until}->($self->{ct}->{sysid}, q[">],
2901                                  length $self->{ct}->{sysid});
2902    
2903          ## Stay in the state          ## Stay in the state
2904          !!!next-input-character;          !!!next-input-character;
2905          redo A;          redo A;
2906        }        }
2907      } elsif ($self->{state} eq 'DOCTYPE system identifier (single-quoted)') {      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {
2908        if ($self->{next_input_character} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
2909          $self->{state} = 'after DOCTYPE system identifier';          !!!cp (211);
2910            $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2911          !!!next-input-character;          !!!next-input-character;
2912          redo A;          redo A;
2913        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == 0x003E) { # >
2914            !!!cp (212);
2915          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2916    
2917          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2918            !!!next-input-character;
2919    
2920            $self->{ct}->{quirks} = 1;
2921            !!!emit ($self->{ct}); # DOCTYPE
2922    
2923            redo A;
2924          } elsif ($self->{nc} == -1) {
2925            !!!cp (213);
2926            !!!parse-error (type => 'unclosed SYSTEM literal');
2927    
2928            $self->{state} = DATA_STATE;
2929          ## reconsume          ## reconsume
2930    
2931          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2932          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2933    
2934          redo A;          redo A;
2935        } else {        } else {
2936          $self->{current_token}->{system_identifier} # DOCTYPE          !!!cp (214);
2937              .= chr $self->{next_input_character};          $self->{ct}->{sysid} # DOCTYPE
2938                .= chr $self->{nc};
2939            $self->{read_until}->($self->{ct}->{sysid}, q['>],
2940                                  length $self->{ct}->{sysid});
2941    
2942          ## Stay in the state          ## Stay in the state
2943          !!!next-input-character;          !!!next-input-character;
2944          redo A;          redo A;
2945        }        }
2946      } elsif ($self->{state} eq 'after DOCTYPE system identifier') {      } elsif ($self->{state} == AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2947        if ({        if ($is_space->{$self->{nc}}) {
2948              0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,          !!!cp (215);
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_input_character}}) {  
2949          ## Stay in the state          ## Stay in the state
2950          !!!next-input-character;          !!!next-input-character;
2951          redo A;          redo A;
2952        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2953          $self->{state} = 'data';          !!!cp (216);
2954            $self->{state} = DATA_STATE;
2955          !!!next-input-character;          !!!next-input-character;
2956    
2957          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2958    
2959          redo A;          redo A;
2960        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2961            !!!cp (217);
2962          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2963            $self->{state} = DATA_STATE;
         $self->{state} = 'data';  
2964          ## reconsume          ## reconsume
2965    
2966          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2967          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2968    
2969          redo A;          redo A;
2970        } else {        } else {
2971            !!!cp (218);
2972          !!!parse-error (type => 'string after SYSTEM literal');          !!!parse-error (type => 'string after SYSTEM literal');
2973          $self->{state} = 'bogus DOCTYPE';          #$self->{ct}->{quirks} = 1;
2974    
2975            $self->{state} = BOGUS_DOCTYPE_STATE;
2976          !!!next-input-character;          !!!next-input-character;
2977          redo A;          redo A;
2978        }        }
2979      } elsif ($self->{state} eq 'bogus DOCTYPE') {      } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {
2980        if ($self->{next_input_character} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2981          $self->{state} = 'data';          !!!cp (219);
2982            $self->{state} = DATA_STATE;
2983          !!!next-input-character;          !!!next-input-character;
2984    
2985          delete $self->{current_token}->{correct};          !!!emit ($self->{ct}); # DOCTYPE
         !!!emit ($self->{current_token}); # DOCTYPE  
2986    
2987          redo A;          redo A;
2988        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2989            !!!cp (220);
2990          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2991          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2992          ## reconsume          ## reconsume
2993    
2994          delete $self->{current_token}->{correct};          !!!emit ($self->{ct}); # DOCTYPE
         !!!emit ($self->{current_token}); # DOCTYPE  
2995    
2996          redo A;          redo A;
2997        } else {        } else {
2998            !!!cp (221);
2999            my $s = '';
3000            $self->{read_until}->($s, q[>], 0);
3001    
3002          ## Stay in the state          ## Stay in the state
3003          !!!next-input-character;          !!!next-input-character;
3004          redo A;          redo A;
3005        }        }
3006      } else {      } elsif ($self->{state} == CDATA_SECTION_STATE) {
3007        die "$0: $self->{state}: Unknown state";        ## NOTE: "CDATA section state" in the state is jointly implemented
3008      }        ## by three states, |CDATA_SECTION_STATE|, |CDATA_SECTION_MSE1_STATE|,
3009    } # A          ## and |CDATA_SECTION_MSE2_STATE|.
3010          
3011    die "$0: _get_next_token: unexpected case";        if ($self->{nc} == 0x005D) { # ]
3012  } # _get_next_token          !!!cp (221.1);
3013            $self->{state} = CDATA_SECTION_MSE1_STATE;
3014            !!!next-input-character;
3015            redo A;
3016          } elsif ($self->{nc} == -1) {
3017            $self->{state} = DATA_STATE;
3018            !!!next-input-character;
3019            if (length $self->{ct}->{data}) { # character
3020              !!!cp (221.2);
3021              !!!emit ($self->{ct}); # character
3022            } else {
3023              !!!cp (221.3);
3024              ## No token to emit. $self->{ct} is discarded.
3025            }        
3026            redo A;
3027          } else {
3028            !!!cp (221.4);
3029            $self->{ct}->{data} .= chr $self->{nc};
3030            $self->{read_until}->($self->{ct}->{data},
3031                                  q<]>,
3032                                  length $self->{ct}->{data});
3033    
3034  sub _tokenize_attempt_to_consume_an_entity ($$) {          ## Stay in the state.
3035    my ($self, $in_attr) = @_;          !!!next-input-character;
3036            redo A;
3037          }
3038    
3039    if ({        ## ISSUE: "text tokens" in spec.
3040         0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, # HT, LF, VT, FF,      } elsif ($self->{state} == CDATA_SECTION_MSE1_STATE) {
3041         0x0020 => 1, 0x003C => 1, 0x0026 => 1, -1 => 1, # SP, <, & # 0x000D # CR        if ($self->{nc} == 0x005D) { # ]
3042        }->{$self->{next_input_character}}) {          !!!cp (221.5);
3043      ## Don't consume          $self->{state} = CDATA_SECTION_MSE2_STATE;
3044      ## No error          !!!next-input-character;
3045      return undef;          redo A;
3046    } elsif ($self->{next_input_character} == 0x0023) { # #        } else {
3047      !!!next-input-character;          !!!cp (221.6);
3048      if ($self->{next_input_character} == 0x0078 or # x          $self->{ct}->{data} .= ']';
3049          $self->{next_input_character} == 0x0058) { # X          $self->{state} = CDATA_SECTION_STATE;
3050        my $code;          ## Reconsume.
3051        X: {          redo A;
3052          my $x_char = $self->{next_input_character};        }
3053          !!!next-input-character;      } elsif ($self->{state} == CDATA_SECTION_MSE2_STATE) {
3054          if (0x0030 <= $self->{next_input_character} and        if ($self->{nc} == 0x003E) { # >
3055              $self->{next_input_character} <= 0x0039) { # 0..9          $self->{state} = DATA_STATE;
3056            $code ||= 0;          !!!next-input-character;
3057            $code *= 0x10;          if (length $self->{ct}->{data}) { # character
3058            $code += $self->{next_input_character} - 0x0030;            !!!cp (221.7);
3059            redo X;            !!!emit ($self->{ct}); # character
         } elsif (0x0061 <= $self->{next_input_character} and  
                  $self->{next_input_character} <= 0x0066) { # a..f  
           $code ||= 0;  
           $code *= 0x10;  
           $code += $self->{next_input_character} - 0x0060 + 9;  
           redo X;  
         } elsif (0x0041 <= $self->{next_input_character} and  
                  $self->{next_input_character} <= 0x0046) { # A..F  
           $code ||= 0;  
           $code *= 0x10;  
           $code += $self->{next_input_character} - 0x0040 + 9;  
           redo X;  
         } elsif (not defined $code) { # no hexadecimal digit  
           !!!parse-error (type => 'bare hcro');  
           !!!back-next-input-character ($x_char, $self->{next_input_character});  
           $self->{next_input_character} = 0x0023; # #  
           return undef;  
         } elsif ($self->{next_input_character} == 0x003B) { # ;  
           !!!next-input-character;  
3060          } else {          } else {
3061            !!!parse-error (type => 'no refc');            !!!cp (221.8);
3062              ## No token to emit. $self->{ct} is discarded.
3063          }          }
3064            redo A;
3065          if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {        } elsif ($self->{nc} == 0x005D) { # ]
3066            !!!parse-error (type => sprintf 'invalid character reference:U+%04X', $code);          !!!cp (221.9); # character
3067            $code = 0xFFFD;          $self->{ct}->{data} .= ']'; ## Add first "]" of "]]]".
3068          } elsif ($code > 0x10FFFF) {          ## Stay in the state.
3069            !!!parse-error (type => sprintf 'invalid character reference:U-%08X', $code);          !!!next-input-character;
3070            $code = 0xFFFD;          redo A;
3071          } elsif ($code == 0x000D) {        } else {
3072            !!!parse-error (type => 'CR character reference');          !!!cp (221.11);
3073            $code = 0x000A;          $self->{ct}->{data} .= ']]'; # character
3074          } elsif (0x80 <= $code and $code <= 0x9F) {          $self->{state} = CDATA_SECTION_STATE;
3075            !!!parse-error (type => sprintf 'C1 character reference:U+%04X', $code);          ## Reconsume.
3076            $code = $c1_entity_char->{$code};          redo A;
3077          }        }
3078        } elsif ($self->{state} == ENTITY_STATE) {
3079          return {type => 'character', data => chr $code};        if ($is_space->{$self->{nc}} or
3080        } # X            {
3081      } elsif (0x0030 <= $self->{next_input_character} and              0x003C => 1, 0x0026 => 1, -1 => 1, # <, &
3082               $self->{next_input_character} <= 0x0039) { # 0..9              $self->{entity_add} => 1,
3083        my $code = $self->{next_input_character} - 0x0030;            }->{$self->{nc}}) {
3084        !!!next-input-character;          !!!cp (1001);
3085                  ## Don't consume
3086        while (0x0030 <= $self->{next_input_character} and          ## No error
3087                  $self->{next_input_character} <= 0x0039) { # 0..9          ## Return nothing.
3088          $code *= 10;          #
3089          $code += $self->{next_input_character} - 0x0030;        } elsif ($self->{nc} == 0x0023) { # #
3090                    !!!cp (999);
3091            $self->{state} = ENTITY_HASH_STATE;
3092            $self->{s_kwd} = '#';
3093            !!!next-input-character;
3094            redo A;
3095          } elsif ((0x0041 <= $self->{nc} and
3096                    $self->{nc} <= 0x005A) or # A..Z
3097                   (0x0061 <= $self->{nc} and
3098                    $self->{nc} <= 0x007A)) { # a..z
3099            !!!cp (998);
3100            require Whatpm::_NamedEntityList;
3101            $self->{state} = ENTITY_NAME_STATE;
3102            $self->{s_kwd} = chr $self->{nc};
3103            $self->{entity__value} = $self->{s_kwd};
3104            $self->{entity__match} = 0;
3105          !!!next-input-character;          !!!next-input-character;
3106            redo A;
3107          } else {
3108            !!!cp (1027);
3109            !!!parse-error (type => 'bare ero');
3110            ## Return nothing.
3111            #
3112        }        }
3113    
3114        if ($self->{next_input_character} == 0x003B) { # ;        ## NOTE: No character is consumed by the "consume a character
3115          ## reference" algorithm.  In other word, there is an "&" character
3116          ## that does not introduce a character reference, which would be
3117          ## appended to the parent element or the attribute value in later
3118          ## process of the tokenizer.
3119    
3120          if ($self->{prev_state} == DATA_STATE) {
3121            !!!cp (997);
3122            $self->{state} = $self->{prev_state};
3123            ## Reconsume.
3124            !!!emit ({type => CHARACTER_TOKEN, data => '&',
3125                      line => $self->{line_prev},
3126                      column => $self->{column_prev},
3127                     });
3128            redo A;
3129          } else {
3130            !!!cp (996);
3131            $self->{ca}->{value} .= '&';
3132            $self->{state} = $self->{prev_state};
3133            ## Reconsume.
3134            redo A;
3135          }
3136        } elsif ($self->{state} == ENTITY_HASH_STATE) {
3137          if ($self->{nc} == 0x0078 or # x
3138              $self->{nc} == 0x0058) { # X
3139            !!!cp (995);
3140            $self->{state} = HEXREF_X_STATE;
3141            $self->{s_kwd} .= chr $self->{nc};
3142            !!!next-input-character;
3143            redo A;
3144          } elsif (0x0030 <= $self->{nc} and
3145                   $self->{nc} <= 0x0039) { # 0..9
3146            !!!cp (994);
3147            $self->{state} = NCR_NUM_STATE;
3148            $self->{s_kwd} = $self->{nc} - 0x0030;
3149            !!!next-input-character;
3150            redo A;
3151          } else {
3152            !!!parse-error (type => 'bare nero',
3153                            line => $self->{line_prev},
3154                            column => $self->{column_prev} - 1);
3155    
3156            ## NOTE: According to the spec algorithm, nothing is returned,
3157            ## and then "&#" is appended to the parent element or the attribute
3158            ## value in the later processing.
3159    
3160            if ($self->{prev_state} == DATA_STATE) {
3161              !!!cp (1019);
3162              $self->{state} = $self->{prev_state};
3163              ## Reconsume.
3164              !!!emit ({type => CHARACTER_TOKEN,
3165                        data => '&#',
3166                        line => $self->{line_prev},
3167                        column => $self->{column_prev} - 1,
3168                       });
3169              redo A;
3170            } else {
3171              !!!cp (993);
3172              $self->{ca}->{value} .= '&#';
3173              $self->{state} = $self->{prev_state};
3174              ## Reconsume.
3175              redo A;
3176            }
3177          }
3178        } elsif ($self->{state} == NCR_NUM_STATE) {
3179          if (0x0030 <= $self->{nc} and
3180              $self->{nc} <= 0x0039) { # 0..9
3181            !!!cp (1012);
3182            $self->{s_kwd} *= 10;
3183            $self->{s_kwd} += $self->{nc} - 0x0030;
3184            
3185            ## Stay in the state.
3186            !!!next-input-character;
3187            redo A;
3188          } elsif ($self->{nc} == 0x003B) { # ;
3189            !!!cp (1013);
3190          !!!next-input-character;          !!!next-input-character;
3191            #
3192        } else {        } else {
3193            !!!cp (1014);
3194          !!!parse-error (type => 'no refc');          !!!parse-error (type => 'no refc');
3195            ## Reconsume.
3196            #
3197        }        }
3198    
3199        if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {        my $code = $self->{s_kwd};
3200          !!!parse-error (type => sprintf 'invalid character reference:U+%04X', $code);        my $l = $self->{line_prev};
3201          my $c = $self->{column_prev};
3202          if ($charref_map->{$code}) {
3203            !!!cp (1015);
3204            !!!parse-error (type => 'invalid character reference',
3205                            text => (sprintf 'U+%04X', $code),
3206                            line => $l, column => $c);
3207            $code = $charref_map->{$code};
3208          } elsif ($code > 0x10FFFF) {
3209            !!!cp (1016);
3210            !!!parse-error (type => 'invalid character reference',
3211                            text => (sprintf 'U-%08X', $code),
3212                            line => $l, column => $c);
3213          $code = 0xFFFD;          $code = 0xFFFD;
3214          }
3215    
3216          if ($self->{prev_state} == DATA_STATE) {
3217            !!!cp (992);
3218            $self->{state} = $self->{prev_state};
3219            ## Reconsume.
3220            !!!emit ({type => CHARACTER_TOKEN, data => chr $code,
3221                      line => $l, column => $c,
3222                     });
3223            redo A;
3224          } else {
3225            !!!cp (991);
3226            $self->{ca}->{value} .= chr $code;
3227            $self->{ca}->{has_reference} = 1;
3228            $self->{state} = $self->{prev_state};
3229            ## Reconsume.
3230            redo A;
3231          }
3232        } elsif ($self->{state} == HEXREF_X_STATE) {
3233          if ((0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) or
3234              (0x0041 <= $self->{nc} and $self->{nc} <= 0x0046) or
3235              (0x0061 <= $self->{nc} and $self->{nc} <= 0x0066)) {
3236            # 0..9, A..F, a..f
3237            !!!cp (990);
3238            $self->{state} = HEXREF_HEX_STATE;
3239            $self->{s_kwd} = 0;
3240            ## Reconsume.
3241            redo A;
3242          } else {
3243            !!!parse-error (type => 'bare hcro',
3244                            line => $self->{line_prev},
3245                            column => $self->{column_prev} - 2);
3246    
3247            ## NOTE: According to the spec algorithm, nothing is returned,
3248            ## and then "&#" followed by "X" or "x" is appended to the parent
3249            ## element or the attribute value in the later processing.
3250    
3251            if ($self->{prev_state} == DATA_STATE) {
3252              !!!cp (1005);
3253              $self->{state} = $self->{prev_state};
3254              ## Reconsume.
3255              !!!emit ({type => CHARACTER_TOKEN,
3256                        data => '&' . $self->{s_kwd},
3257                        line => $self->{line_prev},
3258                        column => $self->{column_prev} - length $self->{s_kwd},
3259                       });
3260              redo A;
3261            } else {
3262              !!!cp (989);
3263              $self->{ca}->{value} .= '&' . $self->{s_kwd};
3264              $self->{state} = $self->{prev_state};
3265              ## Reconsume.
3266              redo A;
3267            }
3268          }
3269        } elsif ($self->{state} == HEXREF_HEX_STATE) {
3270          if (0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) {
3271            # 0..9
3272            !!!cp (1002);
3273            $self->{s_kwd} *= 0x10;
3274            $self->{s_kwd} += $self->{nc} - 0x0030;
3275            ## Stay in the state.
3276            !!!next-input-character;
3277            redo A;
3278          } elsif (0x0061 <= $self->{nc} and
3279                   $self->{nc} <= 0x0066) { # a..f
3280            !!!cp (1003);
3281            $self->{s_kwd} *= 0x10;
3282            $self->{s_kwd} += $self->{nc} - 0x0060 + 9;
3283            ## Stay in the state.
3284            !!!next-input-character;
3285            redo A;
3286          } elsif (0x0041 <= $self->{nc} and
3287                   $self->{nc} <= 0x0046) { # A..F
3288            !!!cp (1004);
3289            $self->{s_kwd} *= 0x10;
3290            $self->{s_kwd} += $self->{nc} - 0x0040 + 9;
3291            ## Stay in the state.
3292            !!!next-input-character;
3293            redo A;
3294          } elsif ($self->{nc} == 0x003B) { # ;
3295            !!!cp (1006);
3296            !!!next-input-character;
3297            #
3298          } else {
3299            !!!cp (1007);
3300            !!!parse-error (type => 'no refc',
3301                            line => $self->{line},
3302                            column => $self->{column});
3303            ## Reconsume.
3304            #
3305          }
3306    
3307          my $code = $self->{s_kwd};
3308          my $l = $self->{line_prev};
3309          my $c = $self->{column_prev};
3310          if ($charref_map->{$code}) {
3311            !!!cp (1008);
3312            !!!parse-error (type => 'invalid character reference',
3313                            text => (sprintf 'U+%04X', $code),
3314                            line => $l, column => $c);
3315            $code = $charref_map->{$code};
3316        } elsif ($code > 0x10FFFF) {        } elsif ($code > 0x10FFFF) {
3317          !!!parse-error (type => sprintf 'invalid character reference:U-%08X', $code);          !!!cp (1009);
3318            !!!parse-error (type => 'invalid character reference',
3319                            text => (sprintf 'U-%08X', $code),
3320                            line => $l, column => $c);
3321          $code = 0xFFFD;          $code = 0xFFFD;
       } elsif ($code == 0x000D) {  
         !!!parse-error (type => 'CR character reference');  
         $code = 0x000A;  
       } elsif (0x80 <= $code and $code <= 0x9F) {  
         !!!parse-error (type => sprintf 'C1 character reference:U+%04X', $code);  
         $code = $c1_entity_char->{$code};  
3322        }        }
3323          
3324        return {type => 'character', data => chr $code};        if ($self->{prev_state} == DATA_STATE) {
3325      } else {          !!!cp (988);
3326        !!!parse-error (type => 'bare nero');          $self->{state} = $self->{prev_state};
3327        !!!back-next-input-character ($self->{next_input_character});          ## Reconsume.
3328        $self->{next_input_character} = 0x0023; # #          !!!emit ({type => CHARACTER_TOKEN, data => chr $code,
3329        return undef;                    line => $l, column => $c,
3330      }                   });
3331    } elsif ((0x0041 <= $self->{next_input_character} and          redo A;
3332              $self->{next_input_character} <= 0x005A) or        } else {
3333             (0x0061 <= $self->{next_input_character} and          !!!cp (987);
3334              $self->{next_input_character} <= 0x007A)) {          $self->{ca}->{value} .= chr $code;
3335      my $entity_name = chr $self->{next_input_character};          $self->{ca}->{has_reference} = 1;
3336      !!!next-input-character;          $self->{state} = $self->{prev_state};
3337            ## Reconsume.
3338      my $value = $entity_name;          redo A;
3339      my $match = 0;        }
3340      require Whatpm::_NamedEntityList;      } elsif ($self->{state} == ENTITY_NAME_STATE) {
3341      our $EntityChar;        if (length $self->{s_kwd} < 30 and
3342              ## NOTE: Some number greater than the maximum length of entity name
3343      while (length $entity_name < 10 and            ((0x0041 <= $self->{nc} and # a
3344             ## NOTE: Some number greater than the maximum length of entity name              $self->{nc} <= 0x005A) or # x
3345             ((0x0041 <= $self->{next_input_character} and # a             (0x0061 <= $self->{nc} and # a
3346               $self->{next_input_character} <= 0x005A) or # x              $self->{nc} <= 0x007A) or # z
3347              (0x0061 <= $self->{next_input_character} and # a             (0x0030 <= $self->{nc} and # 0
3348               $self->{next_input_character} <= 0x007A) or # z              $self->{nc} <= 0x0039) or # 9
3349              (0x0030 <= $self->{next_input_character} and # 0             $self->{nc} == 0x003B)) { # ;
3350               $self->{next_input_character} <= 0x0039) or # 9          our $EntityChar;
3351              $self->{next_input_character} == 0x003B)) { # ;          $self->{s_kwd} .= chr $self->{nc};
3352        $entity_name .= chr $self->{next_input_character};          if (defined $EntityChar->{$self->{s_kwd}}) {
3353        if (defined $EntityChar->{$entity_name}) {            if ($self->{nc} == 0x003B) { # ;
3354          if ($self->{next_input_character} == 0x003B) { # ;              !!!cp (1020);
3355            $value = $EntityChar->{$entity_name};              $self->{entity__value} = $EntityChar->{$self->{s_kwd}};
3356            $match = 1;              $self->{entity__match} = 1;
3357            !!!next-input-character;              !!!next-input-character;
3358            last;              #
3359              } else {
3360                !!!cp (1021);
3361                $self->{entity__value} = $EntityChar->{$self->{s_kwd}};
3362                $self->{entity__match} = -1;
3363                ## Stay in the state.
3364                !!!next-input-character;
3365                redo A;
3366              }
3367          } else {          } else {
3368            $value = $EntityChar->{$entity_name};            !!!cp (1022);
3369            $match = -1;            $self->{entity__value} .= chr $self->{nc};
3370              $self->{entity__match} *= 2;
3371              ## Stay in the state.
3372            !!!next-input-character;            !!!next-input-character;
3373              redo A;
3374          }          }
       } else {  
         $value .= chr $self->{next_input_character};  
         $match *= 2;  
         !!!next-input-character;  
3375        }        }
3376      }  
3377              my $data;
3378      if ($match > 0) {        my $has_ref;
3379        return {type => 'character', data => $value};        if ($self->{entity__match} > 0) {
3380      } elsif ($match < 0) {          !!!cp (1023);
3381        !!!parse-error (type => 'no refc');          $data = $self->{entity__value};
3382        if ($in_attr and $match < -1) {          $has_ref = 1;
3383          return {type => 'character', data => '&'.$entity_name};          #
3384          } elsif ($self->{entity__match} < 0) {
3385            !!!parse-error (type => 'no refc');
3386            if ($self->{prev_state} != DATA_STATE and # in attribute
3387                $self->{entity__match} < -1) {
3388              !!!cp (1024);
3389              $data = '&' . $self->{s_kwd};
3390              #
3391            } else {
3392              !!!cp (1025);
3393              $data = $self->{entity__value};
3394              $has_ref = 1;
3395              #
3396            }
3397        } else {        } else {
3398          return {type => 'character', data => $value};          !!!cp (1026);
3399            !!!parse-error (type => 'bare ero',
3400                            line => $self->{line_prev},
3401                            column => $self->{column_prev} - length $self->{s_kwd});
3402            $data = '&' . $self->{s_kwd};
3403            #
3404          }
3405      
3406          ## NOTE: In these cases, when a character reference is found,
3407          ## it is consumed and a character token is returned, or, otherwise,
3408          ## nothing is consumed and returned, according to the spec algorithm.
3409          ## In this implementation, anything that has been examined by the
3410          ## tokenizer is appended to the parent element or the attribute value
3411          ## as string, either literal string when no character reference or
3412          ## entity-replaced string otherwise, in this stage, since any characters
3413          ## that would not be consumed are appended in the data state or in an
3414          ## appropriate attribute value state anyway.
3415    
3416          if ($self->{prev_state} == DATA_STATE) {
3417            !!!cp (986);
3418            $self->{state} = $self->{prev_state};
3419            ## Reconsume.
3420            !!!emit ({type => CHARACTER_TOKEN,
3421                      data => $data,
3422                      line => $self->{line_prev},
3423                      column => $self->{column_prev} + 1 - length $self->{s_kwd},
3424                     });
3425            redo A;
3426          } else {
3427            !!!cp (985);
3428            $self->{ca}->{value} .= $data;
3429            $self->{ca}->{has_reference} = 1 if $has_ref;
3430            $self->{state} = $self->{prev_state};
3431            ## Reconsume.
3432            redo A;
3433        }        }
3434      } else {      } else {
3435        !!!parse-error (type => 'bare ero');        die "$0: $self->{state}: Unknown state";
       ## NOTE: No characters are consumed in the spec.  
       return {type => 'character', data => '&'.$value};  
3436      }      }
3437    } else {    } # A  
3438      ## no characters are consumed  
3439      !!!parse-error (type => 'bare ero');    die "$0: _get_next_token: unexpected case";
3440      return undef;  } # _get_next_token
   }  
 } # _tokenize_attempt_to_consume_an_entity  
3441    
3442  sub _initialize_tree_constructor ($) {  sub _initialize_tree_constructor ($) {
3443    my $self = shift;    my $self = shift;
# Line 1780  sub _initialize_tree_constructor ($) { Line 3446  sub _initialize_tree_constructor ($) {
3446    ## TODO: Turn mutation events off # MUST    ## TODO: Turn mutation events off # MUST
3447    ## TODO: Turn loose Document option (manakai extension) on    ## TODO: Turn loose Document option (manakai extension) on
3448    $self->{document}->manakai_is_html (1); # MUST    $self->{document}->manakai_is_html (1); # MUST
3449      $self->{document}->set_user_data (manakai_source_line => 1);
3450      $self->{document}->set_user_data (manakai_source_column => 1);
3451  } # _initialize_tree_constructor  } # _initialize_tree_constructor
3452    
3453  sub _terminate_tree_constructor ($) {  sub _terminate_tree_constructor ($) {
# Line 1806  sub _construct_tree ($) { Line 3474  sub _construct_tree ($) {
3474        
3475    !!!next-token;    !!!next-token;
3476    
   $self->{insertion_mode} = 'before head';  
3477    undef $self->{form_element};    undef $self->{form_element};
3478    undef $self->{head_element};    undef $self->{head_element};
3479      undef $self->{head_element_inserted};
3480    $self->{open_elements} = [];    $self->{open_elements} = [];
3481    undef $self->{inner_html_node};    undef $self->{inner_html_node};
3482    
3483      ## NOTE: The "initial" insertion mode.
3484    $self->_tree_construction_initial; # MUST    $self->_tree_construction_initial; # MUST
3485    
3486      ## NOTE: The "before html" insertion mode.
3487    $self->_tree_construction_root_element;    $self->_tree_construction_root_element;
3488      $self->{insertion_mode} = BEFORE_HEAD_IM;
3489    
3490      ## NOTE: The "before head" insertion mode and so on.
3491    $self->_tree_construction_main;    $self->_tree_construction_main;
3492  } # _construct_tree  } # _construct_tree
3493    
3494  sub _tree_construction_initial ($) {  sub _tree_construction_initial ($) {
3495    my $self = shift;    my $self = shift;
3496    
3497      ## NOTE: "initial" insertion mode
3498    
3499    INITIAL: {    INITIAL: {
3500      if ($token->{type} eq 'DOCTYPE') {      if ($token->{type} == DOCTYPE_TOKEN) {
3501        ## NOTE: Conformance checkers MAY, instead of reporting "not HTML5"        ## NOTE: Conformance checkers MAY, instead of reporting "not HTML5"
3502        ## error, switch to a conformance checking mode for another        ## error, switch to a conformance checking mode for another
3503        ## language.        ## language.
3504        my $doctype_name = $token->{name};        my $doctype_name = $token->{name};
3505        $doctype_name = '' unless defined $doctype_name;        $doctype_name = '' unless defined $doctype_name;
3506        $doctype_name =~ tr/a-z/A-Z/;        $doctype_name =~ tr/a-z/A-Z/; # ASCII case-insensitive
3507        if (not defined $token->{name} or # <!DOCTYPE>        if (not defined $token->{name} or # <!DOCTYPE>
3508            defined $token->{public_identifier} or            defined $token->{sysid}) {
3509            defined $token->{system_identifier}) {          !!!cp ('t1');
3510          !!!parse-error (type => 'not HTML5');          !!!parse-error (type => 'not HTML5', token => $token);
3511        } elsif ($doctype_name ne 'HTML') {        } elsif ($doctype_name ne 'HTML') {
3512          ## ISSUE: ASCII case-insensitive? (in fact it does not matter)          !!!cp ('t2');
3513          !!!parse-error (type => 'not HTML5');          !!!parse-error (type => 'not HTML5', token => $token);
3514          } elsif (defined $token->{pubid}) {
3515            if ($token->{pubid} eq 'XSLT-compat') {
3516              !!!cp ('t1.2');
3517              !!!parse-error (type => 'XSLT-compat', token => $token,
3518                              level => $self->{level}->{should});
3519            } else {
3520              !!!parse-error (type => 'not HTML5', token => $token);
3521            }
3522          } else {
3523            !!!cp ('t3');
3524            #
3525        }        }
3526                
3527        my $doctype = $self->{document}->create_document_type_definition        my $doctype = $self->{document}->create_document_type_definition
3528          ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?          ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?
3529        $doctype->public_id ($token->{public_identifier})        ## NOTE: Default value for both |public_id| and |system_id| attributes
3530            if defined $token->{public_identifier};        ## are empty strings, so that we don't set any value in missing cases.
3531        $doctype->system_id ($token->{system_identifier})        $doctype->public_id ($token->{pubid}) if defined $token->{pubid};
3532            if defined $token->{system_identifier};        $doctype->system_id ($token->{sysid}) if defined $token->{sysid};
3533        ## NOTE: Other DocumentType attributes are null or empty lists.        ## NOTE: Other DocumentType attributes are null or empty lists.
3534        ## ISSUE: internalSubset = null??        ## ISSUE: internalSubset = null??
3535        $self->{document}->append_child ($doctype);        $self->{document}->append_child ($doctype);
3536                
3537        if (not $token->{correct} or $doctype_name ne 'HTML') {        if ($token->{quirks} or $doctype_name ne 'HTML') {
3538            !!!cp ('t4');
3539          $self->{document}->manakai_compat_mode ('quirks');          $self->{document}->manakai_compat_mode ('quirks');
3540        } elsif (defined $token->{public_identifier}) {        } elsif (defined $token->{pubid}) {
3541          my $pubid = $token->{public_identifier};          my $pubid = $token->{pubid};
3542          $pubid =~ tr/a-z/A-z/;          $pubid =~ tr/a-z/A-z/;
3543          if ({          my $prefix = [
3544            "+//SILMARIL//DTD HTML PRO V0R11 19970101//EN" => 1,            "+//SILMARIL//DTD HTML PRO V0R11 19970101//",
3545            "-//ADVASOFT LTD//DTD HTML 3.0 ASWEDIT + EXTENSIONS//EN" => 1,            "-//ADVASOFT LTD//DTD HTML 3.0 ASWEDIT + EXTENSIONS//",
3546            "-//AS//DTD HTML 3.0 ASWEDIT + EXTENSIONS//EN" => 1,            "-//AS//DTD HTML 3.0 ASWEDIT + EXTENSIONS//",
3547            "-//IETF//DTD HTML 2.0 LEVEL 1//EN" => 1,            "-//IETF//DTD HTML 2.0 LEVEL 1//",
3548            "-//IETF//DTD HTML 2.0 LEVEL 2//EN" => 1,            "-//IETF//DTD HTML 2.0 LEVEL 2//",
3549            "-//IETF//DTD HTML 2.0 STRICT LEVEL 1//EN" => 1,            "-//IETF//DTD HTML 2.0 STRICT LEVEL 1//",
3550            "-//IETF//DTD HTML 2.0 STRICT LEVEL 2//EN" => 1,            "-//IETF//DTD HTML 2.0 STRICT LEVEL 2//",
3551            "-//IETF//DTD HTML 2.0 STRICT//EN" => 1,            "-//IETF//DTD HTML 2.0 STRICT//",
3552            "-//IETF//DTD HTML 2.0//EN" => 1,            "-//IETF//DTD HTML 2.0//",
3553            "-//IETF//DTD HTML 2.1E//EN" => 1,            "-//IETF//DTD HTML 2.1E//",
3554            "-//IETF//DTD HTML 3.0//EN" => 1,            "-//IETF//DTD HTML 3.0//",
3555            "-//IETF//DTD HTML 3.0//EN//" => 1,            "-//IETF//DTD HTML 3.2 FINAL//",
3556            "-//IETF//DTD HTML 3.2 FINAL//EN" => 1,            "-//IETF//DTD HTML 3.2//",
3557            "-//IETF//DTD HTML 3.2//EN" => 1,            "-//IETF//DTD HTML 3//",
3558            "-//IETF//DTD HTML 3//EN" => 1,            "-//IETF//DTD HTML LEVEL 0//",
3559            "-//IETF//DTD HTML LEVEL 0//EN" => 1,            "-//IETF//DTD HTML LEVEL 1//",
3560            "-//IETF//DTD HTML LEVEL 0//EN//2.0" => 1,            "-//IETF//DTD HTML LEVEL 2//",
3561            "-//IETF//DTD HTML LEVEL 1//EN" => 1,            "-//IETF//DTD HTML LEVEL 3//",
3562            "-//IETF//DTD HTML LEVEL 1//EN//2.0" => 1,            "-//IETF//DTD HTML STRICT LEVEL 0//",
3563            "-//IETF//DTD HTML LEVEL 2//EN" => 1,            "-//IETF//DTD HTML STRICT LEVEL 1//",
3564            "-//IETF//DTD HTML LEVEL 2//EN//2.0" => 1,            "-//IETF//DTD HTML STRICT LEVEL 2//",
3565            "-//IETF//DTD HTML LEVEL 3//EN" => 1,            "-//IETF//DTD HTML STRICT LEVEL 3//",
3566            "-//IETF//DTD HTML LEVEL 3//EN//3.0" => 1,            "-//IETF//DTD HTML STRICT//",
3567            "-//IETF//DTD HTML STRICT LEVEL 0//EN" => 1,            "-//IETF//DTD HTML//",
3568            "-//IETF//DTD HTML STRICT LEVEL 0//EN//2.0" => 1,            "-//METRIUS//DTD METRIUS PRESENTATIONAL//",
3569            "-//IETF//DTD HTML STRICT LEVEL 1//EN" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML STRICT//",
3570            "-//IETF//DTD HTML STRICT LEVEL 1//EN//2.0" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML//",
3571            "-//IETF//DTD HTML STRICT LEVEL 2//EN" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 TABLES//",
3572            "-//IETF//DTD HTML STRICT LEVEL 2//EN//2.0" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML STRICT//",
3573            "-//IETF//DTD HTML STRICT LEVEL 3//EN" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML//",
3574            "-//IETF//DTD HTML STRICT LEVEL 3//EN//3.0" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 TABLES//",
3575            "-//IETF//DTD HTML STRICT//EN" => 1,            "-//NETSCAPE COMM. CORP.//DTD HTML//",
3576            "-//IETF//DTD HTML STRICT//EN//2.0" => 1,            "-//NETSCAPE COMM. CORP.//DTD STRICT HTML//",
3577            "-//IETF//DTD HTML STRICT//EN//3.0" => 1,            "-//O'REILLY AND ASSOCIATES//DTD HTML 2.0//",
3578            "-//IETF//DTD HTML//EN" => 1,            "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED 1.0//",
3579            "-//IETF//DTD HTML//EN//2.0" => 1,            "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED RELAXED 1.0//",
3580            "-//IETF//DTD HTML//EN//3.0" => 1,            "-//SOFTQUAD SOFTWARE//DTD HOTMETAL PRO 6.0::19990601::EXTENSIONS TO HTML 4.0//",
3581            "-//METRIUS//DTD METRIUS PRESENTATIONAL//EN" => 1,            "-//SOFTQUAD//DTD HOTMETAL PRO 4.0::19971010::EXTENSIONS TO HTML 4.0//",
3582            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML STRICT//EN" => 1,            "-//SPYGLASS//DTD HTML 2.0 EXTENDED//",
3583            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML//EN" => 1,            "-//SQ//DTD HTML 2.0 HOTMETAL + EXTENSIONS//",
3584            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 TABLES//EN" => 1,            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA HTML//",
3585            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML STRICT//EN" => 1,            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA STRICT HTML//",
3586            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML//EN" => 1,            "-//W3C//DTD HTML 3 1995-03-24//",
3587            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 TABLES//EN" => 1,            "-//W3C//DTD HTML 3.2 DRAFT//",
3588            "-//NETSCAPE COMM. CORP.//DTD HTML//EN" => 1,            "-//W3C//DTD HTML 3.2 FINAL//",
3589            "-//NETSCAPE COMM. CORP.//DTD STRICT HTML//EN" => 1,            "-//W3C//DTD HTML 3.2//",
3590            "-//O'REILLY AND ASSOCIATES//DTD HTML 2.0//EN" => 1,            "-//W3C//DTD HTML 3.2S DRAFT//",
3591            "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED 1.0//EN" => 1,            "-//W3C//DTD HTML 4.0 FRAMESET//",
3592            "-//SPYGLASS//DTD HTML 2.0 EXTENDED//EN" => 1,            "-//W3C//DTD HTML 4.0 TRANSITIONAL//",
3593            "-//SQ//DTD HTML 2.0 HOTMETAL + EXTENSIONS//EN" => 1,            "-//W3C//DTD HTML EXPERIMETNAL 19960712//",
3594            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA HTML//EN" => 1,            "-//W3C//DTD HTML EXPERIMENTAL 970421//",
3595            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA STRICT HTML//EN" => 1,            "-//W3C//DTD W3 HTML//",
3596            "-//W3C//DTD HTML 3 1995-03-24//EN" => 1,            "-//W3O//DTD W3 HTML 3.0//",
3597            "-//W3C//DTD HTML 3.2 DRAFT//EN" => 1,            "-//WEBTECHS//DTD MOZILLA HTML 2.0//",
3598            "-//W3C//DTD HTML 3.2 FINAL//EN" => 1,            "-//WEBTECHS//DTD MOZILLA HTML//",
3599            "-//W3C//DTD HTML 3.2//EN" => 1,          ]; # $prefix
3600            "-//W3C//DTD HTML 3.2S DRAFT//EN" => 1,          my $match;
3601            "-//W3C//DTD HTML 4.0 FRAMESET//EN" => 1,          for (@$prefix) {
3602            "-//W3C//DTD HTML 4.0 TRANSITIONAL//EN" => 1,            if (substr ($prefix, 0, length $_) eq $_) {
3603            "-//W3C//DTD HTML EXPERIMETNAL 19960712//EN" => 1,              $match = 1;
3604            "-//W3C//DTD HTML EXPERIMENTAL 970421//EN" => 1,              last;
3605            "-//W3C//DTD W3 HTML//EN" => 1,            }
3606            "-//W3O//DTD W3 HTML 3.0//EN" => 1,          }
3607            "-//W3O//DTD W3 HTML 3.0//EN//" => 1,          if ($match or
3608            "-//W3O//DTD W3 HTML STRICT 3.0//EN//" => 1,              $pubid eq "-//W3O//DTD W3 HTML STRICT 3.0//EN//" or
3609            "-//WEBTECHS//DTD MOZILLA HTML 2.0//EN" => 1,              $pubid eq "-/W3C/DTD HTML 4.0 TRANSITIONAL/EN" or
3610            "-//WEBTECHS//DTD MOZILLA HTML//EN" => 1,              $pubid eq "HTML") {
3611            "-/W3C/DTD HTML 4.0 TRANSITIONAL/EN" => 1,            !!!cp ('t5');
           "HTML" => 1,  
         }->{$pubid}) {  
3612            $self->{document}->manakai_compat_mode ('quirks');            $self->{document}->manakai_compat_mode ('quirks');
3613          } elsif ($pubid eq "-//W3C//DTD HTML 4.01 FRAMESET//EN" or          } elsif ($pubid =~ m[^-//W3C//DTD HTML 4.01 FRAMESET//] or
3614                   $pubid eq "-//W3C//DTD HTML 4.01 TRANSITIONAL//EN") {                   $pubid =~ m[^-//W3C//DTD HTML 4.01 TRANSITIONAL//]) {
3615            if (defined $token->{system_identifier}) {            if (defined $token->{sysid}) {
3616                !!!cp ('t6');
3617              $self->{document}->manakai_compat_mode ('quirks');              $self->{document}->manakai_compat_mode ('quirks');
3618            } else {            } else {
3619                !!!cp ('t7');
3620              $self->{document}->manakai_compat_mode ('limited quirks');              $self->{document}->manakai_compat_mode ('limited quirks');
3621            }            }
3622          } elsif ($pubid eq "-//W3C//DTD XHTML 1.0 Frameset//EN" or          } elsif ($pubid =~ m[^-//W3C//DTD XHTML 1.0 FRAMESET//] or
3623                   $pubid eq "-//W3C//DTD XHTML 1.0 Transitional//EN") {                   $pubid =~ m[^-//W3C//DTD XHTML 1.0 TRANSITIONAL//]) {
3624              !!!cp ('t8');
3625            $self->{document}->manakai_compat_mode ('limited quirks');            $self->{document}->manakai_compat_mode ('limited quirks');
3626            } else {
3627              !!!cp ('t9');
3628          }          }
3629          } else {
3630            !!!cp ('t10');
3631        }        }
3632        if (defined $token->{system_identifier}) {        if (defined $token->{sysid}) {
3633          my $sysid = $token->{system_identifier};          my $sysid = $token->{sysid};
3634          $sysid =~ tr/A-Z/a-z/;          $sysid =~ tr/A-Z/a-z/;
3635          if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {          if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {
3636              ## NOTE: Ensure that |PUBLIC "(limited quirks)" "(quirks)"| is
3637              ## marked as quirks.
3638            $self->{document}->manakai_compat_mode ('quirks');            $self->{document}->manakai_compat_mode ('quirks');
3639              !!!cp ('t11');
3640            } else {
3641              !!!cp ('t12');
3642          }          }
3643          } else {
3644            !!!cp ('t13');
3645        }        }
3646                
3647        ## Go to the root element phase.        ## Go to the "before html" insertion mode.
3648        !!!next-token;        !!!next-token;
3649        return;        return;
3650      } elsif ({      } elsif ({
3651                'start tag' => 1,                START_TAG_TOKEN, 1,
3652                'end tag' => 1,                END_TAG_TOKEN, 1,
3653                'end-of-file' => 1,                END_OF_FILE_TOKEN, 1,
3654               }->{$token->{type}}) {               }->{$token->{type}}) {
3655        !!!parse-error (type => 'no DOCTYPE');        !!!cp ('t14');
3656          !!!parse-error (type => 'no DOCTYPE', token => $token);
3657        $self->{document}->manakai_compat_mode ('quirks');        $self->{document}->manakai_compat_mode ('quirks');
3658        ## Go to the root element phase        ## Go to the "before html" insertion mode.
3659        ## reprocess        ## reprocess
3660          !!!ack-later;
3661        return;        return;
3662      } elsif ($token->{type} eq 'character') {      } elsif ($token->{type} == CHARACTER_TOKEN) {
3663        if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D        if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3664          ## Ignore the token          ## Ignore the token
3665    
3666          unless (length $token->{data}) {          unless (length $token->{data}) {
3667            ## Stay in the phase            !!!cp ('t15');
3668              ## Stay in the insertion mode.
3669            !!!next-token;            !!!next-token;
3670            redo INITIAL;            redo INITIAL;
3671            } else {
3672              !!!cp ('t16');
3673          }          }
3674          } else {
3675            !!!cp ('t17');
3676        }        }
3677    
3678        !!!parse-error (type => 'no DOCTYPE');        !!!parse-error (type => 'no DOCTYPE', token => $token);
3679        $self->{document}->manakai_compat_mode ('quirks');        $self->{document}->manakai_compat_mode ('quirks');
3680        ## Go to the root element phase        ## Go to the "before html" insertion mode.
3681        ## reprocess        ## reprocess
3682        return;        return;
3683      } elsif ($token->{type} eq 'comment') {      } elsif ($token->{type} == COMMENT_TOKEN) {
3684          !!!cp ('t18');
3685        my $comment = $self->{document}->create_comment ($token->{data});        my $comment = $self->{document}->create_comment ($token->{data});
3686        $self->{document}->append_child ($comment);        $self->{document}->append_child ($comment);
3687                
3688        ## Stay in the phase.        ## Stay in the insertion mode.
3689        !!!next-token;        !!!next-token;
3690        redo INITIAL;        redo INITIAL;
3691      } else {      } else {
3692        die "$0: $token->{type}: Unknown token";        die "$0: $token->{type}: Unknown token type";
3693      }      }
3694    } # INITIAL    } # INITIAL
3695    
3696      die "$0: _tree_construction_initial: This should be never reached";
3697  } # _tree_construction_initial  } # _tree_construction_initial
3698    
3699  sub _tree_construction_root_element ($) {  sub _tree_construction_root_element ($) {
3700    my $self = shift;    my $self = shift;
3701    
3702      ## NOTE: "before html" insertion mode.
3703        
3704    B: {    B: {
3705        if ($token->{type} eq 'DOCTYPE') {        if ($token->{type} == DOCTYPE_TOKEN) {
3706          !!!parse-error (type => 'in html:#DOCTYPE');          !!!cp ('t19');
3707            !!!parse-error (type => 'in html:#DOCTYPE', token => $token);
3708          ## Ignore the token          ## Ignore the token
3709          ## Stay in the phase          ## Stay in the insertion mode.
3710          !!!next-token;          !!!next-token;
3711          redo B;          redo B;
3712        } elsif ($token->{type} eq 'comment') {        } elsif ($token->{type} == COMMENT_TOKEN) {
3713            !!!cp ('t20');
3714          my $comment = $self->{document}->create_comment ($token->{data});          my $comment = $self->{document}->create_comment ($token->{data});
3715          $self->{document}->append_child ($comment);          $self->{document}->append_child ($comment);
3716          ## Stay in the phase          ## Stay in the insertion mode.
3717          !!!next-token;          !!!next-token;
3718          redo B;          redo B;
3719        } elsif ($token->{type} eq 'character') {        } elsif ($token->{type} == CHARACTER_TOKEN) {
3720          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3721            ## Ignore the token.            ## Ignore the token.
3722    
3723            unless (length $token->{data}) {            unless (length $token->{data}) {
3724              ## Stay in the phase              !!!cp ('t21');
3725                ## Stay in the insertion mode.
3726              !!!next-token;              !!!next-token;
3727              redo B;              redo B;
3728              } else {
3729                !!!cp ('t22');
3730            }            }
3731            } else {
3732              !!!cp ('t23');
3733          }          }
3734    
3735            $self->{application_cache_selection}->(undef);
3736    
3737          #          #
3738          } elsif ($token->{type} == START_TAG_TOKEN) {
3739            if ($token->{tag_name} eq 'html') {
3740              my $root_element;
3741              !!!create-element ($root_element, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);
3742              $self->{document}->append_child ($root_element);
3743              push @{$self->{open_elements}},
3744                  [$root_element, $el_category->{html}];
3745    
3746              if ($token->{attributes}->{manifest}) {
3747                !!!cp ('t24');
3748                $self->{application_cache_selection}
3749                    ->($token->{attributes}->{manifest}->{value});
3750                ## ISSUE: Spec is unclear on relative references.
3751                ## According to Hixie (#whatwg 2008-03-19), it should be
3752                ## resolved against the base URI of the document in HTML
3753                ## or xml:base of the element in XHTML.
3754              } else {
3755                !!!cp ('t25');
3756                $self->{application_cache_selection}->(undef);
3757              }
3758    
3759              !!!nack ('t25c');
3760    
3761              !!!next-token;
3762              return; ## Go to the "before head" insertion mode.
3763            } else {
3764              !!!cp ('t25.1');
3765              #
3766            }
3767        } elsif ({        } elsif ({
3768                  'start tag' => 1,                  END_TAG_TOKEN, 1,
3769                  'end tag' => 1,                  END_OF_FILE_TOKEN, 1,
                 'end-of-file' => 1,  
3770                 }->{$token->{type}}) {                 }->{$token->{type}}) {
3771          ## ISSUE: There is an issue in the spec          !!!cp ('t26');
3772          #          #
3773        } else {        } else {
3774          die "$0: $token->{type}: Unknown token";          die "$0: $token->{type}: Unknown token type";
3775        }        }
3776        my $root_element; !!!create-element ($root_element, 'html');  
3777        $self->{document}->append_child ($root_element);      my $root_element;
3778        push @{$self->{open_elements}}, [$root_element, 'html'];      !!!create-element ($root_element, $HTML_NS, 'html',, $token);
3779        ## reprocess      $self->{document}->append_child ($root_element);
3780        #redo B;      push @{$self->{open_elements}}, [$root_element, $el_category->{html}];
3781        return; ## Go to the main phase.  
3782        $self->{application_cache_selection}->(undef);
3783    
3784        ## NOTE: Reprocess the token.
3785        !!!ack-later;
3786        return; ## Go to the "before head" insertion mode.
3787    } # B    } # B
3788    
3789      die "$0: _tree_construction_root_element: This should never be reached";
3790  } # _tree_construction_root_element  } # _tree_construction_root_element
3791    
3792  sub _reset_insertion_mode ($) {  sub _reset_insertion_mode ($) {
# Line 2043  sub _reset_insertion_mode ($) { Line 3801  sub _reset_insertion_mode ($) {
3801            
3802      ## Step 3      ## Step 3
3803      S3: {      S3: {
       ## ISSUE: Oops! "If node is the first node in the stack of open  
       ## elements, then set last to true. If the context element of the  
       ## HTML fragment parsing algorithm is neither a td element nor a  
       ## th element, then set node to the context element. (fragment case)":  
       ## The second "if" is in the scope of the first "if"!?  
3804        if ($self->{open_elements}->[0]->[0] eq $node->[0]) {        if ($self->{open_elements}->[0]->[0] eq $node->[0]) {
3805          $last = 1;          $last = 1;
3806          if (defined $self->{inner_html_node}) {          if (defined $self->{inner_html_node}) {
3807            if ($self->{inner_html_node}->[1] eq 'td' or            !!!cp ('t28');
3808                $self->{inner_html_node}->[1] eq 'th') {            $node = $self->{inner_html_node};
3809              #          } else {
3810            } else {            die "_reset_insertion_mode: t27";
             $node = $self->{inner_html_node};  
           }  
3811          }          }
3812        }        }
3813              
3814        ## Step 4..13        ## Step 4..14
3815        my $new_mode = {        my $new_mode;
3816                        select => 'in select',        if ($node->[1] & FOREIGN_EL) {
3817                        td => 'in cell',          !!!cp ('t28.1');
3818                        th => 'in cell',          ## NOTE: Strictly spaking, the line below only applies to MathML and
3819                        tr => 'in row',          ## SVG elements.  Currently the HTML syntax supports only MathML and
3820                        tbody => 'in table body',          ## SVG elements as foreigners.
3821                        thead => 'in table head',          $new_mode = IN_BODY_IM | IN_FOREIGN_CONTENT_IM;
3822                        tfoot => 'in table foot',        } elsif ($node->[1] & TABLE_CELL_EL) {
3823                        caption => 'in caption',          if ($last) {
3824                        colgroup => 'in column group',            !!!cp ('t28.2');
3825                        table => 'in table',            #
3826                        head => 'in body', # not in head!          } else {
3827                        body => 'in body',            !!!cp ('t28.3');
3828                        frameset => 'in frameset',            $new_mode = IN_CELL_IM;
3829                       }->{$node->[1]};          }
3830          } else {
3831            !!!cp ('t28.4');
3832            $new_mode = {
3833                          select => IN_SELECT_IM,
3834                          ## NOTE: |option| and |optgroup| do not set
3835                          ## insertion mode to "in select" by themselves.
3836                          tr => IN_ROW_IM,
3837                          tbody => IN_TABLE_BODY_IM,
3838                          thead => IN_TABLE_BODY_IM,
3839                          tfoot => IN_TABLE_BODY_IM,
3840                          caption => IN_CAPTION_IM,
3841                          colgroup => IN_COLUMN_GROUP_IM,
3842                          table => IN_TABLE_IM,
3843                          head => IN_BODY_IM, # not in head!
3844                          body => IN_BODY_IM,
3845                          frameset => IN_FRAMESET_IM,
3846                         }->{$node->[0]->manakai_local_name};
3847          }
3848        $self->{insertion_mode} = $new_mode and return if defined $new_mode;        $self->{insertion_mode} = $new_mode and return if defined $new_mode;
3849                
3850        ## Step 14        ## Step 15
3851        if ($node->[1] eq 'html') {        if ($node->[1] & HTML_EL) {
3852          unless (defined $self->{head_element}) {          unless (defined $self->{head_element}) {
3853            $self->{insertion_mode} = 'before head';            !!!cp ('t29');
3854              $self->{insertion_mode} = BEFORE_HEAD_IM;
3855          } else {          } else {
3856            $self->{insertion_mode} = 'after head';            ## ISSUE: Can this state be reached?
3857              !!!cp ('t30');
3858              $self->{insertion_mode} = AFTER_HEAD_IM;
3859          }          }
3860          return;          return;
3861          } else {
3862            !!!cp ('t31');
3863        }        }
3864                
       ## Step 15  
       $self->{insertion_mode} = 'in body' and return if $last;  
         
3865        ## Step 16        ## Step 16
3866          $self->{insertion_mode} = IN_BODY_IM and return if $last;
3867          
3868          ## Step 17
3869        $i--;        $i--;
3870        $node = $self->{open_elements}->[$i];        $node = $self->{open_elements}->[$i];
3871                
3872        ## Step 17        ## Step 18
3873        redo S3;        redo S3;
3874      } # S3      } # S3
3875    
3876      die "$0: _reset_insertion_mode: This line should never be reached";
3877  } # _reset_insertion_mode  } # _reset_insertion_mode
3878    
3879  sub _tree_construction_main ($) {  sub _tree_construction_main ($) {
3880    my $self = shift;    my $self = shift;
3881    
   my $previous_insertion_mode;  
   
3882    my $active_formatting_elements = [];    my $active_formatting_elements = [];
3883    
3884    my $reconstruct_active_formatting_elements = sub { # MUST    my $reconstruct_active_formatting_elements = sub { # MUST
# Line 2121  sub _tree_construction_main ($) { Line 3895  sub _tree_construction_main ($) {
3895      return if $entry->[0] eq '#marker';      return if $entry->[0] eq '#marker';
3896      for (@{$self->{open_elements}}) {      for (@{$self->{open_elements}}) {
3897        if ($entry->[0] eq $_->[0]) {        if ($entry->[0] eq $_->[0]) {
3898            !!!cp ('t32');
3899          return;          return;
3900        }        }
3901      }      }
# Line 2135  sub _tree_construction_main ($) { Line 3910  sub _tree_construction_main ($) {
3910    
3911        ## Step 6        ## Step 6
3912        if ($entry->[0] eq '#marker') {        if ($entry->[0] eq '#marker') {
3913            !!!cp ('t33_1');
3914          #          #
3915        } else {        } else {
3916          my $in_open_elements;          my $in_open_elements;
3917          OE: for (@{$self->{open_elements}}) {          OE: for (@{$self->{open_elements}}) {
3918            if ($entry->[0] eq $_->[0]) {            if ($entry->[0] eq $_->[0]) {
3919                !!!cp ('t33');
3920              $in_open_elements = 1;              $in_open_elements = 1;
3921              last OE;              last OE;
3922            }            }
3923          }          }
3924          if ($in_open_elements) {          if ($in_open_elements) {
3925              !!!cp ('t34');
3926            #            #
3927          } else {          } else {
3928              ## NOTE: <!DOCTYPE HTML><p><b><i><u></p> <p>X
3929              !!!cp ('t35');
3930            redo S4;            redo S4;
3931          }          }
3932        }        }
# Line 2169  sub _tree_construction_main ($) { Line 3949  sub _tree_construction_main ($) {
3949    
3950        ## Step 11        ## Step 11
3951        unless ($clone->[0] eq $active_formatting_elements->[-1]->[0]) {        unless ($clone->[0] eq $active_formatting_elements->[-1]->[0]) {
3952            !!!cp ('t36');
3953          ## Step 7'          ## Step 7'
3954          $i++;          $i++;
3955          $entry = $active_formatting_elements->[$i];          $entry = $active_formatting_elements->[$i];
3956                    
3957          redo S7;          redo S7;
3958        }        }
3959    
3960          !!!cp ('t37');
3961      } # S7      } # S7
3962    }; # $reconstruct_active_formatting_elements    }; # $reconstruct_active_formatting_elements
3963    
3964    my $clear_up_to_marker = sub {    my $clear_up_to_marker = sub {
3965      for (reverse 0..$#$active_formatting_elements) {      for (reverse 0..$#$active_formatting_elements) {
3966        if ($active_formatting_elements->[$_]->[0] eq '#marker') {        if ($active_formatting_elements->[$_]->[0] eq '#marker') {
3967            !!!cp ('t38');
3968          splice @$active_formatting_elements, $_;          splice @$active_formatting_elements, $_;
3969          return;          return;
3970        }        }
3971      }      }
3972    
3973        !!!cp ('t39');
3974    }; # $clear_up_to_marker    }; # $clear_up_to_marker
3975    
3976    my $parse_rcdata = sub ($$) {    my $insert;
3977      my ($content_model_flag, $insert) = @_;  
3978      my $parse_rcdata = sub ($) {
3979        my ($content_model_flag) = @_;
3980    
3981      ## Step 1      ## Step 1
3982      my $start_tag_name = $token->{tag_name};      my $start_tag_name = $token->{tag_name};
3983      my $el;      my $el;
3984      !!!create-element ($el, $start_tag_name, $token->{attributes});      !!!create-element ($el, $HTML_NS, $start_tag_name, $token->{attributes}, $token);
3985    
3986      ## Step 2      ## Step 2
3987      $insert->($el); # /context node/->append_child ($el)      $insert->($el);
3988    
3989      ## Step 3      ## Step 3
3990      $self->{content_model} = $content_model_flag; # CDATA or RCDATA      $self->{content_model} = $content_model_flag; # CDATA or RCDATA
# Line 2204  sub _tree_construction_main ($) { Line 3992  sub _tree_construction_main ($) {
3992    
3993      ## Step 4      ## Step 4
3994      my $text = '';      my $text = '';
3995        !!!nack ('t40.1');
3996      !!!next-token;      !!!next-token;
3997      while ($token->{type} eq 'character') { # or until stop tokenizing      while ($token->{type} == CHARACTER_TOKEN) { # or until stop tokenizing
3998          !!!cp ('t40');
3999        $text .= $token->{data};        $text .= $token->{data};
4000        !!!next-token;        !!!next-token;
4001      }      }
4002    
4003      ## Step 5      ## Step 5
4004      if (length $text) {      if (length $text) {
4005          !!!cp ('t41');
4006        my $text = $self->{document}->create_text_node ($text);        my $text = $self->{document}->create_text_node ($text);
4007        $el->append_child ($text);        $el->append_child ($text);
4008      }      }
# Line 2220  sub _tree_construction_main ($) { Line 4011  sub _tree_construction_main ($) {
4011      $self->{content_model} = PCDATA_CONTENT_MODEL;      $self->{content_model} = PCDATA_CONTENT_MODEL;
4012    
4013      ## Step 7      ## Step 7
4014      if ($token->{type} eq 'end tag' and $token->{tag_name} eq $start_tag_name) {      if ($token->{type} == END_TAG_TOKEN and
4015            $token->{tag_name} eq $start_tag_name) {
4016          !!!cp ('t42');
4017        ## Ignore the token        ## Ignore the token
     } elsif ($content_model_flag == CDATA_CONTENT_MODEL) {  
       !!!parse-error (type => 'in CDATA:#'.$token->{type});  
     } elsif ($content_model_flag == RCDATA_CONTENT_MODEL) {  
       !!!parse-error (type => 'in RCDATA:#'.$token->{type});  
4018      } else {      } else {
4019        die "$0: $content_model_flag in parse_rcdata";        ## NOTE: An end-of-file token.
4020          if ($content_model_flag == CDATA_CONTENT_MODEL) {
4021            !!!cp ('t43');
4022            !!!parse-error (type => 'in CDATA:#eof', token => $token);
4023          } elsif ($content_model_flag == RCDATA_CONTENT_MODEL) {
4024            !!!cp ('t44');
4025            !!!parse-error (type => 'in RCDATA:#eof', token => $token);
4026          } else {
4027            die "$0: $content_model_flag in parse_rcdata";
4028          }
4029      }      }
4030      !!!next-token;      !!!next-token;
4031    }; # $parse_rcdata    }; # $parse_rcdata
4032    
4033    my $script_start_tag = sub ($) {    my $script_start_tag = sub () {
     my $insert = $_[0];  
4034      my $script_el;      my $script_el;
4035      !!!create-element ($script_el, 'script', $token->{attributes});      !!!create-element ($script_el, $HTML_NS, 'script', $token->{attributes}, $token);
4036      ## TODO: mark as "parser-inserted"      ## TODO: mark as "parser-inserted"
4037    
4038      $self->{content_model} = CDATA_CONTENT_MODEL;      $self->{content_model} = CDATA_CONTENT_MODEL;
4039      delete $self->{escape}; # MUST      delete $self->{escape}; # MUST
4040            
4041      my $text = '';      my $text = '';
4042        !!!nack ('t45.1');
4043      !!!next-token;      !!!next-token;
4044      while ($token->{type} eq 'character') {      while ($token->{type} == CHARACTER_TOKEN) {
4045          !!!cp ('t45');
4046        $text .= $token->{data};        $text .= $token->{data};
4047        !!!next-token;        !!!next-token;
4048      } # stop if non-character token or tokenizer stops tokenising      } # stop if non-character token or tokenizer stops tokenising
4049      if (length $text) {      if (length $text) {
4050          !!!cp ('t46');
4051        $script_el->manakai_append_text ($text);        $script_el->manakai_append_text ($text);
4052      }      }
4053                                
4054      $self->{content_model} = PCDATA_CONTENT_MODEL;      $self->{content_model} = PCDATA_CONTENT_MODEL;
4055    
4056      if ($token->{type} eq 'end tag' and      if ($token->{type} == END_TAG_TOKEN and
4057          $token->{tag_name} eq 'script') {          $token->{tag_name} eq 'script') {
4058          !!!cp ('t47');
4059        ## Ignore the token        ## Ignore the token
4060      } else {      } else {
4061        !!!parse-error (type => 'in CDATA:#'.$token->{type});        !!!cp ('t48');
4062          !!!parse-error (type => 'in CDATA:#eof', token => $token);
4063        ## ISSUE: And ignore?        ## ISSUE: And ignore?
4064        ## TODO: mark as "already executed"        ## TODO: mark as "already executed"
4065      }      }
4066            
4067      if (defined $self->{inner_html_node}) {      if (defined $self->{inner_html_node}) {
4068          !!!cp ('t49');
4069        ## TODO: mark as "already executed"        ## TODO: mark as "already executed"
4070      } else {      } else {
4071          !!!cp ('t50');
4072        ## TODO: $old_insertion_point = current insertion point        ## TODO: $old_insertion_point = current insertion point
4073        ## TODO: insertion point = just before the next input character        ## TODO: insertion point = just before the next input character
4074    
# Line 2278  sub _tree_construction_main ($) { Line 4082  sub _tree_construction_main ($) {
4082      !!!next-token;      !!!next-token;
4083    }; # $script_start_tag    }; # $script_start_tag
4084    
4085      ## NOTE: $open_tables->[-1]->[0] is the "current table" element node.
4086      ## NOTE: $open_tables->[-1]->[1] is the "tainted" flag.
4087      ## NOTE: $open_tables->[-1]->[2] is set false when non-Text node inserted.
4088      my $open_tables = [[$self->{open_elements}->[0]->[0]]];
4089    
4090    my $formatting_end_tag = sub {    my $formatting_end_tag = sub {
4091      my $tag_name = shift;      my $end_tag_token = shift;
4092        my $tag_name = $end_tag_token->{tag_name};
4093    
4094        ## NOTE: The adoption agency algorithm (AAA).
4095    
4096      FET: {      FET: {
4097        ## Step 1        ## Step 1
4098        my $formatting_element;        my $formatting_element;
4099        my $formatting_element_i_in_active;        my $formatting_element_i_in_active;
4100        AFE: for (reverse 0..$#$active_formatting_elements) {        AFE: for (reverse 0..$#$active_formatting_elements) {
4101          if ($active_formatting_elements->[$_]->[1] eq $tag_name) {          if ($active_formatting_elements->[$_]->[0] eq '#marker') {
4102              !!!cp ('t52');
4103              last AFE;
4104            } elsif ($active_formatting_elements->[$_]->[0]->manakai_local_name
4105                         eq $tag_name) {
4106              !!!cp ('t51');
4107            $formatting_element = $active_formatting_elements->[$_];            $formatting_element = $active_formatting_elements->[$_];
4108            $formatting_element_i_in_active = $_;            $formatting_element_i_in_active = $_;
4109            last AFE;            last AFE;
         } elsif ($active_formatting_elements->[$_]->[0] eq '#marker') {  
           last AFE;  
4110          }          }
4111        } # AFE        } # AFE
4112        unless (defined $formatting_element) {        unless (defined $formatting_element) {
4113          !!!parse-error (type => 'unmatched end tag:'.$tag_name);          !!!cp ('t53');
4114            !!!parse-error (type => 'unmatched end tag', text => $tag_name, token => $end_tag_token);
4115          ## Ignore the token          ## Ignore the token
4116          !!!next-token;          !!!next-token;
4117          return;          return;
# Line 2307  sub _tree_construction_main ($) { Line 4123  sub _tree_construction_main ($) {
4123          my $node = $self->{open_elements}->[$_];          my $node = $self->{open_elements}->[$_];
4124          if ($node->[0] eq $formatting_element->[0]) {          if ($node->[0] eq $formatting_element->[0]) {
4125            if ($in_scope) {            if ($in_scope) {
4126                !!!cp ('t54');
4127              $formatting_element_i_in_open = $_;              $formatting_element_i_in_open = $_;
4128              last INSCOPE;              last INSCOPE;
4129            } else { # in open elements but not in scope            } else { # in open elements but not in scope
4130              !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});              !!!cp ('t55');
4131                !!!parse-error (type => 'unmatched end tag',
4132                                text => $token->{tag_name},
4133                                token => $end_tag_token);
4134              ## Ignore the token              ## Ignore the token
4135              !!!next-token;              !!!next-token;
4136              return;              return;
4137            }            }
4138          } elsif ({          } elsif ($node->[1] & SCOPING_EL) {
4139                    table => 1, caption => 1, td => 1, th => 1,            !!!cp ('t56');
                   button => 1, marquee => 1, object => 1, html => 1,  
                  }->{$node->[1]}) {  
4140            $in_scope = 0;            $in_scope = 0;
4141          }          }
4142        } # INSCOPE        } # INSCOPE
4143        unless (defined $formatting_element_i_in_open) {        unless (defined $formatting_element_i_in_open) {
4144          !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});          !!!cp ('t57');
4145            !!!parse-error (type => 'unmatched end tag',
4146                            text => $token->{tag_name},
4147                            token => $end_tag_token);
4148          pop @$active_formatting_elements; # $formatting_element          pop @$active_formatting_elements; # $formatting_element
4149          !!!next-token; ## TODO: ok?          !!!next-token; ## TODO: ok?
4150          return;          return;
4151        }        }
4152        if (not $self->{open_elements}->[-1]->[0] eq $formatting_element->[0]) {        if (not $self->{open_elements}->[-1]->[0] eq $formatting_element->[0]) {
4153          !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);          !!!cp ('t58');
4154            !!!parse-error (type => 'not closed',
4155                            text => $self->{open_elements}->[-1]->[0]
4156                                ->manakai_local_name,
4157                            token => $end_tag_token);
4158        }        }
4159                
4160        ## Step 2        ## Step 2
# Line 2337  sub _tree_construction_main ($) { Line 4162  sub _tree_construction_main ($) {
4162        my $furthest_block_i_in_open;        my $furthest_block_i_in_open;
4163        OE: for (reverse 0..$#{$self->{open_elements}}) {        OE: for (reverse 0..$#{$self->{open_elements}}) {
4164          my $node = $self->{open_elements}->[$_];          my $node = $self->{open_elements}->[$_];
4165          if (not $formatting_category->{$node->[1]} and          if (not ($node->[1] & FORMATTING_EL) and
4166              #not $phrasing_category->{$node->[1]} and              #not $phrasing_category->{$node->[1]} and
4167              ($special_category->{$node->[1]} or              ($node->[1] & SPECIAL_EL or
4168               $scoping_category->{$node->[1]})) {               $node->[1] & SCOPING_EL)) { ## Scoping is redundant, maybe
4169              !!!cp ('t59');
4170            $furthest_block = $node;            $furthest_block = $node;
4171            $furthest_block_i_in_open = $_;            $furthest_block_i_in_open = $_;
4172          } elsif ($node->[0] eq $formatting_element->[0]) {          } elsif ($node->[0] eq $formatting_element->[0]) {
4173              !!!cp ('t60');
4174            last OE;            last OE;
4175          }          }
4176        } # OE        } # OE
4177                
4178        ## Step 3        ## Step 3
4179        unless (defined $furthest_block) { # MUST        unless (defined $furthest_block) { # MUST
4180            !!!cp ('t61');
4181          splice @{$self->{open_elements}}, $formatting_element_i_in_open;          splice @{$self->{open_elements}}, $formatting_element_i_in_open;
4182          splice @$active_formatting_elements, $formatting_element_i_in_active, 1;          splice @$active_formatting_elements, $formatting_element_i_in_active, 1;
4183          !!!next-token;          !!!next-token;
# Line 2362  sub _tree_construction_main ($) { Line 4190  sub _tree_construction_main ($) {
4190        ## Step 5        ## Step 5
4191        my $furthest_block_parent = $furthest_block->[0]->parent_node;        my $furthest_block_parent = $furthest_block->[0]->parent_node;
4192        if (defined $furthest_block_parent) {        if (defined $furthest_block_parent) {
4193            !!!cp ('t62');
4194          $furthest_block_parent->remove_child ($furthest_block->[0]);          $furthest_block_parent->remove_child ($furthest_block->[0]);
4195        }        }
4196                
# Line 2384  sub _tree_construction_main ($) { Line 4213  sub _tree_construction_main ($) {
4213          S7S2: {          S7S2: {
4214            for (reverse 0..$#$active_formatting_elements) {            for (reverse 0..$#$active_formatting_elements) {
4215              if ($active_formatting_elements->[$_]->[0] eq $node->[0]) {              if ($active_formatting_elements->[$_]->[0] eq $node->[0]) {
4216                  !!!cp ('t63');
4217                $node_i_in_active = $_;                $node_i_in_active = $_;
4218                last S7S2;                last S7S2;
4219              }              }
# Line 2397  sub _tree_construction_main ($) { Line 4227  sub _tree_construction_main ($) {
4227                    
4228          ## Step 4          ## Step 4
4229          if ($last_node->[0] eq $furthest_block->[0]) {          if ($last_node->[0] eq $furthest_block->[0]) {
4230              !!!cp ('t64');
4231            $bookmark_prev_el = $node->[0];            $bookmark_prev_el = $node->[0];
4232          }          }
4233                    
4234          ## Step 5          ## Step 5
4235          if ($node->[0]->has_child_nodes ()) {          if ($node->[0]->has_child_nodes ()) {
4236              !!!cp ('t65');
4237            my $clone = [$node->[0]->clone_node (0), $node->[1]];            my $clone = [$node->[0]->clone_node (0), $node->[1]];
4238            $active_formatting_elements->[$node_i_in_active] = $clone;            $active_formatting_elements->[$node_i_in_active] = $clone;
4239            $self->{open_elements}->[$node_i_in_open] = $clone;            $self->{open_elements}->[$node_i_in_open] = $clone;
# Line 2419  sub _tree_construction_main ($) { Line 4251  sub _tree_construction_main ($) {
4251        } # S7          } # S7  
4252                
4253        ## Step 8        ## Step 8
4254        $common_ancestor_node->[0]->append_child ($last_node->[0]);        if ($common_ancestor_node->[1] & TABLE_ROWS_EL) {
4255            my $foster_parent_element;
4256            my $next_sibling;
4257            OE: for (reverse 0..$#{$self->{open_elements}}) {
4258              if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {
4259                                 my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
4260                                 if (defined $parent and $parent->node_type == 1) {
4261                                   !!!cp ('t65.1');
4262                                   $foster_parent_element = $parent;
4263                                   $next_sibling = $self->{open_elements}->[$_]->[0];
4264                                 } else {
4265                                   !!!cp ('t65.2');
4266                                   $foster_parent_element
4267                                     = $self->{open_elements}->[$_ - 1]->[0];
4268                                 }
4269                                 last OE;
4270                               }
4271                             } # OE
4272                             $foster_parent_element = $self->{open_elements}->[0]->[0]
4273                               unless defined $foster_parent_element;
4274            $foster_parent_element->insert_before ($last_node->[0], $next_sibling);
4275            $open_tables->[-1]->[1] = 1; # tainted
4276          } else {
4277            !!!cp ('t65.3');
4278            $common_ancestor_node->[0]->append_child ($last_node->[0]);
4279          }
4280                
4281        ## Step 9        ## Step 9
4282        my $clone = [$formatting_element->[0]->clone_node (0),        my $clone = [$formatting_element->[0]->clone_node (0),
# Line 2436  sub _tree_construction_main ($) { Line 4293  sub _tree_construction_main ($) {
4293        my $i;        my $i;
4294        AFE: for (reverse 0..$#$active_formatting_elements) {        AFE: for (reverse 0..$#$active_formatting_elements) {
4295          if ($active_formatting_elements->[$_]->[0] eq $formatting_element->[0]) {          if ($active_formatting_elements->[$_]->[0] eq $formatting_element->[0]) {
4296              !!!cp ('t66');
4297            splice @$active_formatting_elements, $_, 1;            splice @$active_formatting_elements, $_, 1;
4298            $i-- and last AFE if defined $i;            $i-- and last AFE if defined $i;
4299          } elsif ($active_formatting_elements->[$_]->[0] eq $bookmark_prev_el) {          } elsif ($active_formatting_elements->[$_]->[0] eq $bookmark_prev_el) {
4300              !!!cp ('t67');
4301            $i = $_;            $i = $_;
4302          }          }
4303        } # AFE        } # AFE
# Line 2448  sub _tree_construction_main ($) { Line 4307  sub _tree_construction_main ($) {
4307        undef $i;        undef $i;
4308        OE: for (reverse 0..$#{$self->{open_elements}}) {        OE: for (reverse 0..$#{$self->{open_elements}}) {
4309          if ($self->{open_elements}->[$_]->[0] eq $formatting_element->[0]) {          if ($self->{open_elements}->[$_]->[0] eq $formatting_element->[0]) {
4310              !!!cp ('t68');
4311            splice @{$self->{open_elements}}, $_, 1;            splice @{$self->{open_elements}}, $_, 1;
4312            $i-- and last OE if defined $i;            $i-- and last OE if defined $i;
4313          } elsif ($self->{open_elements}->[$_]->[0] eq $furthest_block->[0]) {          } elsif ($self->{open_elements}->[$_]->[0] eq $furthest_block->[0]) {
4314              !!!cp ('t69');
4315            $i = $_;            $i = $_;
4316          }          }
4317        } # OE        } # OE
# Line 2461  sub _tree_construction_main ($) { Line 4322  sub _tree_construction_main ($) {
4322      } # FET      } # FET
4323    }; # $formatting_end_tag    }; # $formatting_end_tag
4324    
4325    my $insert_to_current = sub {    $insert = my $insert_to_current = sub {
4326      $self->{open_elements}->[-1]->[0]->append_child ($_[0]);      $self->{open_elements}->[-1]->[0]->append_child ($_[0]);
4327    }; # $insert_to_current    }; # $insert_to_current
4328    
4329    my $insert_to_foster = sub {    my $insert_to_foster = sub {
4330                         my $child = shift;      my $child = shift;
4331                         if ({      if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {
4332                              table => 1, tbody => 1, tfoot => 1,        # MUST
4333                              thead => 1, tr => 1,        my $foster_parent_element;
4334                             }->{$self->{open_elements}->[-1]->[1]}) {        my $next_sibling;
4335                           # MUST        OE: for (reverse 0..$#{$self->{open_elements}}) {
4336                           my $foster_parent_element;          if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {
                          my $next_sibling;  
                          OE: for (reverse 0..$#{$self->{open_elements}}) {  
                            if ($self->{open_elements}->[$_]->[1] eq 'table') {  
4337                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
4338                               if (defined $parent and $parent->node_type == 1) {                               if (defined $parent and $parent->node_type == 1) {
4339                                   !!!cp ('t70');
4340                                 $foster_parent_element = $parent;                                 $foster_parent_element = $parent;
4341                                 $next_sibling = $self->{open_elements}->[$_]->[0];                                 $next_sibling = $self->{open_elements}->[$_]->[0];
4342                               } else {                               } else {
4343                                   !!!cp ('t71');
4344                                 $foster_parent_element                                 $foster_parent_element
4345                                   = $self->{open_elements}->[$_ - 1]->[0];                                   = $self->{open_elements}->[$_ - 1]->[0];
4346                               }                               }
# Line 2491  sub _tree_construction_main ($) { Line 4351  sub _tree_construction_main ($) {
4351                             unless defined $foster_parent_element;                             unless defined $foster_parent_element;
4352                           $foster_parent_element->insert_before                           $foster_parent_element->insert_before
4353                             ($child, $next_sibling);                             ($child, $next_sibling);
4354                         } else {        $open_tables->[-1]->[1] = 1; # tainted
4355                           $self->{open_elements}->[-1]->[0]->append_child ($child);      } else {
4356                         }        !!!cp ('t72');
4357          $self->{open_elements}->[-1]->[0]->append_child ($child);
4358        }
4359    }; # $insert_to_foster    }; # $insert_to_foster
4360    
4361    my $in_body = sub {    ## NOTE: When a character is inserted, if the last node that was
4362      my $insert = shift;    ## inserted by the parser is a Text node and the character has to be
4363      if ($token->{type} eq 'start tag') {    ## inserted after that node, then the character is appended to the
4364        if ($token->{tag_name} eq 'script') {    ## Text node.  However, if any other node is inserted by the parser,
4365          ## NOTE: This is an "as if in head" code clone    ## then a new Text node is created and the character is appended as
4366          $script_start_tag->($insert);    ## that Text node.  If I'm not wrong, there are only two cases where
4367          return;    ## this occurs.  One is the case where an element node is inserted
4368        } elsif ($token->{tag_name} eq 'style') {    ## to the |head| element.  This is covered by using the
4369          ## NOTE: This is an "as if in head" code clone    ## |$self->{head_element_inserted}| flag.  Another is the case where
4370          $parse_rcdata->(CDATA_CONTENT_MODEL, $insert);    ## an element or comment is inserted into the |table| subtree while
4371          return;    ## foster parenting happens.  This is covered by using the [2] flag
4372        } elsif ({    ## of the |$open_tables| structure.  All other cases are handled
4373                  base => 1, link => 1,    ## simply by calling |manakai_append_text| method.
4374                 }->{$token->{tag_name}}) {  
4375          ## NOTE: This is an "as if in head" code clone, only "-t" differs    B: while (1) {
4376          !!!insert-element-t ($token->{tag_name}, $token->{attributes});      if ($token->{type} == DOCTYPE_TOKEN) {
4377          pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.        !!!cp ('t73');
4378          !!!next-token;        !!!parse-error (type => 'in html:#DOCTYPE', token => $token);
4379          return;        ## Ignore the token
4380        } elsif ($token->{tag_name} eq 'meta') {        ## Stay in the phase
4381          ## NOTE: This is an "as if in head" code clone, only "-t" differs        !!!next-token;
4382          !!!insert-element-t ($token->{tag_name}, $token->{attributes});        next B;
4383          pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.      } elsif ($token->{type} == START_TAG_TOKEN and
4384                 $token->{tag_name} eq 'html') {
4385          unless ($self->{confident}) {        if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
4386            my $charset;          !!!cp ('t79');
4387            if ($token->{attributes}->{charset}) { ## TODO: And if supported          !!!parse-error (type => 'after html', text => 'html', token => $token);
4388              $charset = $token->{attributes}->{charset}->{value};          $self->{insertion_mode} = AFTER_BODY_IM;
4389            }        } elsif ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {
4390            if ($token->{attributes}->{'http-equiv'}) {          !!!cp ('t80');
4391              ## ISSUE: Algorithm name in the spec was incorrect so that not linked to the definition.          !!!parse-error (type => 'after html', text => 'html', token => $token);
4392              if ($token->{attributes}->{'http-equiv'}->{value}          $self->{insertion_mode} = AFTER_FRAMESET_IM;
4393                  =~ /\A[^;]*;[\x09-\x0D\x20]*charset[\x09-\x0D\x20]*=        } else {
4394                      [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|          !!!cp ('t81');
4395                      ([^"'\x09-\x0D\x20][^\x09-\x0D\x20]*))/x) {        }
               $charset = defined $1 ? $1 : defined $2 ? $2 : $3;  
             } ## TODO: And if supported  
           }  
           ## TODO: Change the encoding  
         }  
4396    
4397          !!!next-token;        !!!cp ('t82');
4398          return;        !!!parse-error (type => 'not first start tag', token => $token);
4399        } elsif ($token->{tag_name} eq 'title') {        my $top_el = $self->{open_elements}->[0]->[0];
4400          !!!parse-error (type => 'in body:title');        for my $attr_name (keys %{$token->{attributes}}) {
4401          ## NOTE: This is an "as if in head" code clone          unless ($top_el->has_attribute_ns (undef, $attr_name)) {
4402          $parse_rcdata->(RCDATA_CONTENT_MODEL, sub {            !!!cp ('t84');
4403            if (defined $self->{head_element}) {            $top_el->set_attribute_ns
4404              $self->{head_element}->append_child ($_[0]);              (undef, [undef, $attr_name],
4405            } else {               $token->{attributes}->{$attr_name}->{value});
             $insert->($_[0]);  
           }  
         });  
         return;  
       } elsif ($token->{tag_name} eq 'body') {  
         !!!parse-error (type => 'in body:body');  
                 
         if (@{$self->{open_elements}} == 1 or  
             $self->{open_elements}->[1]->[1] ne 'body') {  
           ## Ignore the token  
         } else {  
           my $body_el = $self->{open_elements}->[1]->[0];  
           for my $attr_name (keys %{$token->{attributes}}) {  
             unless ($body_el->has_attribute_ns (undef, $attr_name)) {  
               $body_el->set_attribute_ns  
                 (undef, [undef, $attr_name],  
                  $token->{attributes}->{$attr_name}->{value});  
             }  
           }  
4406          }          }
4407          }
4408          !!!nack ('t84.1');
4409          !!!next-token;
4410          next B;
4411        } elsif ($token->{type} == COMMENT_TOKEN) {
4412          my $comment = $self->{document}->create_comment ($token->{data});
4413          if ($self->{insertion_mode} & AFTER_HTML_IMS) {
4414            !!!cp ('t85');
4415            $self->{document}->append_child ($comment);
4416          } elsif ($self->{insertion_mode} == AFTER_BODY_IM) {
4417            !!!cp ('t86');
4418            $self->{open_elements}->[0]->[0]->append_child ($comment);
4419          } else {
4420            !!!cp ('t87');
4421            $self->{open_elements}->[-1]->[0]->append_child ($comment);
4422            $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
4423          }
4424          !!!next-token;
4425          next B;
4426        } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {
4427          if ($token->{type} == CHARACTER_TOKEN) {
4428            !!!cp ('t87.1');
4429            $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
4430          !!!next-token;          !!!next-token;
4431          return;          next B;
4432        } elsif ({        } elsif ($token->{type} == START_TAG_TOKEN) {
4433                  address => 1, blockquote => 1, center => 1, dir => 1,          if ((not {mglyph => 1, malignmark => 1}->{$token->{tag_name}} and
4434                  div => 1, dl => 1, fieldset => 1, listing => 1,               $self->{open_elements}->[-1]->[1] & FOREIGN_FLOW_CONTENT_EL) or
4435                  menu => 1, ol => 1, p => 1, ul => 1,              not ($self->{open_elements}->[-1]->[1] & FOREIGN_EL) or
4436                  pre => 1,              ($token->{tag_name} eq 'svg' and
4437                 }->{$token->{tag_name}}) {               $self->{open_elements}->[-1]->[1] & MML_AXML_EL)) {
4438          ## has a p element in scope            ## NOTE: "using the rules for secondary insertion mode"then"continue"
4439          INSCOPE: for (reverse @{$self->{open_elements}}) {            !!!cp ('t87.2');
4440            if ($_->[1] eq 'p') {            #
4441              !!!back-token;          } elsif ({
4442              $token = {type => 'end tag', tag_name => 'p'};                    b => 1, big => 1, blockquote => 1, body => 1, br => 1,
4443              return;                    center => 1, code => 1, dd => 1, div => 1, dl => 1, dt => 1,
4444            } elsif ({                    em => 1, embed => 1, font => 1, h1 => 1, h2 => 1, h3 => 1,
4445                      table => 1, caption => 1, td => 1, th => 1,                    h4 => 1, h5 => 1, h6 => 1, head => 1, hr => 1, i => 1,
4446                      button => 1, marquee => 1, object => 1, html => 1,                    img => 1, li => 1, listing => 1, menu => 1, meta => 1,
4447                     }->{$_->[1]}) {                    nobr => 1, ol => 1, p => 1, pre => 1, ruby => 1, s => 1,
4448              last INSCOPE;                    small => 1, span => 1, strong => 1, strike => 1, sub => 1,
4449                      sup => 1, table => 1, tt => 1, u => 1, ul => 1, var => 1,
4450                     }->{$token->{tag_name}}) {
4451              !!!cp ('t87.2');
4452              !!!parse-error (type => 'not closed',
4453                              text => $self->{open_elements}->[-1]->[0]
4454                                  ->manakai_local_name,
4455                              token => $token);
4456    
4457              pop @{$self->{open_elements}}
4458                  while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;
4459    
4460              $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
4461              ## Reprocess.
4462              next B;
4463            } else {
4464              my $nsuri = $self->{open_elements}->[-1]->[0]->namespace_uri;
4465              my $tag_name = $token->{tag_name};
4466              if ($nsuri eq $SVG_NS) {
4467                $tag_name = {
4468                   altglyph => 'altGlyph',
4469                   altglyphdef => 'altGlyphDef',
4470                   altglyphitem => 'altGlyphItem',
4471                   animatecolor => 'animateColor',
4472                   animatemotion => 'animateMotion',
4473                   animatetransform => 'animateTransform',
4474                   clippath => 'clipPath',
4475                   feblend => 'feBlend',
4476                   fecolormatrix => 'feColorMatrix',
4477                   fecomponenttransfer => 'feComponentTransfer',
4478                   fecomposite => 'feComposite',
4479                   feconvolvematrix => 'feConvolveMatrix',
4480                   fediffuselighting => 'feDiffuseLighting',
4481                   fedisplacementmap => 'feDisplacementMap',
4482                   fedistantlight => 'feDistantLight',
4483                   feflood => 'feFlood',
4484                   fefunca => 'feFuncA',
4485                   fefuncb => 'feFuncB',
4486                   fefuncg => 'feFuncG',
4487                   fefuncr => 'feFuncR',
4488                   fegaussianblur => 'feGaussianBlur',
4489                   feimage => 'feImage',
4490                   femerge => 'feMerge',
4491                   femergenode => 'feMergeNode',
4492                   femorphology => 'feMorphology',
4493                   feoffset => 'feOffset',
4494                   fepointlight => 'fePointLight',
4495                   fespecularlighting => 'feSpecularLighting',
4496                   fespotlight => 'feSpotLight',
4497                   fetile => 'feTile',
4498                   feturbulence => 'feTurbulence',
4499                   foreignobject => 'foreignObject',
4500                   glyphref => 'glyphRef',
4501                   lineargradient => 'linearGradient',
4502                   radialgradient => 'radialGradient',
4503                   #solidcolor => 'solidColor', ## NOTE: Commented in spec (SVG1.2)
4504                   textpath => 'textPath',  
4505                }->{$tag_name} || $tag_name;
4506            }            }
4507          } # INSCOPE  
4508                        ## "adjust SVG attributes" (SVG only) - done in insert-element-f
4509          !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
4510          if ($token->{tag_name} eq 'pre') {            ## "adjust foreign attributes" - done in insert-element-f
4511            !!!next-token;  
4512            if ($token->{type} eq 'character') {            !!!insert-element-f ($nsuri, $tag_name, $token->{attributes}, $token);
4513              $token->{data} =~ s/^\x0A//;  
4514              unless (length $token->{data}) {            if ($self->{self_closing}) {
4515                !!!next-token;              pop @{$self->{open_elements}};
4516              }              !!!ack ('t87.3');
4517              } else {
4518                !!!cp ('t87.4');
4519            }            }
4520          } else {  
           !!!next-token;  
         }  
         return;  
       } elsif ($token->{tag_name} eq 'form') {  
         if (defined $self->{form_element}) {  
           !!!parse-error (type => 'in form:form');  
           ## Ignore the token  
           !!!next-token;  
           return;  
         } else {  
           ## has a p element in scope  
           INSCOPE: for (reverse @{$self->{open_elements}}) {  
             if ($_->[1] eq 'p') {  
               !!!back-token;  
               $token = {type => 'end tag', tag_name => 'p'};  
               return;  
             } elsif ({  
                       table => 1, caption => 1, td => 1, th => 1,  
                       button => 1, marquee => 1, object => 1, html => 1,  
                      }->{$_->[1]}) {  
               last INSCOPE;  
             }  
           } # INSCOPE  
               
           !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
           $self->{form_element} = $self->{open_elements}->[-1]->[0];  
4521            !!!next-token;            !!!next-token;
4522            return;            next B;
4523          }          }
4524        } elsif ($token->{tag_name} eq 'li') {        } elsif ($token->{type} == END_TAG_TOKEN) {
4525          ## has a p element in scope          ## NOTE: "using the rules for secondary insertion mode" then "continue"
4526          INSCOPE: for (reverse @{$self->{open_elements}}) {          !!!cp ('t87.5');
4527            if ($_->[1] eq 'p') {          #
4528              !!!back-token;        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
4529              $token = {type => 'end tag', tag_name => 'p'};          !!!cp ('t87.6');
4530              return;          !!!parse-error (type => 'not closed',
4531            } elsif ({                          text => $self->{open_elements}->[-1]->[0]
4532                      table => 1, caption => 1, td => 1, th => 1,                              ->manakai_local_name,
4533                      button => 1, marquee => 1, object => 1, html => 1,                          token => $token);
4534                     }->{$_->[1]}) {  
4535              last INSCOPE;          pop @{$self->{open_elements}}
4536            }              while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;
4537          } # INSCOPE  
4538                      ## NOTE: |<span><svg>| ... two parse errors, |<svg>| ... a parse error.
4539          ## Step 1  
4540          my $i = -1;          $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
4541          my $node = $self->{open_elements}->[$i];          ## Reprocess.
4542          LI: {          next B;
4543            ## Step 2        } else {
4544            if ($node->[1] eq 'li') {          die "$0: $token->{type}: Unknown token type";        
4545              if ($i != -1) {        }
4546                !!!parse-error (type => 'end tag missing:'.      }
4547                                $self->{open_elements}->[-1]->[1]);  
4548              }      if ($self->{insertion_mode} & HEAD_IMS) {
4549              splice @{$self->{open_elements}}, $i;        if ($token->{type} == CHARACTER_TOKEN) {
4550              last LI;          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
4551            }            unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4552                          if ($self->{head_element_inserted}) {
4553            ## Step 3                !!!cp ('t88.3');
4554            if (not $formatting_category->{$node->[1]} and                $self->{open_elements}->[-1]->[0]->append_child
4555                #not $phrasing_category->{$node->[1]} and                  ($self->{document}->create_text_node ($1));
4556                ($special_category->{$node->[1]} or                delete $self->{head_element_inserted};
4557                 $scoping_category->{$node->[1]}) and                ## NOTE: |</head> <link> |
4558                $node->[1] ne 'address' and $node->[1] ne 'div') {                #
4559              last LI;              } else {
4560            }                !!!cp ('t88.2');
4561                            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
4562            ## Step 4                ## NOTE: |</head> &#x20;|
4563            $i--;                #
           $node = $self->{open_elements}->[$i];  
           redo LI;  
         } # LI  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'dd' or $token->{tag_name} eq 'dt') {  
         ## has a p element in scope  
         INSCOPE: for (reverse @{$self->{open_elements}}) {  
           if ($_->[1] eq 'p') {  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'p'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$_->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         ## Step 1  
         my $i = -1;  
         my $node = $self->{open_elements}->[$i];  
         LI: {  
           ## Step 2  
           if ($node->[1] eq 'dt' or $node->[1] eq 'dd') {  
             if ($i != -1) {  
               !!!parse-error (type => 'end tag missing:'.  
                               $self->{open_elements}->[-1]->[1]);  
4564              }              }
4565              splice @{$self->{open_elements}}, $i;            } else {
4566              last LI;              !!!cp ('t88.1');
4567            }              ## Ignore the token.
4568                          #
           ## Step 3  
           if (not $formatting_category->{$node->[1]} and  
               #not $phrasing_category->{$node->[1]} and  
               ($special_category->{$node->[1]} or  
                $scoping_category->{$node->[1]}) and  
               $node->[1] ne 'address' and $node->[1] ne 'div') {  
             last LI;  
           }  
             
           ## Step 4  
           $i--;  
           $node = $self->{open_elements}->[$i];  
           redo LI;  
         } # LI  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'plaintext') {  
         ## has a p element in scope  
         INSCOPE: for (reverse @{$self->{open_elements}}) {  
           if ($_->[1] eq 'p') {  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'p'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$_->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
             
         $self->{content_model} = PLAINTEXT_CONTENT_MODEL;  
             
         !!!next-token;  
         return;  
       } elsif ({  
                 h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,  
                }->{$token->{tag_name}}) {  
         ## has a p element in scope  
         INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
           my $node = $self->{open_elements}->[$_];  
           if ($node->[1] eq 'p') {  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'p'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
             last INSCOPE;  
4569            }            }
4570          } # INSCOPE            unless (length $token->{data}) {
4571                          !!!cp ('t88');
4572          ## NOTE: See <http://html5.org/tools/web-apps-tracker?from=925&to=926>              !!!next-token;
4573          ## has an element in scope              next B;
         #my $i;  
         #INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
         #  my $node = $self->{open_elements}->[$_];  
         #  if ({  
         #       h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,  
         #      }->{$node->[1]}) {  
         #    $i = $_;  
         #    last INSCOPE;  
         #  } elsif ({  
         #            table => 1, caption => 1, td => 1, th => 1,  
         #            button => 1, marquee => 1, object => 1, html => 1,  
         #           }->{$node->[1]}) {  
         #    last INSCOPE;  
         #  }  
         #} # INSCOPE  
         #    
         #if (defined $i) {  
         #  !!! parse-error (type => 'in hn:hn');  
         #  splice @{$self->{open_elements}}, $i;  
         #}  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
             
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'a') {  
         AFE: for my $i (reverse 0..$#$active_formatting_elements) {  
           my $node = $active_formatting_elements->[$i];  
           if ($node->[1] eq 'a') {  
             !!!parse-error (type => 'in a:a');  
               
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'a'};  
             $formatting_end_tag->($token->{tag_name});  
               
             AFE2: for (reverse 0..$#$active_formatting_elements) {  
               if ($active_formatting_elements->[$_]->[0] eq $node->[0]) {  
                 splice @$active_formatting_elements, $_, 1;  
                 last AFE2;  
               }  
             } # AFE2  
             OE: for (reverse 0..$#{$self->{open_elements}}) {  
               if ($self->{open_elements}->[$_]->[0] eq $node->[0]) {  
                 splice @{$self->{open_elements}}, $_, 1;  
                 last OE;  
               }  
             } # OE  
             last AFE;  
           } elsif ($node->[0] eq '#marker') {  
             last AFE;  
4574            }            }
4575          } # AFE  ## TODO: set $token->{column} appropriately
4576                      }
         $reconstruct_active_formatting_elements->($insert_to_current);  
4577    
4578          !!!insert-element-t ($token->{tag_name}, $token->{attributes});          if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4579          push @$active_formatting_elements, $self->{open_elements}->[-1];            !!!cp ('t89');
4580              ## As if <head>
4581              !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4582              $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4583              push @{$self->{open_elements}},
4584                  [$self->{head_element}, $el_category->{head}];
4585    
4586          !!!next-token;            ## Reprocess in the "in head" insertion mode...
4587          return;            pop @{$self->{open_elements}};
       } elsif ({  
                 b => 1, big => 1, em => 1, font => 1, i => 1,  
                 s => 1, small => 1, strile => 1,  
                 strong => 1, tt => 1, u => 1,  
                }->{$token->{tag_name}}) {  
         $reconstruct_active_formatting_elements->($insert_to_current);  
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         push @$active_formatting_elements, $self->{open_elements}->[-1];  
           
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'nobr') {  
         $reconstruct_active_formatting_elements->($insert_to_current);  
4588    
4589          ## has a |nobr| element in scope            ## Reprocess in the "after head" insertion mode...
4590          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4591            my $node = $self->{open_elements}->[$_];            !!!cp ('t90');
4592            if ($node->[1] eq 'nobr') {            ## As if </noscript>
4593              !!!parse-error (type => 'not closed:nobr');            pop @{$self->{open_elements}};
4594              !!!back-token;            !!!parse-error (type => 'in noscript:#text', token => $token);
             $token = {type => 'end tag', tag_name => 'nobr'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         push @$active_formatting_elements, $self->{open_elements}->[-1];  
           
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'button') {  
         ## has a button element in scope  
         INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
           my $node = $self->{open_elements}->[$_];  
           if ($node->[1] eq 'button') {  
             !!!parse-error (type => 'in button:button');  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'button'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         $reconstruct_active_formatting_elements->($insert_to_current);  
4595                        
4596          !!!insert-element-t ($token->{tag_name}, $token->{attributes});            ## Reprocess in the "in head" insertion mode...
4597          push @$active_formatting_elements, ['#marker', ''];            ## As if </head>
4598              pop @{$self->{open_elements}};
4599    
4600          !!!next-token;            ## Reprocess in the "after head" insertion mode...
4601          return;          } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4602        } elsif ($token->{tag_name} eq 'marquee' or            !!!cp ('t91');
4603                 $token->{tag_name} eq 'object') {            pop @{$self->{open_elements}};
         $reconstruct_active_formatting_elements->($insert_to_current);  
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         push @$active_formatting_elements, ['#marker', ''];  
           
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'xmp') {  
         $reconstruct_active_formatting_elements->($insert_to_current);  
         $parse_rcdata->(CDATA_CONTENT_MODEL, $insert);  
         return;  
       } elsif ($token->{tag_name} eq 'table') {  
         ## has a p element in scope  
         INSCOPE: for (reverse @{$self->{open_elements}}) {  
           if ($_->[1] eq 'p') {  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'p'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$_->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
             
         $self->{insertion_mode} = 'in table';  
             
         !!!next-token;  
         return;  
       } elsif ({  
                 area => 1, basefont => 1, bgsound => 1, br => 1,  
                 embed => 1, img => 1, param => 1, spacer => 1, wbr => 1,  
                 image => 1,  
                }->{$token->{tag_name}}) {  
         if ($token->{tag_name} eq 'image') {  
           !!!parse-error (type => 'image');  
           $token->{tag_name} = 'img';  
         }  
4604    
4605          ## NOTE: There is an "as if <br>" code clone.            ## Reprocess in the "after head" insertion mode...
         $reconstruct_active_formatting_elements->($insert_to_current);  
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         pop @{$self->{open_elements}};  
           
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'hr') {  
         ## has a p element in scope  
         INSCOPE: for (reverse @{$self->{open_elements}}) {  
           if ($_->[1] eq 'p') {  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'p'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$_->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         pop @{$self->{open_elements}};  
             
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'input') {  
         $reconstruct_active_formatting_elements->($insert_to_current);  
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         ## TODO: associate with $self->{form_element} if defined  
         pop @{$self->{open_elements}};  
           
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'isindex') {  
         !!!parse-error (type => 'isindex');  
           
         if (defined $self->{form_element}) {  
           ## Ignore the token  
           !!!next-token;  
           return;  
4606          } else {          } else {
4607            my $at = $token->{attributes};            !!!cp ('t92');
           my $form_attrs;  
           $form_attrs->{action} = $at->{action} if $at->{action};  
           my $prompt_attr = $at->{prompt};  
           $at->{name} = {name => 'name', value => 'isindex'};  
           delete $at->{action};  
           delete $at->{prompt};  
           my @tokens = (  
                         {type => 'start tag', tag_name => 'form',  
                          attributes => $form_attrs},  
                         {type => 'start tag', tag_name => 'hr'},  
                         {type => 'start tag', tag_name => 'p'},  
                         {type => 'start tag', tag_name => 'label'},  
                        );  
           if ($prompt_attr) {  
             push @tokens, {type => 'character', data => $prompt_attr->{value}};  
           } else {  
             push @tokens, {type => 'character',  
                            data => 'This is a searchable index. Insert your search keywords here: '}; # SHOULD  
             ## TODO: make this configurable  
           }  
           push @tokens,  
                         {type => 'start tag', tag_name => 'input', attributes => $at},  
                         #{type => 'character', data => ''}, # SHOULD  
                         {type => 'end tag', tag_name => 'label'},  
                         {type => 'end tag', tag_name => 'p'},  
                         {type => 'start tag', tag_name => 'hr'},  
                         {type => 'end tag', tag_name => 'form'};  
           $token = shift @tokens;  
           !!!back-token (@tokens);  
           return;  
4608          }          }
4609        } elsif ($token->{tag_name} eq 'textarea') {  
4610          my $tag_name = $token->{tag_name};          ## "after head" insertion mode
4611          my $el;          ## As if <body>
4612          !!!create-element ($el, $token->{tag_name}, $token->{attributes});          !!!insert-element ('body',, $token);
4613                    $self->{insertion_mode} = IN_BODY_IM;
4614          ## TODO: $self->{form_element} if defined          ## reprocess
4615          $self->{content_model} = RCDATA_CONTENT_MODEL;          next B;
4616          delete $self->{escape}; # MUST        } elsif ($token->{type} == START_TAG_TOKEN) {
4617                    if ($token->{tag_name} eq 'head') {
4618          $insert->($el);            if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4619                        !!!cp ('t93');
4620          my $text = '';              !!!create-element ($self->{head_element}, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);
4621          !!!next-token;              $self->{open_elements}->[-1]->[0]->append_child
4622          if ($token->{type} eq 'character') {                  ($self->{head_element});
4623            $token->{data} =~ s/^\x0A//;              push @{$self->{open_elements}},
4624            unless (length $token->{data}) {                  [$self->{head_element}, $el_category->{head}];
4625                $self->{insertion_mode} = IN_HEAD_IM;
4626                !!!nack ('t93.1');
4627                !!!next-token;
4628                next B;
4629              } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4630                !!!cp ('t93.2');
4631                !!!parse-error (type => 'after head', text => 'head',
4632                                token => $token);
4633                ## Ignore the token
4634                !!!nack ('t93.3');
4635                !!!next-token;
4636                next B;
4637              } else {
4638                !!!cp ('t95');
4639                !!!parse-error (type => 'in head:head',
4640                                token => $token); # or in head noscript
4641                ## Ignore the token
4642                !!!nack ('t95.1');
4643              !!!next-token;              !!!next-token;
4644                next B;
4645            }            }
4646          }          } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4647          while ($token->{type} eq 'character') {            !!!cp ('t96');
4648            $text .= $token->{data};            ## As if <head>
4649            !!!next-token;            !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4650          }            $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4651          if (length $text) {            push @{$self->{open_elements}},
4652            $el->manakai_append_text ($text);                [$self->{head_element}, $el_category->{head}];
4653          }  
4654                      $self->{insertion_mode} = IN_HEAD_IM;
4655          $self->{content_model} = PCDATA_CONTENT_MODEL;            ## Reprocess in the "in head" insertion mode...
4656                    } else {
4657          if ($token->{type} eq 'end tag' and            !!!cp ('t97');
4658              $token->{tag_name} eq $tag_name) {          }
4659            ## Ignore the token  
4660          } else {          if ($token->{tag_name} eq 'base') {
4661            !!!parse-error (type => 'in RCDATA:#'.$token->{type});            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4662          }              !!!cp ('t98');
4663          !!!next-token;              ## As if </noscript>
4664          return;              pop @{$self->{open_elements}};
4665        } elsif ({              !!!parse-error (type => 'in noscript', text => 'base',
4666                  iframe => 1,                              token => $token);
4667                  noembed => 1,            
4668                  noframes => 1,              $self->{insertion_mode} = IN_HEAD_IM;
4669                  noscript => 0, ## TODO: 1 if scripting is enabled              ## Reprocess in the "in head" insertion mode...
4670                 }->{$token->{tag_name}}) {            } else {
4671          ## NOTE: There are two "as if in body" code clones.              !!!cp ('t99');
         $parse_rcdata->(CDATA_CONTENT_MODEL, $insert);  
         return;  
       } elsif ($token->{tag_name} eq 'select') {  
         $reconstruct_active_formatting_elements->($insert_to_current);  
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
           
         $self->{insertion_mode} = 'in select';  
         !!!next-token;  
         return;  
       } elsif ({  
                 caption => 1, col => 1, colgroup => 1, frame => 1,  
                 frameset => 1, head => 1, option => 1, optgroup => 1,  
                 tbody => 1, td => 1, tfoot => 1, th => 1,  
                 thead => 1, tr => 1,  
                }->{$token->{tag_name}}) {  
         !!!parse-error (type => 'in body:'.$token->{tag_name});  
         ## Ignore the token  
         !!!next-token;  
         return;  
           
         ## ISSUE: An issue on HTML5 new elements in the spec.  
       } else {  
         $reconstruct_active_formatting_elements->($insert_to_current);  
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
           
         !!!next-token;  
         return;  
       }  
     } elsif ($token->{type} eq 'end tag') {  
       if ($token->{tag_name} eq 'body') {  
         if (@{$self->{open_elements}} > 1 and  
             $self->{open_elements}->[1]->[1] eq 'body') {  
           for (@{$self->{open_elements}}) {  
             unless ({  
                        dd => 1, dt => 1, li => 1, p => 1, td => 1,  
                        th => 1, tr => 1, body => 1, html => 1,  
                      tbody => 1, tfoot => 1, thead => 1,  
                     }->{$_->[1]}) {  
               !!!parse-error (type => 'not closed:'.$_->[1]);  
             }  
4672            }            }
4673    
4674            $self->{insertion_mode} = 'after body';            ## NOTE: There is a "as if in head" code clone.
4675            !!!next-token;            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4676            return;              !!!cp ('t100');
4677          } else {              !!!parse-error (type => 'after head',
4678            !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                              text => $token->{tag_name}, token => $token);
4679            ## Ignore the token              push @{$self->{open_elements}},
4680            !!!next-token;                  [$self->{head_element}, $el_category->{head}];
4681            return;              $self->{head_element_inserted} = 1;
4682          }            } else {
4683        } elsif ($token->{tag_name} eq 'html') {              !!!cp ('t101');
         if (@{$self->{open_elements}} > 1 and $self->{open_elements}->[1]->[1] eq 'body') {  
           ## ISSUE: There is an issue in the spec.  
           if ($self->{open_elements}->[-1]->[1] ne 'body') {  
             !!!parse-error (type => 'not closed:'.$self->{open_elements}->[1]->[1]);  
4684            }            }
4685            $self->{insertion_mode} = 'after body';            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4686            ## reprocess            pop @{$self->{open_elements}};
4687            return;            pop @{$self->{open_elements}} # <head>
4688          } else {                if $self->{insertion_mode} == AFTER_HEAD_IM;
4689            !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});            !!!nack ('t101.1');
           ## Ignore the token  
4690            !!!next-token;            !!!next-token;
4691            return;            next B;
4692          }          } elsif ($token->{tag_name} eq 'link') {
4693        } elsif ({            ## NOTE: There is a "as if in head" code clone.
4694                  address => 1, blockquote => 1, center => 1, dir => 1,            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4695                  div => 1, dl => 1, fieldset => 1, listing => 1,              !!!cp ('t102');
4696                  menu => 1, ol => 1, pre => 1, ul => 1,              !!!parse-error (type => 'after head',
4697                  p => 1,                              text => $token->{tag_name}, token => $token);
4698                  dd => 1, dt => 1, li => 1,              push @{$self->{open_elements}},
4699                  button => 1, marquee => 1, object => 1,                  [$self->{head_element}, $el_category->{head}];
4700                 }->{$token->{tag_name}}) {              $self->{head_element_inserted} = 1;
         ## has an element in scope  
         my $i;  
         INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
           my $node = $self->{open_elements}->[$_];  
           if ($node->[1] eq $token->{tag_name}) {  
             ## generate implied end tags  
             if ({  
                  dd => ($token->{tag_name} ne 'dd'),  
                  dt => ($token->{tag_name} ne 'dt'),  
                  li => ($token->{tag_name} ne 'li'),  
                  p => ($token->{tag_name} ne 'p'),  
                  td => 1, th => 1, tr => 1,  
                  tbody => 1, tfoot=> 1, thead => 1,  
                 }->{$self->{open_elements}->[-1]->[1]}) {  
               !!!back-token;  
               $token = {type => 'end tag',  
                         tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
               return;  
             }  
             $i = $_;  
             last INSCOPE unless $token->{tag_name} eq 'p';  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
           
         if ($self->{open_elements}->[-1]->[1] ne $token->{tag_name}) {  
           if (defined $i) {  
             !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
4701            } else {            } else {
4702              !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});              !!!cp ('t103');
4703            }            }
4704          }            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
           
         if (defined $i) {  
           splice @{$self->{open_elements}}, $i;  
         } elsif ($token->{tag_name} eq 'p') {  
           ## As if <p>, then reprocess the current token  
           my $el;  
           !!!create-element ($el, 'p');  
           $insert->($el);  
         }  
         $clear_up_to_marker->()  
           if {  
             button => 1, marquee => 1, object => 1,  
           }->{$token->{tag_name}};  
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'form') {  
         ## has an element in scope  
         INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
           my $node = $self->{open_elements}->[$_];  
           if ($node->[1] eq $token->{tag_name}) {  
             ## generate implied end tags  
             if ({  
                  dd => 1, dt => 1, li => 1, p => 1,  
                  td => 1, th => 1, tr => 1,  
                  tbody => 1, tfoot=> 1, thead => 1,  
                 }->{$self->{open_elements}->[-1]->[1]}) {  
               !!!back-token;  
               $token = {type => 'end tag',  
                         tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
               return;  
             }  
             last INSCOPE;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
           
         if ($self->{open_elements}->[-1]->[1] eq $token->{tag_name}) {  
4705            pop @{$self->{open_elements}};            pop @{$self->{open_elements}};
4706          } else {            pop @{$self->{open_elements}} # <head>
4707            !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                if $self->{insertion_mode} == AFTER_HEAD_IM;
4708          }            !!!ack ('t103.1');
4709              !!!next-token;
4710              next B;
4711            } elsif ($token->{tag_name} eq 'command' or
4712                     $token->{tag_name} eq 'eventsource') {
4713              if ($self->{insertion_mode} == IN_HEAD_IM) {
4714                ## NOTE: If the insertion mode at the time of the emission
4715                ## of the token was "before head", $self->{insertion_mode}
4716                ## is already changed to |IN_HEAD_IM|.
4717    
4718          undef $self->{form_element};              ## NOTE: There is a "as if in head" code clone.
4719          !!!next-token;              !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4720          return;              pop @{$self->{open_elements}};
4721        } elsif ({              pop @{$self->{open_elements}} # <head>
4722                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,                  if $self->{insertion_mode} == AFTER_HEAD_IM;
4723                 }->{$token->{tag_name}}) {              !!!ack ('t103.2');
4724          ## has an element in scope              !!!next-token;
4725          my $i;              next B;
4726          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {            } else {
4727            my $node = $self->{open_elements}->[$_];              ## NOTE: "in head noscript" or "after head" insertion mode
4728            if ({              ## - in these cases, these tags are treated as same as
4729                 h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,              ## normal in-body tags.
4730                }->{$node->[1]}) {              !!!cp ('t103.3');
4731              ## generate implied end tags              #
             if ({  
                  dd => 1, dt => 1, li => 1, p => 1,  
                  td => 1, th => 1, tr => 1,  
                  tbody => 1, tfoot=> 1, thead => 1,  
                 }->{$self->{open_elements}->[-1]->[1]}) {  
               !!!back-token;  
               $token = {type => 'end tag',  
                         tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
               return;  
             }  
             $i = $_;  
             last INSCOPE;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
             last INSCOPE;  
4732            }            }
4733          } # INSCOPE          } elsif ($token->{tag_name} eq 'meta') {
4734                      ## NOTE: There is a "as if in head" code clone.
4735          if ($self->{open_elements}->[-1]->[1] ne $token->{tag_name}) {            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4736            !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);              !!!cp ('t104');
4737          }              !!!parse-error (type => 'after head',
4738                                        text => $token->{tag_name}, token => $token);
4739          splice @{$self->{open_elements}}, $i if defined $i;              push @{$self->{open_elements}},
4740          !!!next-token;                  [$self->{head_element}, $el_category->{head}];
4741          return;              $self->{head_element_inserted} = 1;
4742        } elsif ({            } else {
4743                  a => 1,              !!!cp ('t105');
4744                  b => 1, big => 1, em => 1, font => 1, i => 1,            }
4745                  nobr => 1, s => 1, small => 1, strile => 1,            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4746                  strong => 1, tt => 1, u => 1,            my $meta_el = pop @{$self->{open_elements}};
                }->{$token->{tag_name}}) {  
         $formatting_end_tag->($token->{tag_name});  
         return;  
       } elsif ($token->{tag_name} eq 'br') {  
         !!!parse-error (type => 'unmatched end tag:br');  
4747    
4748          ## As if <br>                unless ($self->{confident}) {
4749          $reconstruct_active_formatting_elements->($insert_to_current);                  if ($token->{attributes}->{charset}) {
4750                              !!!cp ('t106');
4751          my $el;                    ## NOTE: Whether the encoding is supported or not is handled
4752          !!!create-element ($el, 'br');                    ## in the {change_encoding} callback.
4753          $insert->($el);                    $self->{change_encoding}
4754                                  ->($self, $token->{attributes}->{charset}->{value},
4755          ## Ignore the token.                           $token);
4756          !!!next-token;                    
4757          return;                    $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
4758        } elsif ({                        ->set_user_data (manakai_has_reference =>
4759                  caption => 1, col => 1, colgroup => 1, frame => 1,                                             $token->{attributes}->{charset}
4760                  frameset => 1, head => 1, option => 1, optgroup => 1,                                                 ->{has_reference});
4761                  tbody => 1, td => 1, tfoot => 1, th => 1,                  } elsif ($token->{attributes}->{content}) {
4762                  thead => 1, tr => 1,                    if ($token->{attributes}->{content}->{value}
4763                  area => 1, basefont => 1, bgsound => 1,                        =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
4764                  embed => 1, hr => 1, iframe => 1, image => 1,                            [\x09\x0A\x0C\x0D\x20]*=
4765                  img => 1, input => 1, isindex => 1, noembed => 1,                            [\x09\x0A\x0C\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
4766                  noframes => 1, param => 1, select => 1, spacer => 1,                            ([^"'\x09\x0A\x0C\x0D\x20]
4767                  table => 1, textarea => 1, wbr => 1,                             [^\x09\x0A\x0C\x0D\x20\x3B]*))/x) {
4768                  noscript => 0, ## TODO: if scripting is enabled                      !!!cp ('t107');
4769                 }->{$token->{tag_name}}) {                      ## NOTE: Whether the encoding is supported or not is handled
4770          !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                      ## in the {change_encoding} callback.
4771          ## Ignore the token                      $self->{change_encoding}
4772          !!!next-token;                          ->($self, defined $1 ? $1 : defined $2 ? $2 : $3,
4773          return;                             $token);
4774                                $meta_el->[0]->get_attribute_node_ns (undef, 'content')
4775          ## ISSUE: Issue on HTML5 new elements in spec                          ->set_user_data (manakai_has_reference =>
4776                                                         $token->{attributes}->{content}
4777        } else {                                                     ->{has_reference});
4778          ## Step 1                    } else {
4779          my $node_i = -1;                      !!!cp ('t108');
4780          my $node = $self->{open_elements}->[$node_i];                    }
4781                    }
4782                  } else {
4783                    if ($token->{attributes}->{charset}) {
4784                      !!!cp ('t109');
4785                      $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
4786                          ->set_user_data (manakai_has_reference =>
4787                                               $token->{attributes}->{charset}
4788                                                   ->{has_reference});
4789                    }
4790                    if ($token->{attributes}->{content}) {
4791                      !!!cp ('t110');
4792                      $meta_el->[0]->get_attribute_node_ns (undef, 'content')
4793                          ->set_user_data (manakai_has_reference =>
4794                                               $token->{attributes}->{content}
4795                                                   ->{has_reference});
4796                    }
4797                  }
4798    
4799          ## Step 2                pop @{$self->{open_elements}} # <head>
4800          S2: {                    if $self->{insertion_mode} == AFTER_HEAD_IM;
4801            if ($node->[1] eq $token->{tag_name}) {                !!!ack ('t110.1');
4802              ## Step 1                !!!next-token;
4803              ## generate implied end tags                next B;
4804              if ({          } elsif ($token->{tag_name} eq 'title') {
4805                   dd => 1, dt => 1, li => 1, p => 1,            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4806                   td => 1, th => 1, tr => 1,              !!!cp ('t111');
4807                   tbody => 1, tfoot=> 1, thead => 1,              ## As if </noscript>
4808                  }->{$self->{open_elements}->[-1]->[1]}) {              pop @{$self->{open_elements}};
4809                !!!back-token;              !!!parse-error (type => 'in noscript', text => 'title',
4810                $token = {type => 'end tag',                              token => $token);
4811                          tag_name => $self->{open_elements}->[-1]->[1]}; # MUST            
4812                return;              $self->{insertion_mode} = IN_HEAD_IM;
4813              }              ## Reprocess in the "in head" insertion mode...
4814                      } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4815              ## Step 2              !!!cp ('t112');
4816              if ($token->{tag_name} ne $self->{open_elements}->[-1]->[1]) {              !!!parse-error (type => 'after head',
4817                !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                              text => $token->{tag_name}, token => $token);
4818              }              push @{$self->{open_elements}},
4819                                [$self->{head_element}, $el_category->{head}];
4820              ## Step 3              $self->{head_element_inserted} = 1;
4821              splice @{$self->{open_elements}}, $node_i;            } else {
4822                !!!cp ('t113');
4823              }
4824    
4825              !!!next-token;            ## NOTE: There is a "as if in head" code clone.
4826              last S2;            $parse_rcdata->(RCDATA_CONTENT_MODEL);
4827              pop @{$self->{open_elements}} # <head>
4828                  if $self->{insertion_mode} == AFTER_HEAD_IM;
4829              next B;
4830            } elsif ($token->{tag_name} eq 'style' or
4831                     $token->{tag_name} eq 'noframes') {
4832              ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and
4833              ## insertion mode IN_HEAD_IM)
4834              ## NOTE: There is a "as if in head" code clone.
4835              if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4836                !!!cp ('t114');
4837                !!!parse-error (type => 'after head',
4838                                text => $token->{tag_name}, token => $token);
4839                push @{$self->{open_elements}},
4840                    [$self->{head_element}, $el_category->{head}];
4841                $self->{head_element_inserted} = 1;
4842            } else {            } else {
4843              ## Step 3              !!!cp ('t115');
             if (not $formatting_category->{$node->[1]} and  
                 #not $phrasing_category->{$node->[1]} and  
                 ($special_category->{$node->[1]} or  
                  $scoping_category->{$node->[1]})) {  
               !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
               ## Ignore the token  
               !!!next-token;  
               last S2;  
             }  
4844            }            }
4845              $parse_rcdata->(CDATA_CONTENT_MODEL);
4846              pop @{$self->{open_elements}} # <head>
4847                  if $self->{insertion_mode} == AFTER_HEAD_IM;
4848              next B;
4849                } elsif ($token->{tag_name} eq 'noscript') {
4850                  if ($self->{insertion_mode} == IN_HEAD_IM) {
4851                    !!!cp ('t116');
4852                    ## NOTE: and scripting is disalbed
4853                    !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4854                    $self->{insertion_mode} = IN_HEAD_NOSCRIPT_IM;
4855                    !!!nack ('t116.1');
4856                    !!!next-token;
4857                    next B;
4858                  } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4859                    !!!cp ('t117');
4860                    !!!parse-error (type => 'in noscript', text => 'noscript',
4861                                    token => $token);
4862                    ## Ignore the token
4863                    !!!nack ('t117.1');
4864                    !!!next-token;
4865                    next B;
4866                  } else {
4867                    !!!cp ('t118');
4868                    #
4869                  }
4870            } elsif ($token->{tag_name} eq 'script') {
4871              if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4872                !!!cp ('t119');
4873                ## As if </noscript>
4874                pop @{$self->{open_elements}};
4875                !!!parse-error (type => 'in noscript', text => 'script',
4876                                token => $token);
4877                        
4878            ## Step 4              $self->{insertion_mode} = IN_HEAD_IM;
4879            $node_i--;              ## Reprocess in the "in head" insertion mode...
4880            $node = $self->{open_elements}->[$node_i];            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4881                          !!!cp ('t120');
4882            ## Step 5;              !!!parse-error (type => 'after head',
4883            redo S2;                              text => $token->{tag_name}, token => $token);
4884          } # S2              push @{$self->{open_elements}},
4885          return;                  [$self->{head_element}, $el_category->{head}];
4886        }              $self->{head_element_inserted} = 1;
4887      }            } else {
4888    }; # $in_body              !!!cp ('t121');
4889              }
   B: {  
     if ($token->{type} eq 'DOCTYPE') {  
       !!!parse-error (type => 'DOCTYPE in the middle');  
       ## Ignore the token  
       ## Stay in the phase  
       !!!next-token;  
       redo B;  
     } elsif ($token->{type} eq 'end-of-file') {  
       if ($token->{insertion_mode} ne 'trailing end') {  
         ## Generate implied end tags  
         if ({  
              dd => 1, dt => 1, li => 1, p => 1, td => 1, th => 1, tr => 1,  
              tbody => 1, tfoot=> 1, thead => 1,  
             }->{$self->{open_elements}->[-1]->[1]}) {  
           !!!back-token;  
           $token = {type => 'end tag', tag_name => $self->{open_elements}->[-1]->[1]};  
           redo B;  
         }  
           
         if (@{$self->{open_elements}} > 2 or  
             (@{$self->{open_elements}} == 2 and $self->{open_elements}->[1]->[1] ne 'body')) {  
           !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
         } elsif (defined $self->{inner_html_node} and  
                  @{$self->{open_elements}} > 1 and  
                  $self->{open_elements}->[1]->[1] ne 'body') {  
           !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
         }  
   
         ## ISSUE: There is an issue in the spec.  
       }  
4890    
4891        ## Stop parsing            ## NOTE: There is a "as if in head" code clone.
4892        last B;            $script_start_tag->();
4893      } elsif ($token->{type} eq 'start tag' and            pop @{$self->{open_elements}} # <head>
4894               $token->{tag_name} eq 'html') {                if $self->{insertion_mode} == AFTER_HEAD_IM;
4895        if ($self->{insertion_mode} eq 'trailing end') {            next B;
4896          ## Turn into the main phase          } elsif ($token->{tag_name} eq 'body' or
4897          !!!parse-error (type => 'after html:html');                   $token->{tag_name} eq 'frameset') {
4898          $self->{insertion_mode} = $previous_insertion_mode;                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4899        }                  !!!cp ('t122');
4900                    ## As if </noscript>
4901                    pop @{$self->{open_elements}};
4902                    !!!parse-error (type => 'in noscript',
4903                                    text => $token->{tag_name}, token => $token);
4904                    
4905                    ## Reprocess in the "in head" insertion mode...
4906                    ## As if </head>
4907                    pop @{$self->{open_elements}};
4908                    
4909                    ## Reprocess in the "after head" insertion mode...
4910                  } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4911                    !!!cp ('t124');
4912                    pop @{$self->{open_elements}};
4913                    
4914                    ## Reprocess in the "after head" insertion mode...
4915                  } else {
4916                    !!!cp ('t125');
4917                  }
4918    
4919  ## ISSUE: "aa<html>" is not a parse error.                ## "after head" insertion mode
4920  ## ISSUE: "<html>" in fragment is not a parse error.                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4921        unless ($token->{first_start_tag}) {                if ($token->{tag_name} eq 'body') {
4922          !!!parse-error (type => 'not first start tag');                  !!!cp ('t126');
4923        }                  $self->{insertion_mode} = IN_BODY_IM;
4924        my $top_el = $self->{open_elements}->[0]->[0];                } elsif ($token->{tag_name} eq 'frameset') {
4925        for my $attr_name (keys %{$token->{attributes}}) {                  !!!cp ('t127');
4926          unless ($top_el->has_attribute_ns (undef, $attr_name)) {                  $self->{insertion_mode} = IN_FRAMESET_IM;
4927            $top_el->set_attribute_ns                } else {
4928              (undef, [undef, $attr_name],                  die "$0: tag name: $self->{tag_name}";
              $token->{attributes}->{$attr_name}->{value});  
         }  
       }  
       !!!next-token;  
       redo B;  
     } elsif ($token->{type} eq 'comment') {  
       my $comment = $self->{document}->create_comment ($token->{data});  
       if ($self->{insertion_mode} eq 'trailing end') {  
         $self->{document}->append_child ($comment);  
       } elsif ($self->{insertion_mode} eq 'after body') {  
         $self->{open_elements}->[0]->[0]->append_child ($comment);  
       } else {  
         $self->{open_elements}->[-1]->[0]->append_child ($comment);  
       }  
       !!!next-token;  
       redo B;  
     } elsif ($self->{insertion_mode} eq 'before head') {  
           if ($token->{type} eq 'character') {  
             if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {  
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);  
               unless (length $token->{data}) {  
                 !!!next-token;  
                 redo B;  
4929                }                }
4930              }                !!!nack ('t127.1');
             ## As if <head>  
             !!!create-element ($self->{head_element}, 'head');  
             $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});  
             push @{$self->{open_elements}}, [$self->{head_element}, 'head'];  
             $self->{insertion_mode} = 'in head';  
             ## reprocess  
             redo B;  
           } elsif ($token->{type} eq 'start tag') {  
             my $attr = $token->{tag_name} eq 'head' ? $token->{attributes} : {};  
             !!!create-element ($self->{head_element}, 'head', $attr);  
             $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});  
             push @{$self->{open_elements}}, [$self->{head_element}, 'head'];  
             $self->{insertion_mode} = 'in head';  
             if ($token->{tag_name} eq 'head') {  
4931                !!!next-token;                !!!next-token;
4932              #} elsif ({                next B;
             #          base => 1, link => 1, meta => 1,  
             #          script => 1, style => 1, title => 1,  
             #         }->{$token->{tag_name}}) {  
             #  ## reprocess  
4933              } else {              } else {
4934                ## reprocess                !!!cp ('t128');
4935                  #
4936              }              }
4937              redo B;  
4938            } elsif ($token->{type} eq 'end tag') {              if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4939              if ({                !!!cp ('t129');
4940                   head => 1, body => 1, html => 1,                ## As if </noscript>
4941                   p => 1, br => 1,                pop @{$self->{open_elements}};
4942                  }->{$token->{tag_name}}) {                !!!parse-error (type => 'in noscript:/',
4943                ## As if <head>                                text => $token->{tag_name}, token => $token);
4944                !!!create-element ($self->{head_element}, 'head');                
4945                $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});                ## Reprocess in the "in head" insertion mode...
4946                push @{$self->{open_elements}}, [$self->{head_element}, 'head'];                ## As if </head>
4947                $self->{insertion_mode} = 'in head';                pop @{$self->{open_elements}};
4948                ## reprocess  
4949                redo B;                ## Reprocess in the "after head" insertion mode...
4950                } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4951                  !!!cp ('t130');
4952                  ## As if </head>
4953                  pop @{$self->{open_elements}};
4954    
4955                  ## Reprocess in the "after head" insertion mode...
4956              } else {              } else {
4957                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                !!!cp ('t131');
               ## Ignore the token ## ISSUE: An issue in the spec.  
               !!!next-token;  
               redo B;  
             }  
           } else {  
             die "$0: $token->{type}: Unknown type";  
           }  
         } elsif ($self->{insertion_mode} eq 'in head' or  
                  $self->{insertion_mode} eq 'in head noscript' or  
                  $self->{insertion_mode} eq 'after head') {  
           if ($token->{type} eq 'character') {  
             if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {  
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);  
               unless (length $token->{data}) {  
                 !!!next-token;  
                 redo B;  
               }  
4958              }              }
               
             #  
           } elsif ($token->{type} eq 'start tag') {  
             if ({base => ($self->{insertion_mode} eq 'in head' or  
                           $self->{insertion_mode} eq 'after head'),  
                  link => 1}->{$token->{tag_name}}) {  
               ## NOTE: There is a "as if in head" code clone.  
               if ($self->{insertion_mode} eq 'after head') {  
                 !!!parse-error (type => 'after head:'.$token->{tag_name});  
                 push @{$self->{open_elements}}, [$self->{head_element}, 'head'];  
               }  
               !!!insert-element ($token->{tag_name}, $token->{attributes});  
               pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.  
               pop @{$self->{open_elements}}  
                   if $self->{insertion_mode} eq 'after head';  
               !!!next-token;  
               redo B;  
             } elsif ($token->{tag_name} eq 'meta') {  
               ## NOTE: There is a "as if in head" code clone.  
               if ($self->{insertion_mode} eq 'after head') {  
                 !!!parse-error (type => 'after head:'.$token->{tag_name});  
                 push @{$self->{open_elements}}, [$self->{head_element}, 'head'];  
               }  
               !!!insert-element ($token->{tag_name}, $token->{attributes});  
               pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.  
4959    
4960                unless ($self->{confident}) {              ## "after head" insertion mode
4961                  my $charset;              ## As if <body>
4962                  if ($token->{attributes}->{charset}) { ## TODO: And if supported              !!!insert-element ('body',, $token);
4963                    $charset = $token->{attributes}->{charset}->{value};              $self->{insertion_mode} = IN_BODY_IM;
4964                  }              ## reprocess
4965                  if ($token->{attributes}->{'http-equiv'}) {              !!!ack-later;
4966                    ## ISSUE: Algorithm name in the spec was incorrect so that not linked to the definition.              next B;
4967                    if ($token->{attributes}->{'http-equiv'}->{value}            } elsif ($token->{type} == END_TAG_TOKEN) {
4968                        =~ /\A[^;]*;[\x09-\x0D\x20]*charset[\x09-\x0D\x20]*=              if ($token->{tag_name} eq 'head') {
4969                            [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|                if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4970                            ([^"'\x09-\x0D\x20][^\x09-\x0D\x20]*))/x) {                  !!!cp ('t132');
4971                      $charset = defined $1 ? $1 : defined $2 ? $2 : $3;                  ## As if <head>
4972                    } ## TODO: And if supported                  !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4973                  }                  $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4974                  ## TODO: Change the encoding                  push @{$self->{open_elements}},
4975                }                      [$self->{head_element}, $el_category->{head}];
4976    
4977                ## TODO: Extracting |charset| from |meta|.                  ## Reprocess in the "in head" insertion mode...
4978                pop @{$self->{open_elements}}                  pop @{$self->{open_elements}};
4979                    if $self->{insertion_mode} eq 'after head';                  $self->{insertion_mode} = AFTER_HEAD_IM;
4980                !!!next-token;                  !!!next-token;
4981                redo B;                  next B;
4982              } elsif ($token->{tag_name} eq 'title' and                } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4983                       $self->{insertion_mode} eq 'in head') {                  !!!cp ('t133');
4984                ## NOTE: There is a "as if in head" code clone.                  ## As if </noscript>
4985                if ($self->{insertion_mode} eq 'after head') {                  pop @{$self->{open_elements}};
4986                  !!!parse-error (type => 'after head:'.$token->{tag_name});                  !!!parse-error (type => 'in noscript:/',
4987                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];                                  text => 'head', token => $token);
4988                }                  
4989                my $parent = defined $self->{head_element} ? $self->{head_element}                  ## Reprocess in the "in head" insertion mode...
4990                    : $self->{open_elements}->[-1]->[0];                  pop @{$self->{open_elements}};
4991                $parse_rcdata->(RCDATA_CONTENT_MODEL,                  $self->{insertion_mode} = AFTER_HEAD_IM;
4992                                sub { $parent->append_child ($_[0]) });                  !!!next-token;
4993                pop @{$self->{open_elements}}                  next B;
4994                    if $self->{insertion_mode} eq 'after head';                } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4995                redo B;                  !!!cp ('t134');
4996              } elsif ($token->{tag_name} eq 'style') {                  pop @{$self->{open_elements}};
4997                ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and                  $self->{insertion_mode} = AFTER_HEAD_IM;
4998                ## insertion mode 'in head')                  !!!next-token;
4999                ## NOTE: There is a "as if in head" code clone.                  next B;
5000                if ($self->{insertion_mode} eq 'after head') {                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
5001                  !!!parse-error (type => 'after head:'.$token->{tag_name});                  !!!cp ('t134.1');
5002                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];                  !!!parse-error (type => 'unmatched end tag', text => 'head',
5003                                    token => $token);
5004                    ## Ignore the token
5005                    !!!next-token;
5006                    next B;
5007                  } else {
5008                    die "$0: $self->{insertion_mode}: Unknown insertion mode";
5009                }                }
               $parse_rcdata->(CDATA_CONTENT_MODEL, $insert_to_current);  
               pop @{$self->{open_elements}}  
                   if $self->{insertion_mode} eq 'after head';  
               redo B;  
5010              } elsif ($token->{tag_name} eq 'noscript') {              } elsif ($token->{tag_name} eq 'noscript') {
5011                if ($self->{insertion_mode} eq 'in head') {                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5012                  ## NOTE: and scripting is disalbed                  !!!cp ('t136');
5013                  !!!insert-element ($token->{tag_name}, $token->{attributes});                  pop @{$self->{open_elements}};
5014                  $self->{insertion_mode} = 'in head noscript';                  $self->{insertion_mode} = IN_HEAD_IM;
5015                  !!!next-token;                  !!!next-token;
5016                  redo B;                  next B;
5017                } elsif ($self->{insertion_mode} eq 'in head noscript') {                } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM or
5018                  !!!parse-error (type => 'in noscript:noscript');                         $self->{insertion_mode} == AFTER_HEAD_IM) {
5019                  ## Ignore the token                  !!!cp ('t137');
5020                    !!!parse-error (type => 'unmatched end tag',
5021                                    text => 'noscript', token => $token);
5022                    ## Ignore the token ## ISSUE: An issue in the spec.
5023                  !!!next-token;                  !!!next-token;
5024                  redo B;                  next B;
5025                } else {                } else {
5026                    !!!cp ('t138');
5027                  #                  #
5028                }                }
5029              } elsif ($token->{tag_name} eq 'head' and              } elsif ({
5030                       $self->{insertion_mode} ne 'after head') {                        body => 1, html => 1,
5031                !!!parse-error (type => 'in head:head'); # or in head noscript                       }->{$token->{tag_name}}) {
5032                  if ($self->{insertion_mode} == BEFORE_HEAD_IM or
5033                      $self->{insertion_mode} == IN_HEAD_IM or
5034                      $self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5035                    !!!cp ('t140');
5036                    !!!parse-error (type => 'unmatched end tag',
5037                                    text => $token->{tag_name}, token => $token);
5038                    ## Ignore the token
5039                    !!!next-token;
5040                    next B;
5041                  } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
5042                    !!!cp ('t140.1');
5043                    !!!parse-error (type => 'unmatched end tag',
5044                                    text => $token->{tag_name}, token => $token);
5045                    ## Ignore the token
5046                    !!!next-token;
5047                    next B;
5048                  } else {
5049                    die "$0: $self->{insertion_mode}: Unknown insertion mode";
5050                  }
5051                } elsif ($token->{tag_name} eq 'p') {
5052                  !!!cp ('t142');
5053                  !!!parse-error (type => 'unmatched end tag',
5054                                  text => $token->{tag_name}, token => $token);
5055                ## Ignore the token                ## Ignore the token
5056                !!!next-token;                !!!next-token;
5057                redo B;                next B;
5058              } elsif ($self->{insertion_mode} ne 'in head noscript' and              } elsif ($token->{tag_name} eq 'br') {
5059                       $token->{tag_name} eq 'script') {                if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
5060                if ($self->{insertion_mode} eq 'after head') {                  !!!cp ('t142.2');
5061                  !!!parse-error (type => 'after head:'.$token->{tag_name});                  ## (before head) as if <head>, (in head) as if </head>
5062                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];                  !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
5063                    $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
5064                    $self->{insertion_mode} = AFTER_HEAD_IM;
5065      
5066                    ## Reprocess in the "after head" insertion mode...
5067                  } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
5068                    !!!cp ('t143.2');
5069                    ## As if </head>
5070                    pop @{$self->{open_elements}};
5071                    $self->{insertion_mode} = AFTER_HEAD_IM;
5072      
5073                    ## Reprocess in the "after head" insertion mode...
5074                  } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5075                    !!!cp ('t143.3');
5076                    ## ISSUE: Two parse errors for <head><noscript></br>
5077                    !!!parse-error (type => 'unmatched end tag',
5078                                    text => 'br', token => $token);
5079                    ## As if </noscript>
5080                    pop @{$self->{open_elements}};
5081                    $self->{insertion_mode} = IN_HEAD_IM;
5082    
5083                    ## Reprocess in the "in head" insertion mode...
5084                    ## As if </head>
5085                    pop @{$self->{open_elements}};
5086                    $self->{insertion_mode} = AFTER_HEAD_IM;
5087    
5088                    ## Reprocess in the "after head" insertion mode...
5089                  } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
5090                    !!!cp ('t143.4');
5091                    #
5092                  } else {
5093                    die "$0: $self->{insertion_mode}: Unknown insertion mode";
5094                }                }
5095                ## NOTE: There is a "as if in head" code clone.  
5096                $script_start_tag->($insert_to_current);                ## ISSUE: does not agree with IE7 - it doesn't ignore </br>.
5097                pop @{$self->{open_elements}}                !!!parse-error (type => 'unmatched end tag',
5098                    if $self->{insertion_mode} eq 'after head';                                text => 'br', token => $token);
5099                redo B;                ## Ignore the token
             } elsif ($self->{insertion_mode} eq 'after head' and  
                      $token->{tag_name} eq 'body') {  
               !!!insert-element ('body', $token->{attributes});  
               $self->{insertion_mode} = 'in body';  
               !!!next-token;  
               redo B;  
             } elsif ($self->{insertion_mode} eq 'after head' and  
                      $token->{tag_name} eq 'frameset') {  
               !!!insert-element ('frameset', $token->{attributes});  
               $self->{insertion_mode} = 'in frameset';  
5100                !!!next-token;                !!!next-token;
5101                redo B;                next B;
5102              } else {              } else {
5103                #                !!!cp ('t145');
5104                  !!!parse-error (type => 'unmatched end tag',
5105                                  text => $token->{tag_name}, token => $token);
5106                  ## Ignore the token
5107                  !!!next-token;
5108                  next B;
5109              }              }
5110            } elsif ($token->{type} eq 'end tag') {  
5111              if ($self->{insertion_mode} eq 'in head' and              if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5112                  $token->{tag_name} eq 'head') {                !!!cp ('t146');
5113                  ## As if </noscript>
5114                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
5115                $self->{insertion_mode} = 'after head';                !!!parse-error (type => 'in noscript:/',
5116                !!!next-token;                                text => $token->{tag_name}, token => $token);
5117                redo B;                
5118              } elsif ($self->{insertion_mode} eq 'in head noscript' and                ## Reprocess in the "in head" insertion mode...
5119                  $token->{tag_name} eq 'noscript') {                ## As if </head>
5120                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
5121                $self->{insertion_mode} = 'in head';  
5122                !!!next-token;                ## Reprocess in the "after head" insertion mode...
5123                redo B;              } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
5124              } elsif ($self->{insertion_mode} eq 'in head' and                !!!cp ('t147');
5125                       {                ## As if </head>
5126                        body => 1, html => 1,                pop @{$self->{open_elements}};
5127                        p => 1, br => 1,  
5128                       }->{$token->{tag_name}}) {                ## Reprocess in the "after head" insertion mode...
5129                #              } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {
5130              } elsif ($self->{insertion_mode} eq 'in head noscript' and  ## ISSUE: This case cannot be reached?
5131                       {                !!!cp ('t148');
5132                        p => 1, br => 1,                !!!parse-error (type => 'unmatched end tag',
5133                       }->{$token->{tag_name}}) {                                text => $token->{tag_name}, token => $token);
5134                #                ## Ignore the token ## ISSUE: An issue in the spec.
             } elsif ($self->{insertion_mode} ne 'after head') {  
               !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
               ## Ignore the token  
5135                !!!next-token;                !!!next-token;
5136                redo B;                next B;
5137              } else {              } else {
5138                #                !!!cp ('t149');
5139              }              }
           } else {  
             #  
           }  
5140    
5141            ## As if </head> or </noscript> or <body>              ## "after head" insertion mode
5142            if ($self->{insertion_mode} eq 'in head') {              ## As if <body>
5143              pop @{$self->{open_elements}};              !!!insert-element ('body',, $token);
5144              $self->{insertion_mode} = 'after head';              $self->{insertion_mode} = IN_BODY_IM;
5145            } elsif ($self->{insertion_mode} eq 'in head noscript') {              ## reprocess
5146              pop @{$self->{open_elements}};              next B;
5147              !!!parse-error (type => 'in noscript:'.(defined $token->{tag_name} ? ($token->{type} eq 'end tag' ? '/' : '') . $token->{tag_name} : '#' . $token->{type}));        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
5148              $self->{insertion_mode} = 'in head';          if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
5149            } else { # 'after head'            !!!cp ('t149.1');
5150              !!!insert-element ('body');  
5151              $self->{insertion_mode} = 'in body';            ## NOTE: As if <head>
5152            }            !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
5153            ## reprocess            $self->{open_elements}->[-1]->[0]->append_child
5154            redo B;                ($self->{head_element});
5155              #push @{$self->{open_elements}},
5156              #    [$self->{head_element}, $el_category->{head}];
5157              #$self->{insertion_mode} = IN_HEAD_IM;
5158              ## NOTE: Reprocess.
5159    
5160              ## NOTE: As if </head>
5161              #pop @{$self->{open_elements}};
5162              #$self->{insertion_mode} = IN_AFTER_HEAD_IM;
5163              ## NOTE: Reprocess.
5164              
5165              #
5166            } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
5167              !!!cp ('t149.2');
5168    
5169            ## ISSUE: An issue in the spec.            ## NOTE: As if </head>
5170          } elsif ($self->{insertion_mode} eq 'in body' or            pop @{$self->{open_elements}};
5171                   $self->{insertion_mode} eq 'in cell' or            #$self->{insertion_mode} = IN_AFTER_HEAD_IM;
5172                   $self->{insertion_mode} eq 'in caption') {            ## NOTE: Reprocess.
5173            if ($token->{type} eq 'character') {  
5174              #
5175            } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5176              !!!cp ('t149.3');
5177    
5178              !!!parse-error (type => 'in noscript:#eof', token => $token);
5179    
5180              ## As if </noscript>
5181              pop @{$self->{open_elements}};
5182              #$self->{insertion_mode} = IN_HEAD_IM;
5183              ## NOTE: Reprocess.
5184    
5185              ## NOTE: As if </head>
5186              pop @{$self->{open_elements}};
5187              #$self->{insertion_mode} = IN_AFTER_HEAD_IM;
5188              ## NOTE: Reprocess.
5189    
5190              #
5191            } else {
5192              !!!cp ('t149.4');
5193              #
5194            }
5195    
5196            ## NOTE: As if <body>
5197            !!!insert-element ('body',, $token);
5198            $self->{insertion_mode} = IN_BODY_IM;
5199            ## NOTE: Reprocess.
5200            next B;
5201          } else {
5202            die "$0: $token->{type}: Unknown token type";
5203          }
5204        } elsif ($self->{insertion_mode} & BODY_IMS) {
5205              if ($token->{type} == CHARACTER_TOKEN) {
5206                !!!cp ('t150');
5207              ## NOTE: There is a code clone of "character in body".              ## NOTE: There is a code clone of "character in body".
5208              $reconstruct_active_formatting_elements->($insert_to_current);              $reconstruct_active_formatting_elements->($insert_to_current);
5209                            
5210              $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});              $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
5211    
5212              !!!next-token;              !!!next-token;
5213              redo B;              next B;
5214            } elsif ($token->{type} eq 'start tag') {            } elsif ($token->{type} == START_TAG_TOKEN) {
5215              if ({              if ({
5216                   caption => 1, col => 1, colgroup => 1, tbody => 1,                   caption => 1, col => 1, colgroup => 1, tbody => 1,
5217                   td => 1, tfoot => 1, th => 1, thead => 1, tr => 1,                   td => 1, tfoot => 1, th => 1, thead => 1, tr => 1,
5218                  }->{$token->{tag_name}}) {                  }->{$token->{tag_name}}) {
5219                if ($self->{insertion_mode} eq 'in cell') {                if ($self->{insertion_mode} == IN_CELL_IM) {
5220                  ## have an element in table scope                  ## have an element in table scope
5221                  my $tn;                  for (reverse 0..$#{$self->{open_elements}}) {
                 INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
5222                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
5223                    if ($node->[1] eq 'td' or $node->[1] eq 'th') {                    if ($node->[1] & TABLE_CELL_EL) {
5224                      $tn = $node->[1];                      !!!cp ('t151');
5225                      last INSCOPE;  
5226                    } elsif ({                      ## Close the cell
5227                              table => 1, html => 1,                      !!!back-token; # <x>
5228                             }->{$node->[1]}) {                      $token = {type => END_TAG_TOKEN,
5229                      last INSCOPE;                                tag_name => $node->[0]->manakai_local_name,
5230                    }                                line => $token->{line},
5231                  } # INSCOPE                                column => $token->{column}};
5232                    unless (defined $tn) {                      next B;
5233                      !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                    } elsif ($node->[1] & TABLE_SCOPING_EL) {
5234                      ## Ignore the token                      !!!cp ('t152');
5235                      !!!next-token;                      ## ISSUE: This case can never be reached, maybe.
5236                      redo B;                      last;
5237                    }                    }
5238                    }
5239    
5240                    !!!cp ('t153');
5241                    !!!parse-error (type => 'start tag not allowed',
5242                        text => $token->{tag_name}, token => $token);
5243                    ## Ignore the token
5244                    !!!nack ('t153.1');
5245                    !!!next-token;
5246                    next B;
5247                  } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {
5248                    !!!parse-error (type => 'not closed', text => 'caption',
5249                                    token => $token);
5250                                    
5251                  ## Close the cell                  ## NOTE: As if </caption>.
                 !!!back-token; # <?>  
                 $token = {type => 'end tag', tag_name => $tn};  
                 redo B;  
               } elsif ($self->{insertion_mode} eq 'in caption') {  
                 !!!parse-error (type => 'not closed:caption');  
                   
                 ## As if </caption>  
5252                  ## have a table element in table scope                  ## have a table element in table scope
5253                  my $i;                  my $i;
5254                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: {
5255                    my $node = $self->{open_elements}->[$_];                    for (reverse 0..$#{$self->{open_elements}}) {
5256                    if ($node->[1] eq 'caption') {                      my $node = $self->{open_elements}->[$_];
5257                      $i = $_;                      if ($node->[1] & CAPTION_EL) {
5258                      last INSCOPE;                        !!!cp ('t155');
5259                    } elsif ({                        $i = $_;
5260                              table => 1, html => 1,                        last INSCOPE;
5261                             }->{$node->[1]}) {                      } elsif ($node->[1] & TABLE_SCOPING_EL) {
5262                      last INSCOPE;                        !!!cp ('t156');
5263                          last;
5264                        }
5265                    }                    }
5266    
5267                      !!!cp ('t157');
5268                      !!!parse-error (type => 'start tag not allowed',
5269                                      text => $token->{tag_name}, token => $token);
5270                      ## Ignore the token
5271                      !!!nack ('t157.1');
5272                      !!!next-token;
5273                      next B;
5274                  } # INSCOPE                  } # INSCOPE
                   unless (defined $i) {  
                     !!!parse-error (type => 'unmatched end tag:caption');  
                     ## Ignore the token  
                     !!!next-token;  
                     redo B;  
                   }  
5275                                    
5276                  ## generate implied end tags                  ## generate implied end tags
5277                  if ({                  while ($self->{open_elements}->[-1]->[1]
5278                       dd => 1, dt => 1, li => 1, p => 1,                             & END_TAG_OPTIONAL_EL) {
5279                       td => 1, th => 1, tr => 1,                    !!!cp ('t158');
5280                       tbody => 1, tfoot=> 1, thead => 1,                    pop @{$self->{open_elements}};
                     }->{$self->{open_elements}->[-1]->[1]}) {  
                   !!!back-token; # <?>  
                   $token = {type => 'end tag', tag_name => 'caption'};  
                   !!!back-token;  
                   $token = {type => 'end tag',  
                             tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
                   redo B;  
5281                  }                  }
5282    
5283                  if ($self->{open_elements}->[-1]->[1] ne 'caption') {                  unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
5284                    !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                    !!!cp ('t159');
5285                      !!!parse-error (type => 'not closed',
5286                                      text => $self->{open_elements}->[-1]->[0]
5287                                          ->manakai_local_name,
5288                                      token => $token);
5289                    } else {
5290                      !!!cp ('t160');
5291                  }                  }
5292                                    
5293                  splice @{$self->{open_elements}}, $i;                  splice @{$self->{open_elements}}, $i;
5294                                    
5295                  $clear_up_to_marker->();                  $clear_up_to_marker->();
5296                                    
5297                  $self->{insertion_mode} = 'in table';                  $self->{insertion_mode} = IN_TABLE_IM;
5298                                    
5299                  ## reprocess                  ## reprocess
5300                  redo B;                  !!!ack-later;
5301                    next B;
5302                } else {                } else {
5303                    !!!cp ('t161');
5304                  #                  #
5305                }                }
5306              } else {              } else {
5307                  !!!cp ('t162');
5308                #                #
5309              }              }
5310            } elsif ($token->{type} eq 'end tag') {            } elsif ($token->{type} == END_TAG_TOKEN) {
5311              if ($token->{tag_name} eq 'td' or $token->{tag_name} eq 'th') {              if ($token->{tag_name} eq 'td' or $token->{tag_name} eq 'th') {
5312                if ($self->{insertion_mode} eq 'in cell') {                if ($self->{insertion_mode} == IN_CELL_IM) {
5313                  ## have an element in table scope                  ## have an element in table scope
5314                  my $i;                  my $i;
5315                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5316                    my $node = $self->{open_elements}->[$_];                    my $node = $self->{open_elements}->[$_];
5317                    if ($node->[1] eq $token->{tag_name}) {                    if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
5318                        !!!cp ('t163');
5319                      $i = $_;                      $i = $_;
5320                      last INSCOPE;                      last INSCOPE;
5321                    } elsif ({                    } elsif ($node->[1] & TABLE_SCOPING_EL) {
5322                              table => 1, html => 1,                      !!!cp ('t164');
                            }->{$node->[1]}) {  
5323                      last INSCOPE;                      last INSCOPE;
5324                    }                    }
5325                  } # INSCOPE                  } # INSCOPE
5326                    unless (defined $i) {                    unless (defined $i) {
5327                      !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                      !!!cp ('t165');
5328                        !!!parse-error (type => 'unmatched end tag',
5329                                        text => $token->{tag_name},
5330                                        token => $token);
5331                      ## Ignore the token                      ## Ignore the token
5332                      !!!next-token;                      !!!next-token;
5333                      redo B;                      next B;
5334                    }                    }
5335                                    
5336                  ## generate implied end tags                  ## generate implied end tags
5337                  if ({                  while ($self->{open_elements}->[-1]->[1]
5338                       dd => 1, dt => 1, li => 1, p => 1,                             & END_TAG_OPTIONAL_EL) {
5339                       td => ($token->{tag_name} eq 'th'),                    !!!cp ('t166');
5340                       th => ($token->{tag_name} eq 'td'),                    pop @{$self->{open_elements}};
                      tr => 1,  
                      tbody => 1, tfoot=> 1, thead => 1,  
                     }->{$self->{open_elements}->[-1]->[1]}) {  
                   !!!back-token;  
                   $token = {type => 'end tag',  
                             tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
                   redo B;  
5341                  }                  }
5342                    
5343                  if ($self->{open_elements}->[-1]->[1] ne $token->{tag_name}) {                  if ($self->{open_elements}->[-1]->[0]->manakai_local_name
5344                    !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                          ne $token->{tag_name}) {
5345                      !!!cp ('t167');
5346                      !!!parse-error (type => 'not closed',
5347                                      text => $self->{open_elements}->[-1]->[0]
5348                                          ->manakai_local_name,
5349                                      token => $token);
5350                    } else {
5351                      !!!cp ('t168');
5352                  }                  }
5353                                    
5354                  splice @{$self->{open_elements}}, $i;                  splice @{$self->{open_elements}}, $i;
5355                                    
5356                  $clear_up_to_marker->();                  $clear_up_to_marker->();
5357                                    
5358                  $self->{insertion_mode} = 'in row';                  $self->{insertion_mode} = IN_ROW_IM;
5359                                    
5360                  !!!next-token;                  !!!next-token;
5361                  redo B;                  next B;
5362                } elsif ($self->{insertion_mode} eq 'in caption') {                } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {
5363                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                  !!!cp ('t169');
5364                    !!!parse-error (type => 'unmatched end tag',
5365                                    text => $token->{tag_name}, token => $token);
5366                  ## Ignore the token                  ## Ignore the token
5367                  !!!next-token;                  !!!next-token;
5368                  redo B;                  next B;
5369                } else {                } else {
5370                    !!!cp ('t170');
5371                  #                  #
5372                }                }
5373              } elsif ($token->{tag_name} eq 'caption') {              } elsif ($token->{tag_name} eq 'caption') {
5374                if ($self->{insertion_mode} eq 'in caption') {                if ($self->{insertion_mode} == IN_CAPTION_IM) {
5375                  ## have a table element in table scope                  ## have a table element in table scope
5376                  my $i;                  my $i;
5377                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  INSCOPE: {
5378                    my $node = $self->{open_elements}->[$_];                    for (reverse 0..$#{$self->{open_elements}}) {
5379                    if ($node->[1] eq $token->{tag_name}) {                      my $node = $self->{open_elements}->[$_];
5380                      $i = $_;                      if ($node->[1] & CAPTION_EL) {
5381                      last INSCOPE;                        !!!cp ('t171');
5382                    } elsif ({                        $i = $_;
5383                              table => 1, html => 1,                        last INSCOPE;
5384                             }->{$node->[1]}) {                      } elsif ($node->[1] & TABLE_SCOPING_EL) {
5385                      last INSCOPE;                        !!!cp ('t172');
5386                          last;
5387                        }
5388                    }                    }
5389    
5390                      !!!cp ('t173');
5391                      !!!parse-error (type => 'unmatched end tag',
5392                                      text => $token->{tag_name}, token => $token);
5393                      ## Ignore the token
5394                      !!!next-token;
5395                      next B;
5396                  } # INSCOPE                  } # INSCOPE
                   unless (defined $i) {  
                     !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
                     ## Ignore the token  
                     !!!next-token;  
                     redo B;  
                   }  
5397                                    
5398                  ## generate implied end tags                  ## generate implied end tags
5399                  if ({                  while ($self->{open_elements}->[-1]->[1]
5400                       dd => 1, dt => 1, li => 1, p => 1,                             & END_TAG_OPTIONAL_EL) {
5401                       td => 1, th => 1, tr => 1,                    !!!cp ('t174');
5402                       tbody => 1, tfoot=> 1, thead => 1,                    pop @{$self->{open_elements}};
                     }->{$self->{open_elements}->[-1]->[1]}) {  
                   !!!back-token;  
                   $token = {type => 'end tag',  
                             tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
                   redo B;  
5403                  }                  }
5404                                    
5405                  if ($self->{open_elements}->[-1]->[1] ne 'caption') {                  unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
5406                    !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                    !!!cp ('t175');
5407                      !!!parse-error (type => 'not closed',
5408                                      text => $self->{open_elements}->[-1]->[0]
5409                                          ->manakai_local_name,
5410                                      token => $token);
5411                    } else {
5412                      !!!cp ('t176');
5413                  }                  }
5414                                    
5415                  splice @{$self->{open_elements}}, $i;                  splice @{$self->{open_elements}}, $i;
5416                                    
5417                  $clear_up_to_marker->();                  $clear_up_to_marker->();
5418                                    
5419                  $self->{insertion_mode} = 'in table';                  $self->{insertion_mode} = IN_TABLE_IM;
5420                                    
5421                  !!!next-token;                  !!!next-token;
5422                  redo B;                  next B;
5423                } elsif ($self->{insertion_mode} eq 'in cell') {                } elsif ($self->{insertion_mode} == IN_CELL_IM) {
5424                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                  !!!cp ('t177');
5425                    !!!parse-error (type => 'unmatched end tag',
5426                                    text => $token->{tag_name}, token => $token);
5427                  ## Ignore the token                  ## Ignore the token
5428                  !!!next-token;                  !!!next-token;
5429                  redo B;                  next B;
5430                } else {                } else {
5431                    !!!cp ('t178');
5432                  #                  #
5433                }                }
5434              } elsif ({              } elsif ({
5435                        table => 1, tbody => 1, tfoot => 1,                        table => 1, tbody => 1, tfoot => 1,
5436                        thead => 1, tr => 1,                        thead => 1, tr => 1,
5437                       }->{$token->{tag_name}} and                       }->{$token->{tag_name}} and
5438                       $self->{insertion_mode} eq 'in cell') {                       $self->{insertion_mode} == IN_CELL_IM) {
5439                ## have an element in table scope                ## have an element in table scope
5440                my $i;                my $i;
5441                my $tn;                my $tn;
5442                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: {
5443                  my $node = $self->{open_elements}->[$_];                  for (reverse 0..$#{$self->{open_elements}}) {
5444                  if ($node->[1] eq $token->{tag_name}) {                    my $node = $self->{open_elements}->[$_];
5445                    $i = $_;                    if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
5446                    last INSCOPE;                      !!!cp ('t179');
5447                  } elsif ($node->[1] eq 'td' or $node->[1] eq 'th') {                      $i = $_;
5448                    $tn = $node->[1];  
5449                    ## NOTE: There is exactly one |td| or |th| element                      ## Close the cell
5450                    ## in scope in the stack of open elements by definition.                      !!!back-token; # </x>
5451                  } elsif ({                      $token = {type => END_TAG_TOKEN, tag_name => $tn,
5452                            table => 1, html => 1,                                line => $token->{line},
5453                           }->{$node->[1]}) {                                column => $token->{column}};
5454                    last INSCOPE;                      next B;
5455                      } elsif ($node->[1] & TABLE_CELL_EL) {
5456                        !!!cp ('t180');
5457                        $tn = $node->[0]->manakai_local_name;
5458                        ## NOTE: There is exactly one |td| or |th| element
5459                        ## in scope in the stack of open elements by definition.
5460                      } elsif ($node->[1] & TABLE_SCOPING_EL) {
5461                        ## ISSUE: Can this be reached?
5462                        !!!cp ('t181');
5463                        last;
5464                      }
5465                  }                  }
5466                } # INSCOPE  
5467                unless (defined $i) {                  !!!cp ('t182');
5468                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                  !!!parse-error (type => 'unmatched end tag',
5469                        text => $token->{tag_name}, token => $token);
5470                  ## Ignore the token                  ## Ignore the token
5471                  !!!next-token;                  !!!next-token;
5472                  redo B;                  next B;
5473                }                } # INSCOPE
   
               ## Close the cell  
               !!!back-token; # </?>  
               $token = {type => 'end tag', tag_name => $tn};  
               redo B;  
5474              } elsif ($token->{tag_name} eq 'table' and              } elsif ($token->{tag_name} eq 'table' and
5475                       $self->{insertion_mode} eq 'in caption') {                       $self->{insertion_mode} == IN_CAPTION_IM) {
5476                !!!parse-error (type => 'not closed:caption');                !!!parse-error (type => 'not closed', text => 'caption',
5477                                  token => $token);
5478    
5479                ## As if </caption>                ## As if </caption>
5480                ## have a table element in table scope                ## have a table element in table scope
5481                my $i;                my $i;
5482                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5483                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
5484                  if ($node->[1] eq 'caption') {                  if ($node->[1] & CAPTION_EL) {
5485                      !!!cp ('t184');
5486                    $i = $_;                    $i = $_;
5487                    last INSCOPE;                    last INSCOPE;
5488                  } elsif ({                  } elsif ($node->[1] & TABLE_SCOPING_EL) {
5489                            table => 1, html => 1,                    !!!cp ('t185');
                          }->{$node->[1]}) {  
5490                    last INSCOPE;                    last INSCOPE;
5491                  }                  }
5492                } # INSCOPE                } # INSCOPE
5493                unless (defined $i) {                unless (defined $i) {
5494                  !!!parse-error (type => 'unmatched end tag:caption');                  !!!cp ('t186');
5495                    !!!parse-error (type => 'unmatched end tag',
5496                                    text => 'caption', token => $token);
5497                  ## Ignore the token                  ## Ignore the token
5498                  !!!next-token;                  !!!next-token;
5499                  redo B;                  next B;
5500                }                }
5501                                
5502                ## generate implied end tags                ## generate implied end tags
5503                if ({                while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
5504                     dd => 1, dt => 1, li => 1, p => 1,                  !!!cp ('t187');
5505                     td => 1, th => 1, tr => 1,                  pop @{$self->{open_elements}};
                    tbody => 1, tfoot=> 1, thead => 1,  
                   }->{$self->{open_elements}->[-1]->[1]}) {  
                 !!!back-token; # </table>  
                 $token = {type => 'end tag', tag_name => 'caption'};  
                 !!!back-token;  
                 $token = {type => 'end tag',  
                           tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
                 redo B;  
5506                }                }
5507    
5508                if ($self->{open_elements}->[-1]->[1] ne 'caption') {                unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
5509                  !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                  !!!cp ('t188');
5510                    !!!parse-error (type => 'not closed',
5511                                    text => $self->{open_elements}->[-1]->[0]
5512                                        ->manakai_local_name,
5513                                    token => $token);
5514                  } else {
5515                    !!!cp ('t189');
5516                }                }
5517    
5518                splice @{$self->{open_elements}}, $i;                splice @{$self->{open_elements}}, $i;
5519    
5520                $clear_up_to_marker->();                $clear_up_to_marker->();
5521    
5522                $self->{insertion_mode} = 'in table';                $self->{insertion_mode} = IN_TABLE_IM;
5523    
5524                ## reprocess                ## reprocess
5525                redo B;                next B;
5526              } elsif ({              } elsif ({
5527                        body => 1, col => 1, colgroup => 1, html => 1,                        body => 1, col => 1, colgroup => 1, html => 1,
5528                       }->{$token->{tag_name}}) {                       }->{$token->{tag_name}}) {
5529                if ($self->{insertion_mode} eq 'in cell' or                if ($self->{insertion_mode} & BODY_TABLE_IMS) {
5530                    $self->{insertion_mode} eq 'in caption') {                  !!!cp ('t190');
5531                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                  !!!parse-error (type => 'unmatched end tag',
5532                                    text => $token->{tag_name}, token => $token);
5533                  ## Ignore the token                  ## Ignore the token
5534                  !!!next-token;                  !!!next-token;
5535                  redo B;                  next B;
5536                } else {                } else {
5537                    !!!cp ('t191');
5538                  #                  #
5539                }                }
5540              } elsif ({              } elsif ({
5541                        tbody => 1, tfoot => 1,                        tbody => 1, tfoot => 1,
5542                        thead => 1, tr => 1,                        thead => 1, tr => 1,
5543                       }->{$token->{tag_name}} and                       }->{$token->{tag_name}} and
5544                       $self->{insertion_mode} eq 'in caption') {                       $self->{insertion_mode} == IN_CAPTION_IM) {
5545                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                !!!cp ('t192');
5546                  !!!parse-error (type => 'unmatched end tag',
5547                                  text => $token->{tag_name}, token => $token);
5548                ## Ignore the token                ## Ignore the token
5549                !!!next-token;                !!!next-token;
5550                redo B;                next B;
5551              } else {              } else {
5552                  !!!cp ('t193');
5553                #                #
5554              }              }
5555            } else {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
5556              #          for my $entry (@{$self->{open_elements}}) {
5557              unless ($entry->[1] & ALL_END_TAG_OPTIONAL_EL) {
5558                !!!cp ('t75');
5559                !!!parse-error (type => 'in body:#eof', token => $token);
5560                last;
5561            }            }
5562                      }
5563            $in_body->($insert_to_current);  
5564            redo B;          ## Stop parsing.
5565          } elsif ($self->{insertion_mode} eq 'in table') {          last B;
5566            if ($token->{type} eq 'character') {        } else {
5567              ## NOTE: There are "character in table" code clones.          die "$0: $token->{type}: Unknown token type";
5568              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {        }
5569                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);  
5570          $insert = $insert_to_current;
5571          #
5572        } elsif ($self->{insertion_mode} & TABLE_IMS) {
5573          if ($token->{type} == CHARACTER_TOKEN) {
5574            if (not $open_tables->[-1]->[1] and # tainted
5575                $token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
5576              $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
5577                                
5578                unless (length $token->{data}) {            unless (length $token->{data}) {
5579                  !!!next-token;              !!!cp ('t194');
5580                  redo B;              !!!next-token;
5581                }              next B;
5582              }            } else {
5583                !!!cp ('t195');
5584              }
5585            }
5586    
5587              !!!parse-error (type => 'in table:#character');          !!!parse-error (type => 'in table:#text', token => $token);
5588    
5589              ## As if in body, but insert into foster parent element          ## NOTE: As if in body, but insert into the foster parent element.
5590              ## ISSUE: Spec says that "whenever a node would be inserted          $reconstruct_active_formatting_elements->($insert_to_foster);
             ## into the current node" while characters might not be  
             ## result in a new Text node.  
             $reconstruct_active_formatting_elements->($insert_to_foster);  
5591                            
5592              if ({          if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {
5593                   table => 1, tbody => 1, tfoot => 1,            # MUST
5594                   thead => 1, tr => 1,            my $foster_parent_element;
5595                  }->{$self->{open_elements}->[-1]->[1]}) {            my $next_sibling;
5596                # MUST            my $prev_sibling;
5597                my $foster_parent_element;            OE: for (reverse 0..$#{$self->{open_elements}}) {
5598                my $next_sibling;              if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {
5599                my $prev_sibling;                my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
5600                OE: for (reverse 0..$#{$self->{open_elements}}) {                if (defined $parent and $parent->node_type == 1) {
5601                  if ($self->{open_elements}->[$_]->[1] eq 'table') {                  $foster_parent_element = $parent;
5602                    my $parent = $self->{open_elements}->[$_]->[0]->parent_node;                  !!!cp ('t196');
5603                    if (defined $parent and $parent->node_type == 1) {                  $next_sibling = $self->{open_elements}->[$_]->[0];
5604                      $foster_parent_element = $parent;                  $prev_sibling = $next_sibling->previous_sibling;
5605                      $next_sibling = $self->{open_elements}->[$_]->[0];                  #
                     $prev_sibling = $next_sibling->previous_sibling;  
                   } else {  
                     $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];  
                     $prev_sibling = $foster_parent_element->last_child;  
                   }  
                   last OE;  
                 }  
               } # OE  
               $foster_parent_element = $self->{open_elements}->[0]->[0] and  
               $prev_sibling = $foster_parent_element->last_child  
                 unless defined $foster_parent_element;  
               if (defined $prev_sibling and  
                   $prev_sibling->node_type == 3) {  
                 $prev_sibling->manakai_append_text ($token->{data});  
5606                } else {                } else {
5607                  $foster_parent_element->insert_before                  !!!cp ('t197');
5608                    ($self->{document}->create_text_node ($token->{data}),                  $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];
5609                     $next_sibling);                  $prev_sibling = $foster_parent_element->last_child;
5610                    #
5611                }                }
5612              } else {                last OE;
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});  
5613              }              }
5614              } # OE
5615              $foster_parent_element = $self->{open_elements}->[0]->[0] and
5616              $prev_sibling = $foster_parent_element->last_child
5617                  unless defined $foster_parent_element;
5618              undef $prev_sibling unless $open_tables->[-1]->[2]; # ~node inserted
5619              if (defined $prev_sibling and
5620                  $prev_sibling->node_type == 3) {
5621                !!!cp ('t198');
5622                $prev_sibling->manakai_append_text ($token->{data});
5623              } else {
5624                !!!cp ('t199');
5625                $foster_parent_element->insert_before
5626                    ($self->{document}->create_text_node ($token->{data}),
5627                     $next_sibling);
5628              }
5629              $open_tables->[-1]->[1] = 1; # tainted
5630              $open_tables->[-1]->[2] = 1; # ~node inserted
5631            } else {
5632              ## NOTE: Fragment case or in a foster parent'ed element
5633              ## (e.g. |<table><span>a|).  In fragment case, whether the
5634              ## character is appended to existing node or a new node is
5635              ## created is irrelevant, since the foster parent'ed nodes
5636              ## are discarded and fragment parsing does not invoke any
5637              ## script.
5638              !!!cp ('t200');
5639              $self->{open_elements}->[-1]->[0]->manakai_append_text
5640                  ($token->{data});
5641            }
5642                            
5643              !!!next-token;          !!!next-token;
5644              redo B;          next B;
5645            } elsif ($token->{type} eq 'start tag') {        } elsif ($token->{type} == START_TAG_TOKEN) {
5646              if ({          if ({
5647                   caption => 1,               tr => ($self->{insertion_mode} != IN_ROW_IM),
5648                   colgroup => 1,               th => 1, td => 1,
5649                   tbody => 1, tfoot => 1, thead => 1,              }->{$token->{tag_name}}) {
5650                  }->{$token->{tag_name}}) {            if ($self->{insertion_mode} == IN_TABLE_IM) {
5651                ## Clear back to table context              ## Clear back to table context
5652                while ($self->{open_elements}->[-1]->[1] ne 'table' and              while (not ($self->{open_elements}->[-1]->[1]
5653                       $self->{open_elements}->[-1]->[1] ne 'html') {                              & TABLE_SCOPING_EL)) {
5654                  !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                !!!cp ('t201');
5655                  pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
               }  
   
               push @$active_formatting_elements, ['#marker', '']  
                 if $token->{tag_name} eq 'caption';  
   
               !!!insert-element ($token->{tag_name}, $token->{attributes});  
               $self->{insertion_mode} = {  
                                  caption => 'in caption',  
                                  colgroup => 'in column group',  
                                  tbody => 'in table body',  
                                  tfoot => 'in table body',  
                                  thead => 'in table body',  
                                 }->{$token->{tag_name}};  
               !!!next-token;  
               redo B;  
             } elsif ({  
                       col => 1,  
                       td => 1, th => 1, tr => 1,  
                      }->{$token->{tag_name}}) {  
               ## Clear back to table context  
               while ($self->{open_elements}->[-1]->[1] ne 'table' and  
                      $self->{open_elements}->[-1]->[1] ne 'html') {  
                 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
                 pop @{$self->{open_elements}};  
               }  
   
               !!!insert-element ($token->{tag_name} eq 'col' ? 'colgroup' : 'tbody');  
               $self->{insertion_mode} = $token->{tag_name} eq 'col'  
                 ? 'in column group' : 'in table body';  
               ## reprocess  
               redo B;  
             } elsif ($token->{tag_name} eq 'table') {  
               ## NOTE: There are code clones for this "table in table"  
               !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
   
               ## As if </table>  
               ## have a table element in table scope  
               my $i;  
               INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
                 my $node = $self->{open_elements}->[$_];  
                 if ($node->[1] eq 'table') {  
                   $i = $_;  
                   last INSCOPE;  
                 } elsif ({  
                           table => 1, html => 1,  
                          }->{$node->[1]}) {  
                   last INSCOPE;  
                 }  
               } # INSCOPE  
               unless (defined $i) {  
                 !!!parse-error (type => 'unmatched end tag:table');  
                 ## Ignore tokens </table><table>  
                 !!!next-token;  
                 redo B;  
               }  
                 
               ## generate implied end tags  
               if ({  
                    dd => 1, dt => 1, li => 1, p => 1,  
                    td => 1, th => 1, tr => 1,  
                    tbody => 1, tfoot=> 1, thead => 1,  
                   }->{$self->{open_elements}->[-1]->[1]}) {  
                 !!!back-token; # <table>  
                 $token = {type => 'end tag', tag_name => 'table'};  
                 !!!back-token;  
                 $token = {type => 'end tag',  
                           tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
                 redo B;  
               }  
   
               if ($self->{open_elements}->[-1]->[1] ne 'table') {  
                 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
               }  
   
               splice @{$self->{open_elements}}, $i;  
   
               $self->_reset_insertion_mode;  
   
               ## reprocess  
               redo B;  
             } else {  
               #  
             }  
           } elsif ($token->{type} eq 'end tag') {  
             if ($token->{tag_name} eq 'table') {  
               ## have a table element in table scope  
               my $i;  
               INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
                 my $node = $self->{open_elements}->[$_];  
                 if ($node->[1] eq $token->{tag_name}) {  
                   $i = $_;  
                   last INSCOPE;  
                 } elsif ({  
                           table => 1, html => 1,  
                          }->{$node->[1]}) {  
                   last INSCOPE;  
                 }  
               } # INSCOPE  
               unless (defined $i) {  
                 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
                 ## Ignore the token  
                 !!!next-token;  
                 redo B;  
               }  
                 
               ## generate implied end tags  
               if ({  
                    dd => 1, dt => 1, li => 1, p => 1,  
                    td => 1, th => 1, tr => 1,  
                    tbody => 1, tfoot=> 1, thead => 1,  
                   }->{$self->{open_elements}->[-1]->[1]}) {  
                 !!!back-token;  
                 $token = {type => 'end tag',  
                           tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
                 redo B;  
               }  
   
               if ($self->{open_elements}->[-1]->[1] ne 'table') {  
                 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
               }  
   
               splice @{$self->{open_elements}}, $i;  
   
               $self->_reset_insertion_mode;  
   
               !!!next-token;  
               redo B;  
             } elsif ({  
                       body => 1, caption => 1, col => 1, colgroup => 1,  
                       html => 1, tbody => 1, td => 1, tfoot => 1, th => 1,  
                       thead => 1, tr => 1,  
                      }->{$token->{tag_name}}) {  
               !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
               ## Ignore the token  
               !!!next-token;  
               redo B;  
             } else {  
               #  
5656              }              }
5657            } else {              
5658              #              !!!insert-element ('tbody',, $token);
5659                $self->{insertion_mode} = IN_TABLE_BODY_IM;
5660                ## reprocess in the "in table body" insertion mode...
5661            }            }
5662              
5663            !!!parse-error (type => 'in table:'.$token->{tag_name});            if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {
5664            $in_body->($insert_to_foster);              unless ($token->{tag_name} eq 'tr') {
5665            redo B;                !!!cp ('t202');
5666          } elsif ($self->{insertion_mode} eq 'in column group') {                !!!parse-error (type => 'missing start tag:tr', token => $token);
           if ($token->{type} eq 'character') {  
             if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {  
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);  
               unless (length $token->{data}) {  
                 !!!next-token;  
                 redo B;  
               }  
5667              }              }
5668                                
5669              #              ## Clear back to table body context
5670            } elsif ($token->{type} eq 'start tag') {              while (not ($self->{open_elements}->[-1]->[1]
5671              if ($token->{tag_name} eq 'col') {                              & TABLE_ROWS_SCOPING_EL)) {
5672                !!!insert-element ($token->{tag_name}, $token->{attributes});                !!!cp ('t203');
5673                  ## ISSUE: Can this case be reached?
5674                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
               !!!next-token;  
               redo B;  
             } else {  
               #  
5675              }              }
5676            } elsif ($token->{type} eq 'end tag') {                  
5677              if ($token->{tag_name} eq 'colgroup') {              $self->{insertion_mode} = IN_ROW_IM;
5678                if ($self->{open_elements}->[-1]->[1] eq 'html') {              if ($token->{tag_name} eq 'tr') {
5679                  !!!parse-error (type => 'unmatched end tag:colgroup');                !!!cp ('t204');
5680                  ## Ignore the token                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5681                  !!!next-token;                $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5682                  redo B;                !!!nack ('t204');
               } else {  
                 pop @{$self->{open_elements}}; # colgroup  
                 $self->{insertion_mode} = 'in table';  
                 !!!next-token;  
                 redo B;              
               }  
             } elsif ($token->{tag_name} eq 'col') {  
               !!!parse-error (type => 'unmatched end tag:col');  
               ## Ignore the token  
5683                !!!next-token;                !!!next-token;
5684                redo B;                next B;
5685              } else {              } else {
5686                #                !!!cp ('t205');
5687                  !!!insert-element ('tr',, $token);
5688                  ## reprocess in the "in row" insertion mode
5689              }              }
5690            } else {            } else {
5691              #              !!!cp ('t206');
5692            }            }
5693    
5694            ## As if </colgroup>                ## Clear back to table row context
5695            if ($self->{open_elements}->[-1]->[1] eq 'html') {                while (not ($self->{open_elements}->[-1]->[1]
5696              !!!parse-error (type => 'unmatched end tag:colgroup');                                & TABLE_ROW_SCOPING_EL)) {
5697              ## Ignore the token                  !!!cp ('t207');
5698              !!!next-token;                  pop @{$self->{open_elements}};
             redo B;  
           } else {  
             pop @{$self->{open_elements}}; # colgroup  
             $self->{insertion_mode} = 'in table';  
             ## reprocess  
             redo B;  
           }  
         } elsif ($self->{insertion_mode} eq 'in table body') {  
           if ($token->{type} eq 'character') {  
             ## NOTE: This is a "character in table" code clone.  
             if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {  
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);  
                 
               unless (length $token->{data}) {  
                 !!!next-token;  
                 redo B;  
5699                }                }
5700              }                
5701              !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5702              !!!parse-error (type => 'in table:#character');            $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5703              $self->{insertion_mode} = IN_CELL_IM;
             ## As if in body, but insert into foster parent element  
             ## ISSUE: Spec says that "whenever a node would be inserted  
             ## into the current node" while characters might not be  
             ## result in a new Text node.  
             $reconstruct_active_formatting_elements->($insert_to_foster);  
5704    
5705              if ({            push @$active_formatting_elements, ['#marker', ''];
5706                   table => 1, tbody => 1, tfoot => 1,                
5707                   thead => 1, tr => 1,            !!!nack ('t207.1');
5708                  }->{$self->{open_elements}->[-1]->[1]}) {            !!!next-token;
5709                # MUST            next B;
5710                my $foster_parent_element;          } elsif ({
5711                my $next_sibling;                    caption => 1, col => 1, colgroup => 1,
5712                my $prev_sibling;                    tbody => 1, tfoot => 1, thead => 1,
5713                OE: for (reverse 0..$#{$self->{open_elements}}) {                    tr => 1, # $self->{insertion_mode} == IN_ROW_IM
5714                  if ($self->{open_elements}->[$_]->[1] eq 'table') {                   }->{$token->{tag_name}}) {
5715                    my $parent = $self->{open_elements}->[$_]->[0]->parent_node;            if ($self->{insertion_mode} == IN_ROW_IM) {
5716                    if (defined $parent and $parent->node_type == 1) {              ## As if </tr>
5717                      $foster_parent_element = $parent;              ## have an element in table scope
5718                      $next_sibling = $self->{open_elements}->[$_]->[0];              my $i;
5719                      $prev_sibling = $next_sibling->previous_sibling;              INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5720                    } else {                my $node = $self->{open_elements}->[$_];
5721                      $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];                if ($node->[1] & TABLE_ROW_EL) {
5722                      $prev_sibling = $foster_parent_element->last_child;                  !!!cp ('t208');
5723                    }                  $i = $_;
5724                    last OE;                  last INSCOPE;
5725                  }                } elsif ($node->[1] & TABLE_SCOPING_EL) {
5726                } # OE                  !!!cp ('t209');
5727                $foster_parent_element = $self->{open_elements}->[0]->[0] and                  last INSCOPE;
5728                $prev_sibling = $foster_parent_element->last_child                }
5729                  unless defined $foster_parent_element;              } # INSCOPE
5730                if (defined $prev_sibling and              unless (defined $i) {
5731                    $prev_sibling->node_type == 3) {                !!!cp ('t210');
5732                  $prev_sibling->manakai_append_text ($token->{data});                ## TODO: This type is wrong.
5733                } else {                !!!parse-error (type => 'unmacthed end tag',
5734                  $foster_parent_element->insert_before                                text => $token->{tag_name}, token => $token);
5735                    ($self->{document}->create_text_node ($token->{data}),                ## Ignore the token
5736                     $next_sibling);                !!!nack ('t210.1');
5737                }                !!!next-token;
5738              } else {                next B;
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});  
5739              }              }
5740                                
5741              !!!next-token;                  ## Clear back to table row context
5742              redo B;                  while (not ($self->{open_elements}->[-1]->[1]
5743            } elsif ($token->{type} eq 'start tag') {                                  & TABLE_ROW_SCOPING_EL)) {
5744              if ({                    !!!cp ('t211');
5745                   tr => 1,                    ## ISSUE: Can this case be reached?
5746                   th => 1, td => 1,                    pop @{$self->{open_elements}};
5747                  }->{$token->{tag_name}}) {                  }
5748                unless ($token->{tag_name} eq 'tr') {                  
5749                  !!!parse-error (type => 'missing start tag:tr');                  pop @{$self->{open_elements}}; # tr
5750                    $self->{insertion_mode} = IN_TABLE_BODY_IM;
5751                    if ($token->{tag_name} eq 'tr') {
5752                      !!!cp ('t212');
5753                      ## reprocess
5754                      !!!ack-later;
5755                      next B;
5756                    } else {
5757                      !!!cp ('t213');
5758                      ## reprocess in the "in table body" insertion mode...
5759                    }
5760                }                }
5761    
5762                ## Clear back to table body context                if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {
5763                while (not {                  ## have an element in table scope
5764                  tbody => 1, tfoot => 1, thead => 1, html => 1,                  my $i;
5765                }->{$self->{open_elements}->[-1]->[1]}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5766                  !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                    my $node = $self->{open_elements}->[$_];
5767                  pop @{$self->{open_elements}};                    if ($node->[1] & TABLE_ROW_GROUP_EL) {
5768                }                      !!!cp ('t214');
5769                                      $i = $_;
5770                $self->{insertion_mode} = 'in row';                      last INSCOPE;
5771                if ($token->{tag_name} eq 'tr') {                    } elsif ($node->[1] & TABLE_SCOPING_EL) {
5772                  !!!insert-element ($token->{tag_name}, $token->{attributes});                      !!!cp ('t215');
5773                  !!!next-token;                      last INSCOPE;
5774                } else {                    }
5775                  !!!insert-element ('tr');                  } # INSCOPE
5776                  ## reprocess                  unless (defined $i) {
5777                }                    !!!cp ('t216');
5778                redo B;  ## TODO: This erorr type is wrong.
5779              } elsif ({                    !!!parse-error (type => 'unmatched end tag',
5780                        caption => 1, col => 1, colgroup => 1,                                    text => $token->{tag_name}, token => $token);
5781                        tbody => 1, tfoot => 1, thead => 1,                    ## Ignore the token
5782                       }->{$token->{tag_name}}) {                    !!!nack ('t216.1');
5783                ## have an element in table scope                    !!!next-token;
5784                my $i;                    next B;
               INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
                 my $node = $self->{open_elements}->[$_];  
                 if ({  
                      tbody => 1, thead => 1, tfoot => 1,  
                     }->{$node->[1]}) {  
                   $i = $_;  
                   last INSCOPE;  
                 } elsif ({  
                           table => 1, html => 1,  
                          }->{$node->[1]}) {  
                   last INSCOPE;  
5785                  }                  }
               } # INSCOPE  
               unless (defined $i) {  
                 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
                 ## Ignore the token  
                 !!!next-token;  
                 redo B;  
               }  
5786    
5787                ## Clear back to table body context                  ## Clear back to table body context
5788                while (not {                  while (not ($self->{open_elements}->[-1]->[1]
5789                  tbody => 1, tfoot => 1, thead => 1, html => 1,                                  & TABLE_ROWS_SCOPING_EL)) {
5790                }->{$self->{open_elements}->[-1]->[1]}) {                    !!!cp ('t217');
5791                  !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                    ## ISSUE: Can this state be reached?
5792                      pop @{$self->{open_elements}};
5793                    }
5794                    
5795                    ## As if <{current node}>
5796                    ## have an element in table scope
5797                    ## true by definition
5798                    
5799                    ## Clear back to table body context
5800                    ## nop by definition
5801                    
5802                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
5803                    $self->{insertion_mode} = IN_TABLE_IM;
5804                    ## reprocess in "in table" insertion mode...
5805                  } else {
5806                    !!!cp ('t218');
5807                }                }
5808    
5809                ## As if <{current node}>            if ($token->{tag_name} eq 'col') {
5810                ## have an element in table scope              ## Clear back to table context
5811                ## true by definition              while (not ($self->{open_elements}->[-1]->[1]
5812                                & TABLE_SCOPING_EL)) {
5813                ## Clear back to table body context                !!!cp ('t219');
5814                ## nop by definition                ## ISSUE: Can this state be reached?
   
5815                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
5816                $self->{insertion_mode} = 'in table';              }
5817                ## reprocess              
5818                redo B;              !!!insert-element ('colgroup',, $token);
5819                $self->{insertion_mode} = IN_COLUMN_GROUP_IM;
5820                ## reprocess
5821                $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5822                !!!ack-later;
5823                next B;
5824              } elsif ({
5825                        caption => 1,
5826                        colgroup => 1,
5827                        tbody => 1, tfoot => 1, thead => 1,
5828                       }->{$token->{tag_name}}) {
5829                ## Clear back to table context
5830                    while (not ($self->{open_elements}->[-1]->[1]
5831                                    & TABLE_SCOPING_EL)) {
5832                      !!!cp ('t220');
5833                      ## ISSUE: Can this state be reached?
5834                      pop @{$self->{open_elements}};
5835                    }
5836                    
5837                push @$active_formatting_elements, ['#marker', '']
5838                    if $token->{tag_name} eq 'caption';
5839                    
5840                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5841                $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5842                $self->{insertion_mode} = {
5843                                           caption => IN_CAPTION_IM,
5844                                           colgroup => IN_COLUMN_GROUP_IM,
5845                                           tbody => IN_TABLE_BODY_IM,
5846                                           tfoot => IN_TABLE_BODY_IM,
5847                                           thead => IN_TABLE_BODY_IM,
5848                                          }->{$token->{tag_name}};
5849                !!!next-token;
5850                !!!nack ('t220.1');
5851                next B;
5852              } else {
5853                die "$0: in table: <>: $token->{tag_name}";
5854              }
5855              } elsif ($token->{tag_name} eq 'table') {              } elsif ($token->{tag_name} eq 'table') {
5856                ## NOTE: This is a code clone of "table in table"                !!!parse-error (type => 'not closed',
5857                !!!parse-error (type => 'not closed:table');                                text => $self->{open_elements}->[-1]->[0]
5858                                      ->manakai_local_name,
5859                                  token => $token);
5860    
5861                ## As if </table>                ## As if </table>
5862                ## have a table element in table scope                ## have a table element in table scope
5863                my $i;                my $i;
5864                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5865                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
5866                  if ($node->[1] eq 'table') {                  if ($node->[1] & TABLE_EL) {
5867                      !!!cp ('t221');
5868                    $i = $_;                    $i = $_;
5869                    last INSCOPE;                    last INSCOPE;
5870                  } elsif ({                  } elsif ($node->[1] & TABLE_SCOPING_EL) {
5871                            table => 1, html => 1,                    !!!cp ('t222');
                          }->{$node->[1]}) {  
5872                    last INSCOPE;                    last INSCOPE;
5873                  }                  }
5874                } # INSCOPE                } # INSCOPE
5875                unless (defined $i) {                unless (defined $i) {
5876                  !!!parse-error (type => 'unmatched end tag:table');                  !!!cp ('t223');
5877    ## TODO: The following is wrong, maybe.
5878                    !!!parse-error (type => 'unmatched end tag', text => 'table',
5879                                    token => $token);
5880                  ## Ignore tokens </table><table>                  ## Ignore tokens </table><table>
5881                    !!!nack ('t223.1');
5882                  !!!next-token;                  !!!next-token;
5883                  redo B;                  next B;
5884                }                }
5885                                
5886    ## TODO: Followings are removed from the latest spec.
5887                ## generate implied end tags                ## generate implied end tags
5888                if ({                while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
5889                     dd => 1, dt => 1, li => 1, p => 1,                  !!!cp ('t224');
5890                     td => 1, th => 1, tr => 1,                  pop @{$self->{open_elements}};
                    tbody => 1, tfoot=> 1, thead => 1,  
                   }->{$self->{open_elements}->[-1]->[1]}) {  
                 !!!back-token; # <table>  
                 $token = {type => 'end tag', tag_name => 'table'};  
                 !!!back-token;  
                 $token = {type => 'end tag',  
                           tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
                 redo B;  
5891                }                }
5892    
5893                if ($self->{open_elements}->[-1]->[1] ne 'table') {                unless ($self->{open_elements}->[-1]->[1] & TABLE_EL) {
5894                  !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                  !!!cp ('t225');
5895                    ## NOTE: |<table><tr><table>|
5896                    !!!parse-error (type => 'not closed',
5897                                    text => $self->{open_elements}->[-1]->[0]
5898                                        ->manakai_local_name,
5899                                    token => $token);
5900                  } else {
5901                    !!!cp ('t226');
5902                }                }
5903    
5904                splice @{$self->{open_elements}}, $i;                splice @{$self->{open_elements}}, $i;
5905                  pop @{$open_tables};
5906    
5907                $self->_reset_insertion_mode;                $self->_reset_insertion_mode;
5908    
5909                ## reprocess            ## reprocess
5910                redo B;            !!!ack-later;
5911              } else {            next B;
5912                #          } elsif ($token->{tag_name} eq 'style') {
5913              }            if (not $open_tables->[-1]->[1]) { # tainted
5914            } elsif ($token->{type} eq 'end tag') {              !!!cp ('t227.8');
5915              if ({              ## NOTE: This is a "as if in head" code clone.
5916                   tbody => 1, tfoot => 1, thead => 1,              $parse_rcdata->(CDATA_CONTENT_MODEL);
5917                  }->{$token->{tag_name}}) {              $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5918                ## have an element in table scope              next B;
5919                my $i;            } else {
5920                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {              !!!cp ('t227.7');
5921                  my $node = $self->{open_elements}->[$_];              #
5922                  if ($node->[1] eq $token->{tag_name}) {            }
5923                    $i = $_;          } elsif ($token->{tag_name} eq 'script') {
5924                    last INSCOPE;            if (not $open_tables->[-1]->[1]) { # tainted
5925                  } elsif ({              !!!cp ('t227.6');
5926                            table => 1, html => 1,              ## NOTE: This is a "as if in head" code clone.
5927                           }->{$node->[1]}) {              $script_start_tag->();
5928                    last INSCOPE;              $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5929                  }              next B;
5930                } # INSCOPE            } else {
5931                unless (defined $i) {              !!!cp ('t227.5');
5932                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});              #
5933                  ## Ignore the token            }
5934                  !!!next-token;          } elsif ($token->{tag_name} eq 'input') {
5935                  redo B;            if (not $open_tables->[-1]->[1]) { # tainted
5936                }              if ($token->{attributes}->{type}) { ## TODO: case
5937                  my $type = lc $token->{attributes}->{type}->{value};
5938                  if ($type eq 'hidden') {
5939                    !!!cp ('t227.3');
5940                    !!!parse-error (type => 'in table',
5941                                    text => $token->{tag_name}, token => $token);
5942    
5943                ## Clear back to table body context                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5944                while (not {                  $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
                 tbody => 1, tfoot => 1, thead => 1, html => 1,  
               }->{$self->{open_elements}->[-1]->[1]}) {  
                 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
                 pop @{$self->{open_elements}};  
               }  
5945    
5946                pop @{$self->{open_elements}};                  ## TODO: form element pointer
               $self->{insertion_mode} = 'in table';  
               !!!next-token;  
               redo B;  
             } elsif ($token->{tag_name} eq 'table') {  
               ## have an element in table scope  
               my $i;  
               INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
                 my $node = $self->{open_elements}->[$_];  
                 if ({  
                      tbody => 1, thead => 1, tfoot => 1,  
                     }->{$node->[1]}) {  
                   $i = $_;  
                   last INSCOPE;  
                 } elsif ({  
                           table => 1, html => 1,  
                          }->{$node->[1]}) {  
                   last INSCOPE;  
                 }  
               } # INSCOPE  
               unless (defined $i) {  
                 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
                 ## Ignore the token  
                 !!!next-token;  
                 redo B;  
               }  
5947    
               ## Clear back to table body context  
               while (not {  
                 tbody => 1, tfoot => 1, thead => 1, html => 1,  
               }->{$self->{open_elements}->[-1]->[1]}) {  
                 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
5948                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
               }  
   
               ## As if <{current node}>  
               ## have an element in table scope  
               ## true by definition  
5949    
5950                ## Clear back to table body context                  !!!next-token;
5951                ## nop by definition                  !!!ack ('t227.2.1');
5952                    next B;
5953                pop @{$self->{open_elements}};                } else {
5954                $self->{insertion_mode} = 'in table';                  !!!cp ('t227.2');
5955                ## reprocess                  #
5956                redo B;                }
             } elsif ({  
                       body => 1, caption => 1, col => 1, colgroup => 1,  
                       html => 1, td => 1, th => 1, tr => 1,  
                      }->{$token->{tag_name}}) {  
               !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
               ## Ignore the token  
               !!!next-token;  
               redo B;  
5957              } else {              } else {
5958                  !!!cp ('t227.1');
5959                #                #
5960              }              }
5961            } else {            } else {
5962                !!!cp ('t227.4');
5963              #              #
5964            }            }
5965                      } else {
5966            ## As if in table            !!!cp ('t227');
5967            !!!parse-error (type => 'in table:'.$token->{tag_name});            #
5968            $in_body->($insert_to_foster);          }
           redo B;  
         } elsif ($self->{insertion_mode} eq 'in row') {  
           if ($token->{type} eq 'character') {  
             ## NOTE: This is a "character in table" code clone.  
             if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {  
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);  
                 
               unless (length $token->{data}) {  
                 !!!next-token;  
                 redo B;  
               }  
             }  
5969    
5970              !!!parse-error (type => 'in table:#character');          !!!parse-error (type => 'in table', text => $token->{tag_name},
5971                            token => $token);
5972    
5973              ## As if in body, but insert into foster parent element          $insert = $insert_to_foster;
5974              ## ISSUE: Spec says that "whenever a node would be inserted          #
5975              ## into the current node" while characters might not be        } elsif ($token->{type} == END_TAG_TOKEN) {
5976              ## result in a new Text node.              if ($token->{tag_name} eq 'tr' and
5977              $reconstruct_active_formatting_elements->($insert_to_foster);                  $self->{insertion_mode} == IN_ROW_IM) {
               
             if ({  
                  table => 1, tbody => 1, tfoot => 1,  
                  thead => 1, tr => 1,  
                 }->{$self->{open_elements}->[-1]->[1]}) {  
               # MUST  
               my $foster_parent_element;  
               my $next_sibling;  
               my $prev_sibling;  
               OE: for (reverse 0..$#{$self->{open_elements}}) {  
                 if ($self->{open_elements}->[$_]->[1] eq 'table') {  
                   my $parent = $self->{open_elements}->[$_]->[0]->parent_node;  
                   if (defined $parent and $parent->node_type == 1) {  
                     $foster_parent_element = $parent;  
                     $next_sibling = $self->{open_elements}->[$_]->[0];  
                     $prev_sibling = $next_sibling->previous_sibling;  
                   } else {  
                     $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];  
                     $prev_sibling = $foster_parent_element->last_child;  
                   }  
                   last OE;  
                 }  
               } # OE  
               $foster_parent_element = $self->{open_elements}->[0]->[0] and  
               $prev_sibling = $foster_parent_element->last_child  
                 unless defined $foster_parent_element;  
               if (defined $prev_sibling and  
                   $prev_sibling->node_type == 3) {  
                 $prev_sibling->manakai_append_text ($token->{data});  
               } else {  
                 $foster_parent_element->insert_before  
                   ($self->{document}->create_text_node ($token->{data}),  
                    $next_sibling);  
               }  
             } else {  
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});  
             }  
               
             !!!next-token;  
             redo B;  
           } elsif ($token->{type} eq 'start tag') {  
             if ($token->{tag_name} eq 'th' or  
                 $token->{tag_name} eq 'td') {  
               ## Clear back to table row context  
               while (not {  
                 tr => 1, html => 1,  
               }->{$self->{open_elements}->[-1]->[1]}) {  
                 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
                 pop @{$self->{open_elements}};  
               }  
                 
               !!!insert-element ($token->{tag_name}, $token->{attributes});  
               $self->{insertion_mode} = 'in cell';  
   
               push @$active_formatting_elements, ['#marker', ''];  
                 
               !!!next-token;  
               redo B;  
             } elsif ({  
                       caption => 1, col => 1, colgroup => 1,  
                       tbody => 1, tfoot => 1, thead => 1, tr => 1,  
                      }->{$token->{tag_name}}) {  
               ## As if </tr>  
5978                ## have an element in table scope                ## have an element in table scope
5979                my $i;                my $i;
5980                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5981                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
5982                  if ($node->[1] eq 'tr') {                  if ($node->[1] & TABLE_ROW_EL) {
5983                      !!!cp ('t228');
5984                    $i = $_;                    $i = $_;
5985                    last INSCOPE;                    last INSCOPE;
5986                  } elsif ({                  } elsif ($node->[1] & TABLE_SCOPING_EL) {
5987                            table => 1, html => 1,                    !!!cp ('t229');
                          }->{$node->[1]}) {  
5988                    last INSCOPE;                    last INSCOPE;
5989                  }                  }
5990                } # INSCOPE                } # INSCOPE
5991                unless (defined $i) {                unless (defined $i) {
5992                  !!!parse-error (type => 'unmacthed end tag:'.$token->{tag_name});                  !!!cp ('t230');
5993                    !!!parse-error (type => 'unmatched end tag',
5994                                    text => $token->{tag_name}, token => $token);
5995                  ## Ignore the token                  ## Ignore the token
5996                    !!!nack ('t230.1');
5997                  !!!next-token;                  !!!next-token;
5998                  redo B;                  next B;
5999                  } else {
6000                    !!!cp ('t232');
6001                }                }
6002    
6003                ## Clear back to table row context                ## Clear back to table row context
6004                while (not {                while (not ($self->{open_elements}->[-1]->[1]
6005                  tr => 1, html => 1,                                & TABLE_ROW_SCOPING_EL)) {
6006                }->{$self->{open_elements}->[-1]->[1]}) {                  !!!cp ('t231');
6007                  !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  ## ISSUE: Can this state be reached?
6008                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
6009                }                }
6010    
6011                pop @{$self->{open_elements}}; # tr                pop @{$self->{open_elements}}; # tr
6012                $self->{insertion_mode} = 'in table body';                $self->{insertion_mode} = IN_TABLE_BODY_IM;
6013                ## reprocess                !!!next-token;
6014                redo B;                !!!nack ('t231.1');
6015                  next B;
6016              } elsif ($token->{tag_name} eq 'table') {              } elsif ($token->{tag_name} eq 'table') {
6017                ## NOTE: This is a code clone of "table in table"                if ($self->{insertion_mode} == IN_ROW_IM) {
6018                !!!parse-error (type => 'not closed:table');                  ## As if </tr>
6019                    ## have an element in table scope
6020                ## As if </table>                  my $i;
6021                ## have a table element in table scope                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6022                my $i;                    my $node = $self->{open_elements}->[$_];
6023                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                    if ($node->[1] & TABLE_ROW_EL) {
6024                  my $node = $self->{open_elements}->[$_];                      !!!cp ('t233');
6025                  if ($node->[1] eq 'table') {                      $i = $_;
6026                    $i = $_;                      last INSCOPE;
6027                    last INSCOPE;                    } elsif ($node->[1] & TABLE_SCOPING_EL) {
6028                  } elsif ({                      !!!cp ('t234');
6029                            table => 1, html => 1,                      last INSCOPE;
6030                           }->{$node->[1]}) {                    }
6031                    last INSCOPE;                  } # INSCOPE
6032                    unless (defined $i) {
6033                      !!!cp ('t235');
6034    ## TODO: The following is wrong.
6035                      !!!parse-error (type => 'unmatched end tag',
6036                                      text => $token->{type}, token => $token);
6037                      ## Ignore the token
6038                      !!!nack ('t236.1');
6039                      !!!next-token;
6040                      next B;
6041                  }                  }
6042                } # INSCOPE                  
6043                unless (defined $i) {                  ## Clear back to table row context
6044                  !!!parse-error (type => 'unmatched end tag:table');                  while (not ($self->{open_elements}->[-1]->[1]
6045                  ## Ignore tokens </table><table>                                  & TABLE_ROW_SCOPING_EL)) {
6046                  !!!next-token;                    !!!cp ('t236');
6047                  redo B;  ## ISSUE: Can this state be reached?
6048                }                    pop @{$self->{open_elements}};
6049                                  }
6050                ## generate implied end tags                  
6051                if ({                  pop @{$self->{open_elements}}; # tr
6052                     dd => 1, dt => 1, li => 1, p => 1,                  $self->{insertion_mode} = IN_TABLE_BODY_IM;
6053                     td => 1, th => 1, tr => 1,                  ## reprocess in the "in table body" insertion mode...
                    tbody => 1, tfoot=> 1, thead => 1,  
                   }->{$self->{open_elements}->[-1]->[1]}) {  
                 !!!back-token; # <table>  
                 $token = {type => 'end tag', tag_name => 'table'};  
                 !!!back-token;  
                 $token = {type => 'end tag',  
                           tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
                 redo B;  
6054                }                }
6055    
6056                if ($self->{open_elements}->[-1]->[1] ne 'table') {                if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {
6057                  !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                  ## have an element in table scope
6058                    my $i;
6059                    INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6060                      my $node = $self->{open_elements}->[$_];
6061                      if ($node->[1] & TABLE_ROW_GROUP_EL) {
6062                        !!!cp ('t237');
6063                        $i = $_;
6064                        last INSCOPE;
6065                      } elsif ($node->[1] & TABLE_SCOPING_EL) {
6066                        !!!cp ('t238');
6067                        last INSCOPE;
6068                      }
6069                    } # INSCOPE
6070                    unless (defined $i) {
6071                      !!!cp ('t239');
6072                      !!!parse-error (type => 'unmatched end tag',
6073                                      text => $token->{tag_name}, token => $token);
6074                      ## Ignore the token
6075                      !!!nack ('t239.1');
6076                      !!!next-token;
6077                      next B;
6078                    }
6079                    
6080                    ## Clear back to table body context
6081                    while (not ($self->{open_elements}->[-1]->[1]
6082                                    & TABLE_ROWS_SCOPING_EL)) {
6083                      !!!cp ('t240');
6084                      pop @{$self->{open_elements}};
6085                    }
6086                    
6087                    ## As if <{current node}>
6088                    ## have an element in table scope
6089                    ## true by definition
6090                    
6091                    ## Clear back to table body context
6092                    ## nop by definition
6093                    
6094                    pop @{$self->{open_elements}};
6095                    $self->{insertion_mode} = IN_TABLE_IM;
6096                    ## reprocess in the "in table" insertion mode...
6097                }                }
6098    
6099                splice @{$self->{open_elements}}, $i;                ## NOTE: </table> in the "in table" insertion mode.
6100                  ## When you edit the code fragment below, please ensure that
6101                $self->_reset_insertion_mode;                ## the code for <table> in the "in table" insertion mode
6102                  ## is synced with it.
6103    
6104                ## reprocess                ## have a table element in table scope
               redo B;  
             } else {  
               #  
             }  
           } elsif ($token->{type} eq 'end tag') {  
             if ($token->{tag_name} eq 'tr') {  
               ## have an element in table scope  
6105                my $i;                my $i;
6106                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6107                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
6108                  if ($node->[1] eq $token->{tag_name}) {                  if ($node->[1] & TABLE_EL) {
6109                      !!!cp ('t241');
6110                    $i = $_;                    $i = $_;
6111                    last INSCOPE;                    last INSCOPE;
6112                  } elsif ({                  } elsif ($node->[1] & TABLE_SCOPING_EL) {
6113                            table => 1, html => 1,                    !!!cp ('t242');
                          }->{$node->[1]}) {  
6114                    last INSCOPE;                    last INSCOPE;
6115                  }                  }
6116                } # INSCOPE                } # INSCOPE
6117                unless (defined $i) {                unless (defined $i) {
6118                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                  !!!cp ('t243');
6119                    !!!parse-error (type => 'unmatched end tag',
6120                                    text => $token->{tag_name}, token => $token);
6121                  ## Ignore the token                  ## Ignore the token
6122                    !!!nack ('t243.1');
6123                  !!!next-token;                  !!!next-token;
6124                  redo B;                  next B;
6125                }                }
6126                    
6127                ## Clear back to table row context                splice @{$self->{open_elements}}, $i;
6128                while (not {                pop @{$open_tables};
6129                  tr => 1, html => 1,                
6130                }->{$self->{open_elements}->[-1]->[1]}) {                $self->_reset_insertion_mode;
6131                  !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                
6132                  pop @{$self->{open_elements}};                !!!next-token;
6133                  next B;
6134                } elsif ({
6135                          tbody => 1, tfoot => 1, thead => 1,
6136                         }->{$token->{tag_name}} and
6137                         $self->{insertion_mode} & ROW_IMS) {
6138                  if ($self->{insertion_mode} == IN_ROW_IM) {
6139                    ## have an element in table scope
6140                    my $i;
6141                    INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6142                      my $node = $self->{open_elements}->[$_];
6143                      if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
6144                        !!!cp ('t247');
6145                        $i = $_;
6146                        last INSCOPE;
6147                      } elsif ($node->[1] & TABLE_SCOPING_EL) {
6148                        !!!cp ('t248');
6149                        last INSCOPE;
6150                      }
6151                    } # INSCOPE
6152                      unless (defined $i) {
6153                        !!!cp ('t249');
6154                        !!!parse-error (type => 'unmatched end tag',
6155                                        text => $token->{tag_name}, token => $token);
6156                        ## Ignore the token
6157                        !!!nack ('t249.1');
6158                        !!!next-token;
6159                        next B;
6160                      }
6161                    
6162                    ## As if </tr>
6163                    ## have an element in table scope
6164                    my $i;
6165                    INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6166                      my $node = $self->{open_elements}->[$_];
6167                      if ($node->[1] & TABLE_ROW_EL) {
6168                        !!!cp ('t250');
6169                        $i = $_;
6170                        last INSCOPE;
6171                      } elsif ($node->[1] & TABLE_SCOPING_EL) {
6172                        !!!cp ('t251');
6173                        last INSCOPE;
6174                      }
6175                    } # INSCOPE
6176                      unless (defined $i) {
6177                        !!!cp ('t252');
6178                        !!!parse-error (type => 'unmatched end tag',
6179                                        text => 'tr', token => $token);
6180                        ## Ignore the token
6181                        !!!nack ('t252.1');
6182                        !!!next-token;
6183                        next B;
6184                      }
6185                    
6186                    ## Clear back to table row context
6187                    while (not ($self->{open_elements}->[-1]->[1]
6188                                    & TABLE_ROW_SCOPING_EL)) {
6189                      !!!cp ('t253');
6190    ## ISSUE: Can this case be reached?
6191                      pop @{$self->{open_elements}};
6192                    }
6193                    
6194                    pop @{$self->{open_elements}}; # tr
6195                    $self->{insertion_mode} = IN_TABLE_BODY_IM;
6196                    ## reprocess in the "in table body" insertion mode...
6197                }                }
6198    
               pop @{$self->{open_elements}}; # tr  
               $self->{insertion_mode} = 'in table body';  
               !!!next-token;  
               redo B;  
             } elsif ($token->{tag_name} eq 'table') {  
               ## As if </tr>  
6199                ## have an element in table scope                ## have an element in table scope
6200                my $i;                my $i;
6201                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6202                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
6203                  if ($node->[1] eq 'tr') {                  if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
6204                      !!!cp ('t254');
6205                    $i = $_;                    $i = $_;
6206                    last INSCOPE;                    last INSCOPE;
6207                  } elsif ({                  } elsif ($node->[1] & TABLE_SCOPING_EL) {
6208                            table => 1, html => 1,                    !!!cp ('t255');
                          }->{$node->[1]}) {  
6209                    last INSCOPE;                    last INSCOPE;
6210                  }                  }
6211                } # INSCOPE                } # INSCOPE
6212                unless (defined $i) {                unless (defined $i) {
6213                  !!!parse-error (type => 'unmatched end tag:'.$token->{type});                  !!!cp ('t256');
6214                    !!!parse-error (type => 'unmatched end tag',
6215                                    text => $token->{tag_name}, token => $token);
6216                  ## Ignore the token                  ## Ignore the token
6217                    !!!nack ('t256.1');
6218                  !!!next-token;                  !!!next-token;
6219                  redo B;                  next B;
6220                }                }
6221    
6222                ## Clear back to table row context                ## Clear back to table body context
6223                while (not {                while (not ($self->{open_elements}->[-1]->[1]
6224                  tr => 1, html => 1,                                & TABLE_ROWS_SCOPING_EL)) {
6225                }->{$self->{open_elements}->[-1]->[1]}) {                  !!!cp ('t257');
6226                  !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  ## ISSUE: Can this case be reached?
6227                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
6228                }                }
6229    
6230                pop @{$self->{open_elements}}; # tr                pop @{$self->{open_elements}};
6231                $self->{insertion_mode} = 'in table body';                $self->{insertion_mode} = IN_TABLE_IM;
6232                ## reprocess                !!!nack ('t257.1');
6233                redo B;                !!!next-token;
6234                  next B;
6235              } elsif ({              } elsif ({
6236                        tbody => 1, tfoot => 1, thead => 1,                        body => 1, caption => 1, col => 1, colgroup => 1,
6237                          html => 1, td => 1, th => 1,
6238                          tr => 1, # $self->{insertion_mode} == IN_ROW_IM
6239                          tbody => 1, tfoot => 1, thead => 1, # $self->{insertion_mode} == IN_TABLE_IM
6240                       }->{$token->{tag_name}}) {                       }->{$token->{tag_name}}) {
6241                ## have an element in table scope            !!!cp ('t258');
6242                my $i;            !!!parse-error (type => 'unmatched end tag',
6243                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                            text => $token->{tag_name}, token => $token);
6244                  my $node = $self->{open_elements}->[$_];            ## Ignore the token
6245                  if ($node->[1] eq $token->{tag_name}) {            !!!nack ('t258.1');
6246                    $i = $_;             !!!next-token;
6247                    last INSCOPE;            next B;
6248                  } elsif ({          } else {
6249                            table => 1, html => 1,            !!!cp ('t259');
6250                           }->{$node->[1]}) {            !!!parse-error (type => 'in table:/',
6251                    last INSCOPE;                            text => $token->{tag_name}, token => $token);
6252                  }  
6253                } # INSCOPE            $insert = $insert_to_foster;
6254                unless (defined $i) {            #
6255                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});          }
6256                  ## Ignore the token        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6257            unless ($self->{open_elements}->[-1]->[1] & HTML_EL and
6258                    @{$self->{open_elements}} == 1) { # redundant, maybe
6259              !!!parse-error (type => 'in body:#eof', token => $token);
6260              !!!cp ('t259.1');
6261              #
6262            } else {
6263              !!!cp ('t259.2');
6264              #
6265            }
6266    
6267            ## Stop parsing
6268            last B;
6269          } else {
6270            die "$0: $token->{type}: Unknown token type";
6271          }
6272        } elsif ($self->{insertion_mode} == IN_COLUMN_GROUP_IM) {
6273              if ($token->{type} == CHARACTER_TOKEN) {
6274                if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6275                  $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6276                  unless (length $token->{data}) {
6277                    !!!cp ('t260');
6278                  !!!next-token;                  !!!next-token;
6279                  redo B;                  next B;
6280                }                }
6281                }
6282                ## As if </tr>              
6283                ## have an element in table scope              !!!cp ('t261');
6284                my $i;              #
6285                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {            } elsif ($token->{type} == START_TAG_TOKEN) {
6286                  my $node = $self->{open_elements}->[$_];              if ($token->{tag_name} eq 'col') {
6287                  if ($node->[1] eq 'tr') {                !!!cp ('t262');
6288                    $i = $_;                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6289                    last INSCOPE;                pop @{$self->{open_elements}};
6290                  } elsif ({                !!!ack ('t262.1');
6291                            table => 1, html => 1,                !!!next-token;
6292                           }->{$node->[1]}) {                next B;
6293                    last INSCOPE;              } else {
6294                  }                !!!cp ('t263');
6295                } # INSCOPE                #
6296                unless (defined $i) {              }
6297                  !!!parse-error (type => 'unmatched end tag:tr');            } elsif ($token->{type} == END_TAG_TOKEN) {
6298                if ($token->{tag_name} eq 'colgroup') {
6299                  if ($self->{open_elements}->[-1]->[1] & HTML_EL) {
6300                    !!!cp ('t264');
6301                    !!!parse-error (type => 'unmatched end tag',
6302                                    text => 'colgroup', token => $token);
6303                  ## Ignore the token                  ## Ignore the token
6304                  !!!next-token;                  !!!next-token;
6305                  redo B;                  next B;
6306                }                } else {
6307                    !!!cp ('t265');
6308                ## Clear back to table row context                  pop @{$self->{open_elements}}; # colgroup
6309                while (not {                  $self->{insertion_mode} = IN_TABLE_IM;
6310                  tr => 1, html => 1,                  !!!next-token;
6311                }->{$self->{open_elements}->[-1]->[1]}) {                  next B;            
                 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
                 pop @{$self->{open_elements}};  
6312                }                }
6313                } elsif ($token->{tag_name} eq 'col') {
6314                pop @{$self->{open_elements}}; # tr                !!!cp ('t266');
6315                $self->{insertion_mode} = 'in table body';                !!!parse-error (type => 'unmatched end tag',
6316                ## reprocess                                text => 'col', token => $token);
               redo B;  
             } elsif ({  
                       body => 1, caption => 1, col => 1,  
                       colgroup => 1, html => 1, td => 1, th => 1,  
                      }->{$token->{tag_name}}) {  
               !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
6317                ## Ignore the token                ## Ignore the token
6318                !!!next-token;                !!!next-token;
6319                redo B;                next B;
6320              } else {              } else {
6321                #                !!!cp ('t267');
6322                  #
6323              }              }
6324            } else {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6325              #          if ($self->{open_elements}->[-1]->[1] & HTML_EL and
6326            }              @{$self->{open_elements}} == 1) { # redundant, maybe
6327              !!!cp ('t270.2');
6328              ## Stop parsing.
6329              last B;
6330            } else {
6331              ## NOTE: As if </colgroup>.
6332              !!!cp ('t270.1');
6333              pop @{$self->{open_elements}}; # colgroup
6334              $self->{insertion_mode} = IN_TABLE_IM;
6335              ## Reprocess.
6336              next B;
6337            }
6338          } else {
6339            die "$0: $token->{type}: Unknown token type";
6340          }
6341    
6342            ## As if in table            ## As if </colgroup>
6343            !!!parse-error (type => 'in table:'.$token->{tag_name});            if ($self->{open_elements}->[-1]->[1] & HTML_EL) {
6344            $in_body->($insert_to_foster);              !!!cp ('t269');
6345            redo B;  ## TODO: Wrong error type?
6346          } elsif ($self->{insertion_mode} eq 'in select') {              !!!parse-error (type => 'unmatched end tag',
6347            if ($token->{type} eq 'character') {                              text => 'colgroup', token => $token);
6348              $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});              ## Ignore the token
6349                !!!nack ('t269.1');
6350              !!!next-token;              !!!next-token;
6351              redo B;              next B;
6352            } elsif ($token->{type} eq 'start tag') {            } else {
6353              if ($token->{tag_name} eq 'option') {              !!!cp ('t270');
6354                if ($self->{open_elements}->[-1]->[1] eq 'option') {              pop @{$self->{open_elements}}; # colgroup
6355                  ## As if </option>              $self->{insertion_mode} = IN_TABLE_IM;
6356                  pop @{$self->{open_elements}};              !!!ack-later;
6357                }              ## reprocess
6358                next B;
6359              }
6360        } elsif ($self->{insertion_mode} & SELECT_IMS) {
6361          if ($token->{type} == CHARACTER_TOKEN) {
6362            !!!cp ('t271');
6363            $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
6364            !!!next-token;
6365            next B;
6366          } elsif ($token->{type} == START_TAG_TOKEN) {
6367            if ($token->{tag_name} eq 'option') {
6368              if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {
6369                !!!cp ('t272');
6370                ## As if </option>
6371                pop @{$self->{open_elements}};
6372              } else {
6373                !!!cp ('t273');
6374              }
6375    
6376                !!!insert-element ($token->{tag_name}, $token->{attributes});            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6377                !!!next-token;            !!!nack ('t273.1');
6378                redo B;            !!!next-token;
6379              } elsif ($token->{tag_name} eq 'optgroup') {            next B;
6380                if ($self->{open_elements}->[-1]->[1] eq 'option') {          } elsif ($token->{tag_name} eq 'optgroup') {
6381                  ## As if </option>            if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {
6382                  pop @{$self->{open_elements}};              !!!cp ('t274');
6383                }              ## As if </option>
6384                pop @{$self->{open_elements}};
6385              } else {
6386                !!!cp ('t275');
6387              }
6388    
6389                if ($self->{open_elements}->[-1]->[1] eq 'optgroup') {            if ($self->{open_elements}->[-1]->[1] & OPTGROUP_EL) {
6390                  ## As if </optgroup>              !!!cp ('t276');
6391                  pop @{$self->{open_elements}};              ## As if </optgroup>
6392                }              pop @{$self->{open_elements}};
6393              } else {
6394                !!!cp ('t277');
6395              }
6396    
6397                !!!insert-element ($token->{tag_name}, $token->{attributes});            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6398                !!!next-token;            !!!nack ('t277.1');
6399                redo B;            !!!next-token;
6400              } elsif ($token->{tag_name} eq 'select') {            next B;
6401                !!!parse-error (type => 'not closed:select');          } elsif ({
6402                ## As if </select> instead                     select => 1, input => 1, textarea => 1,
6403                ## have an element in table scope                   }->{$token->{tag_name}} or
6404                my $i;                   ($self->{insertion_mode} == IN_SELECT_IN_TABLE_IM and
6405                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                    {
6406                  my $node = $self->{open_elements}->[$_];                     caption => 1, table => 1,
6407                  if ($node->[1] eq $token->{tag_name}) {                     tbody => 1, tfoot => 1, thead => 1,
6408                    $i = $_;                     tr => 1, td => 1, th => 1,
6409                    last INSCOPE;                    }->{$token->{tag_name}})) {
6410                  } elsif ({            ## TODO: The type below is not good - <select> is replaced by </select>
6411                            table => 1, html => 1,            !!!parse-error (type => 'not closed', text => 'select',
6412                           }->{$node->[1]}) {                            token => $token);
6413                    last INSCOPE;            ## NOTE: As if the token were </select> (<select> case) or
6414                  }            ## as if there were </select> (otherwise).
6415                } # INSCOPE            ## have an element in table scope
6416                unless (defined $i) {            my $i;
6417                  !!!parse-error (type => 'unmatched end tag:select');            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6418                  ## Ignore the token              my $node = $self->{open_elements}->[$_];
6419                  !!!next-token;              if ($node->[1] & SELECT_EL) {
6420                  redo B;                !!!cp ('t278');
6421                }                $i = $_;
6422                  last INSCOPE;
6423                } elsif ($node->[1] & TABLE_SCOPING_EL) {
6424                  !!!cp ('t279');
6425                  last INSCOPE;
6426                }
6427              } # INSCOPE
6428              unless (defined $i) {
6429                !!!cp ('t280');
6430                !!!parse-error (type => 'unmatched end tag',
6431                                text => 'select', token => $token);
6432                ## Ignore the token
6433                !!!nack ('t280.1');
6434                !!!next-token;
6435                next B;
6436              }
6437                                
6438                splice @{$self->{open_elements}}, $i;            !!!cp ('t281');
6439              splice @{$self->{open_elements}}, $i;
6440    
6441                $self->_reset_insertion_mode;            $self->_reset_insertion_mode;
6442    
6443                !!!next-token;            if ($token->{tag_name} eq 'select') {
6444                redo B;              !!!nack ('t281.2');
6445              } else {              !!!next-token;
6446                #              next B;
6447              } else {
6448                !!!cp ('t281.1');
6449                !!!ack-later;
6450                ## Reprocess the token.
6451                next B;
6452              }
6453            } else {
6454              !!!cp ('t282');
6455              !!!parse-error (type => 'in select',
6456                              text => $token->{tag_name}, token => $token);
6457              ## Ignore the token
6458              !!!nack ('t282.1');
6459              !!!next-token;
6460              next B;
6461            }
6462          } elsif ($token->{type} == END_TAG_TOKEN) {
6463            if ($token->{tag_name} eq 'optgroup') {
6464              if ($self->{open_elements}->[-1]->[1] & OPTION_EL and
6465                  $self->{open_elements}->[-2]->[1] & OPTGROUP_EL) {
6466                !!!cp ('t283');
6467                ## As if </option>
6468                splice @{$self->{open_elements}}, -2;
6469              } elsif ($self->{open_elements}->[-1]->[1] & OPTGROUP_EL) {
6470                !!!cp ('t284');
6471                pop @{$self->{open_elements}};
6472              } else {
6473                !!!cp ('t285');
6474                !!!parse-error (type => 'unmatched end tag',
6475                                text => $token->{tag_name}, token => $token);
6476                ## Ignore the token
6477              }
6478              !!!nack ('t285.1');
6479              !!!next-token;
6480              next B;
6481            } elsif ($token->{tag_name} eq 'option') {
6482              if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {
6483                !!!cp ('t286');
6484                pop @{$self->{open_elements}};
6485              } else {
6486                !!!cp ('t287');
6487                !!!parse-error (type => 'unmatched end tag',
6488                                text => $token->{tag_name}, token => $token);
6489                ## Ignore the token
6490              }
6491              !!!nack ('t287.1');
6492              !!!next-token;
6493              next B;
6494            } elsif ($token->{tag_name} eq 'select') {
6495              ## have an element in table scope
6496              my $i;
6497              INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6498                my $node = $self->{open_elements}->[$_];
6499                if ($node->[1] & SELECT_EL) {
6500                  !!!cp ('t288');
6501                  $i = $_;
6502                  last INSCOPE;
6503                } elsif ($node->[1] & TABLE_SCOPING_EL) {
6504                  !!!cp ('t289');
6505                  last INSCOPE;
6506              }              }
6507            } elsif ($token->{type} eq 'end tag') {            } # INSCOPE
6508              if ($token->{tag_name} eq 'optgroup') {            unless (defined $i) {
6509                if ($self->{open_elements}->[-1]->[1] eq 'option' and              !!!cp ('t290');
6510                    $self->{open_elements}->[-2]->[1] eq 'optgroup') {              !!!parse-error (type => 'unmatched end tag',
6511                  ## As if </option>                              text => $token->{tag_name}, token => $token);
6512                  splice @{$self->{open_elements}}, -2;              ## Ignore the token
6513                } elsif ($self->{open_elements}->[-1]->[1] eq 'optgroup') {              !!!nack ('t290.1');
6514                  pop @{$self->{open_elements}};              !!!next-token;
6515                } else {              next B;
6516                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});            }
                 ## Ignore the token  
               }  
               !!!next-token;  
               redo B;  
             } elsif ($token->{tag_name} eq 'option') {  
               if ($self->{open_elements}->[-1]->[1] eq 'option') {  
                 pop @{$self->{open_elements}};  
               } else {  
                 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
                 ## Ignore the token  
               }  
               !!!next-token;  
               redo B;  
             } elsif ($token->{tag_name} eq 'select') {  
               ## have an element in table scope  
               my $i;  
               INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
                 my $node = $self->{open_elements}->[$_];  
                 if ($node->[1] eq $token->{tag_name}) {  
                   $i = $_;  
                   last INSCOPE;  
                 } elsif ({  
                           table => 1, html => 1,  
                          }->{$node->[1]}) {  
                   last INSCOPE;  
                 }  
               } # INSCOPE  
               unless (defined $i) {  
                 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
                 ## Ignore the token  
                 !!!next-token;  
                 redo B;  
               }  
6517                                
6518                splice @{$self->{open_elements}}, $i;            !!!cp ('t291');
6519              splice @{$self->{open_elements}}, $i;
6520    
6521                $self->_reset_insertion_mode;            $self->_reset_insertion_mode;
6522    
6523                !!!next-token;            !!!nack ('t291.1');
6524                redo B;            !!!next-token;
6525              } elsif ({            next B;
6526                        caption => 1, table => 1, tbody => 1,          } elsif ($self->{insertion_mode} == IN_SELECT_IN_TABLE_IM and
6527                        tfoot => 1, thead => 1, tr => 1, td => 1, th => 1,                   {
6528                       }->{$token->{tag_name}}) {                    caption => 1, table => 1, tbody => 1,
6529                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                    tfoot => 1, thead => 1, tr => 1, td => 1, th => 1,
6530                                   }->{$token->{tag_name}}) {
6531                ## have an element in table scope  ## TODO: The following is wrong?
6532                my $i;            !!!parse-error (type => 'unmatched end tag',
6533                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                            text => $token->{tag_name}, token => $token);
                 my $node = $self->{open_elements}->[$_];  
                 if ($node->[1] eq $token->{tag_name}) {  
                   $i = $_;  
                   last INSCOPE;  
                 } elsif ({  
                           table => 1, html => 1,  
                          }->{$node->[1]}) {  
                   last INSCOPE;  
                 }  
               } # INSCOPE  
               unless (defined $i) {  
                 ## Ignore the token  
                 !!!next-token;  
                 redo B;  
               }  
6534                                
6535                ## As if </select>            ## have an element in table scope
6536                ## have an element in table scope            my $i;
6537                undef $i;            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6538                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {              my $node = $self->{open_elements}->[$_];
6539                  my $node = $self->{open_elements}->[$_];              if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
6540                  if ($node->[1] eq 'select') {                !!!cp ('t292');
6541                    $i = $_;                $i = $_;
6542                    last INSCOPE;                last INSCOPE;
6543                  } elsif ({              } elsif ($node->[1] & TABLE_SCOPING_EL) {
6544                            table => 1, html => 1,                !!!cp ('t293');
6545                           }->{$node->[1]}) {                last INSCOPE;
6546                    last INSCOPE;              }
6547                  }            } # INSCOPE
6548                } # INSCOPE            unless (defined $i) {
6549                unless (defined $i) {              !!!cp ('t294');
6550                  !!!parse-error (type => 'unmatched end tag:select');              ## Ignore the token
6551                  ## Ignore the </select> token              !!!nack ('t294.1');
6552                  !!!next-token; ## TODO: ok?              !!!next-token;
6553                  redo B;              next B;
6554                }            }
6555                                
6556                splice @{$self->{open_elements}}, $i;            ## As if </select>
6557              ## have an element in table scope
6558                $self->_reset_insertion_mode;            undef $i;
6559              INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6560                ## reprocess              my $node = $self->{open_elements}->[$_];
6561                redo B;              if ($node->[1] & SELECT_EL) {
6562              } else {                !!!cp ('t295');
6563                #                $i = $_;
6564                  last INSCOPE;
6565                } elsif ($node->[1] & TABLE_SCOPING_EL) {
6566    ## ISSUE: Can this state be reached?
6567                  !!!cp ('t296');
6568                  last INSCOPE;
6569              }              }
6570            } else {            } # INSCOPE
6571              #            unless (defined $i) {
6572                !!!cp ('t297');
6573    ## TODO: The following error type is correct?
6574                !!!parse-error (type => 'unmatched end tag',
6575                                text => 'select', token => $token);
6576                ## Ignore the </select> token
6577                !!!nack ('t297.1');
6578                !!!next-token; ## TODO: ok?
6579                next B;
6580            }            }
6581                  
6582              !!!cp ('t298');
6583              splice @{$self->{open_elements}}, $i;
6584    
6585              $self->_reset_insertion_mode;
6586    
6587            !!!parse-error (type => 'in select:'.$token->{tag_name});            !!!ack-later;
6588              ## reprocess
6589              next B;
6590            } else {
6591              !!!cp ('t299');
6592              !!!parse-error (type => 'in select:/',
6593                              text => $token->{tag_name}, token => $token);
6594            ## Ignore the token            ## Ignore the token
6595              !!!nack ('t299.3');
6596            !!!next-token;            !!!next-token;
6597            redo B;            next B;
6598          } elsif ($self->{insertion_mode} eq 'after body') {          }
6599            if ($token->{type} eq 'character') {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6600              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          unless ($self->{open_elements}->[-1]->[1] & HTML_EL and
6601                my $data = $1;                  @{$self->{open_elements}} == 1) { # redundant, maybe
6602                ## As if in body            !!!cp ('t299.1');
6603                $reconstruct_active_formatting_elements->($insert_to_current);            !!!parse-error (type => 'in body:#eof', token => $token);
6604            } else {
6605              !!!cp ('t299.2');
6606            }
6607    
6608            ## Stop parsing.
6609            last B;
6610          } else {
6611            die "$0: $token->{type}: Unknown token type";
6612          }
6613        } elsif ($self->{insertion_mode} & BODY_AFTER_IMS) {
6614          if ($token->{type} == CHARACTER_TOKEN) {
6615            if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6616              my $data = $1;
6617              ## As if in body
6618              $reconstruct_active_formatting_elements->($insert_to_current);
6619                                
6620                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6621              
6622              unless (length $token->{data}) {
6623                !!!cp ('t300');
6624                !!!next-token;
6625                next B;
6626              }
6627            }
6628            
6629            if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6630              !!!cp ('t301');
6631              !!!parse-error (type => 'after html:#text', token => $token);
6632              #
6633            } else {
6634              !!!cp ('t302');
6635              ## "after body" insertion mode
6636              !!!parse-error (type => 'after body:#text', token => $token);
6637              #
6638            }
6639    
6640                unless (length $token->{data}) {          $self->{insertion_mode} = IN_BODY_IM;
6641                  !!!next-token;          ## reprocess
6642                  redo B;          next B;
6643                }        } elsif ($token->{type} == START_TAG_TOKEN) {
6644              }          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6645                          !!!cp ('t303');
6646              #            !!!parse-error (type => 'after html',
6647              !!!parse-error (type => 'after body:#character');                            text => $token->{tag_name}, token => $token);
6648            } elsif ($token->{type} eq 'start tag') {            #
6649              !!!parse-error (type => 'after body:'.$token->{tag_name});          } else {
6650              #            !!!cp ('t304');
6651            } elsif ($token->{type} eq 'end tag') {            ## "after body" insertion mode
6652              if ($token->{tag_name} eq 'html') {            !!!parse-error (type => 'after body',
6653                if (defined $self->{inner_html_node}) {                            text => $token->{tag_name}, token => $token);
6654                  !!!parse-error (type => 'unmatched end tag:html');            #
6655                  ## Ignore the token          }
6656                  !!!next-token;  
6657                  redo B;          $self->{insertion_mode} = IN_BODY_IM;
6658                } else {          !!!ack-later;
6659                  $previous_insertion_mode = $self->{insertion_mode};          ## reprocess
6660                  $self->{insertion_mode} = 'trailing end';          next B;
6661                  !!!next-token;        } elsif ($token->{type} == END_TAG_TOKEN) {
6662                  redo B;          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6663                }            !!!cp ('t305');
6664              } else {            !!!parse-error (type => 'after html:/',
6665                !!!parse-error (type => 'after body:/'.$token->{tag_name});                            text => $token->{tag_name}, token => $token);
6666              }            
6667              $self->{insertion_mode} = IN_BODY_IM;
6668              ## Reprocess.
6669              next B;
6670            } else {
6671              !!!cp ('t306');
6672            }
6673    
6674            ## "after body" insertion mode
6675            if ($token->{tag_name} eq 'html') {
6676              if (defined $self->{inner_html_node}) {
6677                !!!cp ('t307');
6678                !!!parse-error (type => 'unmatched end tag',
6679                                text => 'html', token => $token);
6680                ## Ignore the token
6681                !!!next-token;
6682                next B;
6683            } else {            } else {
6684              die "$0: $token->{type}: Unknown token type";              !!!cp ('t308');
6685                $self->{insertion_mode} = AFTER_HTML_BODY_IM;
6686                !!!next-token;
6687                next B;
6688            }            }
6689            } else {
6690              !!!cp ('t309');
6691              !!!parse-error (type => 'after body:/',
6692                              text => $token->{tag_name}, token => $token);
6693    
6694            $self->{insertion_mode} = 'in body';            $self->{insertion_mode} = IN_BODY_IM;
6695            ## reprocess            ## reprocess
6696            redo B;            next B;
6697      } elsif ($self->{insertion_mode} eq 'in frameset') {          }
6698        if ($token->{type} eq 'character') {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6699          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          !!!cp ('t309.2');
6700            ## Stop parsing
6701            last B;
6702          } else {
6703            die "$0: $token->{type}: Unknown token type";
6704          }
6705        } elsif ($self->{insertion_mode} & FRAME_IMS) {
6706          if ($token->{type} == CHARACTER_TOKEN) {
6707            if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6708            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);            $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6709              
6710            unless (length $token->{data}) {            unless (length $token->{data}) {
6711                !!!cp ('t310');
6712              !!!next-token;              !!!next-token;
6713              redo B;              next B;
6714            }            }
6715          }          }
6716            
6717          !!!parse-error (type => 'in frameset:#character');          if ($token->{data} =~ s/^[^\x09\x0A\x0C\x20]+//) {
6718          ## Ignore the token            if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6719          !!!next-token;              !!!cp ('t311');
6720          redo B;              !!!parse-error (type => 'in frameset:#text', token => $token);
6721        } elsif ($token->{type} eq 'start tag') {            } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6722          if ($token->{tag_name} eq 'frameset') {              !!!cp ('t312');
6723            !!!insert-element ($token->{tag_name}, $token->{attributes});              !!!parse-error (type => 'after frameset:#text', token => $token);
6724              } else { # "after after frameset"
6725                !!!cp ('t313');
6726                !!!parse-error (type => 'after html:#text', token => $token);
6727              }
6728              
6729              ## Ignore the token.
6730              if (length $token->{data}) {
6731                !!!cp ('t314');
6732                ## reprocess the rest of characters
6733              } else {
6734                !!!cp ('t315');
6735                !!!next-token;
6736              }
6737              next B;
6738            }
6739            
6740            die qq[$0: Character "$token->{data}"];
6741          } elsif ($token->{type} == START_TAG_TOKEN) {
6742            if ($token->{tag_name} eq 'frameset' and
6743                $self->{insertion_mode} == IN_FRAMESET_IM) {
6744              !!!cp ('t318');
6745              !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6746              !!!nack ('t318.1');
6747            !!!next-token;            !!!next-token;
6748            redo B;            next B;
6749          } elsif ($token->{tag_name} eq 'frame') {          } elsif ($token->{tag_name} eq 'frame' and
6750            !!!insert-element ($token->{tag_name}, $token->{attributes});                   $self->{insertion_mode} == IN_FRAMESET_IM) {
6751              !!!cp ('t319');
6752              !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6753            pop @{$self->{open_elements}};            pop @{$self->{open_elements}};
6754              !!!ack ('t319.1');
6755            !!!next-token;            !!!next-token;
6756            redo B;            next B;
6757          } elsif ($token->{tag_name} eq 'noframes') {          } elsif ($token->{tag_name} eq 'noframes') {
6758            ## NOTE: As if in body.            !!!cp ('t320');
6759            $parse_rcdata->(CDATA_CONTENT_MODEL, $insert_to_current);            ## NOTE: As if in head.
6760            redo B;            $parse_rcdata->(CDATA_CONTENT_MODEL);
6761          } else {            next B;
6762            !!!parse-error (type => 'in frameset:'.$token->{tag_name});  
6763              ## NOTE: |<!DOCTYPE HTML><frameset></frameset></html><noframes></noframes>|
6764              ## has no parse error.
6765            } else {
6766              if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6767                !!!cp ('t321');
6768                !!!parse-error (type => 'in frameset',
6769                                text => $token->{tag_name}, token => $token);
6770              } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6771                !!!cp ('t322');
6772                !!!parse-error (type => 'after frameset',
6773                                text => $token->{tag_name}, token => $token);
6774              } else { # "after after frameset"
6775                !!!cp ('t322.2');
6776                !!!parse-error (type => 'after after frameset',
6777                                text => $token->{tag_name}, token => $token);
6778              }
6779            ## Ignore the token            ## Ignore the token
6780              !!!nack ('t322.1');
6781            !!!next-token;            !!!next-token;
6782            redo B;            next B;
6783          }          }
6784        } elsif ($token->{type} eq 'end tag') {        } elsif ($token->{type} == END_TAG_TOKEN) {
6785          if ($token->{tag_name} eq 'frameset') {          if ($token->{tag_name} eq 'frameset' and
6786            if ($self->{open_elements}->[-1]->[1] eq 'html' and              $self->{insertion_mode} == IN_FRAMESET_IM) {
6787              if ($self->{open_elements}->[-1]->[1] & HTML_EL and
6788                @{$self->{open_elements}} == 1) {                @{$self->{open_elements}} == 1) {
6789              !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});              !!!cp ('t325');
6790                !!!parse-error (type => 'unmatched end tag',
6791                                text => $token->{tag_name}, token => $token);
6792              ## Ignore the token              ## Ignore the token
6793              !!!next-token;              !!!next-token;
6794            } else {            } else {
6795                !!!cp ('t326');
6796              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
6797              !!!next-token;              !!!next-token;
6798            }            }
6799    
6800            if (not defined $self->{inner_html_node} and            if (not defined $self->{inner_html_node} and
6801                $self->{open_elements}->[-1]->[1] ne 'frameset') {                not ($self->{open_elements}->[-1]->[1] & FRAMESET_EL)) {
6802              $self->{insertion_mode} = 'after frameset';              !!!cp ('t327');
6803                $self->{insertion_mode} = AFTER_FRAMESET_IM;
6804              } else {
6805                !!!cp ('t328');
6806            }            }
6807            redo B;            next B;
6808            } elsif ($token->{tag_name} eq 'html' and
6809                     $self->{insertion_mode} == AFTER_FRAMESET_IM) {
6810              !!!cp ('t329');
6811              $self->{insertion_mode} = AFTER_HTML_FRAMESET_IM;
6812              !!!next-token;
6813              next B;
6814          } else {          } else {
6815            !!!parse-error (type => 'in frameset:/'.$token->{tag_name});            if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6816                !!!cp ('t330');
6817                !!!parse-error (type => 'in frameset:/',
6818                                text => $token->{tag_name}, token => $token);
6819              } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6820                !!!cp ('t330.1');
6821                !!!parse-error (type => 'after frameset:/',
6822                                text => $token->{tag_name}, token => $token);
6823              } else { # "after after html"
6824                !!!cp ('t331');
6825                !!!parse-error (type => 'after after frameset:/',
6826                                text => $token->{tag_name}, token => $token);
6827              }
6828            ## Ignore the token            ## Ignore the token
6829            !!!next-token;            !!!next-token;
6830            redo B;            next B;
6831          }          }
6832          } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6833            unless ($self->{open_elements}->[-1]->[1] & HTML_EL and
6834                    @{$self->{open_elements}} == 1) { # redundant, maybe
6835              !!!cp ('t331.1');
6836              !!!parse-error (type => 'in body:#eof', token => $token);
6837            } else {
6838              !!!cp ('t331.2');
6839            }
6840            
6841            ## Stop parsing
6842            last B;
6843        } else {        } else {
6844          die "$0: $token->{type}: Unknown token type";          die "$0: $token->{type}: Unknown token type";
6845        }        }
6846      } elsif ($self->{insertion_mode} eq 'after frameset') {      } else {
6847        if ($token->{type} eq 'character') {        die "$0: $self->{insertion_mode}: Unknown insertion mode";
6848              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {      }
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);  
6849    
6850                unless (length $token->{data}) {      ## "in body" insertion mode
6851                  !!!next-token;      if ($token->{type} == START_TAG_TOKEN) {
6852                  redo B;        if ($token->{tag_name} eq 'script') {
6853            !!!cp ('t332');
6854            ## NOTE: This is an "as if in head" code clone
6855            $script_start_tag->();
6856            next B;
6857          } elsif ($token->{tag_name} eq 'style') {
6858            !!!cp ('t333');
6859            ## NOTE: This is an "as if in head" code clone
6860            $parse_rcdata->(CDATA_CONTENT_MODEL);
6861            next B;
6862          } elsif ({
6863                    base => 1, command => 1, eventsource => 1, link => 1,
6864                   }->{$token->{tag_name}}) {
6865            !!!cp ('t334');
6866            ## NOTE: This is an "as if in head" code clone, only "-t" differs
6867            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6868            pop @{$self->{open_elements}};
6869            !!!ack ('t334.1');
6870            !!!next-token;
6871            next B;
6872          } elsif ($token->{tag_name} eq 'meta') {
6873            ## NOTE: This is an "as if in head" code clone, only "-t" differs
6874            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6875            my $meta_el = pop @{$self->{open_elements}};
6876    
6877            unless ($self->{confident}) {
6878              if ($token->{attributes}->{charset}) {
6879                !!!cp ('t335');
6880                ## NOTE: Whether the encoding is supported or not is handled
6881                ## in the {change_encoding} callback.
6882                $self->{change_encoding}
6883                    ->($self, $token->{attributes}->{charset}->{value}, $token);
6884                
6885                $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
6886                    ->set_user_data (manakai_has_reference =>
6887                                         $token->{attributes}->{charset}
6888                                             ->{has_reference});
6889              } elsif ($token->{attributes}->{content}) {
6890                if ($token->{attributes}->{content}->{value}
6891                    =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
6892                        [\x09\x0A\x0C\x0D\x20]*=
6893                        [\x09\x0A\x0C\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
6894                        ([^"'\x09\x0A\x0C\x0D\x20][^\x09\x0A\x0C\x0D\x20\x3B]*))
6895                       /x) {
6896                  !!!cp ('t336');
6897                  ## NOTE: Whether the encoding is supported or not is handled
6898                  ## in the {change_encoding} callback.
6899                  $self->{change_encoding}
6900                      ->($self, defined $1 ? $1 : defined $2 ? $2 : $3, $token);
6901                  $meta_el->[0]->get_attribute_node_ns (undef, 'content')
6902                      ->set_user_data (manakai_has_reference =>
6903                                           $token->{attributes}->{content}
6904                                                 ->{has_reference});
6905                }
6906              }
6907            } else {
6908              if ($token->{attributes}->{charset}) {
6909                !!!cp ('t337');
6910                $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
6911                    ->set_user_data (manakai_has_reference =>
6912                                         $token->{attributes}->{charset}
6913                                             ->{has_reference});
6914              }
6915              if ($token->{attributes}->{content}) {
6916                !!!cp ('t338');
6917                $meta_el->[0]->get_attribute_node_ns (undef, 'content')
6918                    ->set_user_data (manakai_has_reference =>
6919                                         $token->{attributes}->{content}
6920                                             ->{has_reference});
6921              }
6922            }
6923    
6924            !!!ack ('t338.1');
6925            !!!next-token;
6926            next B;
6927          } elsif ($token->{tag_name} eq 'title') {
6928            !!!cp ('t341');
6929            ## NOTE: This is an "as if in head" code clone
6930            $parse_rcdata->(RCDATA_CONTENT_MODEL);
6931            next B;
6932          } elsif ($token->{tag_name} eq 'body') {
6933            !!!parse-error (type => 'in body', text => 'body', token => $token);
6934                  
6935            if (@{$self->{open_elements}} == 1 or
6936                not ($self->{open_elements}->[1]->[1] & BODY_EL)) {
6937              !!!cp ('t342');
6938              ## Ignore the token
6939            } else {
6940              my $body_el = $self->{open_elements}->[1]->[0];
6941              for my $attr_name (keys %{$token->{attributes}}) {
6942                unless ($body_el->has_attribute_ns (undef, $attr_name)) {
6943                  !!!cp ('t343');
6944                  $body_el->set_attribute_ns
6945                    (undef, [undef, $attr_name],
6946                     $token->{attributes}->{$attr_name}->{value});
6947                }
6948              }
6949            }
6950            !!!nack ('t343.1');
6951            !!!next-token;
6952            next B;
6953          } elsif ({
6954                    ## NOTE: Start tags for non-phrasing flow content elements
6955    
6956                    ## NOTE: The normal one
6957                    address => 1, article => 1, aside => 1, blockquote => 1,
6958                    center => 1, datagrid => 1, details => 1, dialog => 1,
6959                    dir => 1, div => 1, dl => 1, fieldset => 1, figure => 1,
6960                    footer => 1, h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1,
6961                    h6 => 1, header => 1, menu => 1, nav => 1, ol => 1, p => 1,
6962                    section => 1, ul => 1,
6963                    ## NOTE: As normal, but drops leading newline
6964                    pre => 1, listing => 1,
6965                    ## NOTE: As normal, but interacts with the form element pointer
6966                    form => 1,
6967                    
6968                    table => 1,
6969                    hr => 1,
6970                   }->{$token->{tag_name}}) {
6971            if ($token->{tag_name} eq 'form' and defined $self->{form_element}) {
6972              !!!cp ('t350');
6973              !!!parse-error (type => 'in form:form', token => $token);
6974              ## Ignore the token
6975              !!!nack ('t350.1');
6976              !!!next-token;
6977              next B;
6978            }
6979    
6980            ## has a p element in scope
6981            INSCOPE: for (reverse @{$self->{open_elements}}) {
6982              if ($_->[1] & P_EL) {
6983                !!!cp ('t344');
6984                !!!back-token; # <form>
6985                $token = {type => END_TAG_TOKEN, tag_name => 'p',
6986                          line => $token->{line}, column => $token->{column}};
6987                next B;
6988              } elsif ($_->[1] & SCOPING_EL) {
6989                !!!cp ('t345');
6990                last INSCOPE;
6991              }
6992            } # INSCOPE
6993              
6994            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6995            if ($token->{tag_name} eq 'pre' or $token->{tag_name} eq 'listing') {
6996              !!!nack ('t346.1');
6997              !!!next-token;
6998              if ($token->{type} == CHARACTER_TOKEN) {
6999                $token->{data} =~ s/^\x0A//;
7000                unless (length $token->{data}) {
7001                  !!!cp ('t346');
7002                  !!!next-token;
7003                } else {
7004                  !!!cp ('t349');
7005                }
7006              } else {
7007                !!!cp ('t348');
7008              }
7009            } elsif ($token->{tag_name} eq 'form') {
7010              !!!cp ('t347.1');
7011              $self->{form_element} = $self->{open_elements}->[-1]->[0];
7012    
7013              !!!nack ('t347.2');
7014              !!!next-token;
7015            } elsif ($token->{tag_name} eq 'table') {
7016              !!!cp ('t382');
7017              push @{$open_tables}, [$self->{open_elements}->[-1]->[0]];
7018              
7019              $self->{insertion_mode} = IN_TABLE_IM;
7020    
7021              !!!nack ('t382.1');
7022              !!!next-token;
7023            } elsif ($token->{tag_name} eq 'hr') {
7024              !!!cp ('t386');
7025              pop @{$self->{open_elements}};
7026            
7027              !!!nack ('t386.1');
7028              !!!next-token;
7029            } else {
7030              !!!nack ('t347.1');
7031              !!!next-token;
7032            }
7033            next B;
7034          } elsif ($token->{tag_name} eq 'li') {
7035            ## NOTE: As normal, but imply </li> when there's another <li> ...
7036    
7037            ## NOTE: Special, Scope (<li><foo><li> == <li><foo><li/></foo></li>)
7038              ## Interpreted as <li><foo/></li><li/> (non-conforming)
7039              ## blockquote (O9.27), center (O), dd (Fx3, O, S3.1.2, IE7),
7040              ## dt (Fx, O, S, IE), dl (O), fieldset (O, S, IE), form (Fx, O, S),
7041              ## hn (O), pre (O), applet (O, S), button (O, S), marquee (Fx, O, S),
7042              ## object (Fx)
7043              ## Generate non-tree (non-conforming)
7044              ## basefont (IE7 (where basefont is non-void)), center (IE),
7045              ## form (IE), hn (IE)
7046            ## address, div, p (<li><foo><li> == <li><foo/></li><li/>)
7047              ## Interpreted as <li><foo><li/></foo></li> (non-conforming)
7048              ## div (Fx, S)
7049    
7050            my $non_optional;
7051            my $i = -1;
7052    
7053            ## 1.
7054            for my $node (reverse @{$self->{open_elements}}) {
7055              if ($node->[1] & LI_EL) {
7056                ## 2. (a) As if </li>
7057                {
7058                  ## If no </li> - not applied
7059                  #
7060    
7061                  ## Otherwise
7062    
7063                  ## 1. generate implied end tags, except for </li>
7064                  #
7065    
7066                  ## 2. If current node != "li", parse error
7067                  if ($non_optional) {
7068                    !!!parse-error (type => 'not closed',
7069                                    text => $non_optional->[0]->manakai_local_name,
7070                                    token => $token);
7071                    !!!cp ('t355');
7072                  } else {
7073                    !!!cp ('t356');
7074                }                }
7075    
7076                  ## 3. Pop
7077                  splice @{$self->{open_elements}}, $i;
7078              }              }
7079    
7080              if ($token->{data} =~ s/^[^\x09\x0A\x0B\x0C\x20]+//) {              last; ## 2. (b) goto 5.
7081                !!!parse-error (type => 'after frameset:#character');            } elsif (
7082                       ## NOTE: not "formatting" and not "phrasing"
7083                       ($node->[1] & SPECIAL_EL or
7084                        $node->[1] & SCOPING_EL) and
7085                       ## NOTE: "li", "dt", and "dd" are in |SPECIAL_EL|.
7086    
7087                       (not $node->[1] & ADDRESS_EL) &
7088                       (not $node->[1] & DIV_EL) &
7089                       (not $node->[1] & P_EL)) {
7090                ## 3.
7091                !!!cp ('t357');
7092                last; ## goto 5.
7093              } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
7094                !!!cp ('t358');
7095                #
7096              } else {
7097                !!!cp ('t359');
7098                $non_optional ||= $node;
7099                #
7100              }
7101              ## 4.
7102              ## goto 2.
7103              $i--;
7104            }
7105    
7106            ## 5. (a) has a |p| element in scope
7107            INSCOPE: for (reverse @{$self->{open_elements}}) {
7108              if ($_->[1] & P_EL) {
7109                !!!cp ('t353');
7110    
7111                ## NOTE: |<p><li>|, for example.
7112    
7113                !!!back-token; # <x>
7114                $token = {type => END_TAG_TOKEN, tag_name => 'p',
7115                          line => $token->{line}, column => $token->{column}};
7116                next B;
7117              } elsif ($_->[1] & SCOPING_EL) {
7118                !!!cp ('t354');
7119                last INSCOPE;
7120              }
7121            } # INSCOPE
7122    
7123                ## Ignore the token.          ## 5. (b) insert
7124                if (length $token->{data}) {          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7125                  ## reprocess the rest of characters          !!!nack ('t359.1');
7126            !!!next-token;
7127            next B;
7128          } elsif ($token->{tag_name} eq 'dt' or
7129                   $token->{tag_name} eq 'dd') {
7130            ## NOTE: As normal, but imply </dt> or </dd> when ...
7131    
7132            my $non_optional;
7133            my $i = -1;
7134    
7135            ## 1.
7136            for my $node (reverse @{$self->{open_elements}}) {
7137              if ($node->[1] & DT_EL or $node->[1] & DD_EL) {
7138                ## 2. (a) As if </li>
7139                {
7140                  ## If no </li> - not applied
7141                  #
7142    
7143                  ## Otherwise
7144    
7145                  ## 1. generate implied end tags, except for </dt> or </dd>
7146                  #
7147    
7148                  ## 2. If current node != "dt"|"dd", parse error
7149                  if ($non_optional) {
7150                    !!!parse-error (type => 'not closed',
7151                                    text => $non_optional->[0]->manakai_local_name,
7152                                    token => $token);
7153                    !!!cp ('t355.1');
7154                } else {                } else {
7155                  !!!next-token;                  !!!cp ('t356.1');
7156                }                }
7157                redo B;  
7158                  ## 3. Pop
7159                  splice @{$self->{open_elements}}, $i;
7160              }              }
7161    
7162          die qq[$0: Character "$token->{data}"];              last; ## 2. (b) goto 5.
7163        } elsif ($token->{type} eq 'start tag') {            } elsif (
7164          if ($token->{tag_name} eq 'noframes') {                     ## NOTE: not "formatting" and not "phrasing"
7165            ## NOTE: As if in body.                     ($node->[1] & SPECIAL_EL or
7166            $parse_rcdata->(CDATA_CONTENT_MODEL, $insert_to_current);                      $node->[1] & SCOPING_EL) and
7167            redo B;                     ## NOTE: "li", "dt", and "dd" are in |SPECIAL_EL|.
7168    
7169                       (not $node->[1] & ADDRESS_EL) &
7170                       (not $node->[1] & DIV_EL) &
7171                       (not $node->[1] & P_EL)) {
7172                ## 3.
7173                !!!cp ('t357.1');
7174                last; ## goto 5.
7175              } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
7176                !!!cp ('t358.1');
7177                #
7178              } else {
7179                !!!cp ('t359.1');
7180                $non_optional ||= $node;
7181                #
7182              }
7183              ## 4.
7184              ## goto 2.
7185              $i--;
7186            }
7187    
7188            ## 5. (a) has a |p| element in scope
7189            INSCOPE: for (reverse @{$self->{open_elements}}) {
7190              if ($_->[1] & P_EL) {
7191                !!!cp ('t353.1');
7192                !!!back-token; # <x>
7193                $token = {type => END_TAG_TOKEN, tag_name => 'p',
7194                          line => $token->{line}, column => $token->{column}};
7195                next B;
7196              } elsif ($_->[1] & SCOPING_EL) {
7197                !!!cp ('t354.1');
7198                last INSCOPE;
7199              }
7200            } # INSCOPE
7201    
7202            ## 5. (b) insert
7203            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7204            !!!nack ('t359.2');
7205            !!!next-token;
7206            next B;
7207          } elsif ($token->{tag_name} eq 'plaintext') {
7208            ## NOTE: As normal, but effectively ends parsing
7209    
7210            ## has a p element in scope
7211            INSCOPE: for (reverse @{$self->{open_elements}}) {
7212              if ($_->[1] & P_EL) {
7213                !!!cp ('t367');
7214                !!!back-token; # <plaintext>
7215                $token = {type => END_TAG_TOKEN, tag_name => 'p',
7216                          line => $token->{line}, column => $token->{column}};
7217                next B;
7218              } elsif ($_->[1] & SCOPING_EL) {
7219                !!!cp ('t368');
7220                last INSCOPE;
7221              }
7222            } # INSCOPE
7223              
7224            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7225              
7226            $self->{content_model} = PLAINTEXT_CONTENT_MODEL;
7227              
7228            !!!nack ('t368.1');
7229            !!!next-token;
7230            next B;
7231          } elsif ($token->{tag_name} eq 'a') {
7232            AFE: for my $i (reverse 0..$#$active_formatting_elements) {
7233              my $node = $active_formatting_elements->[$i];
7234              if ($node->[1] & A_EL) {
7235                !!!cp ('t371');
7236                !!!parse-error (type => 'in a:a', token => $token);
7237                
7238                !!!back-token; # <a>
7239                $token = {type => END_TAG_TOKEN, tag_name => 'a',
7240                          line => $token->{line}, column => $token->{column}};
7241                $formatting_end_tag->($token);
7242                
7243                AFE2: for (reverse 0..$#$active_formatting_elements) {
7244                  if ($active_formatting_elements->[$_]->[0] eq $node->[0]) {
7245                    !!!cp ('t372');
7246                    splice @$active_formatting_elements, $_, 1;
7247                    last AFE2;
7248                  }
7249                } # AFE2
7250                OE: for (reverse 0..$#{$self->{open_elements}}) {
7251                  if ($self->{open_elements}->[$_]->[0] eq $node->[0]) {
7252                    !!!cp ('t373');
7253                    splice @{$self->{open_elements}}, $_, 1;
7254                    last OE;
7255                  }
7256                } # OE
7257                last AFE;
7258              } elsif ($node->[0] eq '#marker') {
7259                !!!cp ('t374');
7260                last AFE;
7261              }
7262            } # AFE
7263              
7264            $reconstruct_active_formatting_elements->($insert_to_current);
7265    
7266            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7267            push @$active_formatting_elements, $self->{open_elements}->[-1];
7268    
7269            !!!nack ('t374.1');
7270            !!!next-token;
7271            next B;
7272          } elsif ($token->{tag_name} eq 'nobr') {
7273            $reconstruct_active_formatting_elements->($insert_to_current);
7274    
7275            ## has a |nobr| element in scope
7276            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7277              my $node = $self->{open_elements}->[$_];
7278              if ($node->[1] & NOBR_EL) {
7279                !!!cp ('t376');
7280                !!!parse-error (type => 'in nobr:nobr', token => $token);
7281                !!!back-token; # <nobr>
7282                $token = {type => END_TAG_TOKEN, tag_name => 'nobr',
7283                          line => $token->{line}, column => $token->{column}};
7284                next B;
7285              } elsif ($node->[1] & SCOPING_EL) {
7286                !!!cp ('t377');
7287                last INSCOPE;
7288              }
7289            } # INSCOPE
7290            
7291            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7292            push @$active_formatting_elements, $self->{open_elements}->[-1];
7293            
7294            !!!nack ('t377.1');
7295            !!!next-token;
7296            next B;
7297          } elsif ($token->{tag_name} eq 'button') {
7298            ## has a button element in scope
7299            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7300              my $node = $self->{open_elements}->[$_];
7301              if ($node->[1] & BUTTON_EL) {
7302                !!!cp ('t378');
7303                !!!parse-error (type => 'in button:button', token => $token);
7304                !!!back-token; # <button>
7305                $token = {type => END_TAG_TOKEN, tag_name => 'button',
7306                          line => $token->{line}, column => $token->{column}};
7307                next B;
7308              } elsif ($node->[1] & SCOPING_EL) {
7309                !!!cp ('t379');
7310                last INSCOPE;
7311              }
7312            } # INSCOPE
7313              
7314            $reconstruct_active_formatting_elements->($insert_to_current);
7315              
7316            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7317    
7318            ## TODO: associate with $self->{form_element} if defined
7319    
7320            push @$active_formatting_elements, ['#marker', ''];
7321    
7322            !!!nack ('t379.1');
7323            !!!next-token;
7324            next B;
7325          } elsif ({
7326                    xmp => 1,
7327                    iframe => 1,
7328                    noembed => 1,
7329                    noframes => 1, ## NOTE: This is an "as if in head" code clone.
7330                    noscript => 0, ## TODO: 1 if scripting is enabled
7331                   }->{$token->{tag_name}}) {
7332            if ($token->{tag_name} eq 'xmp') {
7333              !!!cp ('t381');
7334              $reconstruct_active_formatting_elements->($insert_to_current);
7335          } else {          } else {
7336            !!!parse-error (type => 'after frameset:'.$token->{tag_name});            !!!cp ('t399');
7337            }
7338            ## NOTE: There is an "as if in body" code clone.
7339            $parse_rcdata->(CDATA_CONTENT_MODEL);
7340            next B;
7341          } elsif ($token->{tag_name} eq 'isindex') {
7342            !!!parse-error (type => 'isindex', token => $token);
7343            
7344            if (defined $self->{form_element}) {
7345              !!!cp ('t389');
7346            ## Ignore the token            ## Ignore the token
7347              !!!nack ('t389'); ## NOTE: Not acknowledged.
7348            !!!next-token;            !!!next-token;
7349            redo B;            next B;
7350          }          } else {
7351        } elsif ($token->{type} eq 'end tag') {            !!!ack ('t391.1');
7352          if ($token->{tag_name} eq 'html') {  
7353            $previous_insertion_mode = $self->{insertion_mode};            my $at = $token->{attributes};
7354            $self->{insertion_mode} = 'trailing end';            my $form_attrs;
7355              $form_attrs->{action} = $at->{action} if $at->{action};
7356              my $prompt_attr = $at->{prompt};
7357              $at->{name} = {name => 'name', value => 'isindex'};
7358              delete $at->{action};
7359              delete $at->{prompt};
7360              my @tokens = (
7361                            {type => START_TAG_TOKEN, tag_name => 'form',
7362                             attributes => $form_attrs,
7363                             line => $token->{line}, column => $token->{column}},
7364                            {type => START_TAG_TOKEN, tag_name => 'hr',
7365                             line => $token->{line}, column => $token->{column}},
7366                            {type => START_TAG_TOKEN, tag_name => 'p',
7367                             line => $token->{line}, column => $token->{column}},
7368                            {type => START_TAG_TOKEN, tag_name => 'label',
7369                             line => $token->{line}, column => $token->{column}},
7370                           );
7371              if ($prompt_attr) {
7372                !!!cp ('t390');
7373                push @tokens, {type => CHARACTER_TOKEN, data => $prompt_attr->{value},
7374                               #line => $token->{line}, column => $token->{column},
7375                              };
7376              } else {
7377                !!!cp ('t391');
7378                push @tokens, {type => CHARACTER_TOKEN,
7379                               data => 'This is a searchable index. Insert your search keywords here: ',
7380                               #line => $token->{line}, column => $token->{column},
7381                              }; # SHOULD
7382                ## TODO: make this configurable
7383              }
7384              push @tokens,
7385                            {type => START_TAG_TOKEN, tag_name => 'input', attributes => $at,
7386                             line => $token->{line}, column => $token->{column}},
7387                            #{type => CHARACTER_TOKEN, data => ''}, # SHOULD
7388                            {type => END_TAG_TOKEN, tag_name => 'label',
7389                             line => $token->{line}, column => $token->{column}},
7390                            {type => END_TAG_TOKEN, tag_name => 'p',
7391                             line => $token->{line}, column => $token->{column}},
7392                            {type => START_TAG_TOKEN, tag_name => 'hr',
7393                             line => $token->{line}, column => $token->{column}},
7394                            {type => END_TAG_TOKEN, tag_name => 'form',
7395                             line => $token->{line}, column => $token->{column}};
7396              !!!back-token (@tokens);
7397            !!!next-token;            !!!next-token;
7398            redo B;            next B;
7399            }
7400          } elsif ($token->{tag_name} eq 'textarea') {
7401            my $tag_name = $token->{tag_name};
7402            my $el;
7403            !!!create-element ($el, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);
7404            
7405            ## TODO: $self->{form_element} if defined
7406            $self->{content_model} = RCDATA_CONTENT_MODEL;
7407            delete $self->{escape}; # MUST
7408            
7409            $insert->($el);
7410            
7411            my $text = '';
7412            !!!nack ('t392.1');
7413            !!!next-token;
7414            if ($token->{type} == CHARACTER_TOKEN) {
7415              $token->{data} =~ s/^\x0A//;
7416              unless (length $token->{data}) {
7417                !!!cp ('t392');
7418                !!!next-token;
7419              } else {
7420                !!!cp ('t393');
7421              }
7422          } else {          } else {
7423            !!!parse-error (type => 'after frameset:/'.$token->{tag_name});            !!!cp ('t394');
7424            ## Ignore the token          }
7425            while ($token->{type} == CHARACTER_TOKEN) {
7426              !!!cp ('t395');
7427              $text .= $token->{data};
7428            !!!next-token;            !!!next-token;
           redo B;  
7429          }          }
7430            if (length $text) {
7431              !!!cp ('t396');
7432              $el->manakai_append_text ($text);
7433            }
7434            
7435            $self->{content_model} = PCDATA_CONTENT_MODEL;
7436            
7437            if ($token->{type} == END_TAG_TOKEN and
7438                $token->{tag_name} eq $tag_name) {
7439              !!!cp ('t397');
7440              ## Ignore the token
7441            } else {
7442              !!!cp ('t398');
7443              !!!parse-error (type => 'in RCDATA:#eof', token => $token);
7444            }
7445            !!!next-token;
7446            next B;
7447          } elsif ($token->{tag_name} eq 'optgroup' or
7448                   $token->{tag_name} eq 'option') {
7449            ## has an |option| element in scope
7450            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7451              my $node = $self->{open_elements}->[$_];
7452              if ($node->[1] & OPTION_EL) {
7453                !!!cp ('t397.1');
7454                ## NOTE: As if </option>
7455                !!!back-token; # <option> or <optgroup>
7456                $token = {type => END_TAG_TOKEN, tag_name => 'option',
7457                          line => $token->{line}, column => $token->{column}};
7458                next B;
7459              } elsif ($node->[1] & SCOPING_EL) {
7460                !!!cp ('t397.2');
7461                last INSCOPE;
7462              }
7463            } # INSCOPE
7464    
7465            $reconstruct_active_formatting_elements->($insert_to_current);
7466    
7467            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7468    
7469            !!!nack ('t397.3');
7470            !!!next-token;
7471            redo B;
7472          } elsif ($token->{tag_name} eq 'rt' or
7473                   $token->{tag_name} eq 'rp') {
7474            ## has a |ruby| element in scope
7475            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7476              my $node = $self->{open_elements}->[$_];
7477              if ($node->[1] & RUBY_EL) {
7478                !!!cp ('t398.1');
7479                ## generate implied end tags
7480                while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7481                  !!!cp ('t398.2');
7482                  pop @{$self->{open_elements}};
7483                }
7484                unless ($self->{open_elements}->[-1]->[1] & RUBY_EL) {
7485                  !!!cp ('t398.3');
7486                  !!!parse-error (type => 'not closed',
7487                                  text => $self->{open_elements}->[-1]->[0]
7488                                      ->manakai_local_name,
7489                                  token => $token);
7490                  pop @{$self->{open_elements}}
7491                      while not $self->{open_elements}->[-1]->[1] & RUBY_EL;
7492                }
7493                last INSCOPE;
7494              } elsif ($node->[1] & SCOPING_EL) {
7495                !!!cp ('t398.4');
7496                last INSCOPE;
7497              }
7498            } # INSCOPE
7499    
7500            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7501    
7502            !!!nack ('t398.5');
7503            !!!next-token;
7504            redo B;
7505          } elsif ($token->{tag_name} eq 'math' or
7506                   $token->{tag_name} eq 'svg') {
7507            $reconstruct_active_formatting_elements->($insert_to_current);
7508    
7509            ## "Adjust MathML attributes" ('math' only) - done in insert-element-f
7510    
7511            ## "adjust SVG attributes" ('svg' only) - done in insert-element-f
7512    
7513            ## "adjust foreign attributes" - done in insert-element-f
7514            
7515            !!!insert-element-f ($token->{tag_name} eq 'math' ? $MML_NS : $SVG_NS, $token->{tag_name}, $token->{attributes}, $token);
7516            
7517            if ($self->{self_closing}) {
7518              pop @{$self->{open_elements}};
7519              !!!ack ('t398.6');
7520            } else {
7521              !!!cp ('t398.7');
7522              $self->{insertion_mode} |= IN_FOREIGN_CONTENT_IM;
7523              ## NOTE: |<body><math><mi><svg>| -> "in foreign content" insertion
7524              ## mode, "in body" (not "in foreign content") secondary insertion
7525              ## mode, maybe.
7526            }
7527    
7528            !!!next-token;
7529            next B;
7530          } elsif ({
7531                    caption => 1, col => 1, colgroup => 1, frame => 1,
7532                    frameset => 1, head => 1,
7533                    tbody => 1, td => 1, tfoot => 1, th => 1,
7534                    thead => 1, tr => 1,
7535                   }->{$token->{tag_name}}) {
7536            !!!cp ('t401');
7537            !!!parse-error (type => 'in body',
7538                            text => $token->{tag_name}, token => $token);
7539            ## Ignore the token
7540            !!!nack ('t401.1'); ## NOTE: |<col/>| or |<frame/>| here is an error.
7541            !!!next-token;
7542            next B;
7543          } elsif ($token->{tag_name} eq 'param' or
7544                   $token->{tag_name} eq 'source') {
7545            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7546            pop @{$self->{open_elements}};
7547    
7548            !!!ack ('t398.5');
7549            !!!next-token;
7550            redo B;
7551        } else {        } else {
7552          die "$0: $token->{type}: Unknown token type";          if ($token->{tag_name} eq 'image') {
7553              !!!cp ('t384');
7554              !!!parse-error (type => 'image', token => $token);
7555              $token->{tag_name} = 'img';
7556            } else {
7557              !!!cp ('t385');
7558            }
7559    
7560            ## NOTE: There is an "as if <br>" code clone.
7561            $reconstruct_active_formatting_elements->($insert_to_current);
7562            
7563            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7564    
7565            if ({
7566                 applet => 1, marquee => 1, object => 1,
7567                }->{$token->{tag_name}}) {
7568              !!!cp ('t380');
7569              push @$active_formatting_elements, ['#marker', ''];
7570              !!!nack ('t380.1');
7571            } elsif ({
7572                      b => 1, big => 1, em => 1, font => 1, i => 1,
7573                      s => 1, small => 1, strike => 1,
7574                      strong => 1, tt => 1, u => 1,
7575                     }->{$token->{tag_name}}) {
7576              !!!cp ('t375');
7577              push @$active_formatting_elements, $self->{open_elements}->[-1];
7578              !!!nack ('t375.1');
7579            } elsif ($token->{tag_name} eq 'input') {
7580              !!!cp ('t388');
7581              ## TODO: associate with $self->{form_element} if defined
7582              pop @{$self->{open_elements}};
7583              !!!ack ('t388.2');
7584            } elsif ({
7585                      area => 1, basefont => 1, bgsound => 1, br => 1,
7586                      embed => 1, img => 1, spacer => 1, wbr => 1,
7587                     }->{$token->{tag_name}}) {
7588              !!!cp ('t388.1');
7589              pop @{$self->{open_elements}};
7590              !!!ack ('t388.3');
7591            } elsif ($token->{tag_name} eq 'select') {
7592              ## TODO: associate with $self->{form_element} if defined
7593            
7594              if ($self->{insertion_mode} & TABLE_IMS or
7595                  $self->{insertion_mode} & BODY_TABLE_IMS or
7596                  $self->{insertion_mode} == IN_COLUMN_GROUP_IM) {
7597                !!!cp ('t400.1');
7598                $self->{insertion_mode} = IN_SELECT_IN_TABLE_IM;
7599              } else {
7600                !!!cp ('t400.2');
7601                $self->{insertion_mode} = IN_SELECT_IM;
7602              }
7603              !!!nack ('t400.3');
7604            } else {
7605              !!!nack ('t402');
7606            }
7607            
7608            !!!next-token;
7609            next B;
7610        }        }
7611        } elsif ($token->{type} == END_TAG_TOKEN) {
7612          if ($token->{tag_name} eq 'body') {
7613            ## has a |body| element in scope
7614            my $i;
7615            INSCOPE: {
7616              for (reverse @{$self->{open_elements}}) {
7617                if ($_->[1] & BODY_EL) {
7618                  !!!cp ('t405');
7619                  $i = $_;
7620                  last INSCOPE;
7621                } elsif ($_->[1] & SCOPING_EL) {
7622                  !!!cp ('t405.1');
7623                  last;
7624                }
7625              }
7626    
7627        ## ISSUE: An issue in spec here            ## NOTE: |<marquee></body>|, |<svg><foreignobject></body>|
7628      } elsif ($self->{insertion_mode} eq 'trailing end') {  
7629        ## states in the main stage is preserved yet # MUST            !!!parse-error (type => 'unmatched end tag',
7630                                    text => $token->{tag_name}, token => $token);
7631        if ($token->{type} eq 'character') {            ## NOTE: Ignore the token.
7632          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {            !!!next-token;
7633            my $data = $1;            next B;
7634            ## As if in the main phase.          } # INSCOPE
7635            ## NOTE: The insertion mode in the main phase  
7636            ## just before the phase has been changed to the trailing          for (@{$self->{open_elements}}) {
7637            ## end phase is either "after body" or "after frameset".            unless ($_->[1] & ALL_END_TAG_OPTIONAL_EL) {
7638            $reconstruct_active_formatting_elements->($insert_to_current);              !!!cp ('t403');
7639                !!!parse-error (type => 'not closed',
7640                                text => $_->[0]->manakai_local_name,
7641                                token => $token);
7642                last;
7643              } else {
7644                !!!cp ('t404');
7645              }
7646            }
7647    
7648            $self->{insertion_mode} = AFTER_BODY_IM;
7649            !!!next-token;
7650            next B;
7651          } elsif ($token->{tag_name} eq 'html') {
7652            ## TODO: Update this code.  It seems that the code below is not
7653            ## up-to-date, though it has same effect as speced.
7654            if (@{$self->{open_elements}} > 1 and
7655                $self->{open_elements}->[1]->[1] & BODY_EL) {
7656              unless ($self->{open_elements}->[-1]->[1] & BODY_EL) {
7657                !!!cp ('t406');
7658                !!!parse-error (type => 'not closed',
7659                                text => $self->{open_elements}->[1]->[0]
7660                                    ->manakai_local_name,
7661                                token => $token);
7662              } else {
7663                !!!cp ('t407');
7664              }
7665              $self->{insertion_mode} = AFTER_BODY_IM;
7666              ## reprocess
7667              next B;
7668            } else {
7669              !!!cp ('t408');
7670              !!!parse-error (type => 'unmatched end tag',
7671                              text => $token->{tag_name}, token => $token);
7672              ## Ignore the token
7673              !!!next-token;
7674              next B;
7675            }
7676          } elsif ({
7677                    ## NOTE: End tags for non-phrasing flow content elements
7678    
7679                    ## NOTE: The normal ones
7680                    address => 1, article => 1, aside => 1, blockquote => 1,
7681                    center => 1, datagrid => 1, details => 1, dialog => 1,
7682                    dir => 1, div => 1, dl => 1, fieldset => 1, figure => 1,
7683                    footer => 1, header => 1, listing => 1, menu => 1, nav => 1,
7684                    ol => 1, pre => 1, section => 1, ul => 1,
7685    
7686                    ## NOTE: As normal, but ... optional tags
7687                    dd => 1, dt => 1, li => 1,
7688    
7689                    applet => 1, button => 1, marquee => 1, object => 1,
7690                   }->{$token->{tag_name}}) {
7691            ## NOTE: Code for <li> start tags includes "as if </li>" code.
7692            ## Code for <dt> or <dd> start tags includes "as if </dt> or
7693            ## </dd>" code.
7694    
7695            ## has an element in scope
7696            my $i;
7697            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7698              my $node = $self->{open_elements}->[$_];
7699              if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
7700                !!!cp ('t410');
7701                $i = $_;
7702                last INSCOPE;
7703              } elsif ($node->[1] & SCOPING_EL) {
7704                !!!cp ('t411');
7705                last INSCOPE;
7706              }
7707            } # INSCOPE
7708    
7709            unless (defined $i) { # has an element in scope
7710              !!!cp ('t413');
7711              !!!parse-error (type => 'unmatched end tag',
7712                              text => $token->{tag_name}, token => $token);
7713              ## NOTE: Ignore the token.
7714            } else {
7715              ## Step 1. generate implied end tags
7716              while ({
7717                      ## END_TAG_OPTIONAL_EL
7718                      dd => ($token->{tag_name} ne 'dd'),
7719                      dt => ($token->{tag_name} ne 'dt'),
7720                      li => ($token->{tag_name} ne 'li'),
7721                      option => 1,
7722                      optgroup => 1,
7723                      p => 1,
7724                      rt => 1,
7725                      rp => 1,
7726                     }->{$self->{open_elements}->[-1]->[0]->manakai_local_name}) {
7727                !!!cp ('t409');
7728                pop @{$self->{open_elements}};
7729              }
7730    
7731              ## Step 2.
7732              if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7733                      ne $token->{tag_name}) {
7734                !!!cp ('t412');
7735                !!!parse-error (type => 'not closed',
7736                                text => $self->{open_elements}->[-1]->[0]
7737                                    ->manakai_local_name,
7738                                token => $token);
7739              } else {
7740                !!!cp ('t414');
7741              }
7742    
7743              ## Step 3.
7744              splice @{$self->{open_elements}}, $i;
7745    
7746              ## Step 4.
7747              $clear_up_to_marker->()
7748                  if {
7749                    applet => 1, button => 1, marquee => 1, object => 1,
7750                  }->{$token->{tag_name}};
7751            }
7752            !!!next-token;
7753            next B;
7754          } elsif ($token->{tag_name} eq 'form') {
7755            ## NOTE: As normal, but interacts with the form element pointer
7756    
7757            undef $self->{form_element};
7758    
7759            ## has an element in scope
7760            my $i;
7761            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7762              my $node = $self->{open_elements}->[$_];
7763              if ($node->[1] & FORM_EL) {
7764                !!!cp ('t418');
7765                $i = $_;
7766                last INSCOPE;
7767              } elsif ($node->[1] & SCOPING_EL) {
7768                !!!cp ('t419');
7769                last INSCOPE;
7770              }
7771            } # INSCOPE
7772    
7773            unless (defined $i) { # has an element in scope
7774              !!!cp ('t421');
7775              !!!parse-error (type => 'unmatched end tag',
7776                              text => $token->{tag_name}, token => $token);
7777              ## NOTE: Ignore the token.
7778            } else {
7779              ## Step 1. generate implied end tags
7780              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7781                !!!cp ('t417');
7782                pop @{$self->{open_elements}};
7783              }
7784                        
7785            $self->{open_elements}->[-1]->[0]->manakai_append_text ($data);            ## Step 2.
7786              if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7787                      ne $token->{tag_name}) {
7788                !!!cp ('t417.1');
7789                !!!parse-error (type => 'not closed',
7790                                text => $self->{open_elements}->[-1]->[0]
7791                                    ->manakai_local_name,
7792                                token => $token);
7793              } else {
7794                !!!cp ('t420');
7795              }  
7796                        
7797            unless (length $token->{data}) {            ## Step 3.
7798              !!!next-token;            splice @{$self->{open_elements}}, $i;
7799              redo B;          }
7800    
7801            !!!next-token;
7802            next B;
7803          } elsif ({
7804                    ## NOTE: As normal, except acts as a closer for any ...
7805                    h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
7806                   }->{$token->{tag_name}}) {
7807            ## has an element in scope
7808            my $i;
7809            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7810              my $node = $self->{open_elements}->[$_];
7811              if ($node->[1] & HEADING_EL) {
7812                !!!cp ('t423');
7813                $i = $_;
7814                last INSCOPE;
7815              } elsif ($node->[1] & SCOPING_EL) {
7816                !!!cp ('t424');
7817                last INSCOPE;
7818              }
7819            } # INSCOPE
7820    
7821            unless (defined $i) { # has an element in scope
7822              !!!cp ('t425.1');
7823              !!!parse-error (type => 'unmatched end tag',
7824                              text => $token->{tag_name}, token => $token);
7825              ## NOTE: Ignore the token.
7826            } else {
7827              ## Step 1. generate implied end tags
7828              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7829                !!!cp ('t422');
7830                pop @{$self->{open_elements}};
7831            }            }
7832              
7833              ## Step 2.
7834              if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7835                      ne $token->{tag_name}) {
7836                !!!cp ('t425');
7837                !!!parse-error (type => 'unmatched end tag',
7838                                text => $token->{tag_name}, token => $token);
7839              } else {
7840                !!!cp ('t426');
7841              }
7842    
7843              ## Step 3.
7844              splice @{$self->{open_elements}}, $i;
7845          }          }
7846            
7847            !!!next-token;
7848            next B;
7849          } elsif ($token->{tag_name} eq 'p') {
7850            ## NOTE: As normal, except </p> implies <p> and ...
7851    
7852          !!!parse-error (type => 'after html:#character');          ## has an element in scope
7853          $self->{insertion_mode} = $previous_insertion_mode;          my $non_optional;
7854          ## reprocess          my $i;
7855          redo B;          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7856        } elsif ($token->{type} eq 'start tag') {            my $node = $self->{open_elements}->[$_];
7857          !!!parse-error (type => 'after html:'.$token->{tag_name});            if ($node->[1] & P_EL) {
7858          $self->{insertion_mode} = $previous_insertion_mode;              !!!cp ('t410.1');
7859          ## reprocess              $i = $_;
7860          redo B;              last INSCOPE;
7861        } elsif ($token->{type} eq 'end tag') {            } elsif ($node->[1] & SCOPING_EL) {
7862          !!!parse-error (type => 'after html:/'.$token->{tag_name});              !!!cp ('t411.1');
7863          $self->{insertion_mode} = $previous_insertion_mode;              last INSCOPE;
7864          ## reprocess            } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
7865          redo B;              ## NOTE: |END_TAG_OPTIONAL_EL| includes "p"
7866                !!!cp ('t411.2');
7867                #
7868              } else {
7869                !!!cp ('t411.3');
7870                $non_optional ||= $node;
7871                #
7872              }
7873            } # INSCOPE
7874    
7875            if (defined $i) {
7876              ## 1. Generate implied end tags
7877              #
7878    
7879              ## 2. If current node != "p", parse error
7880              if ($non_optional) {
7881                !!!cp ('t412.1');
7882                !!!parse-error (type => 'not closed',
7883                                text => $non_optional->[0]->manakai_local_name,
7884                                token => $token);
7885              } else {
7886                !!!cp ('t414.1');
7887              }
7888    
7889              ## 3. Pop
7890              splice @{$self->{open_elements}}, $i;
7891            } else {
7892              !!!cp ('t413.1');
7893              !!!parse-error (type => 'unmatched end tag',
7894                              text => $token->{tag_name}, token => $token);
7895    
7896              !!!cp ('t415.1');
7897              ## As if <p>, then reprocess the current token
7898              my $el;
7899              !!!create-element ($el, $HTML_NS, 'p',, $token);
7900              $insert->($el);
7901              ## NOTE: Not inserted into |$self->{open_elements}|.
7902            }
7903    
7904            !!!next-token;
7905            next B;
7906          } elsif ({
7907                    a => 1,
7908                    b => 1, big => 1, em => 1, font => 1, i => 1,
7909                    nobr => 1, s => 1, small => 1, strike => 1,
7910                    strong => 1, tt => 1, u => 1,
7911                   }->{$token->{tag_name}}) {
7912            !!!cp ('t427');
7913            $formatting_end_tag->($token);
7914            next B;
7915          } elsif ($token->{tag_name} eq 'br') {
7916            !!!cp ('t428');
7917            !!!parse-error (type => 'unmatched end tag',
7918                            text => 'br', token => $token);
7919    
7920            ## As if <br>
7921            $reconstruct_active_formatting_elements->($insert_to_current);
7922            
7923            my $el;
7924            !!!create-element ($el, $HTML_NS, 'br',, $token);
7925            $insert->($el);
7926            
7927            ## Ignore the token.
7928            !!!next-token;
7929            next B;
7930        } else {        } else {
7931          die "$0: $token->{type}: Unknown token";          if ($token->{tag_name} eq 'sarcasm') {
7932              sleep 0.001; # take a deep breath
7933            }
7934    
7935            ## Step 1
7936            my $node_i = -1;
7937            my $node = $self->{open_elements}->[$node_i];
7938    
7939            ## Step 2
7940            S2: {
7941              my $node_tag_name = $node->[0]->manakai_local_name;
7942              $node_tag_name =~ tr/A-Z/a-z/; # for SVG camelCase tag names
7943              if ($node_tag_name eq $token->{tag_name}) {
7944                ## Step 1
7945                ## generate implied end tags
7946                while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7947                  !!!cp ('t430');
7948                  ## NOTE: |<ruby><rt></ruby>|.
7949                  ## ISSUE: <ruby><rt></rt> will also take this code path,
7950                  ## which seems wrong.
7951                  pop @{$self->{open_elements}};
7952                  $node_i++;
7953                }
7954            
7955                ## Step 2
7956                my $current_tag_name
7957                    = $self->{open_elements}->[-1]->[0]->manakai_local_name;
7958                $current_tag_name =~ tr/A-Z/a-z/;
7959                if ($current_tag_name ne $token->{tag_name}) {
7960                  !!!cp ('t431');
7961                  ## NOTE: <x><y></x>
7962                  !!!parse-error (type => 'not closed',
7963                                  text => $self->{open_elements}->[-1]->[0]
7964                                      ->manakai_local_name,
7965                                  token => $token);
7966                } else {
7967                  !!!cp ('t432');
7968                }
7969                
7970                ## Step 3
7971                splice @{$self->{open_elements}}, $node_i if $node_i < 0;
7972    
7973                !!!next-token;
7974                last S2;
7975              } else {
7976                ## Step 3
7977                if (not ($node->[1] & FORMATTING_EL) and
7978                    #not $phrasing_category->{$node->[1]} and
7979                    ($node->[1] & SPECIAL_EL or
7980                     $node->[1] & SCOPING_EL)) {
7981                  !!!cp ('t433');
7982                  !!!parse-error (type => 'unmatched end tag',
7983                                  text => $token->{tag_name}, token => $token);
7984                  ## Ignore the token
7985                  !!!next-token;
7986                  last S2;
7987    
7988                  ## NOTE: |<span><dd></span>a|: In Safari 3.1.2 and Opera
7989                  ## 9.27, "a" is a child of <dd> (conforming).  In
7990                  ## Firefox 3.0.2, "a" is a child of <body>.  In WinIE 7,
7991                  ## "a" is a child of both <body> and <dd>.
7992                }
7993                
7994                !!!cp ('t434');
7995              }
7996              
7997              ## Step 4
7998              $node_i--;
7999              $node = $self->{open_elements}->[$node_i];
8000              
8001              ## Step 5;
8002              redo S2;
8003            } # S2
8004            next B;
8005        }        }
8006      } else {      }
8007        die "$0: $self->{insertion_mode}: Unknown insertion mode";      next B;
8008      } continue { # B
8009        if ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {
8010          ## NOTE: The code below is executed in cases where it does not have
8011          ## to be, but it it is harmless even in those cases.
8012          ## has an element in scope
8013          INSCOPE: {
8014            for (reverse 0..$#{$self->{open_elements}}) {
8015              my $node = $self->{open_elements}->[$_];
8016              if ($node->[1] & FOREIGN_EL) {
8017                last INSCOPE;
8018              } elsif ($node->[1] & SCOPING_EL) {
8019                last;
8020              }
8021            }
8022            
8023            ## NOTE: No foreign element in scope.
8024            $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
8025          } # INSCOPE
8026      }      }
8027    } # B    } # B
8028    
# Line 5208  sub _tree_construction_main ($) { Line 8031  sub _tree_construction_main ($) {
8031    ## TODO: script stuffs    ## TODO: script stuffs
8032  } # _tree_construct_main  } # _tree_construct_main
8033    
8034  sub set_inner_html ($$$) {  sub set_inner_html ($$$$;$) {
8035    my $class = shift;    my $class = shift;
8036    my $node = shift;    my $node = shift;
8037    my $s = \$_[0];    #my $s = \$_[0];
8038    my $onerror = $_[1];    my $onerror = $_[1];
8039      my $get_wrapper = $_[2] || sub ($) { return $_[0] };
8040    
8041      ## ISSUE: Should {confident} be true?
8042    
8043    my $nt = $node->node_type;    my $nt = $node->node_type;
8044    if ($nt == 9) {    if ($nt == 9) {
# Line 5229  sub set_inner_html ($$$) { Line 8055  sub set_inner_html ($$$) {
8055      }      }
8056    
8057      ## Step 3, 4, 5 # MUST      ## Step 3, 4, 5 # MUST
8058      $class->parse_string ($$s => $node, $onerror);      $class->parse_char_string ($_[0] => $node, $onerror, $get_wrapper);
8059    } elsif ($nt == 1) {    } elsif ($nt == 1) {
8060      ## TODO: If non-html element      ## TODO: If non-html element
8061    
8062      ## NOTE: Most of this code is copied from |parse_string|      ## NOTE: Most of this code is copied from |parse_string|
8063    
8064    ## TODO: Support for $get_wrapper
8065    
8066      ## Step 1 # MUST      ## Step 1 # MUST
8067      my $this_doc = $node->owner_document;      my $this_doc = $node->owner_document;
8068      my $doc = $this_doc->implementation->create_document;      my $doc = $this_doc->implementation->create_document;
# Line 5242  sub set_inner_html ($$$) { Line 8070  sub set_inner_html ($$$) {
8070      my $p = $class->new;      my $p = $class->new;
8071      $p->{document} = $doc;      $p->{document} = $doc;
8072    
8073      ## Step 9 # MUST      ## Step 8 # MUST
8074      my $i = 0;      my $i = 0;
8075      my $line = 1;      $p->{line_prev} = $p->{line} = 1;
8076      my $column = 0;      $p->{column_prev} = $p->{column} = 0;
8077      $p->{set_next_input_character} = sub {      require Whatpm::Charset::DecodeHandle;
8078        my $input = Whatpm::Charset::DecodeHandle::CharString->new (\($_[0]));
8079        $input = $get_wrapper->($input);
8080        $p->{set_nc} = sub {
8081        my $self = shift;        my $self = shift;
8082    
8083        pop @{$self->{prev_input_character}};        my $char = '';
8084        unshift @{$self->{prev_input_character}}, $self->{next_input_character};        if (defined $self->{next_nc}) {
8085            $char = $self->{next_nc};
8086            delete $self->{next_nc};
8087            $self->{nc} = ord $char;
8088          } else {
8089            $self->{char_buffer} = '';
8090            $self->{char_buffer_pos} = 0;
8091            
8092            my $count = $input->manakai_read_until
8093                ($self->{char_buffer}, qr/[^\x00\x0A\x0D]/,
8094                 $self->{char_buffer_pos});
8095            if ($count) {
8096              $self->{line_prev} = $self->{line};
8097              $self->{column_prev} = $self->{column};
8098              $self->{column}++;
8099              $self->{nc}
8100                  = ord substr ($self->{char_buffer},
8101                                $self->{char_buffer_pos}++, 1);
8102              return;
8103            }
8104            
8105            if ($input->read ($char, 1)) {
8106              $self->{nc} = ord $char;
8107            } else {
8108              $self->{nc} = -1;
8109              return;
8110            }
8111          }
8112    
8113          ($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column});
8114          $p->{column}++;
8115    
8116        $self->{next_input_character} = -1 and return if $i >= length $$s;        if ($self->{nc} == 0x000A) { # LF
8117        $self->{next_input_character} = ord substr $$s, $i++, 1;          $p->{line}++;
8118        $column++;          $p->{column} = 0;
8119            !!!cp ('i1');
8120        if ($self->{next_input_character} == 0x000A) { # LF        } elsif ($self->{nc} == 0x000D) { # CR
8121          $line++;  ## TODO: support for abort/streaming
8122          $column = 0;          my $next = '';
8123        } elsif ($self->{next_input_character} == 0x000D) { # CR          if ($input->read ($next, 1) and $next ne "\x0A") {
8124          $i++ if substr ($$s, $i, 1) eq "\x0A";            $self->{next_nc} = $next;
8125          $self->{next_input_character} = 0x000A; # LF # MUST          }
8126          $line++;          $self->{nc} = 0x000A; # LF # MUST
8127          $column = 0;          $p->{line}++;
8128        } elsif ($self->{next_input_character} > 0x10FFFF) {          $p->{column} = 0;
8129          $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST          !!!cp ('i2');
8130        } elsif ($self->{next_input_character} == 0x0000) { # NULL        } elsif ($self->{nc} == 0x0000) { # NULL
8131            !!!cp ('i4');
8132          !!!parse-error (type => 'NULL');          !!!parse-error (type => 'NULL');
8133          $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST          $self->{nc} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
8134        }        }
8135      };      };
8136      $p->{prev_input_character} = [-1, -1, -1];  
8137      $p->{next_input_character} = -1;      $p->{read_until} = sub {
8138              #my ($scalar, $specials_range, $offset) = @_;
8139          return 0 if defined $p->{next_nc};
8140    
8141          my $pattern = qr/[^$_[1]\x00\x0A\x0D]/;
8142          my $offset = $_[2] || 0;
8143          
8144          if ($p->{char_buffer_pos} < length $p->{char_buffer}) {
8145            pos ($p->{char_buffer}) = $p->{char_buffer_pos};
8146            if ($p->{char_buffer} =~ /\G(?>$pattern)+/) {
8147              substr ($_[0], $offset)
8148                  = substr ($p->{char_buffer}, $-[0], $+[0] - $-[0]);
8149              my $count = $+[0] - $-[0];
8150              if ($count) {
8151                $p->{column} += $count;
8152                $p->{char_buffer_pos} += $count;
8153                $p->{line_prev} = $p->{line};
8154                $p->{column_prev} = $p->{column} - 1;
8155                $p->{nc} = -1;
8156              }
8157              return $count;
8158            } else {
8159              return 0;
8160            }
8161          } else {
8162            my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]);
8163            if ($count) {
8164              $p->{column} += $count;
8165              $p->{column_prev} += $count;
8166              $p->{nc} = -1;
8167            }
8168            return $count;
8169          }
8170        }; # $p->{read_until}
8171    
8172      my $ponerror = $onerror || sub {      my $ponerror = $onerror || sub {
8173        my (%opt) = @_;        my (%opt) = @_;
8174        warn "Parse error ($opt{type}) at line $opt{line} column $opt{column}\n";        my $line = $opt{line};
8175          my $column = $opt{column};
8176          if (defined $opt{token} and defined $opt{token}->{line}) {
8177            $line = $opt{token}->{line};
8178            $column = $opt{token}->{column};
8179          }
8180          warn "Parse error ($opt{type}) at line $line column $column\n";
8181      };      };
8182      $p->{parse_error} = sub {      $p->{parse_error} = sub {
8183        $ponerror->(@_, line => $line, column => $column);        $ponerror->(line => $p->{line}, column => $p->{column}, @_);
8184      };      };
8185            
8186        my $char_onerror = sub {
8187          my (undef, $type, %opt) = @_;
8188          $ponerror->(layer => 'encode',
8189                      line => $p->{line}, column => $p->{column} + 1,
8190                      %opt, type => $type);
8191        }; # $char_onerror
8192        $input->onerror ($char_onerror);
8193    
8194      $p->_initialize_tokenizer;      $p->_initialize_tokenizer;
8195      $p->_initialize_tree_constructor;      $p->_initialize_tree_constructor;
8196    
8197      ## Step 2      ## Step 2
8198      my $node_ln = $node->local_name;      my $node_ln = $node->manakai_local_name;
8199      $p->{content_model} = {      $p->{content_model} = {
8200        title => RCDATA_CONTENT_MODEL,        title => RCDATA_CONTENT_MODEL,
8201        textarea => RCDATA_CONTENT_MODEL,        textarea => RCDATA_CONTENT_MODEL,
# Line 5303  sub set_inner_html ($$$) { Line 8212  sub set_inner_html ($$$) {
8212          unless defined $p->{content_model};          unless defined $p->{content_model};
8213          ## ISSUE: What is "the name of the element"? local name?          ## ISSUE: What is "the name of the element"? local name?
8214    
8215      $p->{inner_html_node} = [$node, $node_ln];      $p->{inner_html_node} = [$node, $el_category->{$node_ln}];
8216          ## TODO: Foreign element OK?
8217    
8218      ## Step 4      ## Step 3
8219      my $root = $doc->create_element_ns      my $root = $doc->create_element_ns
8220        ('http://www.w3.org/1999/xhtml', [undef, 'html']);        ('http://www.w3.org/1999/xhtml', [undef, 'html']);
8221    
8222      ## Step 5 # MUST      ## Step 4 # MUST
8223      $doc->append_child ($root);      $doc->append_child ($root);
8224    
8225      ## Step 6 # MUST      ## Step 5 # MUST
8226      push @{$p->{open_elements}}, [$root, 'html'];      push @{$p->{open_elements}}, [$root, $el_category->{html}];
8227    
8228      undef $p->{head_element};      undef $p->{head_element};
8229        undef $p->{head_element_inserted};
8230    
8231      ## Step 7 # MUST      ## Step 6 # MUST
8232      $p->_reset_insertion_mode;      $p->_reset_insertion_mode;
8233    
8234      ## Step 8 # MUST      ## Step 7 # MUST
8235      my $anode = $node;      my $anode = $node;
8236      AN: while (defined $anode) {      AN: while (defined $anode) {
8237        if ($anode->node_type == 1) {        if ($anode->node_type == 1) {
8238          my $nsuri = $anode->namespace_uri;          my $nsuri = $anode->namespace_uri;
8239          if (defined $nsuri and $nsuri eq 'http://www.w3.org/1999/xhtml') {          if (defined $nsuri and $nsuri eq 'http://www.w3.org/1999/xhtml') {
8240            if ($anode->local_name eq 'form') { ## TODO: case?            if ($anode->manakai_local_name eq 'form') {
8241                !!!cp ('i5');
8242              $p->{form_element} = $anode;              $p->{form_element} = $anode;
8243              last AN;              last AN;
8244            }            }
# Line 5335  sub set_inner_html ($$$) { Line 8247  sub set_inner_html ($$$) {
8247        $anode = $anode->parent_node;        $anode = $anode->parent_node;
8248      } # AN      } # AN
8249            
8250      ## Step 3 # MUST      ## Step 9 # MUST
     ## Step 10 # MUST  
8251      {      {
8252        my $self = $p;        my $self = $p;
8253        !!!next-token;        !!!next-token;
8254      }      }
8255      $p->_tree_construction_main;      $p->_tree_construction_main;
8256    
8257      ## Step 11 # MUST      ## Step 10 # MUST
8258      my @cn = @{$node->child_nodes};      my @cn = @{$node->child_nodes};
8259      for (@cn) {      for (@cn) {
8260        $node->remove_child ($_);        $node->remove_child ($_);
8261      }      }
8262      ## ISSUE: mutation events? read-only?      ## ISSUE: mutation events? read-only?
8263    
8264      ## Step 12 # MUST      ## Step 11 # MUST
8265      @cn = @{$root->child_nodes};      @cn = @{$root->child_nodes};
8266      for (@cn) {      for (@cn) {
8267        $this_doc->adopt_node ($_);        $this_doc->adopt_node ($_);
# Line 5359  sub set_inner_html ($$$) { Line 8270  sub set_inner_html ($$$) {
8270      ## ISSUE: mutation events?      ## ISSUE: mutation events?
8271    
8272      $p->_terminate_tree_constructor;      $p->_terminate_tree_constructor;
8273    
8274        delete $p->{parse_error}; # delete loop
8275    } else {    } else {
8276      die "$0: |set_inner_html| is not defined for node of type $nt";      die "$0: |set_inner_html| is not defined for node of type $nt";
8277    }    }
# Line 5366  sub set_inner_html ($$$) { Line 8279  sub set_inner_html ($$$) {
8279    
8280  } # tree construction stage  } # tree construction stage
8281    
8282  sub get_inner_html ($$$) {  package Whatpm::HTML::RestartParser;
8283    my (undef, $node, $on_error) = @_;  push our @ISA, 'Error';
   
   ## Step 1  
   my $s = '';  
   
   my $in_cdata;  
   my $parent = $node;  
   while (defined $parent) {  
     if ($parent->node_type == 1 and  
         $parent->namespace_uri eq 'http://www.w3.org/1999/xhtml' and  
         {  
           style => 1, script => 1, xmp => 1, iframe => 1,  
           noembed => 1, noframes => 1, noscript => 1,  
         }->{$parent->local_name}) { ## TODO: case thingy  
       $in_cdata = 1;  
     }  
     $parent = $parent->parent_node;  
   }  
   
   ## Step 2  
   my @node = @{$node->child_nodes};  
   C: while (@node) {  
     my $child = shift @node;  
     unless (ref $child) {  
       if ($child eq 'cdata-out') {  
         $in_cdata = 0;  
       } else {  
         $s .= $child; # end tag  
       }  
       next C;  
     }  
       
     my $nt = $child->node_type;  
     if ($nt == 1) { # Element  
       my $tag_name = $child->tag_name; ## TODO: manakai_tag_name  
       $s .= '<' . $tag_name;  
       ## NOTE: Non-HTML case:  
       ## <http://permalink.gmane.org/gmane.org.w3c.whatwg.discuss/11191>  
   
       my @attrs = @{$child->attributes}; # sort order MUST be stable  
       for my $attr (@attrs) { # order is implementation dependent  
         my $attr_name = $attr->name; ## TODO: manakai_name  
         $s .= ' ' . $attr_name . '="';  
         my $attr_value = $attr->value;  
         ## escape  
         $attr_value =~ s/&/&amp;/g;  
         $attr_value =~ s/</&lt;/g;  
         $attr_value =~ s/>/&gt;/g;  
         $attr_value =~ s/"/&quot;/g;  
         $s .= $attr_value . '"';  
       }  
       $s .= '>';  
         
       next C if {  
         area => 1, base => 1, basefont => 1, bgsound => 1,  
         br => 1, col => 1, embed => 1, frame => 1, hr => 1,  
         img => 1, input => 1, link => 1, meta => 1, param => 1,  
         spacer => 1, wbr => 1,  
       }->{$tag_name};  
   
       $s .= "\x0A" if $tag_name eq 'pre' or $tag_name eq 'textarea';  
   
       if (not $in_cdata and {  
         style => 1, script => 1, xmp => 1, iframe => 1,  
         noembed => 1, noframes => 1, noscript => 1,  
         plaintext => 1,  
       }->{$tag_name}) {  
         unshift @node, 'cdata-out';  
         $in_cdata = 1;  
       }  
   
       unshift @node, @{$child->child_nodes}, '</' . $tag_name . '>';  
     } elsif ($nt == 3 or $nt == 4) {  
       if ($in_cdata) {  
         $s .= $child->data;  
       } else {  
         my $value = $child->data;  
         $value =~ s/&/&amp;/g;  
         $value =~ s/</&lt;/g;  
         $value =~ s/>/&gt;/g;  
         $value =~ s/"/&quot;/g;  
         $s .= $value;  
       }  
     } elsif ($nt == 8) {  
       $s .= '<!--' . $child->data . '-->';  
     } elsif ($nt == 10) {  
       $s .= '<!DOCTYPE ' . $child->name . '>';  
     } elsif ($nt == 5) { # entrefs  
       push @node, @{$child->child_nodes};  
     } else {  
       $on_error->($child) if defined $on_error;  
     }  
     ## ISSUE: This code does not support PIs.  
   } # C  
     
   ## Step 3  
   return \$s;  
 } # get_inner_html  
8284    
8285  1;  1;
8286  # $Date$  # $Date$

Legend:
Removed from v.1.44  
changed lines
  Added in v.1.202

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24