/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.35 by wakaba, Mon Jul 16 03:21:04 2007 UTC revision 1.205 by wakaba, Mon Oct 13 06:18:31 2008 UTC
# Line 1  Line 1 
1  package Whatpm::HTML;  package Whatpm::HTML;
2  use strict;  use strict;
3  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};
4    use Error qw(:try);
5    
6    ## NOTE: This module don't check all HTML5 parse errors; character
7    ## encoding related parse errors are expected to be handled by relevant
8    ## modules.
9    ## Parse errors for control characters that are not allowed in HTML5
10    ## documents, for surrogate code points, and for noncharacter code
11    ## points, as well as U+FFFD substitions for characters whose code points
12    ## is higher than U+10FFFF may be detected by combining the parser with
13    ## the checker implemented by Whatpm::Charset::UnicodeChecker (for its
14    ## usage example, see |t/HTML-tree.t| in the Whatpm package or the
15    ## WebHACC::Language::HTML module in the WebHACC package).
16    
17  ## ISSUE:  ## ISSUE:
18  ## var doc = implementation.createDocument (null, null, null);  ## var doc = implementation.createDocument (null, null, null);
19  ## doc.write ('');  ## doc.write ('');
20  ## alert (doc.compatMode);  ## alert (doc.compatMode);
21    
22  ## ISSUE: HTML5 revision 967 says that the encoding layer MUST NOT  require IO::Handle;
23  ## strip BOM and the HTML layer MUST ignore it.  Whether we can do it  
24  ## is not yet clear.  my $HTML_NS = q<http://www.w3.org/1999/xhtml>;
25  ## "{U+FEFF}..." in UTF-16BE/UTF-16LE is three or four characters?  my $MML_NS = q<http://www.w3.org/1998/Math/MathML>;
26  ## "{U+FEFF}..." in GB18030?  my $SVG_NS = q<http://www.w3.org/2000/svg>;
27    my $XLINK_NS = q<http://www.w3.org/1999/xlink>;
28  my $permitted_slash_tag_name = {  my $XML_NS = q<http://www.w3.org/XML/1998/namespace>;
29    base => 1,  my $XMLNS_NS = q<http://www.w3.org/2000/xmlns/>;
30    link => 1,  
31    meta => 1,  sub A_EL () { 0b1 }
32    hr => 1,  sub ADDRESS_EL () { 0b10 }
33    br => 1,  sub BODY_EL () { 0b100 }
34    img=> 1,  sub BUTTON_EL () { 0b1000 }
35    embed => 1,  sub CAPTION_EL () { 0b10000 }
36    param => 1,  sub DD_EL () { 0b100000 }
37    area => 1,  sub DIV_EL () { 0b1000000 }
38    col => 1,  sub DT_EL () { 0b10000000 }
39    input => 1,  sub FORM_EL () { 0b100000000 }
40    sub FORMATTING_EL () { 0b1000000000 }
41    sub FRAMESET_EL () { 0b10000000000 }
42    sub HEADING_EL () { 0b100000000000 }
43    sub HTML_EL () { 0b1000000000000 }
44    sub LI_EL () { 0b10000000000000 }
45    sub NOBR_EL () { 0b100000000000000 }
46    sub OPTION_EL () { 0b1000000000000000 }
47    sub OPTGROUP_EL () { 0b10000000000000000 }
48    sub P_EL () { 0b100000000000000000 }
49    sub SELECT_EL () { 0b1000000000000000000 }
50    sub TABLE_EL () { 0b10000000000000000000 }
51    sub TABLE_CELL_EL () { 0b100000000000000000000 }
52    sub TABLE_ROW_EL () { 0b1000000000000000000000 }
53    sub TABLE_ROW_GROUP_EL () { 0b10000000000000000000000 }
54    sub MISC_SCOPING_EL () { 0b100000000000000000000000 }
55    sub MISC_SPECIAL_EL () { 0b1000000000000000000000000 }
56    sub FOREIGN_EL () { 0b10000000000000000000000000 }
57    sub FOREIGN_FLOW_CONTENT_EL () { 0b100000000000000000000000000 }
58    sub MML_AXML_EL () { 0b1000000000000000000000000000 }
59    sub RUBY_EL () { 0b10000000000000000000000000000 }
60    sub RUBY_COMPONENT_EL () { 0b100000000000000000000000000000 }
61    
62    sub TABLE_ROWS_EL () {
63      TABLE_EL |
64      TABLE_ROW_EL |
65      TABLE_ROW_GROUP_EL
66    }
67    
68    ## NOTE: Used in "generate implied end tags" algorithm.
69    ## NOTE: There is a code where a modified version of
70    ## END_TAG_OPTIONAL_EL is used in "generate implied end tags"
71    ## implementation (search for the algorithm name).
72    sub END_TAG_OPTIONAL_EL () {
73      DD_EL |
74      DT_EL |
75      LI_EL |
76      OPTION_EL |
77      OPTGROUP_EL |
78      P_EL |
79      RUBY_COMPONENT_EL
80    }
81    
82    ## NOTE: Used in </body> and EOF algorithms.
83    sub ALL_END_TAG_OPTIONAL_EL () {
84      DD_EL |
85      DT_EL |
86      LI_EL |
87      P_EL |
88    
89      ## ISSUE: option, optgroup, rt, rp?
90    
91      BODY_EL |
92      HTML_EL |
93      TABLE_CELL_EL |
94      TABLE_ROW_EL |
95      TABLE_ROW_GROUP_EL
96    }
97    
98    sub SCOPING_EL () {
99      BUTTON_EL |
100      CAPTION_EL |
101      HTML_EL |
102      TABLE_EL |
103      TABLE_CELL_EL |
104      MISC_SCOPING_EL
105    }
106    
107    sub TABLE_SCOPING_EL () {
108      HTML_EL |
109      TABLE_EL
110    }
111    
112    sub TABLE_ROWS_SCOPING_EL () {
113      HTML_EL |
114      TABLE_ROW_GROUP_EL
115    }
116    
117    sub TABLE_ROW_SCOPING_EL () {
118      HTML_EL |
119      TABLE_ROW_EL
120    }
121    
122    sub SPECIAL_EL () {
123      ADDRESS_EL |
124      BODY_EL |
125      DIV_EL |
126    
127      DD_EL |
128      DT_EL |
129      LI_EL |
130      P_EL |
131    
132      FORM_EL |
133      FRAMESET_EL |
134      HEADING_EL |
135      SELECT_EL |
136      TABLE_ROW_EL |
137      TABLE_ROW_GROUP_EL |
138      MISC_SPECIAL_EL
139    }
140    
141    my $el_category = {
142      a => A_EL | FORMATTING_EL,
143      address => ADDRESS_EL,
144      applet => MISC_SCOPING_EL,
145      area => MISC_SPECIAL_EL,
146      article => MISC_SPECIAL_EL,
147      aside => MISC_SPECIAL_EL,
148      b => FORMATTING_EL,
149      base => MISC_SPECIAL_EL,
150      basefont => MISC_SPECIAL_EL,
151      bgsound => MISC_SPECIAL_EL,
152      big => FORMATTING_EL,
153      blockquote => MISC_SPECIAL_EL,
154      body => BODY_EL,
155      br => MISC_SPECIAL_EL,
156      button => BUTTON_EL,
157      caption => CAPTION_EL,
158      center => MISC_SPECIAL_EL,
159      col => MISC_SPECIAL_EL,
160      colgroup => MISC_SPECIAL_EL,
161      command => MISC_SPECIAL_EL,
162      datagrid => MISC_SPECIAL_EL,
163      dd => DD_EL,
164      details => MISC_SPECIAL_EL,
165      dialog => MISC_SPECIAL_EL,
166      dir => MISC_SPECIAL_EL,
167      div => DIV_EL,
168      dl => MISC_SPECIAL_EL,
169      dt => DT_EL,
170      em => FORMATTING_EL,
171      embed => MISC_SPECIAL_EL,
172      eventsource => MISC_SPECIAL_EL,
173      fieldset => MISC_SPECIAL_EL,
174      figure => MISC_SPECIAL_EL,
175      font => FORMATTING_EL,
176      footer => MISC_SPECIAL_EL,
177      form => FORM_EL,
178      frame => MISC_SPECIAL_EL,
179      frameset => FRAMESET_EL,
180      h1 => HEADING_EL,
181      h2 => HEADING_EL,
182      h3 => HEADING_EL,
183      h4 => HEADING_EL,
184      h5 => HEADING_EL,
185      h6 => HEADING_EL,
186      head => MISC_SPECIAL_EL,
187      header => MISC_SPECIAL_EL,
188      hr => MISC_SPECIAL_EL,
189      html => HTML_EL,
190      i => FORMATTING_EL,
191      iframe => MISC_SPECIAL_EL,
192      img => MISC_SPECIAL_EL,
193      #image => MISC_SPECIAL_EL, ## NOTE: Commented out in the spec.
194      input => MISC_SPECIAL_EL,
195      isindex => MISC_SPECIAL_EL,
196      li => LI_EL,
197      link => MISC_SPECIAL_EL,
198      listing => MISC_SPECIAL_EL,
199      marquee => MISC_SCOPING_EL,
200      menu => MISC_SPECIAL_EL,
201      meta => MISC_SPECIAL_EL,
202      nav => MISC_SPECIAL_EL,
203      nobr => NOBR_EL | FORMATTING_EL,
204      noembed => MISC_SPECIAL_EL,
205      noframes => MISC_SPECIAL_EL,
206      noscript => MISC_SPECIAL_EL,
207      object => MISC_SCOPING_EL,
208      ol => MISC_SPECIAL_EL,
209      optgroup => OPTGROUP_EL,
210      option => OPTION_EL,
211      p => P_EL,
212      param => MISC_SPECIAL_EL,
213      plaintext => MISC_SPECIAL_EL,
214      pre => MISC_SPECIAL_EL,
215      rp => RUBY_COMPONENT_EL,
216      rt => RUBY_COMPONENT_EL,
217      ruby => RUBY_EL,
218      s => FORMATTING_EL,
219      script => MISC_SPECIAL_EL,
220      select => SELECT_EL,
221      section => MISC_SPECIAL_EL,
222      small => FORMATTING_EL,
223      spacer => MISC_SPECIAL_EL,
224      strike => FORMATTING_EL,
225      strong => FORMATTING_EL,
226      style => MISC_SPECIAL_EL,
227      table => TABLE_EL,
228      tbody => TABLE_ROW_GROUP_EL,
229      td => TABLE_CELL_EL,
230      textarea => MISC_SPECIAL_EL,
231      tfoot => TABLE_ROW_GROUP_EL,
232      th => TABLE_CELL_EL,
233      thead => TABLE_ROW_GROUP_EL,
234      title => MISC_SPECIAL_EL,
235      tr => TABLE_ROW_EL,
236      tt => FORMATTING_EL,
237      u => FORMATTING_EL,
238      ul => MISC_SPECIAL_EL,
239      wbr => MISC_SPECIAL_EL,
240    };
241    
242    my $el_category_f = {
243      $MML_NS => {
244        'annotation-xml' => MML_AXML_EL,
245        mi => FOREIGN_FLOW_CONTENT_EL,
246        mo => FOREIGN_FLOW_CONTENT_EL,
247        mn => FOREIGN_FLOW_CONTENT_EL,
248        ms => FOREIGN_FLOW_CONTENT_EL,
249        mtext => FOREIGN_FLOW_CONTENT_EL,
250      },
251      $SVG_NS => {
252        foreignObject => FOREIGN_FLOW_CONTENT_EL | MISC_SCOPING_EL,
253        desc => FOREIGN_FLOW_CONTENT_EL,
254        title => FOREIGN_FLOW_CONTENT_EL,
255      },
256      ## NOTE: In addition, FOREIGN_EL is set to non-HTML elements.
257    };
258    
259    my $svg_attr_name = {
260      attributename => 'attributeName',
261      attributetype => 'attributeType',
262      basefrequency => 'baseFrequency',
263      baseprofile => 'baseProfile',
264      calcmode => 'calcMode',
265      clippathunits => 'clipPathUnits',
266      contentscripttype => 'contentScriptType',
267      contentstyletype => 'contentStyleType',
268      diffuseconstant => 'diffuseConstant',
269      edgemode => 'edgeMode',
270      externalresourcesrequired => 'externalResourcesRequired',
271      filterres => 'filterRes',
272      filterunits => 'filterUnits',
273      glyphref => 'glyphRef',
274      gradienttransform => 'gradientTransform',
275      gradientunits => 'gradientUnits',
276      kernelmatrix => 'kernelMatrix',
277      kernelunitlength => 'kernelUnitLength',
278      keypoints => 'keyPoints',
279      keysplines => 'keySplines',
280      keytimes => 'keyTimes',
281      lengthadjust => 'lengthAdjust',
282      limitingconeangle => 'limitingConeAngle',
283      markerheight => 'markerHeight',
284      markerunits => 'markerUnits',
285      markerwidth => 'markerWidth',
286      maskcontentunits => 'maskContentUnits',
287      maskunits => 'maskUnits',
288      numoctaves => 'numOctaves',
289      pathlength => 'pathLength',
290      patterncontentunits => 'patternContentUnits',
291      patterntransform => 'patternTransform',
292      patternunits => 'patternUnits',
293      pointsatx => 'pointsAtX',
294      pointsaty => 'pointsAtY',
295      pointsatz => 'pointsAtZ',
296      preservealpha => 'preserveAlpha',
297      preserveaspectratio => 'preserveAspectRatio',
298      primitiveunits => 'primitiveUnits',
299      refx => 'refX',
300      refy => 'refY',
301      repeatcount => 'repeatCount',
302      repeatdur => 'repeatDur',
303      requiredextensions => 'requiredExtensions',
304      requiredfeatures => 'requiredFeatures',
305      specularconstant => 'specularConstant',
306      specularexponent => 'specularExponent',
307      spreadmethod => 'spreadMethod',
308      startoffset => 'startOffset',
309      stddeviation => 'stdDeviation',
310      stitchtiles => 'stitchTiles',
311      surfacescale => 'surfaceScale',
312      systemlanguage => 'systemLanguage',
313      tablevalues => 'tableValues',
314      targetx => 'targetX',
315      targety => 'targetY',
316      textlength => 'textLength',
317      viewbox => 'viewBox',
318      viewtarget => 'viewTarget',
319      xchannelselector => 'xChannelSelector',
320      ychannelselector => 'yChannelSelector',
321      zoomandpan => 'zoomAndPan',
322  };  };
323    
324  my $c1_entity_char = {  my $foreign_attr_xname = {
325      'xlink:actuate' => [$XLINK_NS, ['xlink', 'actuate']],
326      'xlink:arcrole' => [$XLINK_NS, ['xlink', 'arcrole']],
327      'xlink:href' => [$XLINK_NS, ['xlink', 'href']],
328      'xlink:role' => [$XLINK_NS, ['xlink', 'role']],
329      'xlink:show' => [$XLINK_NS, ['xlink', 'show']],
330      'xlink:title' => [$XLINK_NS, ['xlink', 'title']],
331      'xlink:type' => [$XLINK_NS, ['xlink', 'type']],
332      'xml:base' => [$XML_NS, ['xml', 'base']],
333      'xml:lang' => [$XML_NS, ['xml', 'lang']],
334      'xml:space' => [$XML_NS, ['xml', 'space']],
335      'xmlns' => [$XMLNS_NS, [undef, 'xmlns']],
336      'xmlns:xlink' => [$XMLNS_NS, ['xmlns', 'xlink']],
337    };
338    
339    ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.
340    
341    my $charref_map = {
342      0x0D => 0x000A,
343    0x80 => 0x20AC,    0x80 => 0x20AC,
344    0x81 => 0xFFFD,    0x81 => 0xFFFD,
345    0x82 => 0x201A,    0x82 => 0x201A,
# Line 60  my $c1_entity_char = { Line 372  my $c1_entity_char = {
372    0x9D => 0xFFFD,    0x9D => 0xFFFD,
373    0x9E => 0x017E,    0x9E => 0x017E,
374    0x9F => 0x0178,    0x9F => 0x0178,
375  }; # $c1_entity_char  }; # $charref_map
376    $charref_map->{$_} = 0xFFFD
377        for 0x0000..0x0008, 0x000B, 0x000E..0x001F, 0x007F,
378            0xD800..0xDFFF, 0xFDD0..0xFDDF, ## ISSUE: 0xFDEF
379            0xFFFE, 0xFFFF, 0x1FFFE, 0x1FFFF, 0x2FFFE, 0x2FFFF, 0x3FFFE, 0x3FFFF,
380            0x4FFFE, 0x4FFFF, 0x5FFFE, 0x5FFFF, 0x6FFFE, 0x6FFFF, 0x7FFFE,
381            0x7FFFF, 0x8FFFE, 0x8FFFF, 0x9FFFE, 0x9FFFF, 0xAFFFE, 0xAFFFF,
382            0xBFFFE, 0xBFFFF, 0xCFFFE, 0xCFFFF, 0xDFFFE, 0xDFFFF, 0xEFFFE,
383            0xEFFFF, 0xFFFFE, 0xFFFFF, 0x10FFFE, 0x10FFFF;
384    
385  my $special_category = {  ## TODO: Invoke the reset algorithm when a resettable element is
386    address => 1, area => 1, base => 1, basefont => 1, bgsound => 1,  ## created (cf. HTML5 revision 2259).
387    blockquote => 1, body => 1, br => 1, center => 1, col => 1, colgroup => 1,  
388    dd => 1, dir => 1, div => 1, dl => 1, dt => 1, embed => 1, fieldset => 1,  sub parse_byte_string ($$$$;$) {
389    form => 1, frame => 1, frameset => 1, h1 => 1, h2 => 1, h3 => 1,    my $self = shift;
390    h4 => 1, h5 => 1, h6 => 1, head => 1, hr => 1, iframe => 1, image => 1,    my $charset_name = shift;
391    img => 1, input => 1, isindex => 1, li => 1, link => 1, listing => 1,    open my $input, '<', ref $_[0] ? $_[0] : \($_[0]);
392    menu => 1, meta => 1, noembed => 1, noframes => 1, noscript => 1,    return $self->parse_byte_stream ($charset_name, $input, @_[1..$#_]);
393    ol => 1, optgroup => 1, option => 1, p => 1, param => 1, plaintext => 1,  } # parse_byte_string
394    pre => 1, script => 1, select => 1, spacer => 1, style => 1, tbody => 1,  
395    textarea => 1, tfoot => 1, thead => 1, title => 1, tr => 1, ul => 1, wbr => 1,  sub parse_byte_stream ($$$$;$$) {
396  };    # my ($self, $charset_name, $byte_stream, $doc, $onerror, $get_wrapper) = @_;
397  my $scoping_category = {    my $self = ref $_[0] ? shift : shift->new;
398    button => 1, caption => 1, html => 1, marquee => 1, object => 1,    my $charset_name = shift;
399    table => 1, td => 1, th => 1,    my $byte_stream = $_[0];
400  };  
401  my $formatting_category = {    my $onerror = $_[2] || sub {
402    a => 1, b => 1, big => 1, em => 1, font => 1, i => 1, nobr => 1,      my (%opt) = @_;
403    s => 1, small => 1, strile => 1, strong => 1, tt => 1, u => 1,      warn "Parse error ($opt{type})\n";
404  };    };
405  # $phrasing_category: all other elements    $self->{parse_error} = $onerror; # updated later by parse_char_string
406    
407      my $get_wrapper = $_[3] || sub ($) {
408        return $_[0]; # $_[0] = byte stream handle, returned = arg to char handle
409      };
410    
411      ## HTML5 encoding sniffing algorithm
412      require Message::Charset::Info;
413      my $charset;
414      my $buffer;
415      my ($char_stream, $e_status);
416    
417      SNIFFING: {
418        ## NOTE: By setting |allow_fallback| option true when the
419        ## |get_decode_handle| method is invoked, we ignore what the HTML5
420        ## spec requires, i.e. unsupported encoding should be ignored.
421          ## TODO: We should not do this unless the parser is invoked
422          ## in the conformance checking mode, in which this behavior
423          ## would be useful.
424    
425        ## Step 1
426        if (defined $charset_name) {
427          $charset = Message::Charset::Info->get_by_html_name ($charset_name);
428              ## TODO: Is this ok?  Transfer protocol's parameter should be
429              ## interpreted in its semantics?
430    
431          ($char_stream, $e_status) = $charset->get_decode_handle
432              ($byte_stream, allow_error_reporting => 1,
433               allow_fallback => 1);
434          if ($char_stream) {
435            $self->{confident} = 1;
436            last SNIFFING;
437          } else {
438            !!!parse-error (type => 'charset:not supported',
439                            layer => 'encode',
440                            line => 1, column => 1,
441                            value => $charset_name,
442                            level => $self->{level}->{uncertain});
443          }
444        }
445    
446        ## Step 2
447        my $byte_buffer = '';
448        for (1..1024) {
449          my $char = $byte_stream->getc;
450          last unless defined $char;
451          $byte_buffer .= $char;
452        } ## TODO: timeout
453    
454        ## Step 3
455        if ($byte_buffer =~ /^\xFE\xFF/) {
456          $charset = Message::Charset::Info->get_by_html_name ('utf-16be');
457          ($char_stream, $e_status) = $charset->get_decode_handle
458              ($byte_stream, allow_error_reporting => 1,
459               allow_fallback => 1, byte_buffer => \$byte_buffer);
460          $self->{confident} = 1;
461          last SNIFFING;
462        } elsif ($byte_buffer =~ /^\xFF\xFE/) {
463          $charset = Message::Charset::Info->get_by_html_name ('utf-16le');
464          ($char_stream, $e_status) = $charset->get_decode_handle
465              ($byte_stream, allow_error_reporting => 1,
466               allow_fallback => 1, byte_buffer => \$byte_buffer);
467          $self->{confident} = 1;
468          last SNIFFING;
469        } elsif ($byte_buffer =~ /^\xEF\xBB\xBF/) {
470          $charset = Message::Charset::Info->get_by_html_name ('utf-8');
471          ($char_stream, $e_status) = $charset->get_decode_handle
472              ($byte_stream, allow_error_reporting => 1,
473               allow_fallback => 1, byte_buffer => \$byte_buffer);
474          $self->{confident} = 1;
475          last SNIFFING;
476        }
477    
478        ## Step 4
479        ## TODO: <meta charset>
480    
481        ## Step 5
482        ## TODO: from history
483    
484        ## Step 6
485        require Whatpm::Charset::UniversalCharDet;
486        $charset_name = Whatpm::Charset::UniversalCharDet->detect_byte_string
487            ($byte_buffer);
488        if (defined $charset_name) {
489          $charset = Message::Charset::Info->get_by_html_name ($charset_name);
490    
491          require Whatpm::Charset::DecodeHandle;
492          $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new
493              ($byte_stream);
494          ($char_stream, $e_status) = $charset->get_decode_handle
495              ($buffer, allow_error_reporting => 1,
496               allow_fallback => 1, byte_buffer => \$byte_buffer);
497          if ($char_stream) {
498            $buffer->{buffer} = $byte_buffer;
499            !!!parse-error (type => 'sniffing:chardet',
500                            text => $charset_name,
501                            level => $self->{level}->{info},
502                            layer => 'encode',
503                            line => 1, column => 1);
504            $self->{confident} = 0;
505            last SNIFFING;
506          }
507        }
508    
509        ## Step 7: default
510        ## TODO: Make this configurable.
511        $charset = Message::Charset::Info->get_by_html_name ('windows-1252');
512            ## NOTE: We choose |windows-1252| here, since |utf-8| should be
513            ## detectable in the step 6.
514        require Whatpm::Charset::DecodeHandle;
515        $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new
516            ($byte_stream);
517        ($char_stream, $e_status)
518            = $charset->get_decode_handle ($buffer,
519                                           allow_error_reporting => 1,
520                                           allow_fallback => 1,
521                                           byte_buffer => \$byte_buffer);
522        $buffer->{buffer} = $byte_buffer;
523        !!!parse-error (type => 'sniffing:default',
524                        text => 'windows-1252',
525                        level => $self->{level}->{info},
526                        line => 1, column => 1,
527                        layer => 'encode');
528        $self->{confident} = 0;
529      } # SNIFFING
530    
531      if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
532        $self->{input_encoding} = $charset->get_iana_name; ## TODO: Should we set actual charset decoder's encoding name?
533        !!!parse-error (type => 'chardecode:fallback',
534                        #text => $self->{input_encoding},
535                        level => $self->{level}->{uncertain},
536                        line => 1, column => 1,
537                        layer => 'encode');
538      } elsif (not ($e_status &
539                    Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) {
540        $self->{input_encoding} = $charset->get_iana_name;
541        !!!parse-error (type => 'chardecode:no error',
542                        text => $self->{input_encoding},
543                        level => $self->{level}->{uncertain},
544                        line => 1, column => 1,
545                        layer => 'encode');
546      } else {
547        $self->{input_encoding} = $charset->get_iana_name;
548      }
549    
550      $self->{change_encoding} = sub {
551        my $self = shift;
552        $charset_name = shift;
553        my $token = shift;
554    
555        $charset = Message::Charset::Info->get_by_html_name ($charset_name);
556        ($char_stream, $e_status) = $charset->get_decode_handle
557            ($byte_stream, allow_error_reporting => 1, allow_fallback => 1,
558             byte_buffer => \ $buffer->{buffer});
559        
560        if ($char_stream) { # if supported
561          ## "Change the encoding" algorithm:
562    
563          ## Step 1    
564          if ($charset->{category} &
565              Message::Charset::Info::CHARSET_CATEGORY_UTF16 ()) {
566            $charset = Message::Charset::Info->get_by_html_name ('utf-8');
567            ($char_stream, $e_status) = $charset->get_decode_handle
568                ($byte_stream,
569                 byte_buffer => \ $buffer->{buffer});
570          }
571          $charset_name = $charset->get_iana_name;
572          
573          ## Step 2
574          if (defined $self->{input_encoding} and
575              $self->{input_encoding} eq $charset_name) {
576            !!!parse-error (type => 'charset label:matching',
577                            text => $charset_name,
578                            level => $self->{level}->{info});
579            $self->{confident} = 1;
580            return;
581          }
582    
583          !!!parse-error (type => 'charset label detected',
584                          text => $self->{input_encoding},
585                          value => $charset_name,
586                          level => $self->{level}->{warn},
587                          token => $token);
588          
589          ## Step 3
590          # if (can) {
591            ## change the encoding on the fly.
592            #$self->{confident} = 1;
593            #return;
594          # }
595          
596          ## Step 4
597          throw Whatpm::HTML::RestartParser ();
598        }
599      }; # $self->{change_encoding}
600    
601      my $char_onerror = sub {
602        my (undef, $type, %opt) = @_;
603        !!!parse-error (layer => 'encode',
604                        line => $self->{line}, column => $self->{column} + 1,
605                        %opt, type => $type);
606        if ($opt{octets}) {
607          ${$opt{octets}} = "\x{FFFD}"; # relacement character
608        }
609      };
610    
611      my $wrapped_char_stream = $get_wrapper->($char_stream);
612      $wrapped_char_stream->onerror ($char_onerror);
613    
614      my @args = ($_[1], $_[2]); # $doc, $onerror - $get_wrapper = undef;
615      my $return;
616      try {
617        $return = $self->parse_char_stream ($wrapped_char_stream, @args);  
618      } catch Whatpm::HTML::RestartParser with {
619        ## NOTE: Invoked after {change_encoding}.
620    
621        if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
622          $self->{input_encoding} = $charset->get_iana_name; ## TODO: Should we set actual charset decoder's encoding name?
623          !!!parse-error (type => 'chardecode:fallback',
624                          level => $self->{level}->{uncertain},
625                          #text => $self->{input_encoding},
626                          line => 1, column => 1,
627                          layer => 'encode');
628        } elsif (not ($e_status &
629                      Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) {
630          $self->{input_encoding} = $charset->get_iana_name;
631          !!!parse-error (type => 'chardecode:no error',
632                          text => $self->{input_encoding},
633                          level => $self->{level}->{uncertain},
634                          line => 1, column => 1,
635                          layer => 'encode');
636        } else {
637          $self->{input_encoding} = $charset->get_iana_name;
638        }
639        $self->{confident} = 1;
640    
641  sub parse_string ($$$;$) {      $wrapped_char_stream = $get_wrapper->($char_stream);
642    my $self = shift->new;      $wrapped_char_stream->onerror ($char_onerror);
643    my $s = \$_[0];  
644        $return = $self->parse_char_stream ($wrapped_char_stream, @args);
645      };
646      return $return;
647    } # parse_byte_stream
648    
649    ## NOTE: HTML5 spec says that the encoding layer MUST NOT strip BOM
650    ## and the HTML layer MUST ignore it.  However, we does strip BOM in
651    ## the encoding layer and the HTML layer does not ignore any U+FEFF,
652    ## because the core part of our HTML parser expects a string of character,
653    ## not a string of bytes or code units or anything which might contain a BOM.
654    ## Therefore, any parser interface that accepts a string of bytes,
655    ## such as |parse_byte_string| in this module, must ensure that it does
656    ## strip the BOM and never strip any ZWNBSP.
657    
658    sub parse_char_string ($$$;$$) {
659      #my ($self, $s, $doc, $onerror, $get_wrapper) = @_;
660      my $self = shift;
661      my $s = ref $_[0] ? $_[0] : \($_[0]);
662      require Whatpm::Charset::DecodeHandle;
663      my $input = Whatpm::Charset::DecodeHandle::CharString->new ($s);
664      return $self->parse_char_stream ($input, @_[1..$#_]);
665    } # parse_char_string
666    *parse_string = \&parse_char_string; ## NOTE: Alias for backward compatibility.
667    
668    sub parse_char_stream ($$$;$$) {
669      my $self = ref $_[0] ? shift : shift->new;
670      my $input = $_[0];
671    $self->{document} = $_[1];    $self->{document} = $_[1];
672      @{$self->{document}->child_nodes} = ();
673    
674    ## NOTE: |set_inner_html| copies most of this method's code    ## NOTE: |set_inner_html| copies most of this method's code
675    
676    my $i = 0;    $self->{confident} = 1 unless exists $self->{confident};
677    my $line = 1;    $self->{document}->input_encoding ($self->{input_encoding})
678    my $column = 0;        if defined $self->{input_encoding};
679    $self->{set_next_input_character} = sub {  ## TODO: |{input_encoding}| is needless?
680    
681      $self->{line_prev} = $self->{line} = 1;
682      $self->{column_prev} = -1;
683      $self->{column} = 0;
684      $self->{set_nc} = sub {
685      my $self = shift;      my $self = shift;
686    
687      pop @{$self->{prev_input_character}};      my $char = '';
688      unshift @{$self->{prev_input_character}}, $self->{next_input_character};      if (defined $self->{next_nc}) {
689          $char = $self->{next_nc};
690          delete $self->{next_nc};
691          $self->{nc} = ord $char;
692        } else {
693          $self->{char_buffer} = '';
694          $self->{char_buffer_pos} = 0;
695    
696      $self->{next_input_character} = -1 and return if $i >= length $$s;        my $count = $input->manakai_read_until
697      $self->{next_input_character} = ord substr $$s, $i++, 1;           ($self->{char_buffer}, qr/[^\x00\x0A\x0D]/, $self->{char_buffer_pos});
698      $column++;        if ($count) {
699            $self->{line_prev} = $self->{line};
700            $self->{column_prev} = $self->{column};
701            $self->{column}++;
702            $self->{nc}
703                = ord substr ($self->{char_buffer}, $self->{char_buffer_pos}++, 1);
704            return;
705          }
706    
707          if ($input->read ($char, 1)) {
708            $self->{nc} = ord $char;
709          } else {
710            $self->{nc} = -1;
711            return;
712          }
713        }
714    
715        ($self->{line_prev}, $self->{column_prev})
716            = ($self->{line}, $self->{column});
717        $self->{column}++;
718            
719      if ($self->{next_input_character} == 0x000A) { # LF      if ($self->{nc} == 0x000A) { # LF
720        $line++;        !!!cp ('j1');
721        $column = 0;        $self->{line}++;
722      } elsif ($self->{next_input_character} == 0x000D) { # CR        $self->{column} = 0;
723        $i++ if substr ($$s, $i, 1) eq "\x0A";      } elsif ($self->{nc} == 0x000D) { # CR
724        $self->{next_input_character} = 0x000A; # LF # MUST        !!!cp ('j2');
725        $line++;  ## TODO: support for abort/streaming
726        $column = 0;        my $next = '';
727      } elsif ($self->{next_input_character} > 0x10FFFF) {        if ($input->read ($next, 1) and $next ne "\x0A") {
728        $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST          $self->{next_nc} = $next;
729      } elsif ($self->{next_input_character} == 0x0000) { # NULL        }
730          $self->{nc} = 0x000A; # LF # MUST
731          $self->{line}++;
732          $self->{column} = 0;
733        } elsif ($self->{nc} == 0x0000) { # NULL
734          !!!cp ('j4');
735        !!!parse-error (type => 'NULL');        !!!parse-error (type => 'NULL');
736        $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST        $self->{nc} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
737      }      }
738    };    };
739    $self->{prev_input_character} = [-1, -1, -1];  
740    $self->{next_input_character} = -1;    $self->{read_until} = sub {
741        #my ($scalar, $specials_range, $offset) = @_;
742        return 0 if defined $self->{next_nc};
743    
744        my $pattern = qr/[^$_[1]\x00\x0A\x0D]/;
745        my $offset = $_[2] || 0;
746    
747        if ($self->{char_buffer_pos} < length $self->{char_buffer}) {
748          pos ($self->{char_buffer}) = $self->{char_buffer_pos};
749          if ($self->{char_buffer} =~ /\G(?>$pattern)+/) {
750            substr ($_[0], $offset)
751                = substr ($self->{char_buffer}, $-[0], $+[0] - $-[0]);
752            my $count = $+[0] - $-[0];
753            if ($count) {
754              $self->{column} += $count;
755              $self->{char_buffer_pos} += $count;
756              $self->{line_prev} = $self->{line};
757              $self->{column_prev} = $self->{column} - 1;
758              $self->{nc} = -1;
759            }
760            return $count;
761          } else {
762            return 0;
763          }
764        } else {
765          my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]);
766          if ($count) {
767            $self->{column} += $count;
768            $self->{line_prev} = $self->{line};
769            $self->{column_prev} = $self->{column} - 1;
770            $self->{nc} = -1;
771          }
772          return $count;
773        }
774      }; # $self->{read_until}
775    
776    my $onerror = $_[2] || sub {    my $onerror = $_[2] || sub {
777      my (%opt) = @_;      my (%opt) = @_;
778      warn "Parse error ($opt{type}) at line $opt{line} column $opt{column}\n";      my $line = $opt{token} ? $opt{token}->{line} : $opt{line};
779        my $column = $opt{token} ? $opt{token}->{column} : $opt{column};
780        warn "Parse error ($opt{type}) at line $line column $column\n";
781    };    };
782    $self->{parse_error} = sub {    $self->{parse_error} = sub {
783      $onerror->(@_, line => $line, column => $column);      $onerror->(line => $self->{line}, column => $self->{column}, @_);
784    };    };
785    
786      my $char_onerror = sub {
787        my (undef, $type, %opt) = @_;
788        !!!parse-error (layer => 'encode',
789                        line => $self->{line}, column => $self->{column} + 1,
790                        %opt, type => $type);
791      }; # $char_onerror
792    
793      if ($_[3]) {
794        $input = $_[3]->($input);
795        $input->onerror ($char_onerror);
796      } else {
797        $input->onerror ($char_onerror) unless defined $input->onerror;
798      }
799    
800    $self->_initialize_tokenizer;    $self->_initialize_tokenizer;
801    $self->_initialize_tree_constructor;    $self->_initialize_tree_constructor;
802    $self->_construct_tree;    $self->_construct_tree;
803    $self->_terminate_tree_constructor;    $self->_terminate_tree_constructor;
804    
805      delete $self->{parse_error}; # remove loop
806    
807    return $self->{document};    return $self->{document};
808  } # parse_string  } # parse_char_stream
809    
810  sub new ($) {  sub new ($) {
811    my $class = shift;    my $class = shift;
812    my $self = bless {}, $class;    my $self = bless {
813    $self->{set_next_input_character} = sub {      level => {must => 'm',
814      $self->{next_input_character} = -1;                should => 's',
815                  warn => 'w',
816                  info => 'i',
817                  uncertain => 'u'},
818      }, $class;
819      $self->{set_nc} = sub {
820        $self->{nc} = -1;
821    };    };
822    $self->{parse_error} = sub {    $self->{parse_error} = sub {
823      #      #
824    };    };
825      $self->{change_encoding} = sub {
826        # if ($_[0] is a supported encoding) {
827        #   run "change the encoding" algorithm;
828        #   throw Whatpm::HTML::RestartParser (charset => $new_encoding);
829        # }
830      };
831      $self->{application_cache_selection} = sub {
832        #
833      };
834    return $self;    return $self;
835  } # new  } # new
836    
837    sub CM_ENTITY () { 0b001 } # & markup in data
838    sub CM_LIMITED_MARKUP () { 0b010 } # < markup in data (limited)
839    sub CM_FULL_MARKUP () { 0b100 } # < markup in data (any)
840    
841    sub PLAINTEXT_CONTENT_MODEL () { 0 }
842    sub CDATA_CONTENT_MODEL () { CM_LIMITED_MARKUP }
843    sub RCDATA_CONTENT_MODEL () { CM_ENTITY | CM_LIMITED_MARKUP }
844    sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP }
845    
846    sub DATA_STATE () { 0 }
847    #sub ENTITY_DATA_STATE () { 1 }
848    sub TAG_OPEN_STATE () { 2 }
849    sub CLOSE_TAG_OPEN_STATE () { 3 }
850    sub TAG_NAME_STATE () { 4 }
851    sub BEFORE_ATTRIBUTE_NAME_STATE () { 5 }
852    sub ATTRIBUTE_NAME_STATE () { 6 }
853    sub AFTER_ATTRIBUTE_NAME_STATE () { 7 }
854    sub BEFORE_ATTRIBUTE_VALUE_STATE () { 8 }
855    sub ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE () { 9 }
856    sub ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE () { 10 }
857    sub ATTRIBUTE_VALUE_UNQUOTED_STATE () { 11 }
858    #sub ENTITY_IN_ATTRIBUTE_VALUE_STATE () { 12 }
859    sub MARKUP_DECLARATION_OPEN_STATE () { 13 }
860    sub COMMENT_START_STATE () { 14 }
861    sub COMMENT_START_DASH_STATE () { 15 }
862    sub COMMENT_STATE () { 16 }
863    sub COMMENT_END_STATE () { 17 }
864    sub COMMENT_END_DASH_STATE () { 18 }
865    sub BOGUS_COMMENT_STATE () { 19 }
866    sub DOCTYPE_STATE () { 20 }
867    sub BEFORE_DOCTYPE_NAME_STATE () { 21 }
868    sub DOCTYPE_NAME_STATE () { 22 }
869    sub AFTER_DOCTYPE_NAME_STATE () { 23 }
870    sub BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE () { 24 }
871    sub DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE () { 25 }
872    sub DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE () { 26 }
873    sub AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE () { 27 }
874    sub BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 28 }
875    sub DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE () { 29 }
876    sub DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE () { 30 }
877    sub AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 31 }
878    sub BOGUS_DOCTYPE_STATE () { 32 }
879    sub AFTER_ATTRIBUTE_VALUE_QUOTED_STATE () { 33 }
880    sub SELF_CLOSING_START_TAG_STATE () { 34 }
881    sub CDATA_SECTION_STATE () { 35 }
882    sub MD_HYPHEN_STATE () { 36 } # "markup declaration open state" in the spec
883    sub MD_DOCTYPE_STATE () { 37 } # "markup declaration open state" in the spec
884    sub MD_CDATA_STATE () { 38 } # "markup declaration open state" in the spec
885    sub CDATA_RCDATA_CLOSE_TAG_STATE () { 39 } # "close tag open state" in the spec
886    sub CDATA_SECTION_MSE1_STATE () { 40 } # "CDATA section state" in the spec
887    sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec
888    sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec
889    sub SYSTEM_STATE () { 43 } # "after DOCTYPE name state" in the spec
890    ## NOTE: "Entity data state", "entity in attribute value state", and
891    ## "consume a character reference" algorithm are jointly implemented
892    ## using the following six states:
893    sub ENTITY_STATE () { 44 }
894    sub ENTITY_HASH_STATE () { 45 }
895    sub NCR_NUM_STATE () { 46 }
896    sub HEXREF_X_STATE () { 47 }
897    sub HEXREF_HEX_STATE () { 48 }
898    sub ENTITY_NAME_STATE () { 49 }
899    sub PCDATA_STATE () { 50 } # "data state" in the spec
900    
901    sub DOCTYPE_TOKEN () { 1 }
902    sub COMMENT_TOKEN () { 2 }
903    sub START_TAG_TOKEN () { 3 }
904    sub END_TAG_TOKEN () { 4 }
905    sub END_OF_FILE_TOKEN () { 5 }
906    sub CHARACTER_TOKEN () { 6 }
907    
908    sub AFTER_HTML_IMS () { 0b100 }
909    sub HEAD_IMS ()       { 0b1000 }
910    sub BODY_IMS ()       { 0b10000 }
911    sub BODY_TABLE_IMS () { 0b100000 }
912    sub TABLE_IMS ()      { 0b1000000 }
913    sub ROW_IMS ()        { 0b10000000 }
914    sub BODY_AFTER_IMS () { 0b100000000 }
915    sub FRAME_IMS ()      { 0b1000000000 }
916    sub SELECT_IMS ()     { 0b10000000000 }
917    sub IN_FOREIGN_CONTENT_IM () { 0b100000000000 }
918        ## NOTE: "in foreign content" insertion mode is special; it is combined
919        ## with the secondary insertion mode.  In this parser, they are stored
920        ## together in the bit-or'ed form.
921    sub IN_CDATA_RCDATA_IM () { 0b1000000000000 }
922        ## NOTE: "in CDATA/RCDATA" insertion mode is also special; it is
923        ## combined with the original insertion mode.  In thie parser,
924        ## they are stored together in the bit-or'ed form.
925    
926    ## NOTE: "initial" and "before html" insertion modes have no constants.
927    
928    ## NOTE: "after after body" insertion mode.
929    sub AFTER_HTML_BODY_IM () { AFTER_HTML_IMS | BODY_AFTER_IMS }
930    
931    ## NOTE: "after after frameset" insertion mode.
932    sub AFTER_HTML_FRAMESET_IM () { AFTER_HTML_IMS | FRAME_IMS }
933    
934    sub IN_HEAD_IM () { HEAD_IMS | 0b00 }
935    sub IN_HEAD_NOSCRIPT_IM () { HEAD_IMS | 0b01 }
936    sub AFTER_HEAD_IM () { HEAD_IMS | 0b10 }
937    sub BEFORE_HEAD_IM () { HEAD_IMS | 0b11 }
938    sub IN_BODY_IM () { BODY_IMS }
939    sub IN_CELL_IM () { BODY_IMS | BODY_TABLE_IMS | 0b01 }
940    sub IN_CAPTION_IM () { BODY_IMS | BODY_TABLE_IMS | 0b10 }
941    sub IN_ROW_IM () { TABLE_IMS | ROW_IMS | 0b01 }
942    sub IN_TABLE_BODY_IM () { TABLE_IMS | ROW_IMS | 0b10 }
943    sub IN_TABLE_IM () { TABLE_IMS }
944    sub AFTER_BODY_IM () { BODY_AFTER_IMS }
945    sub IN_FRAMESET_IM () { FRAME_IMS | 0b01 }
946    sub AFTER_FRAMESET_IM () { FRAME_IMS | 0b10 }
947    sub IN_SELECT_IM () { SELECT_IMS | 0b01 }
948    sub IN_SELECT_IN_TABLE_IM () { SELECT_IMS | 0b10 }
949    sub IN_COLUMN_GROUP_IM () { 0b10 }
950    
951  ## Implementations MUST act as if state machine in the spec  ## Implementations MUST act as if state machine in the spec
952    
953  sub _initialize_tokenizer ($) {  sub _initialize_tokenizer ($) {
954    my $self = shift;    my $self = shift;
955    $self->{state} = 'data'; # MUST    $self->{state} = DATA_STATE; # MUST
956    $self->{content_model_flag} = 'PCDATA'; # be    #$self->{s_kwd}; # state keyword - initialized when used
957    undef $self->{current_token}; # start tag, end tag, comment, or DOCTYPE    #$self->{entity__value}; # initialized when used
958    undef $self->{current_attribute};    #$self->{entity__match}; # initialized when used
959    undef $self->{last_emitted_start_tag_name};    $self->{content_model} = PCDATA_CONTENT_MODEL; # be
960    undef $self->{last_attribute_value_state};    undef $self->{ct}; # current token
961    $self->{char} = [];    undef $self->{ca}; # current attribute
962    # $self->{next_input_character}    undef $self->{last_stag_name}; # last emitted start tag name
963      #$self->{prev_state}; # initialized when used
964      delete $self->{self_closing};
965      $self->{char_buffer} = '';
966      $self->{char_buffer_pos} = 0;
967      $self->{nc} = -1; # next input character
968      #$self->{next_nc}
969    !!!next-input-character;    !!!next-input-character;
970    $self->{token} = [];    $self->{token} = [];
971    # $self->{escape}    # $self->{escape}
972  } # _initialize_tokenizer  } # _initialize_tokenizer
973    
974  ## A token has:  ## A token has:
975  ##   ->{type} eq 'DOCTYPE', 'start tag', 'end tag', 'comment',  ##   ->{type} == DOCTYPE_TOKEN, START_TAG_TOKEN, END_TAG_TOKEN, COMMENT_TOKEN,
976  ##       'character', or 'end-of-file'  ##       CHARACTER_TOKEN, or END_OF_FILE_TOKEN
977  ##   ->{name} (DOCTYPE, start tag (tag name), end tag (tag name))  ##   ->{name} (DOCTYPE_TOKEN)
978  ##   ->{public_identifier} (DOCTYPE)  ##   ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)
979  ##   ->{system_identifier} (DOCTYPE)  ##   ->{pubid} (DOCTYPE_TOKEN)
980  ##   ->{correct} == 1 or 0 (DOCTYPE)  ##   ->{sysid} (DOCTYPE_TOKEN)
981  ##   ->{attributes} isa HASH (start tag, end tag)  ##   ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag
982  ##   ->{data} (comment, character)  ##   ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)
983    ##        ->{name}
984    ##        ->{value}
985    ##        ->{has_reference} == 1 or 0
986    ##   ->{data} (COMMENT_TOKEN, CHARACTER_TOKEN)
987    ## NOTE: The "self-closing flag" is hold as |$self->{self_closing}|.
988    ##     |->{self_closing}| is used to save the value of |$self->{self_closing}|
989    ##     while the token is pushed back to the stack.
990    
991  ## Emitted token MUST immediately be handled by the tree construction state.  ## Emitted token MUST immediately be handled by the tree construction state.
992    
# Line 185  sub _initialize_tokenizer ($) { Line 996  sub _initialize_tokenizer ($) {
996  ## has completed loading.  If one has, then it MUST be executed  ## has completed loading.  If one has, then it MUST be executed
997  ## and removed from the list.  ## and removed from the list.
998    
999    ## TODO: Polytheistic slash SHOULD NOT be used. (Applied only to atheists.)
1000    ## (This requirement was dropped from HTML5 spec, unfortunately.)
1001    
1002    my $is_space = {
1003      0x0009 => 1, # CHARACTER TABULATION (HT)
1004      0x000A => 1, # LINE FEED (LF)
1005      #0x000B => 0, # LINE TABULATION (VT)
1006      0x000C => 1, # FORM FEED (FF)
1007      #0x000D => 1, # CARRIAGE RETURN (CR)
1008      0x0020 => 1, # SPACE (SP)
1009    };
1010    
1011  sub _get_next_token ($) {  sub _get_next_token ($) {
1012    my $self = shift;    my $self = shift;
1013    
1014      if ($self->{self_closing}) {
1015        !!!parse-error (type => 'nestc', token => $self->{ct});
1016        ## NOTE: The |self_closing| flag is only set by start tag token.
1017        ## In addition, when a start tag token is emitted, it is always set to
1018        ## |ct|.
1019        delete $self->{self_closing};
1020      }
1021    
1022    if (@{$self->{token}}) {    if (@{$self->{token}}) {
1023        $self->{self_closing} = $self->{token}->[0]->{self_closing};
1024      return shift @{$self->{token}};      return shift @{$self->{token}};
1025    }    }
1026    
1027    A: {    A: {
1028      if ($self->{state} eq 'data') {      if ($self->{state} == PCDATA_STATE) {
1029        if ($self->{next_input_character} == 0x0026) { # &        ## NOTE: Same as |DATA_STATE|, but only for |PCDATA| content model.
1030          if ($self->{content_model_flag} eq 'PCDATA' or  
1031              $self->{content_model_flag} eq 'RCDATA') {        if ($self->{nc} == 0x0026) { # &
1032            $self->{state} = 'entity data';          !!!cp (0.1);
1033            ## NOTE: In the spec, the tokenizer is switched to the
1034            ## "entity data state".  In this implementation, the tokenizer
1035            ## is switched to the |ENTITY_STATE|, which is an implementation
1036            ## of the "consume a character reference" algorithm.
1037            $self->{entity_add} = -1;
1038            $self->{prev_state} = DATA_STATE;
1039            $self->{state} = ENTITY_STATE;
1040            !!!next-input-character;
1041            redo A;
1042          } elsif ($self->{nc} == 0x003C) { # <
1043            !!!cp (0.2);
1044            $self->{state} = TAG_OPEN_STATE;
1045            !!!next-input-character;
1046            redo A;
1047          } elsif ($self->{nc} == -1) {
1048            !!!cp (0.3);
1049            !!!emit ({type => END_OF_FILE_TOKEN,
1050                      line => $self->{line}, column => $self->{column}});
1051            last A; ## TODO: ok?
1052          } else {
1053            !!!cp (0.4);
1054            #
1055          }
1056    
1057          # Anything else
1058          my $token = {type => CHARACTER_TOKEN,
1059                       data => chr $self->{nc},
1060                       line => $self->{line}, column => $self->{column},
1061                      };
1062          $self->{read_until}->($token->{data}, q[<&], length $token->{data});
1063    
1064          ## Stay in the state.
1065          !!!next-input-character;
1066          !!!emit ($token);
1067          redo A;
1068        } elsif ($self->{state} == DATA_STATE) {
1069          $self->{s_kwd} = '' unless defined $self->{s_kwd};
1070          if ($self->{nc} == 0x0026) { # &
1071            $self->{s_kwd} = '';
1072            if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA
1073                not $self->{escape}) {
1074              !!!cp (1);
1075              ## NOTE: In the spec, the tokenizer is switched to the
1076              ## "entity data state".  In this implementation, the tokenizer
1077              ## is switched to the |ENTITY_STATE|, which is an implementation
1078              ## of the "consume a character reference" algorithm.
1079              $self->{entity_add} = -1;
1080              $self->{prev_state} = DATA_STATE;
1081              $self->{state} = ENTITY_STATE;
1082            !!!next-input-character;            !!!next-input-character;
1083            redo A;            redo A;
1084          } else {          } else {
1085              !!!cp (2);
1086            #            #
1087          }          }
1088        } elsif ($self->{next_input_character} == 0x002D) { # -        } elsif ($self->{nc} == 0x002D) { # -
1089          if ($self->{content_model_flag} eq 'RCDATA' or          if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1090              $self->{content_model_flag} eq 'CDATA') {            $self->{s_kwd} .= '-';
1091            unless ($self->{escape}) {            
1092              if ($self->{prev_input_character}->[0] == 0x002D and # -            if ($self->{s_kwd} eq '<!--') {
1093                  $self->{prev_input_character}->[1] == 0x0021 and # !              !!!cp (3);
1094                  $self->{prev_input_character}->[2] == 0x003C) { # <              $self->{escape} = 1; # unless $self->{escape};
1095                $self->{escape} = 1;              $self->{s_kwd} = '--';
1096              }              #
1097              } elsif ($self->{s_kwd} eq '---') {
1098                !!!cp (4);
1099                $self->{s_kwd} = '--';
1100                #
1101              } else {
1102                !!!cp (5);
1103                #
1104            }            }
1105          }          }
1106                    
1107          #          #
1108        } elsif ($self->{next_input_character} == 0x003C) { # <        } elsif ($self->{nc} == 0x0021) { # !
1109          if ($self->{content_model_flag} eq 'PCDATA' or          if (length $self->{s_kwd}) {
1110              (($self->{content_model_flag} eq 'CDATA' or            !!!cp (5.1);
1111                $self->{content_model_flag} eq 'RCDATA') and            $self->{s_kwd} .= '!';
1112              #
1113            } else {
1114              !!!cp (5.2);
1115              #$self->{s_kwd} = '';
1116              #
1117            }
1118            #
1119          } elsif ($self->{nc} == 0x003C) { # <
1120            if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA
1121                (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA
1122               not $self->{escape})) {               not $self->{escape})) {
1123            $self->{state} = 'tag open';            !!!cp (6);
1124              $self->{state} = TAG_OPEN_STATE;
1125            !!!next-input-character;            !!!next-input-character;
1126            redo A;            redo A;
1127          } else {          } else {
1128              !!!cp (7);
1129              $self->{s_kwd} = '';
1130            #            #
1131          }          }
1132        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1133          if ($self->{escape} and          if ($self->{escape} and
1134              ($self->{content_model_flag} eq 'RCDATA' or              ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA
1135               $self->{content_model_flag} eq 'CDATA')) {            if ($self->{s_kwd} eq '--') {
1136            if ($self->{prev_input_character}->[0] == 0x002D and # -              !!!cp (8);
               $self->{prev_input_character}->[1] == 0x002D) { # -  
1137              delete $self->{escape};              delete $self->{escape};
1138              } else {
1139                !!!cp (9);
1140            }            }
1141            } else {
1142              !!!cp (10);
1143          }          }
1144                    
1145            $self->{s_kwd} = '';
1146          #          #
1147        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1148          !!!emit ({type => 'end-of-file'});          !!!cp (11);
1149            $self->{s_kwd} = '';
1150            !!!emit ({type => END_OF_FILE_TOKEN,
1151                      line => $self->{line}, column => $self->{column}});
1152          last A; ## TODO: ok?          last A; ## TODO: ok?
1153          } else {
1154            !!!cp (12);
1155            $self->{s_kwd} = '';
1156            #
1157        }        }
       # Anything else  
       my $token = {type => 'character',  
                    data => chr $self->{next_input_character}};  
       ## Stay in the data state  
       !!!next-input-character;  
   
       !!!emit ($token);  
1158    
1159        redo A;        # Anything else
1160      } elsif ($self->{state} eq 'entity data') {        my $token = {type => CHARACTER_TOKEN,
1161        ## (cannot happen in CDATA state)                     data => chr $self->{nc},
1162                             line => $self->{line}, column => $self->{column},
1163        my $token = $self->_tokenize_attempt_to_consume_an_entity (0);                    };
1164          if ($self->{read_until}->($token->{data}, q[-!<>&],
1165        $self->{state} = 'data';                                  length $token->{data})) {
1166        # next-input-character is already done          $self->{s_kwd} = '';
1167          }
1168    
1169        unless (defined $token) {        ## Stay in the data state.
1170          !!!emit ({type => 'character', data => '&'});        if ($self->{content_model} == PCDATA_CONTENT_MODEL) {
1171            !!!cp (13);
1172            $self->{state} = PCDATA_STATE;
1173        } else {        } else {
1174          !!!emit ($token);          !!!cp (14);
1175            ## Stay in the state.
1176        }        }
1177          !!!next-input-character;
1178          !!!emit ($token);
1179        redo A;        redo A;
1180      } elsif ($self->{state} eq 'tag open') {      } elsif ($self->{state} == TAG_OPEN_STATE) {
1181        if ($self->{content_model_flag} eq 'RCDATA' or        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1182            $self->{content_model_flag} eq 'CDATA') {          if ($self->{nc} == 0x002F) { # /
1183          if ($self->{next_input_character} == 0x002F) { # /            !!!cp (15);
1184            !!!next-input-character;            !!!next-input-character;
1185            $self->{state} = 'close tag open';            $self->{state} = CLOSE_TAG_OPEN_STATE;
1186            redo A;            redo A;
1187            } elsif ($self->{nc} == 0x0021) { # !
1188              !!!cp (15.1);
1189              $self->{s_kwd} = '<' unless $self->{escape};
1190              #
1191          } else {          } else {
1192            ## reconsume            !!!cp (16);
1193            $self->{state} = 'data';            #
   
           !!!emit ({type => 'character', data => '<'});  
   
           redo A;  
1194          }          }
1195        } elsif ($self->{content_model_flag} eq 'PCDATA') {  
1196          if ($self->{next_input_character} == 0x0021) { # !          ## reconsume
1197            $self->{state} = 'markup declaration open';          $self->{state} = DATA_STATE;
1198            !!!emit ({type => CHARACTER_TOKEN, data => '<',
1199                      line => $self->{line_prev},
1200                      column => $self->{column_prev},
1201                     });
1202            redo A;
1203          } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA
1204            if ($self->{nc} == 0x0021) { # !
1205              !!!cp (17);
1206              $self->{state} = MARKUP_DECLARATION_OPEN_STATE;
1207            !!!next-input-character;            !!!next-input-character;
1208            redo A;            redo A;
1209          } elsif ($self->{next_input_character} == 0x002F) { # /          } elsif ($self->{nc} == 0x002F) { # /
1210            $self->{state} = 'close tag open';            !!!cp (18);
1211              $self->{state} = CLOSE_TAG_OPEN_STATE;
1212            !!!next-input-character;            !!!next-input-character;
1213            redo A;            redo A;
1214          } elsif (0x0041 <= $self->{next_input_character} and          } elsif (0x0041 <= $self->{nc} and
1215                   $self->{next_input_character} <= 0x005A) { # A..Z                   $self->{nc} <= 0x005A) { # A..Z
1216            $self->{current_token}            !!!cp (19);
1217              = {type => 'start tag',            $self->{ct}
1218                 tag_name => chr ($self->{next_input_character} + 0x0020)};              = {type => START_TAG_TOKEN,
1219            $self->{state} = 'tag name';                 tag_name => chr ($self->{nc} + 0x0020),
1220                   line => $self->{line_prev},
1221                   column => $self->{column_prev}};
1222              $self->{state} = TAG_NAME_STATE;
1223            !!!next-input-character;            !!!next-input-character;
1224            redo A;            redo A;
1225          } elsif (0x0061 <= $self->{next_input_character} and          } elsif (0x0061 <= $self->{nc} and
1226                   $self->{next_input_character} <= 0x007A) { # a..z                   $self->{nc} <= 0x007A) { # a..z
1227            $self->{current_token} = {type => 'start tag',            !!!cp (20);
1228                              tag_name => chr ($self->{next_input_character})};            $self->{ct} = {type => START_TAG_TOKEN,
1229            $self->{state} = 'tag name';                                      tag_name => chr ($self->{nc}),
1230                                        line => $self->{line_prev},
1231                                        column => $self->{column_prev}};
1232              $self->{state} = TAG_NAME_STATE;
1233            !!!next-input-character;            !!!next-input-character;
1234            redo A;            redo A;
1235          } elsif ($self->{next_input_character} == 0x003E) { # >          } elsif ($self->{nc} == 0x003E) { # >
1236            !!!parse-error (type => 'empty start tag');            !!!cp (21);
1237            $self->{state} = 'data';            !!!parse-error (type => 'empty start tag',
1238                              line => $self->{line_prev},
1239                              column => $self->{column_prev});
1240              $self->{state} = DATA_STATE;
1241            !!!next-input-character;            !!!next-input-character;
1242    
1243            !!!emit ({type => 'character', data => '<>'});            !!!emit ({type => CHARACTER_TOKEN, data => '<>',
1244                        line => $self->{line_prev},
1245                        column => $self->{column_prev},
1246                       });
1247    
1248            redo A;            redo A;
1249          } elsif ($self->{next_input_character} == 0x003F) { # ?          } elsif ($self->{nc} == 0x003F) { # ?
1250            !!!parse-error (type => 'pio');            !!!cp (22);
1251            $self->{state} = 'bogus comment';            !!!parse-error (type => 'pio',
1252            ## $self->{next_input_character} is intentionally left as is                            line => $self->{line_prev},
1253                              column => $self->{column_prev});
1254              $self->{state} = BOGUS_COMMENT_STATE;
1255              $self->{ct} = {type => COMMENT_TOKEN, data => '',
1256                                        line => $self->{line_prev},
1257                                        column => $self->{column_prev},
1258                                       };
1259              ## $self->{nc} is intentionally left as is
1260            redo A;            redo A;
1261          } else {          } else {
1262            !!!parse-error (type => 'bare stago');            !!!cp (23);
1263            $self->{state} = 'data';            !!!parse-error (type => 'bare stago',
1264                              line => $self->{line_prev},
1265                              column => $self->{column_prev});
1266              $self->{state} = DATA_STATE;
1267            ## reconsume            ## reconsume
1268    
1269            !!!emit ({type => 'character', data => '<'});            !!!emit ({type => CHARACTER_TOKEN, data => '<',
1270                        line => $self->{line_prev},
1271                        column => $self->{column_prev},
1272                       });
1273    
1274            redo A;            redo A;
1275          }          }
1276        } else {        } else {
1277          die "$0: $self->{content_model_flag}: Unknown content model flag";          die "$0: $self->{content_model} in tag open";
1278        }        }
1279      } elsif ($self->{state} eq 'close tag open') {      } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {
1280        if ($self->{content_model_flag} eq 'RCDATA' or        ## NOTE: The "close tag open state" in the spec is implemented as
1281            $self->{content_model_flag} eq 'CDATA') {        ## |CLOSE_TAG_OPEN_STATE| and |CDATA_RCDATA_CLOSE_TAG_STATE|.
1282          if (defined $self->{last_emitted_start_tag_name}) {  
1283            ## NOTE: <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>        my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"
1284            my @next_char;        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1285            TAGNAME: for (my $i = 0; $i < length $self->{last_emitted_start_tag_name}; $i++) {          if (defined $self->{last_stag_name}) {
1286              push @next_char, $self->{next_input_character};            $self->{state} = CDATA_RCDATA_CLOSE_TAG_STATE;
1287              my $c = ord substr ($self->{last_emitted_start_tag_name}, $i, 1);            $self->{s_kwd} = '';
1288              my $C = 0x0061 <= $c && $c <= 0x007A ? $c - 0x0020 : $c;            ## Reconsume.
1289              if ($self->{next_input_character} == $c or $self->{next_input_character} == $C) {            redo A;
               !!!next-input-character;  
               next TAGNAME;  
             } else {  
               $self->{next_input_character} = shift @next_char; # reconsume  
               !!!back-next-input-character (@next_char);  
               $self->{state} = 'data';  
   
               !!!emit ({type => 'character', data => '</'});  
     
               redo A;  
             }  
           }  
           push @next_char, $self->{next_input_character};  
         
           unless ($self->{next_input_character} == 0x0009 or # HT  
                   $self->{next_input_character} == 0x000A or # LF  
                   $self->{next_input_character} == 0x000B or # VT  
                   $self->{next_input_character} == 0x000C or # FF  
                   $self->{next_input_character} == 0x0020 or # SP  
                   $self->{next_input_character} == 0x003E or # >  
                   $self->{next_input_character} == 0x002F or # /  
                   $self->{next_input_character} == -1) {  
             $self->{next_input_character} = shift @next_char; # reconsume  
             !!!back-next-input-character (@next_char);  
             $self->{state} = 'data';  
             !!!emit ({type => 'character', data => '</'});  
             redo A;  
           } else {  
             $self->{next_input_character} = shift @next_char;  
             !!!back-next-input-character (@next_char);  
             # and consume...  
           }  
1290          } else {          } else {
1291            ## No start tag token has ever been emitted            ## No start tag token has ever been emitted
1292            # next-input-character is already done            ## NOTE: See <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>.
1293            $self->{state} = 'data';            !!!cp (28);
1294            !!!emit ({type => 'character', data => '</'});            $self->{state} = DATA_STATE;
1295              ## Reconsume.
1296              !!!emit ({type => CHARACTER_TOKEN, data => '</',
1297                        line => $l, column => $c,
1298                       });
1299            redo A;            redo A;
1300          }          }
1301        }        }
1302          
1303        if (0x0041 <= $self->{next_input_character} and        if (0x0041 <= $self->{nc} and
1304            $self->{next_input_character} <= 0x005A) { # A..Z            $self->{nc} <= 0x005A) { # A..Z
1305          $self->{current_token} = {type => 'end tag',          !!!cp (29);
1306                            tag_name => chr ($self->{next_input_character} + 0x0020)};          $self->{ct}
1307          $self->{state} = 'tag name';              = {type => END_TAG_TOKEN,
1308          !!!next-input-character;                 tag_name => chr ($self->{nc} + 0x0020),
1309          redo A;                 line => $l, column => $c};
1310        } elsif (0x0061 <= $self->{next_input_character} and          $self->{state} = TAG_NAME_STATE;
1311                 $self->{next_input_character} <= 0x007A) { # a..z          !!!next-input-character;
1312          $self->{current_token} = {type => 'end tag',          redo A;
1313                            tag_name => chr ($self->{next_input_character})};        } elsif (0x0061 <= $self->{nc} and
1314          $self->{state} = 'tag name';                 $self->{nc} <= 0x007A) { # a..z
1315          !!!next-input-character;          !!!cp (30);
1316          redo A;          $self->{ct} = {type => END_TAG_TOKEN,
1317        } elsif ($self->{next_input_character} == 0x003E) { # >                                    tag_name => chr ($self->{nc}),
1318          !!!parse-error (type => 'empty end tag');                                    line => $l, column => $c};
1319          $self->{state} = 'data';          $self->{state} = TAG_NAME_STATE;
1320            !!!next-input-character;
1321            redo A;
1322          } elsif ($self->{nc} == 0x003E) { # >
1323            !!!cp (31);
1324            !!!parse-error (type => 'empty end tag',
1325                            line => $self->{line_prev}, ## "<" in "</>"
1326                            column => $self->{column_prev} - 1);
1327            $self->{state} = DATA_STATE;
1328          !!!next-input-character;          !!!next-input-character;
1329          redo A;          redo A;
1330        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1331            !!!cp (32);
1332          !!!parse-error (type => 'bare etago');          !!!parse-error (type => 'bare etago');
1333          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1334          # reconsume          # reconsume
1335    
1336          !!!emit ({type => 'character', data => '</'});          !!!emit ({type => CHARACTER_TOKEN, data => '</',
1337                      line => $l, column => $c,
1338                     });
1339    
1340          redo A;          redo A;
1341        } else {        } else {
1342            !!!cp (33);
1343          !!!parse-error (type => 'bogus end tag');          !!!parse-error (type => 'bogus end tag');
1344          $self->{state} = 'bogus comment';          $self->{state} = BOGUS_COMMENT_STATE;
1345          ## $self->{next_input_character} is intentionally left as is          $self->{ct} = {type => COMMENT_TOKEN, data => '',
1346          redo A;                                    line => $self->{line_prev}, # "<" of "</"
1347                                      column => $self->{column_prev} - 1,
1348                                     };
1349            ## NOTE: $self->{nc} is intentionally left as is.
1350            ## Although the "anything else" case of the spec not explicitly
1351            ## states that the next input character is to be reconsumed,
1352            ## it will be included to the |data| of the comment token
1353            ## generated from the bogus end tag, as defined in the
1354            ## "bogus comment state" entry.
1355            redo A;
1356          }
1357        } elsif ($self->{state} == CDATA_RCDATA_CLOSE_TAG_STATE) {
1358          my $ch = substr $self->{last_stag_name}, length $self->{s_kwd}, 1;
1359          if (length $ch) {
1360            my $CH = $ch;
1361            $ch =~ tr/a-z/A-Z/;
1362            my $nch = chr $self->{nc};
1363            if ($nch eq $ch or $nch eq $CH) {
1364              !!!cp (24);
1365              ## Stay in the state.
1366              $self->{s_kwd} .= $nch;
1367              !!!next-input-character;
1368              redo A;
1369            } else {
1370              !!!cp (25);
1371              $self->{state} = DATA_STATE;
1372              ## Reconsume.
1373              !!!emit ({type => CHARACTER_TOKEN,
1374                        data => '</' . $self->{s_kwd},
1375                        line => $self->{line_prev},
1376                        column => $self->{column_prev} - 1 - length $self->{s_kwd},
1377                       });
1378              redo A;
1379            }
1380          } else { # after "<{tag-name}"
1381            unless ($is_space->{$self->{nc}} or
1382                    {
1383                     0x003E => 1, # >
1384                     0x002F => 1, # /
1385                     -1 => 1, # EOF
1386                    }->{$self->{nc}}) {
1387              !!!cp (26);
1388              ## Reconsume.
1389              $self->{state} = DATA_STATE;
1390              !!!emit ({type => CHARACTER_TOKEN,
1391                        data => '</' . $self->{s_kwd},
1392                        line => $self->{line_prev},
1393                        column => $self->{column_prev} - 1 - length $self->{s_kwd},
1394                       });
1395              redo A;
1396            } else {
1397              !!!cp (27);
1398              $self->{ct}
1399                  = {type => END_TAG_TOKEN,
1400                     tag_name => $self->{last_stag_name},
1401                     line => $self->{line_prev},
1402                     column => $self->{column_prev} - 1 - length $self->{s_kwd}};
1403              $self->{state} = TAG_NAME_STATE;
1404              ## Reconsume.
1405              redo A;
1406            }
1407        }        }
1408      } elsif ($self->{state} eq 'tag name') {      } elsif ($self->{state} == TAG_NAME_STATE) {
1409        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
1410            $self->{next_input_character} == 0x000A or # LF          !!!cp (34);
1411            $self->{next_input_character} == 0x000B or # VT          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1412            $self->{next_input_character} == 0x000C or # FF          !!!next-input-character;
1413            $self->{next_input_character} == 0x0020) { # SP          redo A;
1414          $self->{state} = 'before attribute name';        } elsif ($self->{nc} == 0x003E) { # >
1415          !!!next-input-character;          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1416          redo A;            !!!cp (35);
1417        } elsif ($self->{next_input_character} == 0x003E) { # >            $self->{last_stag_name} = $self->{ct}->{tag_name};
1418          if ($self->{current_token}->{type} eq 'start tag') {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1419            $self->{current_token}->{first_start_tag}            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1420                = not defined $self->{last_emitted_start_tag_name};            #if ($self->{ct}->{attributes}) {
1421            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            #  ## NOTE: This should never be reached.
1422          } elsif ($self->{current_token}->{type} eq 'end tag') {            #  !!! cp (36);
1423            $self->{content_model_flag} = 'PCDATA'; # MUST            #  !!! parse-error (type => 'end tag attribute');
1424            if ($self->{current_token}->{attributes}) {            #} else {
1425              !!!parse-error (type => 'end tag attribute');              !!!cp (37);
1426            }            #}
1427          } else {          } else {
1428            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1429          }          }
1430          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1431          !!!next-input-character;          !!!next-input-character;
1432    
1433          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1434    
1435          redo A;          redo A;
1436        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{nc} and
1437                 $self->{next_input_character} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1438          $self->{current_token}->{tag_name} .= chr ($self->{next_input_character} + 0x0020);          !!!cp (38);
1439            $self->{ct}->{tag_name} .= chr ($self->{nc} + 0x0020);
1440            # start tag or end tag            # start tag or end tag
1441          ## Stay in this state          ## Stay in this state
1442          !!!next-input-character;          !!!next-input-character;
1443          redo A;          redo A;
1444        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1445          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1446          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1447            $self->{current_token}->{first_start_tag}            !!!cp (39);
1448                = not defined $self->{last_emitted_start_tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1449            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1450          } elsif ($self->{current_token}->{type} eq 'end tag') {            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1451            $self->{content_model_flag} = 'PCDATA'; # MUST            #if ($self->{ct}->{attributes}) {
1452            if ($self->{current_token}->{attributes}) {            #  ## NOTE: This state should never be reached.
1453              !!!parse-error (type => 'end tag attribute');            #  !!! cp (40);
1454            }            #  !!! parse-error (type => 'end tag attribute');
1455              #} else {
1456                !!!cp (41);
1457              #}
1458          } else {          } else {
1459            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1460          }          }
1461          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1462          # reconsume          # reconsume
1463    
1464          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1465    
1466          redo A;          redo A;
1467        } elsif ($self->{next_input_character} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1468            !!!cp (42);
1469            $self->{state} = SELF_CLOSING_START_TAG_STATE;
1470          !!!next-input-character;          !!!next-input-character;
         if ($self->{next_input_character} == 0x003E and # >  
             $self->{current_token}->{type} eq 'start tag' and  
             $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {  
           # permitted slash  
           #  
         } else {  
           !!!parse-error (type => 'nestc');  
         }  
         $self->{state} = 'before attribute name';  
         # next-input-character is already done  
1471          redo A;          redo A;
1472        } else {        } else {
1473          $self->{current_token}->{tag_name} .= chr $self->{next_input_character};          !!!cp (44);
1474            $self->{ct}->{tag_name} .= chr $self->{nc};
1475            # start tag or end tag            # start tag or end tag
1476          ## Stay in the state          ## Stay in the state
1477          !!!next-input-character;          !!!next-input-character;
1478          redo A;          redo A;
1479        }        }
1480      } elsif ($self->{state} eq 'before attribute name') {      } elsif ($self->{state} == BEFORE_ATTRIBUTE_NAME_STATE) {
1481        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
1482            $self->{next_input_character} == 0x000A or # LF          !!!cp (45);
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
1483          ## Stay in the state          ## Stay in the state
1484          !!!next-input-character;          !!!next-input-character;
1485          redo A;          redo A;
1486        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1487          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1488            $self->{current_token}->{first_start_tag}            !!!cp (46);
1489                = not defined $self->{last_emitted_start_tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1490            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1491          } elsif ($self->{current_token}->{type} eq 'end tag') {            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1492            $self->{content_model_flag} = 'PCDATA'; # MUST            if ($self->{ct}->{attributes}) {
1493            if ($self->{current_token}->{attributes}) {              !!!cp (47);
1494              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1495              } else {
1496                !!!cp (48);
1497            }            }
1498          } else {          } else {
1499            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1500          }          }
1501          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1502          !!!next-input-character;          !!!next-input-character;
1503    
1504          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1505    
1506          redo A;          redo A;
1507        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{nc} and
1508                 $self->{next_input_character} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1509          $self->{current_attribute} = {name => chr ($self->{next_input_character} + 0x0020),          !!!cp (49);
1510                                value => ''};          $self->{ca}
1511          $self->{state} = 'attribute name';              = {name => chr ($self->{nc} + 0x0020),
1512                   value => '',
1513                   line => $self->{line}, column => $self->{column}};
1514            $self->{state} = ATTRIBUTE_NAME_STATE;
1515          !!!next-input-character;          !!!next-input-character;
1516          redo A;          redo A;
1517        } elsif ($self->{next_input_character} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1518            !!!cp (50);
1519            $self->{state} = SELF_CLOSING_START_TAG_STATE;
1520          !!!next-input-character;          !!!next-input-character;
         if ($self->{next_input_character} == 0x003E and # >  
             $self->{current_token}->{type} eq 'start tag' and  
             $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {  
           # permitted slash  
           #  
         } else {  
           !!!parse-error (type => 'nestc');  
         }  
         ## Stay in the state  
         # next-input-character is already done  
1521          redo A;          redo A;
1522        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1523          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1524          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1525            $self->{current_token}->{first_start_tag}            !!!cp (52);
1526                = not defined $self->{last_emitted_start_tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1527            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1528          } elsif ($self->{current_token}->{type} eq 'end tag') {            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1529            $self->{content_model_flag} = 'PCDATA'; # MUST            if ($self->{ct}->{attributes}) {
1530            if ($self->{current_token}->{attributes}) {              !!!cp (53);
1531              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1532              } else {
1533                !!!cp (54);
1534            }            }
1535          } else {          } else {
1536            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1537          }          }
1538          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1539          # reconsume          # reconsume
1540    
1541          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1542    
1543          redo A;          redo A;
1544        } else {        } else {
1545          $self->{current_attribute} = {name => chr ($self->{next_input_character}),          if ({
1546                                value => ''};               0x0022 => 1, # "
1547          $self->{state} = 'attribute name';               0x0027 => 1, # '
1548                 0x003D => 1, # =
1549                }->{$self->{nc}}) {
1550              !!!cp (55);
1551              !!!parse-error (type => 'bad attribute name');
1552            } else {
1553              !!!cp (56);
1554            }
1555            $self->{ca}
1556                = {name => chr ($self->{nc}),
1557                   value => '',
1558                   line => $self->{line}, column => $self->{column}};
1559            $self->{state} = ATTRIBUTE_NAME_STATE;
1560          !!!next-input-character;          !!!next-input-character;
1561          redo A;          redo A;
1562        }        }
1563      } elsif ($self->{state} eq 'attribute name') {      } elsif ($self->{state} == ATTRIBUTE_NAME_STATE) {
1564        my $before_leave = sub {        my $before_leave = sub {
1565          if (exists $self->{current_token}->{attributes} # start tag or end tag          if (exists $self->{ct}->{attributes} # start tag or end tag
1566              ->{$self->{current_attribute}->{name}}) { # MUST              ->{$self->{ca}->{name}}) { # MUST
1567            !!!parse-error (type => 'dupulicate attribute');            !!!cp (57);
1568            ## Discard $self->{current_attribute} # MUST            !!!parse-error (type => 'duplicate attribute', text => $self->{ca}->{name}, line => $self->{ca}->{line}, column => $self->{ca}->{column});
1569              ## Discard $self->{ca} # MUST
1570          } else {          } else {
1571            $self->{current_token}->{attributes}->{$self->{current_attribute}->{name}}            !!!cp (58);
1572              = $self->{current_attribute};            $self->{ct}->{attributes}->{$self->{ca}->{name}}
1573                = $self->{ca};
1574          }          }
1575        }; # $before_leave        }; # $before_leave
1576    
1577        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
1578            $self->{next_input_character} == 0x000A or # LF          !!!cp (59);
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
1579          $before_leave->();          $before_leave->();
1580          $self->{state} = 'after attribute name';          $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;
1581          !!!next-input-character;          !!!next-input-character;
1582          redo A;          redo A;
1583        } elsif ($self->{next_input_character} == 0x003D) { # =        } elsif ($self->{nc} == 0x003D) { # =
1584            !!!cp (60);
1585          $before_leave->();          $before_leave->();
1586          $self->{state} = 'before attribute value';          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1587          !!!next-input-character;          !!!next-input-character;
1588          redo A;          redo A;
1589        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1590          $before_leave->();          $before_leave->();
1591          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1592            $self->{current_token}->{first_start_tag}            !!!cp (61);
1593                = not defined $self->{last_emitted_start_tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1594            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1595          } elsif ($self->{current_token}->{type} eq 'end tag') {            !!!cp (62);
1596            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1597            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1598              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1599            }            }
1600          } else {          } else {
1601            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1602          }          }
1603          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1604          !!!next-input-character;          !!!next-input-character;
1605    
1606          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1607    
1608          redo A;          redo A;
1609        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{nc} and
1610                 $self->{next_input_character} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1611          $self->{current_attribute}->{name} .= chr ($self->{next_input_character} + 0x0020);          !!!cp (63);
1612            $self->{ca}->{name} .= chr ($self->{nc} + 0x0020);
1613          ## Stay in the state          ## Stay in the state
1614          !!!next-input-character;          !!!next-input-character;
1615          redo A;          redo A;
1616        } elsif ($self->{next_input_character} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1617            !!!cp (64);
1618          $before_leave->();          $before_leave->();
1619            $self->{state} = SELF_CLOSING_START_TAG_STATE;
1620          !!!next-input-character;          !!!next-input-character;
         if ($self->{next_input_character} == 0x003E and # >  
             $self->{current_token}->{type} eq 'start tag' and  
             $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {  
           # permitted slash  
           #  
         } else {  
           !!!parse-error (type => 'nestc');  
         }  
         $self->{state} = 'before attribute name';  
         # next-input-character is already done  
1621          redo A;          redo A;
1622        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1623          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1624          $before_leave->();          $before_leave->();
1625          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1626            $self->{current_token}->{first_start_tag}            !!!cp (66);
1627                = not defined $self->{last_emitted_start_tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1628            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1629          } elsif ($self->{current_token}->{type} eq 'end tag') {            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1630            $self->{content_model_flag} = 'PCDATA'; # MUST            if ($self->{ct}->{attributes}) {
1631            if ($self->{current_token}->{attributes}) {              !!!cp (67);
1632              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1633              } else {
1634                ## NOTE: This state should never be reached.
1635                !!!cp (68);
1636            }            }
1637          } else {          } else {
1638            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1639          }          }
1640          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1641          # reconsume          # reconsume
1642    
1643          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1644    
1645          redo A;          redo A;
1646        } else {        } else {
1647          $self->{current_attribute}->{name} .= chr ($self->{next_input_character});          if ($self->{nc} == 0x0022 or # "
1648                $self->{nc} == 0x0027) { # '
1649              !!!cp (69);
1650              !!!parse-error (type => 'bad attribute name');
1651            } else {
1652              !!!cp (70);
1653            }
1654            $self->{ca}->{name} .= chr ($self->{nc});
1655          ## Stay in the state          ## Stay in the state
1656          !!!next-input-character;          !!!next-input-character;
1657          redo A;          redo A;
1658        }        }
1659      } elsif ($self->{state} eq 'after attribute name') {      } elsif ($self->{state} == AFTER_ATTRIBUTE_NAME_STATE) {
1660        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
1661            $self->{next_input_character} == 0x000A or # LF          !!!cp (71);
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
1662          ## Stay in the state          ## Stay in the state
1663          !!!next-input-character;          !!!next-input-character;
1664          redo A;          redo A;
1665        } elsif ($self->{next_input_character} == 0x003D) { # =        } elsif ($self->{nc} == 0x003D) { # =
1666          $self->{state} = 'before attribute value';          !!!cp (72);
1667            $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1668          !!!next-input-character;          !!!next-input-character;
1669          redo A;          redo A;
1670        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1671          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1672            $self->{current_token}->{first_start_tag}            !!!cp (73);
1673                = not defined $self->{last_emitted_start_tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1674            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1675          } elsif ($self->{current_token}->{type} eq 'end tag') {            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1676            $self->{content_model_flag} = 'PCDATA'; # MUST            if ($self->{ct}->{attributes}) {
1677            if ($self->{current_token}->{attributes}) {              !!!cp (74);
1678              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1679              } else {
1680                ## NOTE: This state should never be reached.
1681                !!!cp (75);
1682            }            }
1683          } else {          } else {
1684            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1685          }          }
1686          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1687          !!!next-input-character;          !!!next-input-character;
1688    
1689          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1690    
1691          redo A;          redo A;
1692        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{nc} and
1693                 $self->{next_input_character} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1694          $self->{current_attribute} = {name => chr ($self->{next_input_character} + 0x0020),          !!!cp (76);
1695                                value => ''};          $self->{ca}
1696          $self->{state} = 'attribute name';              = {name => chr ($self->{nc} + 0x0020),
1697                   value => '',
1698                   line => $self->{line}, column => $self->{column}};
1699            $self->{state} = ATTRIBUTE_NAME_STATE;
1700          !!!next-input-character;          !!!next-input-character;
1701          redo A;          redo A;
1702        } elsif ($self->{next_input_character} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1703            !!!cp (77);
1704            $self->{state} = SELF_CLOSING_START_TAG_STATE;
1705          !!!next-input-character;          !!!next-input-character;
         if ($self->{next_input_character} == 0x003E and # >  
             $self->{current_token}->{type} eq 'start tag' and  
             $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {  
           # permitted slash  
           #  
         } else {  
           !!!parse-error (type => 'nestc');  
           ## TODO: Different error type for <aa / bb> than <aa/>  
         }  
         $self->{state} = 'before attribute name';  
         # next-input-character is already done  
1706          redo A;          redo A;
1707        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1708          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1709          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1710            $self->{current_token}->{first_start_tag}            !!!cp (79);
1711                = not defined $self->{last_emitted_start_tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1712            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1713          } elsif ($self->{current_token}->{type} eq 'end tag') {            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1714            $self->{content_model_flag} = 'PCDATA'; # MUST            if ($self->{ct}->{attributes}) {
1715            if ($self->{current_token}->{attributes}) {              !!!cp (80);
1716              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1717              } else {
1718                ## NOTE: This state should never be reached.
1719                !!!cp (81);
1720            }            }
1721          } else {          } else {
1722            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1723          }          }
1724          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1725          # reconsume          # reconsume
1726    
1727          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1728    
1729          redo A;          redo A;
1730        } else {        } else {
1731          $self->{current_attribute} = {name => chr ($self->{next_input_character}),          if ($self->{nc} == 0x0022 or # "
1732                                value => ''};              $self->{nc} == 0x0027) { # '
1733          $self->{state} = 'attribute name';            !!!cp (78);
1734              !!!parse-error (type => 'bad attribute name');
1735            } else {
1736              !!!cp (82);
1737            }
1738            $self->{ca}
1739                = {name => chr ($self->{nc}),
1740                   value => '',
1741                   line => $self->{line}, column => $self->{column}};
1742            $self->{state} = ATTRIBUTE_NAME_STATE;
1743          !!!next-input-character;          !!!next-input-character;
1744          redo A;                  redo A;        
1745        }        }
1746      } elsif ($self->{state} eq 'before attribute value') {      } elsif ($self->{state} == BEFORE_ATTRIBUTE_VALUE_STATE) {
1747        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
1748            $self->{next_input_character} == 0x000A or # LF          !!!cp (83);
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP        
1749          ## Stay in the state          ## Stay in the state
1750          !!!next-input-character;          !!!next-input-character;
1751          redo A;          redo A;
1752        } elsif ($self->{next_input_character} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
1753          $self->{state} = 'attribute value (double-quoted)';          !!!cp (84);
1754            $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;
1755          !!!next-input-character;          !!!next-input-character;
1756          redo A;          redo A;
1757        } elsif ($self->{next_input_character} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1758          $self->{state} = 'attribute value (unquoted)';          !!!cp (85);
1759            $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1760          ## reconsume          ## reconsume
1761          redo A;          redo A;
1762        } elsif ($self->{next_input_character} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
1763          $self->{state} = 'attribute value (single-quoted)';          !!!cp (86);
1764            $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;
1765          !!!next-input-character;          !!!next-input-character;
1766          redo A;          redo A;
1767        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1768          if ($self->{current_token}->{type} eq 'start tag') {          !!!parse-error (type => 'empty unquoted attribute value');
1769            $self->{current_token}->{first_start_tag}          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1770                = not defined $self->{last_emitted_start_tag_name};            !!!cp (87);
1771            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1772          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1773            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1774            if ($self->{current_token}->{attributes}) {            if ($self->{ct}->{attributes}) {
1775                !!!cp (88);
1776              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1777              } else {
1778                ## NOTE: This state should never be reached.
1779                !!!cp (89);
1780            }            }
1781          } else {          } else {
1782            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1783          }          }
1784          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1785          !!!next-input-character;          !!!next-input-character;
1786    
1787          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1788    
1789          redo A;          redo A;
1790        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1791          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1792          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1793            $self->{current_token}->{first_start_tag}            !!!cp (90);
1794                = not defined $self->{last_emitted_start_tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1795            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1796          } elsif ($self->{current_token}->{type} eq 'end tag') {            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1797            $self->{content_model_flag} = 'PCDATA'; # MUST            if ($self->{ct}->{attributes}) {
1798            if ($self->{current_token}->{attributes}) {              !!!cp (91);
1799              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1800              } else {
1801                ## NOTE: This state should never be reached.
1802                !!!cp (92);
1803            }            }
1804          } else {          } else {
1805            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1806          }          }
1807          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1808          ## reconsume          ## reconsume
1809    
1810          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1811    
1812          redo A;          redo A;
1813        } else {        } else {
1814          $self->{current_attribute}->{value} .= chr ($self->{next_input_character});          if ($self->{nc} == 0x003D) { # =
1815          $self->{state} = 'attribute value (unquoted)';            !!!cp (93);
1816              !!!parse-error (type => 'bad attribute value');
1817            } else {
1818              !!!cp (94);
1819            }
1820            $self->{ca}->{value} .= chr ($self->{nc});
1821            $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1822          !!!next-input-character;          !!!next-input-character;
1823          redo A;          redo A;
1824        }        }
1825      } elsif ($self->{state} eq 'attribute value (double-quoted)') {      } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {
1826        if ($self->{next_input_character} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
1827          $self->{state} = 'before attribute name';          !!!cp (95);
1828            $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1829          !!!next-input-character;          !!!next-input-character;
1830          redo A;          redo A;
1831        } elsif ($self->{next_input_character} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1832          $self->{last_attribute_value_state} = 'attribute value (double-quoted)';          !!!cp (96);
1833          $self->{state} = 'entity in attribute value';          ## NOTE: In the spec, the tokenizer is switched to the
1834            ## "entity in attribute value state".  In this implementation, the
1835            ## tokenizer is switched to the |ENTITY_STATE|, which is an
1836            ## implementation of the "consume a character reference" algorithm.
1837            $self->{prev_state} = $self->{state};
1838            $self->{entity_add} = 0x0022; # "
1839            $self->{state} = ENTITY_STATE;
1840          !!!next-input-character;          !!!next-input-character;
1841          redo A;          redo A;
1842        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1843          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
1844          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1845            $self->{current_token}->{first_start_tag}            !!!cp (97);
1846                = not defined $self->{last_emitted_start_tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1847            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1848          } elsif ($self->{current_token}->{type} eq 'end tag') {            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1849            $self->{content_model_flag} = 'PCDATA'; # MUST            if ($self->{ct}->{attributes}) {
1850            if ($self->{current_token}->{attributes}) {              !!!cp (98);
1851              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1852              } else {
1853                ## NOTE: This state should never be reached.
1854                !!!cp (99);
1855            }            }
1856          } else {          } else {
1857            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1858          }          }
1859          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1860          ## reconsume          ## reconsume
1861    
1862          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1863    
1864          redo A;          redo A;
1865        } else {        } else {
1866          $self->{current_attribute}->{value} .= chr ($self->{next_input_character});          !!!cp (100);
1867            $self->{ca}->{value} .= chr ($self->{nc});
1868            $self->{read_until}->($self->{ca}->{value},
1869                                  q["&],
1870                                  length $self->{ca}->{value});
1871    
1872          ## Stay in the state          ## Stay in the state
1873          !!!next-input-character;          !!!next-input-character;
1874          redo A;          redo A;
1875        }        }
1876      } elsif ($self->{state} eq 'attribute value (single-quoted)') {      } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {
1877        if ($self->{next_input_character} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
1878          $self->{state} = 'before attribute name';          !!!cp (101);
1879            $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1880            !!!next-input-character;
1881            redo A;
1882          } elsif ($self->{nc} == 0x0026) { # &
1883            !!!cp (102);
1884            ## NOTE: In the spec, the tokenizer is switched to the
1885            ## "entity in attribute value state".  In this implementation, the
1886            ## tokenizer is switched to the |ENTITY_STATE|, which is an
1887            ## implementation of the "consume a character reference" algorithm.
1888            $self->{entity_add} = 0x0027; # '
1889            $self->{prev_state} = $self->{state};
1890            $self->{state} = ENTITY_STATE;
1891          !!!next-input-character;          !!!next-input-character;
1892          redo A;          redo A;
1893        } elsif ($self->{next_input_character} == 0x0026) { # &        } elsif ($self->{nc} == -1) {
         $self->{last_attribute_value_state} = 'attribute value (single-quoted)';  
         $self->{state} = 'entity in attribute value';  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == -1) {  
1894          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
1895          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1896            $self->{current_token}->{first_start_tag}            !!!cp (103);
1897                = not defined $self->{last_emitted_start_tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1898            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1899          } elsif ($self->{current_token}->{type} eq 'end tag') {            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1900            $self->{content_model_flag} = 'PCDATA'; # MUST            if ($self->{ct}->{attributes}) {
1901            if ($self->{current_token}->{attributes}) {              !!!cp (104);
1902              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1903              } else {
1904                ## NOTE: This state should never be reached.
1905                !!!cp (105);
1906            }            }
1907          } else {          } else {
1908            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1909          }          }
1910          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1911          ## reconsume          ## reconsume
1912    
1913          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1914    
1915          redo A;          redo A;
1916        } else {        } else {
1917          $self->{current_attribute}->{value} .= chr ($self->{next_input_character});          !!!cp (106);
1918            $self->{ca}->{value} .= chr ($self->{nc});
1919            $self->{read_until}->($self->{ca}->{value},
1920                                  q['&],
1921                                  length $self->{ca}->{value});
1922    
1923          ## Stay in the state          ## Stay in the state
1924          !!!next-input-character;          !!!next-input-character;
1925          redo A;          redo A;
1926        }        }
1927      } elsif ($self->{state} eq 'attribute value (unquoted)') {      } elsif ($self->{state} == ATTRIBUTE_VALUE_UNQUOTED_STATE) {
1928        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
1929            $self->{next_input_character} == 0x000A or # LF          !!!cp (107);
1930            $self->{next_input_character} == 0x000B or # HT          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1931            $self->{next_input_character} == 0x000C or # FF          !!!next-input-character;
1932            $self->{next_input_character} == 0x0020) { # SP          redo A;
1933          $self->{state} = 'before attribute name';        } elsif ($self->{nc} == 0x0026) { # &
1934          !!!next-input-character;          !!!cp (108);
1935          redo A;          ## NOTE: In the spec, the tokenizer is switched to the
1936        } elsif ($self->{next_input_character} == 0x0026) { # &          ## "entity in attribute value state".  In this implementation, the
1937          $self->{last_attribute_value_state} = 'attribute value (unquoted)';          ## tokenizer is switched to the |ENTITY_STATE|, which is an
1938          $self->{state} = 'entity in attribute value';          ## implementation of the "consume a character reference" algorithm.
1939          !!!next-input-character;          $self->{entity_add} = -1;
1940          redo A;          $self->{prev_state} = $self->{state};
1941        } elsif ($self->{next_input_character} == 0x003E) { # >          $self->{state} = ENTITY_STATE;
1942          if ($self->{current_token}->{type} eq 'start tag') {          !!!next-input-character;
1943            $self->{current_token}->{first_start_tag}          redo A;
1944                = not defined $self->{last_emitted_start_tag_name};        } elsif ($self->{nc} == 0x003E) { # >
1945            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1946          } elsif ($self->{current_token}->{type} eq 'end tag') {            !!!cp (109);
1947            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{last_stag_name} = $self->{ct}->{tag_name};
1948            if ($self->{current_token}->{attributes}) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1949              $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1950              if ($self->{ct}->{attributes}) {
1951                !!!cp (110);
1952              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1953              } else {
1954                ## NOTE: This state should never be reached.
1955                !!!cp (111);
1956            }            }
1957          } else {          } else {
1958            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1959          }          }
1960          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1961          !!!next-input-character;          !!!next-input-character;
1962    
1963          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1964    
1965          redo A;          redo A;
1966        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1967          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1968          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1969            $self->{current_token}->{first_start_tag}            !!!cp (112);
1970                = not defined $self->{last_emitted_start_tag_name};            $self->{last_stag_name} = $self->{ct}->{tag_name};
1971            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1972          } elsif ($self->{current_token}->{type} eq 'end tag') {            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1973            $self->{content_model_flag} = 'PCDATA'; # MUST            if ($self->{ct}->{attributes}) {
1974            if ($self->{current_token}->{attributes}) {              !!!cp (113);
1975              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1976              } else {
1977                ## NOTE: This state should never be reached.
1978                !!!cp (114);
1979            }            }
1980          } else {          } else {
1981            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1982          }          }
1983          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1984          ## reconsume          ## reconsume
1985    
1986          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1987    
1988          redo A;          redo A;
1989        } else {        } else {
1990          $self->{current_attribute}->{value} .= chr ($self->{next_input_character});          if ({
1991                 0x0022 => 1, # "
1992                 0x0027 => 1, # '
1993                 0x003D => 1, # =
1994                }->{$self->{nc}}) {
1995              !!!cp (115);
1996              !!!parse-error (type => 'bad attribute value');
1997            } else {
1998              !!!cp (116);
1999            }
2000            $self->{ca}->{value} .= chr ($self->{nc});
2001            $self->{read_until}->($self->{ca}->{value},
2002                                  q["'=& >],
2003                                  length $self->{ca}->{value});
2004    
2005          ## Stay in the state          ## Stay in the state
2006          !!!next-input-character;          !!!next-input-character;
2007          redo A;          redo A;
2008        }        }
2009      } elsif ($self->{state} eq 'entity in attribute value') {      } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {
2010        my $token = $self->_tokenize_attempt_to_consume_an_entity (1);        if ($is_space->{$self->{nc}}) {
2011            !!!cp (118);
2012            $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
2013            !!!next-input-character;
2014            redo A;
2015          } elsif ($self->{nc} == 0x003E) { # >
2016            if ($self->{ct}->{type} == START_TAG_TOKEN) {
2017              !!!cp (119);
2018              $self->{last_stag_name} = $self->{ct}->{tag_name};
2019            } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2020              $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
2021              if ($self->{ct}->{attributes}) {
2022                !!!cp (120);
2023                !!!parse-error (type => 'end tag attribute');
2024              } else {
2025                ## NOTE: This state should never be reached.
2026                !!!cp (121);
2027              }
2028            } else {
2029              die "$0: $self->{ct}->{type}: Unknown token type";
2030            }
2031            $self->{state} = DATA_STATE;
2032            !!!next-input-character;
2033    
2034            !!!emit ($self->{ct}); # start tag or end tag
2035    
2036        unless (defined $token) {          redo A;
2037          $self->{current_attribute}->{value} .= '&';        } elsif ($self->{nc} == 0x002F) { # /
2038            !!!cp (122);
2039            $self->{state} = SELF_CLOSING_START_TAG_STATE;
2040            !!!next-input-character;
2041            redo A;
2042          } elsif ($self->{nc} == -1) {
2043            !!!parse-error (type => 'unclosed tag');
2044            if ($self->{ct}->{type} == START_TAG_TOKEN) {
2045              !!!cp (122.3);
2046              $self->{last_stag_name} = $self->{ct}->{tag_name};
2047            } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2048              if ($self->{ct}->{attributes}) {
2049                !!!cp (122.1);
2050                !!!parse-error (type => 'end tag attribute');
2051              } else {
2052                ## NOTE: This state should never be reached.
2053                !!!cp (122.2);
2054              }
2055            } else {
2056              die "$0: $self->{ct}->{type}: Unknown token type";
2057            }
2058            $self->{state} = DATA_STATE;
2059            ## Reconsume.
2060            !!!emit ($self->{ct}); # start tag or end tag
2061            redo A;
2062        } else {        } else {
2063          $self->{current_attribute}->{value} .= $token->{data};          !!!cp ('124.1');
2064          ## ISSUE: spec says "append the returned character token to the current attribute's value"          !!!parse-error (type => 'no space between attributes');
2065            $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
2066            ## reconsume
2067            redo A;
2068        }        }
2069        } elsif ($self->{state} == SELF_CLOSING_START_TAG_STATE) {
2070          if ($self->{nc} == 0x003E) { # >
2071            if ($self->{ct}->{type} == END_TAG_TOKEN) {
2072              !!!cp ('124.2');
2073              !!!parse-error (type => 'nestc', token => $self->{ct});
2074              ## TODO: Different type than slash in start tag
2075              $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
2076              if ($self->{ct}->{attributes}) {
2077                !!!cp ('124.4');
2078                !!!parse-error (type => 'end tag attribute');
2079              } else {
2080                !!!cp ('124.5');
2081              }
2082              ## TODO: Test |<title></title/>|
2083            } else {
2084              !!!cp ('124.3');
2085              $self->{self_closing} = 1;
2086            }
2087    
2088        $self->{state} = $self->{last_attribute_value_state};          $self->{state} = DATA_STATE;
2089        # next-input-character is already done          !!!next-input-character;
       redo A;  
     } elsif ($self->{state} eq 'bogus comment') {  
       ## (only happen if PCDATA state)  
         
       my $token = {type => 'comment', data => ''};  
   
       BC: {  
         if ($self->{next_input_character} == 0x003E) { # >  
           $self->{state} = 'data';  
           !!!next-input-character;  
   
           !!!emit ($token);  
   
           redo A;  
         } elsif ($self->{next_input_character} == -1) {  
           $self->{state} = 'data';  
           ## reconsume  
2090    
2091            !!!emit ($token);          !!!emit ($self->{ct}); # start tag or end tag
2092    
2093            redo A;          redo A;
2094          } elsif ($self->{nc} == -1) {
2095            !!!parse-error (type => 'unclosed tag');
2096            if ($self->{ct}->{type} == START_TAG_TOKEN) {
2097              !!!cp (124.7);
2098              $self->{last_stag_name} = $self->{ct}->{tag_name};
2099            } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2100              if ($self->{ct}->{attributes}) {
2101                !!!cp (124.5);
2102                !!!parse-error (type => 'end tag attribute');
2103              } else {
2104                ## NOTE: This state should never be reached.
2105                !!!cp (124.6);
2106              }
2107          } else {          } else {
2108            $token->{data} .= chr ($self->{next_input_character});            die "$0: $self->{ct}->{type}: Unknown token type";
           !!!next-input-character;  
           redo BC;  
2109          }          }
2110        } # BC          $self->{state} = DATA_STATE;
2111      } elsif ($self->{state} eq 'markup declaration open') {          ## Reconsume.
2112            !!!emit ($self->{ct}); # start tag or end tag
2113            redo A;
2114          } else {
2115            !!!cp ('124.4');
2116            !!!parse-error (type => 'nestc');
2117            ## TODO: This error type is wrong.
2118            $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
2119            ## Reconsume.
2120            redo A;
2121          }
2122        } elsif ($self->{state} == BOGUS_COMMENT_STATE) {
2123        ## (only happen if PCDATA state)        ## (only happen if PCDATA state)
2124    
2125        my @next_char;        ## NOTE: Unlike spec's "bogus comment state", this implementation
2126        push @next_char, $self->{next_input_character};        ## consumes characters one-by-one basis.
2127                
2128        if ($self->{next_input_character} == 0x002D) { # -        if ($self->{nc} == 0x003E) { # >
2129            !!!cp (124);
2130            $self->{state} = DATA_STATE;
2131          !!!next-input-character;          !!!next-input-character;
2132          push @next_char, $self->{next_input_character};  
2133          if ($self->{next_input_character} == 0x002D) { # -          !!!emit ($self->{ct}); # comment
2134            $self->{current_token} = {type => 'comment', data => ''};          redo A;
2135            $self->{state} = 'comment start';        } elsif ($self->{nc} == -1) {
2136            !!!next-input-character;          !!!cp (125);
2137            redo A;          $self->{state} = DATA_STATE;
2138          }          ## reconsume
2139        } elsif ($self->{next_input_character} == 0x0044 or # D  
2140                 $self->{next_input_character} == 0x0064) { # d          !!!emit ($self->{ct}); # comment
2141            redo A;
2142          } else {
2143            !!!cp (126);
2144            $self->{ct}->{data} .= chr ($self->{nc}); # comment
2145            $self->{read_until}->($self->{ct}->{data},
2146                                  q[>],
2147                                  length $self->{ct}->{data});
2148    
2149            ## Stay in the state.
2150          !!!next-input-character;          !!!next-input-character;
2151          push @next_char, $self->{next_input_character};          redo A;
2152          if ($self->{next_input_character} == 0x004F or # O        }
2153              $self->{next_input_character} == 0x006F) { # o      } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {
2154            !!!next-input-character;        ## (only happen if PCDATA state)
2155            push @next_char, $self->{next_input_character};        
2156            if ($self->{next_input_character} == 0x0043 or # C        if ($self->{nc} == 0x002D) { # -
2157                $self->{next_input_character} == 0x0063) { # c          !!!cp (133);
2158              !!!next-input-character;          $self->{state} = MD_HYPHEN_STATE;
2159              push @next_char, $self->{next_input_character};          !!!next-input-character;
2160              if ($self->{next_input_character} == 0x0054 or # T          redo A;
2161                  $self->{next_input_character} == 0x0074) { # t        } elsif ($self->{nc} == 0x0044 or # D
2162                !!!next-input-character;                 $self->{nc} == 0x0064) { # d
2163                push @next_char, $self->{next_input_character};          ## ASCII case-insensitive.
2164                if ($self->{next_input_character} == 0x0059 or # Y          !!!cp (130);
2165                    $self->{next_input_character} == 0x0079) { # y          $self->{state} = MD_DOCTYPE_STATE;
2166                  !!!next-input-character;          $self->{s_kwd} = chr $self->{nc};
2167                  push @next_char, $self->{next_input_character};          !!!next-input-character;
2168                  if ($self->{next_input_character} == 0x0050 or # P          redo A;
2169                      $self->{next_input_character} == 0x0070) { # p        } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and
2170                    !!!next-input-character;                 $self->{open_elements}->[-1]->[1] & FOREIGN_EL and
2171                    push @next_char, $self->{next_input_character};                 $self->{nc} == 0x005B) { # [
2172                    if ($self->{next_input_character} == 0x0045 or # E          !!!cp (135.4);                
2173                        $self->{next_input_character} == 0x0065) { # e          $self->{state} = MD_CDATA_STATE;
2174                      ## ISSUE: What a stupid code this is!          $self->{s_kwd} = '[';
2175                      $self->{state} = 'DOCTYPE';          !!!next-input-character;
2176                      !!!next-input-character;          redo A;
2177                      redo A;        } else {
2178                    }          !!!cp (136);
                 }  
               }  
             }  
           }  
         }  
2179        }        }
2180    
2181        !!!parse-error (type => 'bogus comment');        !!!parse-error (type => 'bogus comment',
2182        $self->{next_input_character} = shift @next_char;                        line => $self->{line_prev},
2183        !!!back-next-input-character (@next_char);                        column => $self->{column_prev} - 1);
2184        $self->{state} = 'bogus comment';        ## Reconsume.
2185          $self->{state} = BOGUS_COMMENT_STATE;
2186          $self->{ct} = {type => COMMENT_TOKEN, data => '',
2187                                    line => $self->{line_prev},
2188                                    column => $self->{column_prev} - 1,
2189                                   };
2190        redo A;        redo A;
2191              } elsif ($self->{state} == MD_HYPHEN_STATE) {
2192        ## ISSUE: typos in spec: chacacters, is is a parse error        if ($self->{nc} == 0x002D) { # -
2193        ## ISSUE: spec is somewhat unclear on "is the first character that will be in the comment"; what is "that will be in the comment" is what the algorithm defines, isn't it?          !!!cp (127);
2194      } elsif ($self->{state} eq 'comment start') {          $self->{ct} = {type => COMMENT_TOKEN, data => '',
2195        if ($self->{next_input_character} == 0x002D) { # -                                    line => $self->{line_prev},
2196          $self->{state} = 'comment start dash';                                    column => $self->{column_prev} - 2,
2197                                     };
2198            $self->{state} = COMMENT_START_STATE;
2199          !!!next-input-character;          !!!next-input-character;
2200          redo A;          redo A;
2201        } elsif ($self->{next_input_character} == 0x003E) { # >        } else {
2202            !!!cp (128);
2203            !!!parse-error (type => 'bogus comment',
2204                            line => $self->{line_prev},
2205                            column => $self->{column_prev} - 2);
2206            $self->{state} = BOGUS_COMMENT_STATE;
2207            ## Reconsume.
2208            $self->{ct} = {type => COMMENT_TOKEN,
2209                                      data => '-',
2210                                      line => $self->{line_prev},
2211                                      column => $self->{column_prev} - 2,
2212                                     };
2213            redo A;
2214          }
2215        } elsif ($self->{state} == MD_DOCTYPE_STATE) {
2216          ## ASCII case-insensitive.
2217          if ($self->{nc} == [
2218                undef,
2219                0x004F, # O
2220                0x0043, # C
2221                0x0054, # T
2222                0x0059, # Y
2223                0x0050, # P
2224              ]->[length $self->{s_kwd}] or
2225              $self->{nc} == [
2226                undef,
2227                0x006F, # o
2228                0x0063, # c
2229                0x0074, # t
2230                0x0079, # y
2231                0x0070, # p
2232              ]->[length $self->{s_kwd}]) {
2233            !!!cp (131);
2234            ## Stay in the state.
2235            $self->{s_kwd} .= chr $self->{nc};
2236            !!!next-input-character;
2237            redo A;
2238          } elsif ((length $self->{s_kwd}) == 6 and
2239                   ($self->{nc} == 0x0045 or # E
2240                    $self->{nc} == 0x0065)) { # e
2241            !!!cp (129);
2242            $self->{state} = DOCTYPE_STATE;
2243            $self->{ct} = {type => DOCTYPE_TOKEN,
2244                                      quirks => 1,
2245                                      line => $self->{line_prev},
2246                                      column => $self->{column_prev} - 7,
2247                                     };
2248            !!!next-input-character;
2249            redo A;
2250          } else {
2251            !!!cp (132);        
2252            !!!parse-error (type => 'bogus comment',
2253                            line => $self->{line_prev},
2254                            column => $self->{column_prev} - 1 - length $self->{s_kwd});
2255            $self->{state} = BOGUS_COMMENT_STATE;
2256            ## Reconsume.
2257            $self->{ct} = {type => COMMENT_TOKEN,
2258                                      data => $self->{s_kwd},
2259                                      line => $self->{line_prev},
2260                                      column => $self->{column_prev} - 1 - length $self->{s_kwd},
2261                                     };
2262            redo A;
2263          }
2264        } elsif ($self->{state} == MD_CDATA_STATE) {
2265          if ($self->{nc} == {
2266                '[' => 0x0043, # C
2267                '[C' => 0x0044, # D
2268                '[CD' => 0x0041, # A
2269                '[CDA' => 0x0054, # T
2270                '[CDAT' => 0x0041, # A
2271              }->{$self->{s_kwd}}) {
2272            !!!cp (135.1);
2273            ## Stay in the state.
2274            $self->{s_kwd} .= chr $self->{nc};
2275            !!!next-input-character;
2276            redo A;
2277          } elsif ($self->{s_kwd} eq '[CDATA' and
2278                   $self->{nc} == 0x005B) { # [
2279            !!!cp (135.2);
2280            $self->{ct} = {type => CHARACTER_TOKEN,
2281                                      data => '',
2282                                      line => $self->{line_prev},
2283                                      column => $self->{column_prev} - 7};
2284            $self->{state} = CDATA_SECTION_STATE;
2285            !!!next-input-character;
2286            redo A;
2287          } else {
2288            !!!cp (135.3);
2289            !!!parse-error (type => 'bogus comment',
2290                            line => $self->{line_prev},
2291                            column => $self->{column_prev} - 1 - length $self->{s_kwd});
2292            $self->{state} = BOGUS_COMMENT_STATE;
2293            ## Reconsume.
2294            $self->{ct} = {type => COMMENT_TOKEN,
2295                                      data => $self->{s_kwd},
2296                                      line => $self->{line_prev},
2297                                      column => $self->{column_prev} - 1 - length $self->{s_kwd},
2298                                     };
2299            redo A;
2300          }
2301        } elsif ($self->{state} == COMMENT_START_STATE) {
2302          if ($self->{nc} == 0x002D) { # -
2303            !!!cp (137);
2304            $self->{state} = COMMENT_START_DASH_STATE;
2305            !!!next-input-character;
2306            redo A;
2307          } elsif ($self->{nc} == 0x003E) { # >
2308            !!!cp (138);
2309          !!!parse-error (type => 'bogus comment');          !!!parse-error (type => 'bogus comment');
2310          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2311          !!!next-input-character;          !!!next-input-character;
2312    
2313          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2314    
2315          redo A;          redo A;
2316        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2317            !!!cp (139);
2318          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2319          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2320          ## reconsume          ## reconsume
2321    
2322          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2323    
2324          redo A;          redo A;
2325        } else {        } else {
2326          $self->{current_token}->{data} # comment          !!!cp (140);
2327              .= chr ($self->{next_input_character});          $self->{ct}->{data} # comment
2328          $self->{state} = 'comment';              .= chr ($self->{nc});
2329            $self->{state} = COMMENT_STATE;
2330          !!!next-input-character;          !!!next-input-character;
2331          redo A;          redo A;
2332        }        }
2333      } elsif ($self->{state} eq 'comment start dash') {      } elsif ($self->{state} == COMMENT_START_DASH_STATE) {
2334        if ($self->{next_input_character} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2335          $self->{state} = 'comment end';          !!!cp (141);
2336            $self->{state} = COMMENT_END_STATE;
2337          !!!next-input-character;          !!!next-input-character;
2338          redo A;          redo A;
2339        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2340            !!!cp (142);
2341          !!!parse-error (type => 'bogus comment');          !!!parse-error (type => 'bogus comment');
2342          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2343          !!!next-input-character;          !!!next-input-character;
2344    
2345          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2346    
2347          redo A;          redo A;
2348        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2349            !!!cp (143);
2350          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2351          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2352          ## reconsume          ## reconsume
2353    
2354          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2355    
2356          redo A;          redo A;
2357        } else {        } else {
2358          $self->{current_token}->{data} # comment          !!!cp (144);
2359              .= '-' . chr ($self->{next_input_character});          $self->{ct}->{data} # comment
2360          $self->{state} = 'comment';              .= '-' . chr ($self->{nc});
2361            $self->{state} = COMMENT_STATE;
2362          !!!next-input-character;          !!!next-input-character;
2363          redo A;          redo A;
2364        }        }
2365      } elsif ($self->{state} eq 'comment') {      } elsif ($self->{state} == COMMENT_STATE) {
2366        if ($self->{next_input_character} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2367          $self->{state} = 'comment end dash';          !!!cp (145);
2368            $self->{state} = COMMENT_END_DASH_STATE;
2369          !!!next-input-character;          !!!next-input-character;
2370          redo A;          redo A;
2371        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2372            !!!cp (146);
2373          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2374          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2375          ## reconsume          ## reconsume
2376    
2377          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2378    
2379          redo A;          redo A;
2380        } else {        } else {
2381          $self->{current_token}->{data} .= chr ($self->{next_input_character}); # comment          !!!cp (147);
2382            $self->{ct}->{data} .= chr ($self->{nc}); # comment
2383            $self->{read_until}->($self->{ct}->{data},
2384                                  q[-],
2385                                  length $self->{ct}->{data});
2386    
2387          ## Stay in the state          ## Stay in the state
2388          !!!next-input-character;          !!!next-input-character;
2389          redo A;          redo A;
2390        }        }
2391      } elsif ($self->{state} eq 'comment end dash') {      } elsif ($self->{state} == COMMENT_END_DASH_STATE) {
2392        if ($self->{next_input_character} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2393          $self->{state} = 'comment end';          !!!cp (148);
2394            $self->{state} = COMMENT_END_STATE;
2395          !!!next-input-character;          !!!next-input-character;
2396          redo A;          redo A;
2397        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2398            !!!cp (149);
2399          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2400          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2401          ## reconsume          ## reconsume
2402    
2403          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2404    
2405          redo A;          redo A;
2406        } else {        } else {
2407          $self->{current_token}->{data} .= '-' . chr ($self->{next_input_character}); # comment          !!!cp (150);
2408          $self->{state} = 'comment';          $self->{ct}->{data} .= '-' . chr ($self->{nc}); # comment
2409            $self->{state} = COMMENT_STATE;
2410          !!!next-input-character;          !!!next-input-character;
2411          redo A;          redo A;
2412        }        }
2413      } elsif ($self->{state} eq 'comment end') {      } elsif ($self->{state} == COMMENT_END_STATE) {
2414        if ($self->{next_input_character} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2415          $self->{state} = 'data';          !!!cp (151);
2416            $self->{state} = DATA_STATE;
2417          !!!next-input-character;          !!!next-input-character;
2418    
2419          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2420    
2421          redo A;          redo A;
2422        } elsif ($self->{next_input_character} == 0x002D) { # -        } elsif ($self->{nc} == 0x002D) { # -
2423          !!!parse-error (type => 'dash in comment');          !!!cp (152);
2424          $self->{current_token}->{data} .= '-'; # comment          !!!parse-error (type => 'dash in comment',
2425                            line => $self->{line_prev},
2426                            column => $self->{column_prev});
2427            $self->{ct}->{data} .= '-'; # comment
2428          ## Stay in the state          ## Stay in the state
2429          !!!next-input-character;          !!!next-input-character;
2430          redo A;          redo A;
2431        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2432            !!!cp (153);
2433          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2434          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2435          ## reconsume          ## reconsume
2436    
2437          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2438    
2439          redo A;          redo A;
2440        } else {        } else {
2441          !!!parse-error (type => 'dash in comment');          !!!cp (154);
2442          $self->{current_token}->{data} .= '--' . chr ($self->{next_input_character}); # comment          !!!parse-error (type => 'dash in comment',
2443          $self->{state} = 'comment';                          line => $self->{line_prev},
2444                            column => $self->{column_prev});
2445            $self->{ct}->{data} .= '--' . chr ($self->{nc}); # comment
2446            $self->{state} = COMMENT_STATE;
2447          !!!next-input-character;          !!!next-input-character;
2448          redo A;          redo A;
2449        }        }
2450      } elsif ($self->{state} eq 'DOCTYPE') {      } elsif ($self->{state} == DOCTYPE_STATE) {
2451        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
2452            $self->{next_input_character} == 0x000A or # LF          !!!cp (155);
2453            $self->{next_input_character} == 0x000B or # VT          $self->{state} = BEFORE_DOCTYPE_NAME_STATE;
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
         $self->{state} = 'before DOCTYPE name';  
2454          !!!next-input-character;          !!!next-input-character;
2455          redo A;          redo A;
2456        } else {        } else {
2457            !!!cp (156);
2458          !!!parse-error (type => 'no space before DOCTYPE name');          !!!parse-error (type => 'no space before DOCTYPE name');
2459          $self->{state} = 'before DOCTYPE name';          $self->{state} = BEFORE_DOCTYPE_NAME_STATE;
2460          ## reconsume          ## reconsume
2461          redo A;          redo A;
2462        }        }
2463      } elsif ($self->{state} eq 'before DOCTYPE name') {      } elsif ($self->{state} == BEFORE_DOCTYPE_NAME_STATE) {
2464        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
2465            $self->{next_input_character} == 0x000A or # LF          !!!cp (157);
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
2466          ## Stay in the state          ## Stay in the state
2467          !!!next-input-character;          !!!next-input-character;
2468          redo A;          redo A;
2469        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2470            !!!cp (158);
2471          !!!parse-error (type => 'no DOCTYPE name');          !!!parse-error (type => 'no DOCTYPE name');
2472          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2473          !!!next-input-character;          !!!next-input-character;
2474    
2475          !!!emit ({type => 'DOCTYPE'}); # incorrect          !!!emit ($self->{ct}); # DOCTYPE (quirks)
2476    
2477          redo A;          redo A;
2478        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2479            !!!cp (159);
2480          !!!parse-error (type => 'no DOCTYPE name');          !!!parse-error (type => 'no DOCTYPE name');
2481          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2482          ## reconsume          ## reconsume
2483    
2484          !!!emit ({type => 'DOCTYPE'}); # incorrect          !!!emit ($self->{ct}); # DOCTYPE (quirks)
2485    
2486          redo A;          redo A;
2487        } else {        } else {
2488          $self->{current_token}          !!!cp (160);
2489              = {type => 'DOCTYPE',          $self->{ct}->{name} = chr $self->{nc};
2490                 name => chr ($self->{next_input_character}),          delete $self->{ct}->{quirks};
2491                 correct => 1};          $self->{state} = DOCTYPE_NAME_STATE;
 ## ISSUE: "Set the token's name name to the" in the spec  
         $self->{state} = 'DOCTYPE name';  
2492          !!!next-input-character;          !!!next-input-character;
2493          redo A;          redo A;
2494        }        }
2495      } elsif ($self->{state} eq 'DOCTYPE name') {      } elsif ($self->{state} == DOCTYPE_NAME_STATE) {
2496  ## ISSUE: Redundant "First," in the spec.  ## ISSUE: Redundant "First," in the spec.
2497        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
2498            $self->{next_input_character} == 0x000A or # LF          !!!cp (161);
2499            $self->{next_input_character} == 0x000B or # VT          $self->{state} = AFTER_DOCTYPE_NAME_STATE;
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
         $self->{state} = 'after DOCTYPE name';  
2500          !!!next-input-character;          !!!next-input-character;
2501          redo A;          redo A;
2502        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2503          $self->{state} = 'data';          !!!cp (162);
2504            $self->{state} = DATA_STATE;
2505          !!!next-input-character;          !!!next-input-character;
2506    
2507          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2508    
2509          redo A;          redo A;
2510        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2511            !!!cp (163);
2512          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2513          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2514          ## reconsume          ## reconsume
2515    
2516          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2517          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2518    
2519          redo A;          redo A;
2520        } else {        } else {
2521          $self->{current_token}->{name}          !!!cp (164);
2522            .= chr ($self->{next_input_character}); # DOCTYPE          $self->{ct}->{name}
2523              .= chr ($self->{nc}); # DOCTYPE
2524          ## Stay in the state          ## Stay in the state
2525          !!!next-input-character;          !!!next-input-character;
2526          redo A;          redo A;
2527        }        }
2528      } elsif ($self->{state} eq 'after DOCTYPE name') {      } elsif ($self->{state} == AFTER_DOCTYPE_NAME_STATE) {
2529        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
2530            $self->{next_input_character} == 0x000A or # LF          !!!cp (165);
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
2531          ## Stay in the state          ## Stay in the state
2532          !!!next-input-character;          !!!next-input-character;
2533          redo A;          redo A;
2534        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2535          $self->{state} = 'data';          !!!cp (166);
2536            $self->{state} = DATA_STATE;
2537          !!!next-input-character;          !!!next-input-character;
2538    
2539          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2540    
2541          redo A;          redo A;
2542        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2543            !!!cp (167);
2544          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2545          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2546          ## reconsume          ## reconsume
2547    
2548          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2549          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2550    
2551          redo A;          redo A;
2552        } elsif ($self->{next_input_character} == 0x0050 or # P        } elsif ($self->{nc} == 0x0050 or # P
2553                 $self->{next_input_character} == 0x0070) { # p                 $self->{nc} == 0x0070) { # p
2554            $self->{state} = PUBLIC_STATE;
2555            $self->{s_kwd} = chr $self->{nc};
2556          !!!next-input-character;          !!!next-input-character;
2557          if ($self->{next_input_character} == 0x0055 or # U          redo A;
2558              $self->{next_input_character} == 0x0075) { # u        } elsif ($self->{nc} == 0x0053 or # S
2559            !!!next-input-character;                 $self->{nc} == 0x0073) { # s
2560            if ($self->{next_input_character} == 0x0042 or # B          $self->{state} = SYSTEM_STATE;
2561                $self->{next_input_character} == 0x0062) { # b          $self->{s_kwd} = chr $self->{nc};
             !!!next-input-character;  
             if ($self->{next_input_character} == 0x004C or # L  
                 $self->{next_input_character} == 0x006C) { # l  
               !!!next-input-character;  
               if ($self->{next_input_character} == 0x0049 or # I  
                   $self->{next_input_character} == 0x0069) { # i  
                 !!!next-input-character;  
                 if ($self->{next_input_character} == 0x0043 or # C  
                     $self->{next_input_character} == 0x0063) { # c  
                   $self->{state} = 'before DOCTYPE public identifier';  
                   !!!next-input-character;  
                   redo A;  
                 }  
               }  
             }  
           }  
         }  
   
         #  
       } elsif ($self->{next_input_character} == 0x0053 or # S  
                $self->{next_input_character} == 0x0073) { # s  
2562          !!!next-input-character;          !!!next-input-character;
2563          if ($self->{next_input_character} == 0x0059 or # Y          redo A;
             $self->{next_input_character} == 0x0079) { # y  
           !!!next-input-character;  
           if ($self->{next_input_character} == 0x0053 or # S  
               $self->{next_input_character} == 0x0073) { # s  
             !!!next-input-character;  
             if ($self->{next_input_character} == 0x0054 or # T  
                 $self->{next_input_character} == 0x0074) { # t  
               !!!next-input-character;  
               if ($self->{next_input_character} == 0x0045 or # E  
                   $self->{next_input_character} == 0x0065) { # e  
                 !!!next-input-character;  
                 if ($self->{next_input_character} == 0x004D or # M  
                     $self->{next_input_character} == 0x006D) { # m  
                   $self->{state} = 'before DOCTYPE system identifier';  
                   !!!next-input-character;  
                   redo A;  
                 }  
               }  
             }  
           }  
         }  
   
         #  
2564        } else {        } else {
2565            !!!cp (180);
2566            !!!parse-error (type => 'string after DOCTYPE name');
2567            $self->{ct}->{quirks} = 1;
2568    
2569            $self->{state} = BOGUS_DOCTYPE_STATE;
2570          !!!next-input-character;          !!!next-input-character;
2571          #          redo A;
2572        }        }
2573        } elsif ($self->{state} == PUBLIC_STATE) {
2574        !!!parse-error (type => 'string after DOCTYPE name');        ## ASCII case-insensitive
2575        $self->{state} = 'bogus DOCTYPE';        if ($self->{nc} == [
2576        # next-input-character is already done              undef,
2577        redo A;              0x0055, # U
2578      } elsif ($self->{state} eq 'before DOCTYPE public identifier') {              0x0042, # B
2579        if ({              0x004C, # L
2580              0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,              0x0049, # I
2581              #0x000D => 1, # HT, LF, VT, FF, SP, CR            ]->[length $self->{s_kwd}] or
2582            }->{$self->{next_input_character}}) {            $self->{nc} == [
2583                undef,
2584                0x0075, # u
2585                0x0062, # b
2586                0x006C, # l
2587                0x0069, # i
2588              ]->[length $self->{s_kwd}]) {
2589            !!!cp (175);
2590            ## Stay in the state.
2591            $self->{s_kwd} .= chr $self->{nc};
2592            !!!next-input-character;
2593            redo A;
2594          } elsif ((length $self->{s_kwd}) == 5 and
2595                   ($self->{nc} == 0x0043 or # C
2596                    $self->{nc} == 0x0063)) { # c
2597            !!!cp (168);
2598            $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2599            !!!next-input-character;
2600            redo A;
2601          } else {
2602            !!!cp (169);
2603            !!!parse-error (type => 'string after DOCTYPE name',
2604                            line => $self->{line_prev},
2605                            column => $self->{column_prev} + 1 - length $self->{s_kwd});
2606            $self->{ct}->{quirks} = 1;
2607    
2608            $self->{state} = BOGUS_DOCTYPE_STATE;
2609            ## Reconsume.
2610            redo A;
2611          }
2612        } elsif ($self->{state} == SYSTEM_STATE) {
2613          ## ASCII case-insensitive
2614          if ($self->{nc} == [
2615                undef,
2616                0x0059, # Y
2617                0x0053, # S
2618                0x0054, # T
2619                0x0045, # E
2620              ]->[length $self->{s_kwd}] or
2621              $self->{nc} == [
2622                undef,
2623                0x0079, # y
2624                0x0073, # s
2625                0x0074, # t
2626                0x0065, # e
2627              ]->[length $self->{s_kwd}]) {
2628            !!!cp (170);
2629            ## Stay in the state.
2630            $self->{s_kwd} .= chr $self->{nc};
2631            !!!next-input-character;
2632            redo A;
2633          } elsif ((length $self->{s_kwd}) == 5 and
2634                   ($self->{nc} == 0x004D or # M
2635                    $self->{nc} == 0x006D)) { # m
2636            !!!cp (171);
2637            $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2638            !!!next-input-character;
2639            redo A;
2640          } else {
2641            !!!cp (172);
2642            !!!parse-error (type => 'string after DOCTYPE name',
2643                            line => $self->{line_prev},
2644                            column => $self->{column_prev} + 1 - length $self->{s_kwd});
2645            $self->{ct}->{quirks} = 1;
2646    
2647            $self->{state} = BOGUS_DOCTYPE_STATE;
2648            ## Reconsume.
2649            redo A;
2650          }
2651        } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2652          if ($is_space->{$self->{nc}}) {
2653            !!!cp (181);
2654          ## Stay in the state          ## Stay in the state
2655          !!!next-input-character;          !!!next-input-character;
2656          redo A;          redo A;
2657        } elsif ($self->{next_input_character} eq 0x0022) { # "        } elsif ($self->{nc} eq 0x0022) { # "
2658          $self->{current_token}->{public_identifier} = ''; # DOCTYPE          !!!cp (182);
2659          $self->{state} = 'DOCTYPE public identifier (double-quoted)';          $self->{ct}->{pubid} = ''; # DOCTYPE
2660            $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;
2661          !!!next-input-character;          !!!next-input-character;
2662          redo A;          redo A;
2663        } elsif ($self->{next_input_character} eq 0x0027) { # '        } elsif ($self->{nc} eq 0x0027) { # '
2664          $self->{current_token}->{public_identifier} = ''; # DOCTYPE          !!!cp (183);
2665          $self->{state} = 'DOCTYPE public identifier (single-quoted)';          $self->{ct}->{pubid} = ''; # DOCTYPE
2666            $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;
2667          !!!next-input-character;          !!!next-input-character;
2668          redo A;          redo A;
2669        } elsif ($self->{next_input_character} eq 0x003E) { # >        } elsif ($self->{nc} eq 0x003E) { # >
2670            !!!cp (184);
2671          !!!parse-error (type => 'no PUBLIC literal');          !!!parse-error (type => 'no PUBLIC literal');
2672    
2673          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2674          !!!next-input-character;          !!!next-input-character;
2675    
2676          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2677          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2678    
2679          redo A;          redo A;
2680        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2681            !!!cp (185);
2682          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2683    
2684          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2685          ## reconsume          ## reconsume
2686    
2687          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2688          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2689    
2690          redo A;          redo A;
2691        } else {        } else {
2692            !!!cp (186);
2693          !!!parse-error (type => 'string after PUBLIC');          !!!parse-error (type => 'string after PUBLIC');
2694          $self->{state} = 'bogus DOCTYPE';          $self->{ct}->{quirks} = 1;
2695    
2696            $self->{state} = BOGUS_DOCTYPE_STATE;
2697          !!!next-input-character;          !!!next-input-character;
2698          redo A;          redo A;
2699        }        }
2700      } elsif ($self->{state} eq 'DOCTYPE public identifier (double-quoted)') {      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2701        if ($self->{next_input_character} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
2702          $self->{state} = 'after DOCTYPE public identifier';          !!!cp (187);
2703            $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2704          !!!next-input-character;          !!!next-input-character;
2705          redo A;          redo A;
2706        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == 0x003E) { # >
2707            !!!cp (188);
2708          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2709    
2710          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2711            !!!next-input-character;
2712    
2713            $self->{ct}->{quirks} = 1;
2714            !!!emit ($self->{ct}); # DOCTYPE
2715    
2716            redo A;
2717          } elsif ($self->{nc} == -1) {
2718            !!!cp (189);
2719            !!!parse-error (type => 'unclosed PUBLIC literal');
2720    
2721            $self->{state} = DATA_STATE;
2722          ## reconsume          ## reconsume
2723    
2724          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2725          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2726    
2727          redo A;          redo A;
2728        } else {        } else {
2729          $self->{current_token}->{public_identifier} # DOCTYPE          !!!cp (190);
2730              .= chr $self->{next_input_character};          $self->{ct}->{pubid} # DOCTYPE
2731                .= chr $self->{nc};
2732            $self->{read_until}->($self->{ct}->{pubid}, q[">],
2733                                  length $self->{ct}->{pubid});
2734    
2735          ## Stay in the state          ## Stay in the state
2736          !!!next-input-character;          !!!next-input-character;
2737          redo A;          redo A;
2738        }        }
2739      } elsif ($self->{state} eq 'DOCTYPE public identifier (single-quoted)') {      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {
2740        if ($self->{next_input_character} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
2741          $self->{state} = 'after DOCTYPE public identifier';          !!!cp (191);
2742            $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2743            !!!next-input-character;
2744            redo A;
2745          } elsif ($self->{nc} == 0x003E) { # >
2746            !!!cp (192);
2747            !!!parse-error (type => 'unclosed PUBLIC literal');
2748    
2749            $self->{state} = DATA_STATE;
2750          !!!next-input-character;          !!!next-input-character;
2751    
2752            $self->{ct}->{quirks} = 1;
2753            !!!emit ($self->{ct}); # DOCTYPE
2754    
2755          redo A;          redo A;
2756        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2757            !!!cp (193);
2758          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2759    
2760          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2761          ## reconsume          ## reconsume
2762    
2763          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2764          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2765    
2766          redo A;          redo A;
2767        } else {        } else {
2768          $self->{current_token}->{public_identifier} # DOCTYPE          !!!cp (194);
2769              .= chr $self->{next_input_character};          $self->{ct}->{pubid} # DOCTYPE
2770                .= chr $self->{nc};
2771            $self->{read_until}->($self->{ct}->{pubid}, q['>],
2772                                  length $self->{ct}->{pubid});
2773    
2774          ## Stay in the state          ## Stay in the state
2775          !!!next-input-character;          !!!next-input-character;
2776          redo A;          redo A;
2777        }        }
2778      } elsif ($self->{state} eq 'after DOCTYPE public identifier') {      } elsif ($self->{state} == AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2779        if ({        if ($is_space->{$self->{nc}}) {
2780              0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,          !!!cp (195);
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_input_character}}) {  
2781          ## Stay in the state          ## Stay in the state
2782          !!!next-input-character;          !!!next-input-character;
2783          redo A;          redo A;
2784        } elsif ($self->{next_input_character} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
2785          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          !!!cp (196);
2786          $self->{state} = 'DOCTYPE system identifier (double-quoted)';          $self->{ct}->{sysid} = ''; # DOCTYPE
2787            $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2788          !!!next-input-character;          !!!next-input-character;
2789          redo A;          redo A;
2790        } elsif ($self->{next_input_character} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
2791          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          !!!cp (197);
2792          $self->{state} = 'DOCTYPE system identifier (single-quoted)';          $self->{ct}->{sysid} = ''; # DOCTYPE
2793            $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2794          !!!next-input-character;          !!!next-input-character;
2795          redo A;          redo A;
2796        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2797          $self->{state} = 'data';          !!!cp (198);
2798            $self->{state} = DATA_STATE;
2799          !!!next-input-character;          !!!next-input-character;
2800    
2801          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2802    
2803          redo A;          redo A;
2804        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2805            !!!cp (199);
2806          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2807    
2808          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2809          ## reconsume          ## reconsume
2810    
2811          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2812          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2813    
2814          redo A;          redo A;
2815        } else {        } else {
2816            !!!cp (200);
2817          !!!parse-error (type => 'string after PUBLIC literal');          !!!parse-error (type => 'string after PUBLIC literal');
2818          $self->{state} = 'bogus DOCTYPE';          $self->{ct}->{quirks} = 1;
2819    
2820            $self->{state} = BOGUS_DOCTYPE_STATE;
2821          !!!next-input-character;          !!!next-input-character;
2822          redo A;          redo A;
2823        }        }
2824      } elsif ($self->{state} eq 'before DOCTYPE system identifier') {      } elsif ($self->{state} == BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2825        if ({        if ($is_space->{$self->{nc}}) {
2826              0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,          !!!cp (201);
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_input_character}}) {  
2827          ## Stay in the state          ## Stay in the state
2828          !!!next-input-character;          !!!next-input-character;
2829          redo A;          redo A;
2830        } elsif ($self->{next_input_character} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
2831          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          !!!cp (202);
2832          $self->{state} = 'DOCTYPE system identifier (double-quoted)';          $self->{ct}->{sysid} = ''; # DOCTYPE
2833            $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2834          !!!next-input-character;          !!!next-input-character;
2835          redo A;          redo A;
2836        } elsif ($self->{next_input_character} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
2837          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          !!!cp (203);
2838          $self->{state} = 'DOCTYPE system identifier (single-quoted)';          $self->{ct}->{sysid} = ''; # DOCTYPE
2839            $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2840          !!!next-input-character;          !!!next-input-character;
2841          redo A;          redo A;
2842        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2843            !!!cp (204);
2844          !!!parse-error (type => 'no SYSTEM literal');          !!!parse-error (type => 'no SYSTEM literal');
2845          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2846          !!!next-input-character;          !!!next-input-character;
2847    
2848          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2849          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2850    
2851          redo A;          redo A;
2852        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2853            !!!cp (205);
2854          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2855    
2856          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2857          ## reconsume          ## reconsume
2858    
2859          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2860          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2861    
2862          redo A;          redo A;
2863        } else {        } else {
2864            !!!cp (206);
2865          !!!parse-error (type => 'string after SYSTEM');          !!!parse-error (type => 'string after SYSTEM');
2866          $self->{state} = 'bogus DOCTYPE';          $self->{ct}->{quirks} = 1;
2867    
2868            $self->{state} = BOGUS_DOCTYPE_STATE;
2869          !!!next-input-character;          !!!next-input-character;
2870          redo A;          redo A;
2871        }        }
2872      } elsif ($self->{state} eq 'DOCTYPE system identifier (double-quoted)') {      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2873        if ($self->{next_input_character} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
2874          $self->{state} = 'after DOCTYPE system identifier';          !!!cp (207);
2875            $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2876          !!!next-input-character;          !!!next-input-character;
2877          redo A;          redo A;
2878        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == 0x003E) { # >
2879            !!!cp (208);
2880            !!!parse-error (type => 'unclosed SYSTEM literal');
2881    
2882            $self->{state} = DATA_STATE;
2883            !!!next-input-character;
2884    
2885            $self->{ct}->{quirks} = 1;
2886            !!!emit ($self->{ct}); # DOCTYPE
2887    
2888            redo A;
2889          } elsif ($self->{nc} == -1) {
2890            !!!cp (209);
2891          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2892    
2893          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2894          ## reconsume          ## reconsume
2895    
2896          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2897          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2898    
2899          redo A;          redo A;
2900        } else {        } else {
2901          $self->{current_token}->{system_identifier} # DOCTYPE          !!!cp (210);
2902              .= chr $self->{next_input_character};          $self->{ct}->{sysid} # DOCTYPE
2903                .= chr $self->{nc};
2904            $self->{read_until}->($self->{ct}->{sysid}, q[">],
2905                                  length $self->{ct}->{sysid});
2906    
2907          ## Stay in the state          ## Stay in the state
2908          !!!next-input-character;          !!!next-input-character;
2909          redo A;          redo A;
2910        }        }
2911      } elsif ($self->{state} eq 'DOCTYPE system identifier (single-quoted)') {      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {
2912        if ($self->{next_input_character} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
2913          $self->{state} = 'after DOCTYPE system identifier';          !!!cp (211);
2914            $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2915          !!!next-input-character;          !!!next-input-character;
2916          redo A;          redo A;
2917        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == 0x003E) { # >
2918            !!!cp (212);
2919          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2920    
2921          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2922            !!!next-input-character;
2923    
2924            $self->{ct}->{quirks} = 1;
2925            !!!emit ($self->{ct}); # DOCTYPE
2926    
2927            redo A;
2928          } elsif ($self->{nc} == -1) {
2929            !!!cp (213);
2930            !!!parse-error (type => 'unclosed SYSTEM literal');
2931    
2932            $self->{state} = DATA_STATE;
2933          ## reconsume          ## reconsume
2934    
2935          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2936          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2937    
2938          redo A;          redo A;
2939        } else {        } else {
2940          $self->{current_token}->{system_identifier} # DOCTYPE          !!!cp (214);
2941              .= chr $self->{next_input_character};          $self->{ct}->{sysid} # DOCTYPE
2942                .= chr $self->{nc};
2943            $self->{read_until}->($self->{ct}->{sysid}, q['>],
2944                                  length $self->{ct}->{sysid});
2945    
2946          ## Stay in the state          ## Stay in the state
2947          !!!next-input-character;          !!!next-input-character;
2948          redo A;          redo A;
2949        }        }
2950      } elsif ($self->{state} eq 'after DOCTYPE system identifier') {      } elsif ($self->{state} == AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2951        if ({        if ($is_space->{$self->{nc}}) {
2952              0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,          !!!cp (215);
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_input_character}}) {  
2953          ## Stay in the state          ## Stay in the state
2954          !!!next-input-character;          !!!next-input-character;
2955          redo A;          redo A;
2956        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2957          $self->{state} = 'data';          !!!cp (216);
2958            $self->{state} = DATA_STATE;
2959          !!!next-input-character;          !!!next-input-character;
2960    
2961          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2962    
2963          redo A;          redo A;
2964        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2965            !!!cp (217);
2966          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2967            $self->{state} = DATA_STATE;
         $self->{state} = 'data';  
2968          ## reconsume          ## reconsume
2969    
2970          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2971          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2972    
2973          redo A;          redo A;
2974        } else {        } else {
2975            !!!cp (218);
2976          !!!parse-error (type => 'string after SYSTEM literal');          !!!parse-error (type => 'string after SYSTEM literal');
2977          $self->{state} = 'bogus DOCTYPE';          #$self->{ct}->{quirks} = 1;
2978    
2979            $self->{state} = BOGUS_DOCTYPE_STATE;
2980          !!!next-input-character;          !!!next-input-character;
2981          redo A;          redo A;
2982        }        }
2983      } elsif ($self->{state} eq 'bogus DOCTYPE') {      } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {
2984        if ($self->{next_input_character} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2985          $self->{state} = 'data';          !!!cp (219);
2986            $self->{state} = DATA_STATE;
2987          !!!next-input-character;          !!!next-input-character;
2988    
2989          delete $self->{current_token}->{correct};          !!!emit ($self->{ct}); # DOCTYPE
         !!!emit ($self->{current_token}); # DOCTYPE  
2990    
2991          redo A;          redo A;
2992        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2993          !!!parse-error (type => 'unclosed DOCTYPE');          !!!cp (220);
2994          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2995          ## reconsume          ## reconsume
2996    
2997          delete $self->{current_token}->{correct};          !!!emit ($self->{ct}); # DOCTYPE
         !!!emit ($self->{current_token}); # DOCTYPE  
2998    
2999          redo A;          redo A;
3000        } else {        } else {
3001            !!!cp (221);
3002            my $s = '';
3003            $self->{read_until}->($s, q[>], 0);
3004    
3005          ## Stay in the state          ## Stay in the state
3006          !!!next-input-character;          !!!next-input-character;
3007          redo A;          redo A;
3008        }        }
3009      } else {      } elsif ($self->{state} == CDATA_SECTION_STATE) {
3010        die "$0: $self->{state}: Unknown state";        ## NOTE: "CDATA section state" in the state is jointly implemented
3011      }        ## by three states, |CDATA_SECTION_STATE|, |CDATA_SECTION_MSE1_STATE|,
3012    } # A          ## and |CDATA_SECTION_MSE2_STATE|.
3013          
3014    die "$0: _get_next_token: unexpected case";        if ($self->{nc} == 0x005D) { # ]
3015  } # _get_next_token          !!!cp (221.1);
3016            $self->{state} = CDATA_SECTION_MSE1_STATE;
3017            !!!next-input-character;
3018            redo A;
3019          } elsif ($self->{nc} == -1) {
3020            $self->{state} = DATA_STATE;
3021            !!!next-input-character;
3022            if (length $self->{ct}->{data}) { # character
3023              !!!cp (221.2);
3024              !!!emit ($self->{ct}); # character
3025            } else {
3026              !!!cp (221.3);
3027              ## No token to emit. $self->{ct} is discarded.
3028            }        
3029            redo A;
3030          } else {
3031            !!!cp (221.4);
3032            $self->{ct}->{data} .= chr $self->{nc};
3033            $self->{read_until}->($self->{ct}->{data},
3034                                  q<]>,
3035                                  length $self->{ct}->{data});
3036    
3037  sub _tokenize_attempt_to_consume_an_entity ($$) {          ## Stay in the state.
3038    my ($self, $in_attr) = @_;          !!!next-input-character;
3039            redo A;
3040          }
3041    
3042    if ({        ## ISSUE: "text tokens" in spec.
3043         0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, # HT, LF, VT, FF,      } elsif ($self->{state} == CDATA_SECTION_MSE1_STATE) {
3044         0x0020 => 1, 0x003C => 1, 0x0026 => 1, -1 => 1, # SP, <, & # 0x000D # CR        if ($self->{nc} == 0x005D) { # ]
3045        }->{$self->{next_input_character}}) {          !!!cp (221.5);
3046      ## Don't consume          $self->{state} = CDATA_SECTION_MSE2_STATE;
3047      ## No error          !!!next-input-character;
3048      return undef;          redo A;
3049    } elsif ($self->{next_input_character} == 0x0023) { # #        } else {
3050      !!!next-input-character;          !!!cp (221.6);
3051      if ($self->{next_input_character} == 0x0078 or # x          $self->{ct}->{data} .= ']';
3052          $self->{next_input_character} == 0x0058) { # X          $self->{state} = CDATA_SECTION_STATE;
3053        my $code;          ## Reconsume.
3054        X: {          redo A;
3055          my $x_char = $self->{next_input_character};        }
3056          !!!next-input-character;      } elsif ($self->{state} == CDATA_SECTION_MSE2_STATE) {
3057          if (0x0030 <= $self->{next_input_character} and        if ($self->{nc} == 0x003E) { # >
3058              $self->{next_input_character} <= 0x0039) { # 0..9          $self->{state} = DATA_STATE;
3059            $code ||= 0;          !!!next-input-character;
3060            $code *= 0x10;          if (length $self->{ct}->{data}) { # character
3061            $code += $self->{next_input_character} - 0x0030;            !!!cp (221.7);
3062            redo X;            !!!emit ($self->{ct}); # character
         } elsif (0x0061 <= $self->{next_input_character} and  
                  $self->{next_input_character} <= 0x0066) { # a..f  
           $code ||= 0;  
           $code *= 0x10;  
           $code += $self->{next_input_character} - 0x0060 + 9;  
           redo X;  
         } elsif (0x0041 <= $self->{next_input_character} and  
                  $self->{next_input_character} <= 0x0046) { # A..F  
           $code ||= 0;  
           $code *= 0x10;  
           $code += $self->{next_input_character} - 0x0040 + 9;  
           redo X;  
         } elsif (not defined $code) { # no hexadecimal digit  
           !!!parse-error (type => 'bare hcro');  
           $self->{next_input_character} = 0x0023; # #  
           !!!back-next-input-character ($x_char);  
           return undef;  
         } elsif ($self->{next_input_character} == 0x003B) { # ;  
           !!!next-input-character;  
3063          } else {          } else {
3064            !!!parse-error (type => 'no refc');            !!!cp (221.8);
3065              ## No token to emit. $self->{ct} is discarded.
3066          }          }
3067            redo A;
3068          if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {        } elsif ($self->{nc} == 0x005D) { # ]
3069            !!!parse-error (type => sprintf 'invalid character reference:U+%04X', $code);          !!!cp (221.9); # character
3070            $code = 0xFFFD;          $self->{ct}->{data} .= ']'; ## Add first "]" of "]]]".
3071          } elsif ($code > 0x10FFFF) {          ## Stay in the state.
3072            !!!parse-error (type => sprintf 'invalid character reference:U-%08X', $code);          !!!next-input-character;
3073            $code = 0xFFFD;          redo A;
3074          } elsif ($code == 0x000D) {        } else {
3075            !!!parse-error (type => 'CR character reference');          !!!cp (221.11);
3076            $code = 0x000A;          $self->{ct}->{data} .= ']]'; # character
3077          } elsif (0x80 <= $code and $code <= 0x9F) {          $self->{state} = CDATA_SECTION_STATE;
3078            !!!parse-error (type => sprintf 'C1 character reference:U+%04X', $code);          ## Reconsume.
3079            $code = $c1_entity_char->{$code};          redo A;
3080          }        }
3081        } elsif ($self->{state} == ENTITY_STATE) {
3082          return {type => 'character', data => chr $code};        if ($is_space->{$self->{nc}} or
3083        } # X            {
3084      } elsif (0x0030 <= $self->{next_input_character} and              0x003C => 1, 0x0026 => 1, -1 => 1, # <, &
3085               $self->{next_input_character} <= 0x0039) { # 0..9              $self->{entity_add} => 1,
3086        my $code = $self->{next_input_character} - 0x0030;            }->{$self->{nc}}) {
3087        !!!next-input-character;          !!!cp (1001);
3088                  ## Don't consume
3089        while (0x0030 <= $self->{next_input_character} and          ## No error
3090                  $self->{next_input_character} <= 0x0039) { # 0..9          ## Return nothing.
3091          $code *= 10;          #
3092          $code += $self->{next_input_character} - 0x0030;        } elsif ($self->{nc} == 0x0023) { # #
3093                    !!!cp (999);
3094            $self->{state} = ENTITY_HASH_STATE;
3095            $self->{s_kwd} = '#';
3096            !!!next-input-character;
3097            redo A;
3098          } elsif ((0x0041 <= $self->{nc} and
3099                    $self->{nc} <= 0x005A) or # A..Z
3100                   (0x0061 <= $self->{nc} and
3101                    $self->{nc} <= 0x007A)) { # a..z
3102            !!!cp (998);
3103            require Whatpm::_NamedEntityList;
3104            $self->{state} = ENTITY_NAME_STATE;
3105            $self->{s_kwd} = chr $self->{nc};
3106            $self->{entity__value} = $self->{s_kwd};
3107            $self->{entity__match} = 0;
3108          !!!next-input-character;          !!!next-input-character;
3109            redo A;
3110          } else {
3111            !!!cp (1027);
3112            !!!parse-error (type => 'bare ero');
3113            ## Return nothing.
3114            #
3115        }        }
3116    
3117        if ($self->{next_input_character} == 0x003B) { # ;        ## NOTE: No character is consumed by the "consume a character
3118          ## reference" algorithm.  In other word, there is an "&" character
3119          ## that does not introduce a character reference, which would be
3120          ## appended to the parent element or the attribute value in later
3121          ## process of the tokenizer.
3122    
3123          if ($self->{prev_state} == DATA_STATE) {
3124            !!!cp (997);
3125            $self->{state} = $self->{prev_state};
3126            ## Reconsume.
3127            !!!emit ({type => CHARACTER_TOKEN, data => '&',
3128                      line => $self->{line_prev},
3129                      column => $self->{column_prev},
3130                     });
3131            redo A;
3132          } else {
3133            !!!cp (996);
3134            $self->{ca}->{value} .= '&';
3135            $self->{state} = $self->{prev_state};
3136            ## Reconsume.
3137            redo A;
3138          }
3139        } elsif ($self->{state} == ENTITY_HASH_STATE) {
3140          if ($self->{nc} == 0x0078 or # x
3141              $self->{nc} == 0x0058) { # X
3142            !!!cp (995);
3143            $self->{state} = HEXREF_X_STATE;
3144            $self->{s_kwd} .= chr $self->{nc};
3145            !!!next-input-character;
3146            redo A;
3147          } elsif (0x0030 <= $self->{nc} and
3148                   $self->{nc} <= 0x0039) { # 0..9
3149            !!!cp (994);
3150            $self->{state} = NCR_NUM_STATE;
3151            $self->{s_kwd} = $self->{nc} - 0x0030;
3152          !!!next-input-character;          !!!next-input-character;
3153            redo A;
3154        } else {        } else {
3155            !!!parse-error (type => 'bare nero',
3156                            line => $self->{line_prev},
3157                            column => $self->{column_prev} - 1);
3158    
3159            ## NOTE: According to the spec algorithm, nothing is returned,
3160            ## and then "&#" is appended to the parent element or the attribute
3161            ## value in the later processing.
3162    
3163            if ($self->{prev_state} == DATA_STATE) {
3164              !!!cp (1019);
3165              $self->{state} = $self->{prev_state};
3166              ## Reconsume.
3167              !!!emit ({type => CHARACTER_TOKEN,
3168                        data => '&#',
3169                        line => $self->{line_prev},
3170                        column => $self->{column_prev} - 1,
3171                       });
3172              redo A;
3173            } else {
3174              !!!cp (993);
3175              $self->{ca}->{value} .= '&#';
3176              $self->{state} = $self->{prev_state};
3177              ## Reconsume.
3178              redo A;
3179            }
3180          }
3181        } elsif ($self->{state} == NCR_NUM_STATE) {
3182          if (0x0030 <= $self->{nc} and
3183              $self->{nc} <= 0x0039) { # 0..9
3184            !!!cp (1012);
3185            $self->{s_kwd} *= 10;
3186            $self->{s_kwd} += $self->{nc} - 0x0030;
3187            
3188            ## Stay in the state.
3189            !!!next-input-character;
3190            redo A;
3191          } elsif ($self->{nc} == 0x003B) { # ;
3192            !!!cp (1013);
3193            !!!next-input-character;
3194            #
3195          } else {
3196            !!!cp (1014);
3197          !!!parse-error (type => 'no refc');          !!!parse-error (type => 'no refc');
3198            ## Reconsume.
3199            #
3200        }        }
3201    
3202        if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {        my $code = $self->{s_kwd};
3203          !!!parse-error (type => sprintf 'invalid character reference:U+%04X', $code);        my $l = $self->{line_prev};
3204          my $c = $self->{column_prev};
3205          if ($charref_map->{$code}) {
3206            !!!cp (1015);
3207            !!!parse-error (type => 'invalid character reference',
3208                            text => (sprintf 'U+%04X', $code),
3209                            line => $l, column => $c);
3210            $code = $charref_map->{$code};
3211          } elsif ($code > 0x10FFFF) {
3212            !!!cp (1016);
3213            !!!parse-error (type => 'invalid character reference',
3214                            text => (sprintf 'U-%08X', $code),
3215                            line => $l, column => $c);
3216          $code = 0xFFFD;          $code = 0xFFFD;
3217          }
3218    
3219          if ($self->{prev_state} == DATA_STATE) {
3220            !!!cp (992);
3221            $self->{state} = $self->{prev_state};
3222            ## Reconsume.
3223            !!!emit ({type => CHARACTER_TOKEN, data => chr $code,
3224                      line => $l, column => $c,
3225                     });
3226            redo A;
3227          } else {
3228            !!!cp (991);
3229            $self->{ca}->{value} .= chr $code;
3230            $self->{ca}->{has_reference} = 1;
3231            $self->{state} = $self->{prev_state};
3232            ## Reconsume.
3233            redo A;
3234          }
3235        } elsif ($self->{state} == HEXREF_X_STATE) {
3236          if ((0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) or
3237              (0x0041 <= $self->{nc} and $self->{nc} <= 0x0046) or
3238              (0x0061 <= $self->{nc} and $self->{nc} <= 0x0066)) {
3239            # 0..9, A..F, a..f
3240            !!!cp (990);
3241            $self->{state} = HEXREF_HEX_STATE;
3242            $self->{s_kwd} = 0;
3243            ## Reconsume.
3244            redo A;
3245          } else {
3246            !!!parse-error (type => 'bare hcro',
3247                            line => $self->{line_prev},
3248                            column => $self->{column_prev} - 2);
3249    
3250            ## NOTE: According to the spec algorithm, nothing is returned,
3251            ## and then "&#" followed by "X" or "x" is appended to the parent
3252            ## element or the attribute value in the later processing.
3253    
3254            if ($self->{prev_state} == DATA_STATE) {
3255              !!!cp (1005);
3256              $self->{state} = $self->{prev_state};
3257              ## Reconsume.
3258              !!!emit ({type => CHARACTER_TOKEN,
3259                        data => '&' . $self->{s_kwd},
3260                        line => $self->{line_prev},
3261                        column => $self->{column_prev} - length $self->{s_kwd},
3262                       });
3263              redo A;
3264            } else {
3265              !!!cp (989);
3266              $self->{ca}->{value} .= '&' . $self->{s_kwd};
3267              $self->{state} = $self->{prev_state};
3268              ## Reconsume.
3269              redo A;
3270            }
3271          }
3272        } elsif ($self->{state} == HEXREF_HEX_STATE) {
3273          if (0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) {
3274            # 0..9
3275            !!!cp (1002);
3276            $self->{s_kwd} *= 0x10;
3277            $self->{s_kwd} += $self->{nc} - 0x0030;
3278            ## Stay in the state.
3279            !!!next-input-character;
3280            redo A;
3281          } elsif (0x0061 <= $self->{nc} and
3282                   $self->{nc} <= 0x0066) { # a..f
3283            !!!cp (1003);
3284            $self->{s_kwd} *= 0x10;
3285            $self->{s_kwd} += $self->{nc} - 0x0060 + 9;
3286            ## Stay in the state.
3287            !!!next-input-character;
3288            redo A;
3289          } elsif (0x0041 <= $self->{nc} and
3290                   $self->{nc} <= 0x0046) { # A..F
3291            !!!cp (1004);
3292            $self->{s_kwd} *= 0x10;
3293            $self->{s_kwd} += $self->{nc} - 0x0040 + 9;
3294            ## Stay in the state.
3295            !!!next-input-character;
3296            redo A;
3297          } elsif ($self->{nc} == 0x003B) { # ;
3298            !!!cp (1006);
3299            !!!next-input-character;
3300            #
3301          } else {
3302            !!!cp (1007);
3303            !!!parse-error (type => 'no refc',
3304                            line => $self->{line},
3305                            column => $self->{column});
3306            ## Reconsume.
3307            #
3308          }
3309    
3310          my $code = $self->{s_kwd};
3311          my $l = $self->{line_prev};
3312          my $c = $self->{column_prev};
3313          if ($charref_map->{$code}) {
3314            !!!cp (1008);
3315            !!!parse-error (type => 'invalid character reference',
3316                            text => (sprintf 'U+%04X', $code),
3317                            line => $l, column => $c);
3318            $code = $charref_map->{$code};
3319        } elsif ($code > 0x10FFFF) {        } elsif ($code > 0x10FFFF) {
3320          !!!parse-error (type => sprintf 'invalid character reference:U-%08X', $code);          !!!cp (1009);
3321            !!!parse-error (type => 'invalid character reference',
3322                            text => (sprintf 'U-%08X', $code),
3323                            line => $l, column => $c);
3324          $code = 0xFFFD;          $code = 0xFFFD;
       } elsif ($code == 0x000D) {  
         !!!parse-error (type => 'CR character reference');  
         $code = 0x000A;  
       } elsif (0x80 <= $code and $code <= 0x9F) {  
         !!!parse-error (type => sprintf 'C1 character reference:U+%04X', $code);  
         $code = $c1_entity_char->{$code};  
3325        }        }
3326          
3327        return {type => 'character', data => chr $code};        if ($self->{prev_state} == DATA_STATE) {
3328      } else {          !!!cp (988);
3329        !!!parse-error (type => 'bare nero');          $self->{state} = $self->{prev_state};
3330        !!!back-next-input-character ($self->{next_input_character});          ## Reconsume.
3331        $self->{next_input_character} = 0x0023; # #          !!!emit ({type => CHARACTER_TOKEN, data => chr $code,
3332        return undef;                    line => $l, column => $c,
3333      }                   });
3334    } elsif ((0x0041 <= $self->{next_input_character} and          redo A;
3335              $self->{next_input_character} <= 0x005A) or        } else {
3336             (0x0061 <= $self->{next_input_character} and          !!!cp (987);
3337              $self->{next_input_character} <= 0x007A)) {          $self->{ca}->{value} .= chr $code;
3338      my $entity_name = chr $self->{next_input_character};          $self->{ca}->{has_reference} = 1;
3339      !!!next-input-character;          $self->{state} = $self->{prev_state};
3340            ## Reconsume.
3341      my $value = $entity_name;          redo A;
3342      my $match;        }
3343      require Whatpm::_NamedEntityList;      } elsif ($self->{state} == ENTITY_NAME_STATE) {
3344      our $EntityChar;        if (length $self->{s_kwd} < 30 and
3345              ## NOTE: Some number greater than the maximum length of entity name
3346      while (length $entity_name < 10 and            ((0x0041 <= $self->{nc} and # a
3347             ## NOTE: Some number greater than the maximum length of entity name              $self->{nc} <= 0x005A) or # x
3348             ((0x0041 <= $self->{next_input_character} and # a             (0x0061 <= $self->{nc} and # a
3349               $self->{next_input_character} <= 0x005A) or # x              $self->{nc} <= 0x007A) or # z
3350              (0x0061 <= $self->{next_input_character} and # a             (0x0030 <= $self->{nc} and # 0
3351               $self->{next_input_character} <= 0x007A) or # z              $self->{nc} <= 0x0039) or # 9
3352              (0x0030 <= $self->{next_input_character} and # 0             $self->{nc} == 0x003B)) { # ;
3353               $self->{next_input_character} <= 0x0039) or # 9          our $EntityChar;
3354              $self->{next_input_character} == 0x003B)) { # ;          $self->{s_kwd} .= chr $self->{nc};
3355        $entity_name .= chr $self->{next_input_character};          if (defined $EntityChar->{$self->{s_kwd}}) {
3356        if (defined $EntityChar->{$entity_name}) {            if ($self->{nc} == 0x003B) { # ;
3357          if ($self->{next_input_character} == 0x003B) { # ;              !!!cp (1020);
3358            $value = $EntityChar->{$entity_name};              $self->{entity__value} = $EntityChar->{$self->{s_kwd}};
3359            $match = 1;              $self->{entity__match} = 1;
3360                !!!next-input-character;
3361                #
3362              } else {
3363                !!!cp (1021);
3364                $self->{entity__value} = $EntityChar->{$self->{s_kwd}};
3365                $self->{entity__match} = -1;
3366                ## Stay in the state.
3367                !!!next-input-character;
3368                redo A;
3369              }
3370            } else {
3371              !!!cp (1022);
3372              $self->{entity__value} .= chr $self->{nc};
3373              $self->{entity__match} *= 2;
3374              ## Stay in the state.
3375            !!!next-input-character;            !!!next-input-character;
3376            last;            redo A;
3377          } elsif (not $in_attr) {          }
3378            $value = $EntityChar->{$entity_name};        }
3379            $match = -1;  
3380          my $data;
3381          my $has_ref;
3382          if ($self->{entity__match} > 0) {
3383            !!!cp (1023);
3384            $data = $self->{entity__value};
3385            $has_ref = 1;
3386            #
3387          } elsif ($self->{entity__match} < 0) {
3388            !!!parse-error (type => 'no refc');
3389            if ($self->{prev_state} != DATA_STATE and # in attribute
3390                $self->{entity__match} < -1) {
3391              !!!cp (1024);
3392              $data = '&' . $self->{s_kwd};
3393              #
3394          } else {          } else {
3395            $value .= chr $self->{next_input_character};            !!!cp (1025);
3396              $data = $self->{entity__value};
3397              $has_ref = 1;
3398              #
3399          }          }
3400        } else {        } else {
3401          $value .= chr $self->{next_input_character};          !!!cp (1026);
3402            !!!parse-error (type => 'bare ero',
3403                            line => $self->{line_prev},
3404                            column => $self->{column_prev} - length $self->{s_kwd});
3405            $data = '&' . $self->{s_kwd};
3406            #
3407          }
3408      
3409          ## NOTE: In these cases, when a character reference is found,
3410          ## it is consumed and a character token is returned, or, otherwise,
3411          ## nothing is consumed and returned, according to the spec algorithm.
3412          ## In this implementation, anything that has been examined by the
3413          ## tokenizer is appended to the parent element or the attribute value
3414          ## as string, either literal string when no character reference or
3415          ## entity-replaced string otherwise, in this stage, since any characters
3416          ## that would not be consumed are appended in the data state or in an
3417          ## appropriate attribute value state anyway.
3418    
3419          if ($self->{prev_state} == DATA_STATE) {
3420            !!!cp (986);
3421            $self->{state} = $self->{prev_state};
3422            ## Reconsume.
3423            !!!emit ({type => CHARACTER_TOKEN,
3424                      data => $data,
3425                      line => $self->{line_prev},
3426                      column => $self->{column_prev} + 1 - length $self->{s_kwd},
3427                     });
3428            redo A;
3429          } else {
3430            !!!cp (985);
3431            $self->{ca}->{value} .= $data;
3432            $self->{ca}->{has_reference} = 1 if $has_ref;
3433            $self->{state} = $self->{prev_state};
3434            ## Reconsume.
3435            redo A;
3436        }        }
       !!!next-input-character;  
     }  
       
     if ($match > 0) {  
       return {type => 'character', data => $value};  
     } elsif ($match < 0) {  
       !!!parse-error (type => 'no refc');  
       return {type => 'character', data => $value};  
3437      } else {      } else {
3438        !!!parse-error (type => 'bare ero');        die "$0: $self->{state}: Unknown state";
       ## NOTE: No characters are consumed in the spec.  
       return {type => 'character', data => '&'.$value};  
3439      }      }
3440    } else {    } # A  
3441      ## no characters are consumed  
3442      !!!parse-error (type => 'bare ero');    die "$0: _get_next_token: unexpected case";
3443      return undef;  } # _get_next_token
   }  
 } # _tokenize_attempt_to_consume_an_entity  
3444    
3445  sub _initialize_tree_constructor ($) {  sub _initialize_tree_constructor ($) {
3446    my $self = shift;    my $self = shift;
# Line 1773  sub _initialize_tree_constructor ($) { Line 3449  sub _initialize_tree_constructor ($) {
3449    ## TODO: Turn mutation events off # MUST    ## TODO: Turn mutation events off # MUST
3450    ## TODO: Turn loose Document option (manakai extension) on    ## TODO: Turn loose Document option (manakai extension) on
3451    $self->{document}->manakai_is_html (1); # MUST    $self->{document}->manakai_is_html (1); # MUST
3452      $self->{document}->set_user_data (manakai_source_line => 1);
3453      $self->{document}->set_user_data (manakai_source_column => 1);
3454  } # _initialize_tree_constructor  } # _initialize_tree_constructor
3455    
3456  sub _terminate_tree_constructor ($) {  sub _terminate_tree_constructor ($) {
# Line 1792  sub _construct_tree ($) { Line 3470  sub _construct_tree ($) {
3470    ## When an interactive UA render the $self->{document} available    ## When an interactive UA render the $self->{document} available
3471    ## to the user, or when it begin accepting user input, are    ## to the user, or when it begin accepting user input, are
3472    ## not defined.    ## not defined.
   
   ## Append a character: collect it and all subsequent consecutive  
   ## characters and insert one Text node whose data is concatenation  
   ## of all those characters. # MUST  
3473        
3474    !!!next-token;    !!!next-token;
3475    
   $self->{insertion_mode} = 'before head';  
3476    undef $self->{form_element};    undef $self->{form_element};
3477    undef $self->{head_element};    undef $self->{head_element};
3478      undef $self->{head_element_inserted};
3479    $self->{open_elements} = [];    $self->{open_elements} = [];
3480    undef $self->{inner_html_node};    undef $self->{inner_html_node};
3481    
3482      ## NOTE: The "initial" insertion mode.
3483    $self->_tree_construction_initial; # MUST    $self->_tree_construction_initial; # MUST
3484    
3485      ## NOTE: The "before html" insertion mode.
3486    $self->_tree_construction_root_element;    $self->_tree_construction_root_element;
3487      $self->{insertion_mode} = BEFORE_HEAD_IM;
3488    
3489      ## NOTE: The "before head" insertion mode and so on.
3490    $self->_tree_construction_main;    $self->_tree_construction_main;
3491  } # _construct_tree  } # _construct_tree
3492    
3493  sub _tree_construction_initial ($) {  sub _tree_construction_initial ($) {
3494    my $self = shift;    my $self = shift;
3495    
3496      ## NOTE: "initial" insertion mode
3497    
3498    INITIAL: {    INITIAL: {
3499      if ($token->{type} eq 'DOCTYPE') {      if ($token->{type} == DOCTYPE_TOKEN) {
3500        ## NOTE: Conformance checkers MAY, instead of reporting "not HTML5"        ## NOTE: Conformance checkers MAY, instead of reporting "not HTML5"
3501        ## error, switch to a conformance checking mode for another        ## error, switch to a conformance checking mode for another
3502        ## language.        ## language.
3503        my $doctype_name = $token->{name};        my $doctype_name = $token->{name};
3504        $doctype_name = '' unless defined $doctype_name;        $doctype_name = '' unless defined $doctype_name;
3505        $doctype_name =~ tr/a-z/A-Z/;        $doctype_name =~ tr/a-z/A-Z/; # ASCII case-insensitive
3506        if (not defined $token->{name} or # <!DOCTYPE>        if (not defined $token->{name} or # <!DOCTYPE>
3507            defined $token->{public_identifier} or            defined $token->{sysid}) {
3508            defined $token->{system_identifier}) {          !!!cp ('t1');
3509          !!!parse-error (type => 'not HTML5');          !!!parse-error (type => 'not HTML5', token => $token);
3510        } elsif ($doctype_name ne 'HTML') {        } elsif ($doctype_name ne 'HTML') {
3511          ## ISSUE: ASCII case-insensitive? (in fact it does not matter)          !!!cp ('t2');
3512          !!!parse-error (type => 'not HTML5');          !!!parse-error (type => 'not HTML5', token => $token);
3513          } elsif (defined $token->{pubid}) {
3514            if ($token->{pubid} eq 'XSLT-compat') {
3515              !!!cp ('t1.2');
3516              !!!parse-error (type => 'XSLT-compat', token => $token,
3517                              level => $self->{level}->{should});
3518            } else {
3519              !!!parse-error (type => 'not HTML5', token => $token);
3520            }
3521          } else {
3522            !!!cp ('t3');
3523            #
3524        }        }
3525                
3526        my $doctype = $self->{document}->create_document_type_definition        my $doctype = $self->{document}->create_document_type_definition
3527          ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?          ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?
3528        $doctype->public_id ($token->{public_identifier})        ## NOTE: Default value for both |public_id| and |system_id| attributes
3529            if defined $token->{public_identifier};        ## are empty strings, so that we don't set any value in missing cases.
3530        $doctype->system_id ($token->{system_identifier})        $doctype->public_id ($token->{pubid}) if defined $token->{pubid};
3531            if defined $token->{system_identifier};        $doctype->system_id ($token->{sysid}) if defined $token->{sysid};
3532        ## NOTE: Other DocumentType attributes are null or empty lists.        ## NOTE: Other DocumentType attributes are null or empty lists.
3533        ## ISSUE: internalSubset = null??        ## ISSUE: internalSubset = null??
3534        $self->{document}->append_child ($doctype);        $self->{document}->append_child ($doctype);
3535                
3536        if (not $token->{correct} or $doctype_name ne 'HTML') {        if ($token->{quirks} or $doctype_name ne 'HTML') {
3537            !!!cp ('t4');
3538          $self->{document}->manakai_compat_mode ('quirks');          $self->{document}->manakai_compat_mode ('quirks');
3539        } elsif (defined $token->{public_identifier}) {        } elsif (defined $token->{pubid}) {
3540          my $pubid = $token->{public_identifier};          my $pubid = $token->{pubid};
3541          $pubid =~ tr/a-z/A-z/;          $pubid =~ tr/a-z/A-z/;
3542          if ({          my $prefix = [
3543            "+//SILMARIL//DTD HTML PRO V0R11 19970101//EN" => 1,            "+//SILMARIL//DTD HTML PRO V0R11 19970101//",
3544            "-//ADVASOFT LTD//DTD HTML 3.0 ASWEDIT + EXTENSIONS//EN" => 1,            "-//ADVASOFT LTD//DTD HTML 3.0 ASWEDIT + EXTENSIONS//",
3545            "-//AS//DTD HTML 3.0 ASWEDIT + EXTENSIONS//EN" => 1,            "-//AS//DTD HTML 3.0 ASWEDIT + EXTENSIONS//",
3546            "-//IETF//DTD HTML 2.0 LEVEL 1//EN" => 1,            "-//IETF//DTD HTML 2.0 LEVEL 1//",
3547            "-//IETF//DTD HTML 2.0 LEVEL 2//EN" => 1,            "-//IETF//DTD HTML 2.0 LEVEL 2//",
3548            "-//IETF//DTD HTML 2.0 STRICT LEVEL 1//EN" => 1,            "-//IETF//DTD HTML 2.0 STRICT LEVEL 1//",
3549            "-//IETF//DTD HTML 2.0 STRICT LEVEL 2//EN" => 1,            "-//IETF//DTD HTML 2.0 STRICT LEVEL 2//",
3550            "-//IETF//DTD HTML 2.0 STRICT//EN" => 1,            "-//IETF//DTD HTML 2.0 STRICT//",
3551            "-//IETF//DTD HTML 2.0//EN" => 1,            "-//IETF//DTD HTML 2.0//",
3552            "-//IETF//DTD HTML 2.1E//EN" => 1,            "-//IETF//DTD HTML 2.1E//",
3553            "-//IETF//DTD HTML 3.0//EN" => 1,            "-//IETF//DTD HTML 3.0//",
3554            "-//IETF//DTD HTML 3.0//EN//" => 1,            "-//IETF//DTD HTML 3.2 FINAL//",
3555            "-//IETF//DTD HTML 3.2 FINAL//EN" => 1,            "-//IETF//DTD HTML 3.2//",
3556            "-//IETF//DTD HTML 3.2//EN" => 1,            "-//IETF//DTD HTML 3//",
3557            "-//IETF//DTD HTML 3//EN" => 1,            "-//IETF//DTD HTML LEVEL 0//",
3558            "-//IETF//DTD HTML LEVEL 0//EN" => 1,            "-//IETF//DTD HTML LEVEL 1//",
3559            "-//IETF//DTD HTML LEVEL 0//EN//2.0" => 1,            "-//IETF//DTD HTML LEVEL 2//",
3560            "-//IETF//DTD HTML LEVEL 1//EN" => 1,            "-//IETF//DTD HTML LEVEL 3//",
3561            "-//IETF//DTD HTML LEVEL 1//EN//2.0" => 1,            "-//IETF//DTD HTML STRICT LEVEL 0//",
3562            "-//IETF//DTD HTML LEVEL 2//EN" => 1,            "-//IETF//DTD HTML STRICT LEVEL 1//",
3563            "-//IETF//DTD HTML LEVEL 2//EN//2.0" => 1,            "-//IETF//DTD HTML STRICT LEVEL 2//",
3564            "-//IETF//DTD HTML LEVEL 3//EN" => 1,            "-//IETF//DTD HTML STRICT LEVEL 3//",
3565            "-//IETF//DTD HTML LEVEL 3//EN//3.0" => 1,            "-//IETF//DTD HTML STRICT//",
3566            "-//IETF//DTD HTML STRICT LEVEL 0//EN" => 1,            "-//IETF//DTD HTML//",
3567            "-//IETF//DTD HTML STRICT LEVEL 0//EN//2.0" => 1,            "-//METRIUS//DTD METRIUS PRESENTATIONAL//",
3568            "-//IETF//DTD HTML STRICT LEVEL 1//EN" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML STRICT//",
3569            "-//IETF//DTD HTML STRICT LEVEL 1//EN//2.0" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML//",
3570            "-//IETF//DTD HTML STRICT LEVEL 2//EN" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 TABLES//",
3571            "-//IETF//DTD HTML STRICT LEVEL 2//EN//2.0" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML STRICT//",
3572            "-//IETF//DTD HTML STRICT LEVEL 3//EN" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML//",
3573            "-//IETF//DTD HTML STRICT LEVEL 3//EN//3.0" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 TABLES//",
3574            "-//IETF//DTD HTML STRICT//EN" => 1,            "-//NETSCAPE COMM. CORP.//DTD HTML//",
3575            "-//IETF//DTD HTML STRICT//EN//2.0" => 1,            "-//NETSCAPE COMM. CORP.//DTD STRICT HTML//",
3576            "-//IETF//DTD HTML STRICT//EN//3.0" => 1,            "-//O'REILLY AND ASSOCIATES//DTD HTML 2.0//",
3577            "-//IETF//DTD HTML//EN" => 1,            "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED 1.0//",
3578            "-//IETF//DTD HTML//EN//2.0" => 1,            "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED RELAXED 1.0//",
3579            "-//IETF//DTD HTML//EN//3.0" => 1,            "-//SOFTQUAD SOFTWARE//DTD HOTMETAL PRO 6.0::19990601::EXTENSIONS TO HTML 4.0//",
3580            "-//METRIUS//DTD METRIUS PRESENTATIONAL//EN" => 1,            "-//SOFTQUAD//DTD HOTMETAL PRO 4.0::19971010::EXTENSIONS TO HTML 4.0//",
3581            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML STRICT//EN" => 1,            "-//SPYGLASS//DTD HTML 2.0 EXTENDED//",
3582            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML//EN" => 1,            "-//SQ//DTD HTML 2.0 HOTMETAL + EXTENSIONS//",
3583            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 TABLES//EN" => 1,            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA HTML//",
3584            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML STRICT//EN" => 1,            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA STRICT HTML//",
3585            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML//EN" => 1,            "-//W3C//DTD HTML 3 1995-03-24//",
3586            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 TABLES//EN" => 1,            "-//W3C//DTD HTML 3.2 DRAFT//",
3587            "-//NETSCAPE COMM. CORP.//DTD HTML//EN" => 1,            "-//W3C//DTD HTML 3.2 FINAL//",
3588            "-//NETSCAPE COMM. CORP.//DTD STRICT HTML//EN" => 1,            "-//W3C//DTD HTML 3.2//",
3589            "-//O'REILLY AND ASSOCIATES//DTD HTML 2.0//EN" => 1,            "-//W3C//DTD HTML 3.2S DRAFT//",
3590            "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED 1.0//EN" => 1,            "-//W3C//DTD HTML 4.0 FRAMESET//",
3591            "-//SPYGLASS//DTD HTML 2.0 EXTENDED//EN" => 1,            "-//W3C//DTD HTML 4.0 TRANSITIONAL//",
3592            "-//SQ//DTD HTML 2.0 HOTMETAL + EXTENSIONS//EN" => 1,            "-//W3C//DTD HTML EXPERIMETNAL 19960712//",
3593            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA HTML//EN" => 1,            "-//W3C//DTD HTML EXPERIMENTAL 970421//",
3594            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA STRICT HTML//EN" => 1,            "-//W3C//DTD W3 HTML//",
3595            "-//W3C//DTD HTML 3 1995-03-24//EN" => 1,            "-//W3O//DTD W3 HTML 3.0//",
3596            "-//W3C//DTD HTML 3.2 DRAFT//EN" => 1,            "-//WEBTECHS//DTD MOZILLA HTML 2.0//",
3597            "-//W3C//DTD HTML 3.2 FINAL//EN" => 1,            "-//WEBTECHS//DTD MOZILLA HTML//",
3598            "-//W3C//DTD HTML 3.2//EN" => 1,          ]; # $prefix
3599            "-//W3C//DTD HTML 3.2S DRAFT//EN" => 1,          my $match;
3600            "-//W3C//DTD HTML 4.0 FRAMESET//EN" => 1,          for (@$prefix) {
3601            "-//W3C//DTD HTML 4.0 TRANSITIONAL//EN" => 1,            if (substr ($prefix, 0, length $_) eq $_) {
3602            "-//W3C//DTD HTML EXPERIMETNAL 19960712//EN" => 1,              $match = 1;
3603            "-//W3C//DTD HTML EXPERIMENTAL 970421//EN" => 1,              last;
3604            "-//W3C//DTD W3 HTML//EN" => 1,            }
3605            "-//W3O//DTD W3 HTML 3.0//EN" => 1,          }
3606            "-//W3O//DTD W3 HTML 3.0//EN//" => 1,          if ($match or
3607            "-//W3O//DTD W3 HTML STRICT 3.0//EN//" => 1,              $pubid eq "-//W3O//DTD W3 HTML STRICT 3.0//EN//" or
3608            "-//WEBTECHS//DTD MOZILLA HTML 2.0//EN" => 1,              $pubid eq "-/W3C/DTD HTML 4.0 TRANSITIONAL/EN" or
3609            "-//WEBTECHS//DTD MOZILLA HTML//EN" => 1,              $pubid eq "HTML") {
3610            "-/W3C/DTD HTML 4.0 TRANSITIONAL/EN" => 1,            !!!cp ('t5');
           "HTML" => 1,  
         }->{$pubid}) {  
3611            $self->{document}->manakai_compat_mode ('quirks');            $self->{document}->manakai_compat_mode ('quirks');
3612          } elsif ($pubid eq "-//W3C//DTD HTML 4.01 FRAMESET//EN" or          } elsif ($pubid =~ m[^-//W3C//DTD HTML 4.01 FRAMESET//] or
3613                   $pubid eq "-//W3C//DTD HTML 4.01 TRANSITIONAL//EN") {                   $pubid =~ m[^-//W3C//DTD HTML 4.01 TRANSITIONAL//]) {
3614            if (defined $token->{system_identifier}) {            if (defined $token->{sysid}) {
3615                !!!cp ('t6');
3616              $self->{document}->manakai_compat_mode ('quirks');              $self->{document}->manakai_compat_mode ('quirks');
3617            } else {            } else {
3618                !!!cp ('t7');
3619              $self->{document}->manakai_compat_mode ('limited quirks');              $self->{document}->manakai_compat_mode ('limited quirks');
3620            }            }
3621          } elsif ($pubid eq "-//W3C//DTD XHTML 1.0 Frameset//EN" or          } elsif ($pubid =~ m[^-//W3C//DTD XHTML 1.0 FRAMESET//] or
3622                   $pubid eq "-//W3C//DTD XHTML 1.0 Transitional//EN") {                   $pubid =~ m[^-//W3C//DTD XHTML 1.0 TRANSITIONAL//]) {
3623              !!!cp ('t8');
3624            $self->{document}->manakai_compat_mode ('limited quirks');            $self->{document}->manakai_compat_mode ('limited quirks');
3625            } else {
3626              !!!cp ('t9');
3627          }          }
3628          } else {
3629            !!!cp ('t10');
3630        }        }
3631        if (defined $token->{system_identifier}) {        if (defined $token->{sysid}) {
3632          my $sysid = $token->{system_identifier};          my $sysid = $token->{sysid};
3633          $sysid =~ tr/A-Z/a-z/;          $sysid =~ tr/A-Z/a-z/;
3634          if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {          if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {
3635              ## NOTE: Ensure that |PUBLIC "(limited quirks)" "(quirks)"| is
3636              ## marked as quirks.
3637            $self->{document}->manakai_compat_mode ('quirks');            $self->{document}->manakai_compat_mode ('quirks');
3638              !!!cp ('t11');
3639            } else {
3640              !!!cp ('t12');
3641          }          }
3642          } else {
3643            !!!cp ('t13');
3644        }        }
3645                
3646        ## Go to the root element phase.        ## Go to the "before html" insertion mode.
3647        !!!next-token;        !!!next-token;
3648        return;        return;
3649      } elsif ({      } elsif ({
3650                'start tag' => 1,                START_TAG_TOKEN, 1,
3651                'end tag' => 1,                END_TAG_TOKEN, 1,
3652                'end-of-file' => 1,                END_OF_FILE_TOKEN, 1,
3653               }->{$token->{type}}) {               }->{$token->{type}}) {
3654        !!!parse-error (type => 'no DOCTYPE');        !!!cp ('t14');
3655          !!!parse-error (type => 'no DOCTYPE', token => $token);
3656        $self->{document}->manakai_compat_mode ('quirks');        $self->{document}->manakai_compat_mode ('quirks');
3657        ## Go to the root element phase        ## Go to the "before html" insertion mode.
3658        ## reprocess        ## reprocess
3659          !!!ack-later;
3660        return;        return;
3661      } elsif ($token->{type} eq 'character') {      } elsif ($token->{type} == CHARACTER_TOKEN) {
3662        if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D        if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3663          ## Ignore the token          ## Ignore the token
3664    
3665          unless (length $token->{data}) {          unless (length $token->{data}) {
3666            ## Stay in the phase            !!!cp ('t15');
3667              ## Stay in the insertion mode.
3668            !!!next-token;            !!!next-token;
3669            redo INITIAL;            redo INITIAL;
3670            } else {
3671              !!!cp ('t16');
3672          }          }
3673          } else {
3674            !!!cp ('t17');
3675        }        }
3676    
3677        !!!parse-error (type => 'no DOCTYPE');        !!!parse-error (type => 'no DOCTYPE', token => $token);
3678        $self->{document}->manakai_compat_mode ('quirks');        $self->{document}->manakai_compat_mode ('quirks');
3679        ## Go to the root element phase        ## Go to the "before html" insertion mode.
3680        ## reprocess        ## reprocess
3681        return;        return;
3682      } elsif ($token->{type} eq 'comment') {      } elsif ($token->{type} == COMMENT_TOKEN) {
3683          !!!cp ('t18');
3684        my $comment = $self->{document}->create_comment ($token->{data});        my $comment = $self->{document}->create_comment ($token->{data});
3685        $self->{document}->append_child ($comment);        $self->{document}->append_child ($comment);
3686                
3687        ## Stay in the phase.        ## Stay in the insertion mode.
3688        !!!next-token;        !!!next-token;
3689        redo INITIAL;        redo INITIAL;
3690      } else {      } else {
3691        die "$0: $token->{type}: Unknown token";        die "$0: $token->{type}: Unknown token type";
3692      }      }
3693    } # INITIAL    } # INITIAL
3694    
3695      die "$0: _tree_construction_initial: This should be never reached";
3696  } # _tree_construction_initial  } # _tree_construction_initial
3697    
3698  sub _tree_construction_root_element ($) {  sub _tree_construction_root_element ($) {
3699    my $self = shift;    my $self = shift;
3700    
3701      ## NOTE: "before html" insertion mode.
3702        
3703    B: {    B: {
3704        if ($token->{type} eq 'DOCTYPE') {        if ($token->{type} == DOCTYPE_TOKEN) {
3705          !!!parse-error (type => 'in html:#DOCTYPE');          !!!cp ('t19');
3706            !!!parse-error (type => 'in html:#DOCTYPE', token => $token);
3707          ## Ignore the token          ## Ignore the token
3708          ## Stay in the phase          ## Stay in the insertion mode.
3709          !!!next-token;          !!!next-token;
3710          redo B;          redo B;
3711        } elsif ($token->{type} eq 'comment') {        } elsif ($token->{type} == COMMENT_TOKEN) {
3712            !!!cp ('t20');
3713          my $comment = $self->{document}->create_comment ($token->{data});          my $comment = $self->{document}->create_comment ($token->{data});
3714          $self->{document}->append_child ($comment);          $self->{document}->append_child ($comment);
3715          ## Stay in the phase          ## Stay in the insertion mode.
3716          !!!next-token;          !!!next-token;
3717          redo B;          redo B;
3718        } elsif ($token->{type} eq 'character') {        } elsif ($token->{type} == CHARACTER_TOKEN) {
3719          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3720            ## Ignore the token.            ## Ignore the token.
3721    
3722            unless (length $token->{data}) {            unless (length $token->{data}) {
3723              ## Stay in the phase              !!!cp ('t21');
3724                ## Stay in the insertion mode.
3725              !!!next-token;              !!!next-token;
3726              redo B;              redo B;
3727              } else {
3728                !!!cp ('t22');
3729            }            }
3730            } else {
3731              !!!cp ('t23');
3732          }          }
3733    
3734            $self->{application_cache_selection}->(undef);
3735    
3736          #          #
3737          } elsif ($token->{type} == START_TAG_TOKEN) {
3738            if ($token->{tag_name} eq 'html') {
3739              my $root_element;
3740              !!!create-element ($root_element, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);
3741              $self->{document}->append_child ($root_element);
3742              push @{$self->{open_elements}},
3743                  [$root_element, $el_category->{html}];
3744    
3745              if ($token->{attributes}->{manifest}) {
3746                !!!cp ('t24');
3747                $self->{application_cache_selection}
3748                    ->($token->{attributes}->{manifest}->{value});
3749                ## ISSUE: Spec is unclear on relative references.
3750                ## According to Hixie (#whatwg 2008-03-19), it should be
3751                ## resolved against the base URI of the document in HTML
3752                ## or xml:base of the element in XHTML.
3753              } else {
3754                !!!cp ('t25');
3755                $self->{application_cache_selection}->(undef);
3756              }
3757    
3758              !!!nack ('t25c');
3759    
3760              !!!next-token;
3761              return; ## Go to the "before head" insertion mode.
3762            } else {
3763              !!!cp ('t25.1');
3764              #
3765            }
3766        } elsif ({        } elsif ({
3767                  'start tag' => 1,                  END_TAG_TOKEN, 1,
3768                  'end tag' => 1,                  END_OF_FILE_TOKEN, 1,
                 'end-of-file' => 1,  
3769                 }->{$token->{type}}) {                 }->{$token->{type}}) {
3770          ## ISSUE: There is an issue in the spec          !!!cp ('t26');
3771          #          #
3772        } else {        } else {
3773          die "$0: $token->{type}: Unknown token";          die "$0: $token->{type}: Unknown token type";
3774        }        }
3775        my $root_element; !!!create-element ($root_element, 'html');  
3776        $self->{document}->append_child ($root_element);      my $root_element;
3777        push @{$self->{open_elements}}, [$root_element, 'html'];      !!!create-element ($root_element, $HTML_NS, 'html',, $token);
3778        ## reprocess      $self->{document}->append_child ($root_element);
3779        #redo B;      push @{$self->{open_elements}}, [$root_element, $el_category->{html}];
3780        return; ## Go to the main phase.  
3781        $self->{application_cache_selection}->(undef);
3782    
3783        ## NOTE: Reprocess the token.
3784        !!!ack-later;
3785        return; ## Go to the "before head" insertion mode.
3786    } # B    } # B
3787    
3788      die "$0: _tree_construction_root_element: This should never be reached";
3789  } # _tree_construction_root_element  } # _tree_construction_root_element
3790    
3791  sub _reset_insertion_mode ($) {  sub _reset_insertion_mode ($) {
# Line 2036  sub _reset_insertion_mode ($) { Line 3800  sub _reset_insertion_mode ($) {
3800            
3801      ## Step 3      ## Step 3
3802      S3: {      S3: {
       ## ISSUE: Oops! "If node is the first node in the stack of open  
       ## elements, then set last to true. If the context element of the  
       ## HTML fragment parsing algorithm is neither a td element nor a  
       ## th element, then set node to the context element. (fragment case)":  
       ## The second "if" is in the scope of the first "if"!?  
3803        if ($self->{open_elements}->[0]->[0] eq $node->[0]) {        if ($self->{open_elements}->[0]->[0] eq $node->[0]) {
3804          $last = 1;          $last = 1;
3805          if (defined $self->{inner_html_node}) {          if (defined $self->{inner_html_node}) {
3806            if ($self->{inner_html_node}->[1] eq 'td' or            !!!cp ('t28');
3807                $self->{inner_html_node}->[1] eq 'th') {            $node = $self->{inner_html_node};
3808              #          } else {
3809            } else {            die "_reset_insertion_mode: t27";
             $node = $self->{inner_html_node};  
           }  
3810          }          }
3811        }        }
3812              
3813        ## Step 4..13        ## Step 4..14
3814        my $new_mode = {        my $new_mode;
3815                        select => 'in select',        if ($node->[1] & FOREIGN_EL) {
3816                        td => 'in cell',          !!!cp ('t28.1');
3817                        th => 'in cell',          ## NOTE: Strictly spaking, the line below only applies to MathML and
3818                        tr => 'in row',          ## SVG elements.  Currently the HTML syntax supports only MathML and
3819                        tbody => 'in table body',          ## SVG elements as foreigners.
3820                        thead => 'in table head',          $new_mode = IN_BODY_IM | IN_FOREIGN_CONTENT_IM;
3821                        tfoot => 'in table foot',        } elsif ($node->[1] & TABLE_CELL_EL) {
3822                        caption => 'in caption',          if ($last) {
3823                        colgroup => 'in column group',            !!!cp ('t28.2');
3824                        table => 'in table',            #
3825                        head => 'in body', # not in head!          } else {
3826                        body => 'in body',            !!!cp ('t28.3');
3827                        frameset => 'in frameset',            $new_mode = IN_CELL_IM;
3828                       }->{$node->[1]};          }
3829          } else {
3830            !!!cp ('t28.4');
3831            $new_mode = {
3832                          select => IN_SELECT_IM,
3833                          ## NOTE: |option| and |optgroup| do not set
3834                          ## insertion mode to "in select" by themselves.
3835                          tr => IN_ROW_IM,
3836                          tbody => IN_TABLE_BODY_IM,
3837                          thead => IN_TABLE_BODY_IM,
3838                          tfoot => IN_TABLE_BODY_IM,
3839                          caption => IN_CAPTION_IM,
3840                          colgroup => IN_COLUMN_GROUP_IM,
3841                          table => IN_TABLE_IM,
3842                          head => IN_BODY_IM, # not in head!
3843                          body => IN_BODY_IM,
3844                          frameset => IN_FRAMESET_IM,
3845                         }->{$node->[0]->manakai_local_name};
3846          }
3847        $self->{insertion_mode} = $new_mode and return if defined $new_mode;        $self->{insertion_mode} = $new_mode and return if defined $new_mode;
3848                
3849        ## Step 14        ## Step 15
3850        if ($node->[1] eq 'html') {        if ($node->[1] & HTML_EL) {
3851          unless (defined $self->{head_element}) {          unless (defined $self->{head_element}) {
3852            $self->{insertion_mode} = 'before head';            !!!cp ('t29');
3853              $self->{insertion_mode} = BEFORE_HEAD_IM;
3854          } else {          } else {
3855            $self->{insertion_mode} = 'after head';            ## ISSUE: Can this state be reached?
3856              !!!cp ('t30');
3857              $self->{insertion_mode} = AFTER_HEAD_IM;
3858          }          }
3859          return;          return;
3860          } else {
3861            !!!cp ('t31');
3862        }        }
3863                
       ## Step 15  
       $self->{insertion_mode} = 'in body' and return if $last;  
         
3864        ## Step 16        ## Step 16
3865          $self->{insertion_mode} = IN_BODY_IM and return if $last;
3866          
3867          ## Step 17
3868        $i--;        $i--;
3869        $node = $self->{open_elements}->[$i];        $node = $self->{open_elements}->[$i];
3870                
3871        ## Step 17        ## Step 18
3872        redo S3;        redo S3;
3873      } # S3      } # S3
3874    
3875      die "$0: _reset_insertion_mode: This line should never be reached";
3876  } # _reset_insertion_mode  } # _reset_insertion_mode
3877    
3878  sub _tree_construction_main ($) {  sub _tree_construction_main ($) {
3879    my $self = shift;    my $self = shift;
3880    
   my $previous_insertion_mode;  
   
3881    my $active_formatting_elements = [];    my $active_formatting_elements = [];
3882    
3883    my $reconstruct_active_formatting_elements = sub { # MUST    my $reconstruct_active_formatting_elements = sub { # MUST
# Line 2114  sub _tree_construction_main ($) { Line 3894  sub _tree_construction_main ($) {
3894      return if $entry->[0] eq '#marker';      return if $entry->[0] eq '#marker';
3895      for (@{$self->{open_elements}}) {      for (@{$self->{open_elements}}) {
3896        if ($entry->[0] eq $_->[0]) {        if ($entry->[0] eq $_->[0]) {
3897            !!!cp ('t32');
3898          return;          return;
3899        }        }
3900      }      }
# Line 2128  sub _tree_construction_main ($) { Line 3909  sub _tree_construction_main ($) {
3909    
3910        ## Step 6        ## Step 6
3911        if ($entry->[0] eq '#marker') {        if ($entry->[0] eq '#marker') {
3912            !!!cp ('t33_1');
3913          #          #
3914        } else {        } else {
3915          my $in_open_elements;          my $in_open_elements;
3916          OE: for (@{$self->{open_elements}}) {          OE: for (@{$self->{open_elements}}) {
3917            if ($entry->[0] eq $_->[0]) {            if ($entry->[0] eq $_->[0]) {
3918                !!!cp ('t33');
3919              $in_open_elements = 1;              $in_open_elements = 1;
3920              last OE;              last OE;
3921            }            }
3922          }          }
3923          if ($in_open_elements) {          if ($in_open_elements) {
3924              !!!cp ('t34');
3925            #            #
3926          } else {          } else {
3927              ## NOTE: <!DOCTYPE HTML><p><b><i><u></p> <p>X
3928              !!!cp ('t35');
3929            redo S4;            redo S4;
3930          }          }
3931        }        }
# Line 2162  sub _tree_construction_main ($) { Line 3948  sub _tree_construction_main ($) {
3948    
3949        ## Step 11        ## Step 11
3950        unless ($clone->[0] eq $active_formatting_elements->[-1]->[0]) {        unless ($clone->[0] eq $active_formatting_elements->[-1]->[0]) {
3951            !!!cp ('t36');
3952          ## Step 7'          ## Step 7'
3953          $i++;          $i++;
3954          $entry = $active_formatting_elements->[$i];          $entry = $active_formatting_elements->[$i];
3955                    
3956          redo S7;          redo S7;
3957        }        }
3958    
3959          !!!cp ('t37');
3960      } # S7      } # S7
3961    }; # $reconstruct_active_formatting_elements    }; # $reconstruct_active_formatting_elements
3962    
3963    my $clear_up_to_marker = sub {    my $clear_up_to_marker = sub {
3964      for (reverse 0..$#$active_formatting_elements) {      for (reverse 0..$#$active_formatting_elements) {
3965        if ($active_formatting_elements->[$_]->[0] eq '#marker') {        if ($active_formatting_elements->[$_]->[0] eq '#marker') {
3966            !!!cp ('t38');
3967          splice @$active_formatting_elements, $_;          splice @$active_formatting_elements, $_;
3968          return;          return;
3969        }        }
3970      }      }
3971    
3972        !!!cp ('t39');
3973    }; # $clear_up_to_marker    }; # $clear_up_to_marker
3974    
3975    my $parse_rcdata = sub ($$) {    my $insert;
3976      my ($content_model_flag, $insert) = @_;  
3977      my $parse_rcdata = sub ($) {
3978        my ($content_model_flag) = @_;
3979    
3980      ## Step 1      ## Step 1
3981      my $start_tag_name = $token->{tag_name};      my $start_tag_name = $token->{tag_name};
3982      my $el;      !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
     !!!create-element ($el, $start_tag_name, $token->{attributes});  
3983    
3984      ## Step 2      ## Step 2
3985      $insert->($el); # /context node/->append_child ($el)      $self->{content_model} = $content_model_flag; # CDATA or RCDATA
   
     ## Step 3  
     $self->{content_model_flag} = $content_model_flag; # CDATA or RCDATA  
3986      delete $self->{escape}; # MUST      delete $self->{escape}; # MUST
3987    
3988      ## Step 4      ## Step 3, 4
3989      my $text = '';      $self->{insertion_mode} |= IN_CDATA_RCDATA_IM;
     !!!next-token;  
     while ($token->{type} eq 'character') { # or until stop tokenizing  
       $text .= $token->{data};  
       !!!next-token;  
     }  
   
     ## Step 5  
     if (length $text) {  
       my $text = $self->{document}->create_text_node ($text);  
       $el->append_child ($text);  
     }  
3990    
3991      ## Step 6      !!!nack ('t40.1');
     $self->{content_model_flag} = 'PCDATA';  
   
     ## Step 7  
     if ($token->{type} eq 'end tag' and $token->{tag_name} eq $start_tag_name) {  
       ## Ignore the token  
     } else {  
       !!!parse-error (type => 'in '.$content_model_flag.':#'.$token->{type});  
     }  
3992      !!!next-token;      !!!next-token;
3993    }; # $parse_rcdata    }; # $parse_rcdata
3994    
3995    my $script_start_tag = sub ($) {    my $script_start_tag = sub () {
3996      my $insert = $_[0];      ## Step 1
3997      my $script_el;      my $script_el;
3998      !!!create-element ($script_el, 'script', $token->{attributes});      !!!create-element ($script_el, $HTML_NS, 'script', $token->{attributes}, $token);
3999    
4000        ## Step 2
4001      ## TODO: mark as "parser-inserted"      ## TODO: mark as "parser-inserted"
4002    
4003      $self->{content_model_flag} = 'CDATA';      ## Step 3
4004        ## TODO: Mark as "already executed", if ...
4005    
4006        ## Step 4
4007        $insert->($script_el);
4008    
4009        ## ISSUE: $script_el is not put into the stack
4010        push @{$self->{open_elements}}, [$script_el, $el_category->{script}];
4011    
4012        ## Step 5
4013        $self->{content_model} = CDATA_CONTENT_MODEL;
4014      delete $self->{escape}; # MUST      delete $self->{escape}; # MUST
       
     my $text = '';  
     !!!next-token;  
     while ($token->{type} eq 'character') {  
       $text .= $token->{data};  
       !!!next-token;  
     } # stop if non-character token or tokenizer stops tokenising  
     if (length $text) {  
       $script_el->manakai_append_text ($text);  
     }  
                 
     $self->{content_model_flag} = 'PCDATA';  
4015    
4016      if ($token->{type} eq 'end tag' and      ## Step 6-7
4017          $token->{tag_name} eq 'script') {      $self->{insertion_mode} |= IN_CDATA_RCDATA_IM;
       ## Ignore the token  
     } else {  
       !!!parse-error (type => 'in CDATA:#'.$token->{type});  
       ## ISSUE: And ignore?  
       ## TODO: mark as "already executed"  
     }  
       
     if (defined $self->{inner_html_node}) {  
       ## TODO: mark as "already executed"  
     } else {  
       ## TODO: $old_insertion_point = current insertion point  
       ## TODO: insertion point = just before the next input character  
4018    
4019        $insert->($script_el);      !!!nack ('t40.2');
         
       ## TODO: insertion point = $old_insertion_point (might be "undefined")  
         
       ## TODO: if there is a script that will execute as soon as the parser resume, then...  
     }  
       
4020      !!!next-token;      !!!next-token;
4021    }; # $script_start_tag    }; # $script_start_tag
4022    
4023      ## NOTE: $open_tables->[-1]->[0] is the "current table" element node.
4024      ## NOTE: $open_tables->[-1]->[1] is the "tainted" flag.
4025      ## NOTE: $open_tables->[-1]->[2] is set false when non-Text node inserted.
4026      my $open_tables = [[$self->{open_elements}->[0]->[0]]];
4027    
4028    my $formatting_end_tag = sub {    my $formatting_end_tag = sub {
4029      my $tag_name = shift;      my $end_tag_token = shift;
4030        my $tag_name = $end_tag_token->{tag_name};
4031    
4032        ## NOTE: The adoption agency algorithm (AAA).
4033    
4034      FET: {      FET: {
4035        ## Step 1        ## Step 1
4036        my $formatting_element;        my $formatting_element;
4037        my $formatting_element_i_in_active;        my $formatting_element_i_in_active;
4038        AFE: for (reverse 0..$#$active_formatting_elements) {        AFE: for (reverse 0..$#$active_formatting_elements) {
4039          if ($active_formatting_elements->[$_]->[1] eq $tag_name) {          if ($active_formatting_elements->[$_]->[0] eq '#marker') {
4040              !!!cp ('t52');
4041              last AFE;
4042            } elsif ($active_formatting_elements->[$_]->[0]->manakai_local_name
4043                         eq $tag_name) {
4044              !!!cp ('t51');
4045            $formatting_element = $active_formatting_elements->[$_];            $formatting_element = $active_formatting_elements->[$_];
4046            $formatting_element_i_in_active = $_;            $formatting_element_i_in_active = $_;
4047            last AFE;            last AFE;
         } elsif ($active_formatting_elements->[$_]->[0] eq '#marker') {  
           last AFE;  
4048          }          }
4049        } # AFE        } # AFE
4050        unless (defined $formatting_element) {        unless (defined $formatting_element) {
4051          !!!parse-error (type => 'unmatched end tag:'.$tag_name);          !!!cp ('t53');
4052            !!!parse-error (type => 'unmatched end tag', text => $tag_name, token => $end_tag_token);
4053          ## Ignore the token          ## Ignore the token
4054          !!!next-token;          !!!next-token;
4055          return;          return;
# Line 2296  sub _tree_construction_main ($) { Line 4061  sub _tree_construction_main ($) {
4061          my $node = $self->{open_elements}->[$_];          my $node = $self->{open_elements}->[$_];
4062          if ($node->[0] eq $formatting_element->[0]) {          if ($node->[0] eq $formatting_element->[0]) {
4063            if ($in_scope) {            if ($in_scope) {
4064                !!!cp ('t54');
4065              $formatting_element_i_in_open = $_;              $formatting_element_i_in_open = $_;
4066              last INSCOPE;              last INSCOPE;
4067            } else { # in open elements but not in scope            } else { # in open elements but not in scope
4068              !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});              !!!cp ('t55');
4069                !!!parse-error (type => 'unmatched end tag',
4070                                text => $token->{tag_name},
4071                                token => $end_tag_token);
4072              ## Ignore the token              ## Ignore the token
4073              !!!next-token;              !!!next-token;
4074              return;              return;
4075            }            }
4076          } elsif ({          } elsif ($node->[1] & SCOPING_EL) {
4077                    table => 1, caption => 1, td => 1, th => 1,            !!!cp ('t56');
                   button => 1, marquee => 1, object => 1, html => 1,  
                  }->{$node->[1]}) {  
4078            $in_scope = 0;            $in_scope = 0;
4079          }          }
4080        } # INSCOPE        } # INSCOPE
4081        unless (defined $formatting_element_i_in_open) {        unless (defined $formatting_element_i_in_open) {
4082          !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});          !!!cp ('t57');
4083            !!!parse-error (type => 'unmatched end tag',
4084                            text => $token->{tag_name},
4085                            token => $end_tag_token);
4086          pop @$active_formatting_elements; # $formatting_element          pop @$active_formatting_elements; # $formatting_element
4087          !!!next-token; ## TODO: ok?          !!!next-token; ## TODO: ok?
4088          return;          return;
4089        }        }
4090        if (not $self->{open_elements}->[-1]->[0] eq $formatting_element->[0]) {        if (not $self->{open_elements}->[-1]->[0] eq $formatting_element->[0]) {
4091          !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);          !!!cp ('t58');
4092            !!!parse-error (type => 'not closed',
4093                            text => $self->{open_elements}->[-1]->[0]
4094                                ->manakai_local_name,
4095                            token => $end_tag_token);
4096        }        }
4097                
4098        ## Step 2        ## Step 2
# Line 2326  sub _tree_construction_main ($) { Line 4100  sub _tree_construction_main ($) {
4100        my $furthest_block_i_in_open;        my $furthest_block_i_in_open;
4101        OE: for (reverse 0..$#{$self->{open_elements}}) {        OE: for (reverse 0..$#{$self->{open_elements}}) {
4102          my $node = $self->{open_elements}->[$_];          my $node = $self->{open_elements}->[$_];
4103          if (not $formatting_category->{$node->[1]} and          if (not ($node->[1] & FORMATTING_EL) and
4104              #not $phrasing_category->{$node->[1]} and              #not $phrasing_category->{$node->[1]} and
4105              ($special_category->{$node->[1]} or              ($node->[1] & SPECIAL_EL or
4106               $scoping_category->{$node->[1]})) {               $node->[1] & SCOPING_EL)) { ## Scoping is redundant, maybe
4107              !!!cp ('t59');
4108            $furthest_block = $node;            $furthest_block = $node;
4109            $furthest_block_i_in_open = $_;            $furthest_block_i_in_open = $_;
4110              ## NOTE: The topmost (eldest) node.
4111          } elsif ($node->[0] eq $formatting_element->[0]) {          } elsif ($node->[0] eq $formatting_element->[0]) {
4112              !!!cp ('t60');
4113            last OE;            last OE;
4114          }          }
4115        } # OE        } # OE
4116                
4117        ## Step 3        ## Step 3
4118        unless (defined $furthest_block) { # MUST        unless (defined $furthest_block) { # MUST
4119            !!!cp ('t61');
4120          splice @{$self->{open_elements}}, $formatting_element_i_in_open;          splice @{$self->{open_elements}}, $formatting_element_i_in_open;
4121          splice @$active_formatting_elements, $formatting_element_i_in_active, 1;          splice @$active_formatting_elements, $formatting_element_i_in_active, 1;
4122          !!!next-token;          !!!next-token;
# Line 2351  sub _tree_construction_main ($) { Line 4129  sub _tree_construction_main ($) {
4129        ## Step 5        ## Step 5
4130        my $furthest_block_parent = $furthest_block->[0]->parent_node;        my $furthest_block_parent = $furthest_block->[0]->parent_node;
4131        if (defined $furthest_block_parent) {        if (defined $furthest_block_parent) {
4132            !!!cp ('t62');
4133          $furthest_block_parent->remove_child ($furthest_block->[0]);          $furthest_block_parent->remove_child ($furthest_block->[0]);
4134        }        }
4135                
# Line 2373  sub _tree_construction_main ($) { Line 4152  sub _tree_construction_main ($) {
4152          S7S2: {          S7S2: {
4153            for (reverse 0..$#$active_formatting_elements) {            for (reverse 0..$#$active_formatting_elements) {
4154              if ($active_formatting_elements->[$_]->[0] eq $node->[0]) {              if ($active_formatting_elements->[$_]->[0] eq $node->[0]) {
4155                  !!!cp ('t63');
4156                $node_i_in_active = $_;                $node_i_in_active = $_;
4157                last S7S2;                last S7S2;
4158              }              }
# Line 2386  sub _tree_construction_main ($) { Line 4166  sub _tree_construction_main ($) {
4166                    
4167          ## Step 4          ## Step 4
4168          if ($last_node->[0] eq $furthest_block->[0]) {          if ($last_node->[0] eq $furthest_block->[0]) {
4169              !!!cp ('t64');
4170            $bookmark_prev_el = $node->[0];            $bookmark_prev_el = $node->[0];
4171          }          }
4172                    
4173          ## Step 5          ## Step 5
4174          if ($node->[0]->has_child_nodes ()) {          if ($node->[0]->has_child_nodes ()) {
4175              !!!cp ('t65');
4176            my $clone = [$node->[0]->clone_node (0), $node->[1]];            my $clone = [$node->[0]->clone_node (0), $node->[1]];
4177            $active_formatting_elements->[$node_i_in_active] = $clone;            $active_formatting_elements->[$node_i_in_active] = $clone;
4178            $self->{open_elements}->[$node_i_in_open] = $clone;            $self->{open_elements}->[$node_i_in_open] = $clone;
# Line 2408  sub _tree_construction_main ($) { Line 4190  sub _tree_construction_main ($) {
4190        } # S7          } # S7  
4191                
4192        ## Step 8        ## Step 8
4193        $common_ancestor_node->[0]->append_child ($last_node->[0]);        if ($common_ancestor_node->[1] & TABLE_ROWS_EL) {
4194            my $foster_parent_element;
4195            my $next_sibling;
4196            OE: for (reverse 0..$#{$self->{open_elements}}) {
4197              if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {
4198                                 my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
4199                                 if (defined $parent and $parent->node_type == 1) {
4200                                   !!!cp ('t65.1');
4201                                   $foster_parent_element = $parent;
4202                                   $next_sibling = $self->{open_elements}->[$_]->[0];
4203                                 } else {
4204                                   !!!cp ('t65.2');
4205                                   $foster_parent_element
4206                                     = $self->{open_elements}->[$_ - 1]->[0];
4207                                 }
4208                                 last OE;
4209                               }
4210                             } # OE
4211                             $foster_parent_element = $self->{open_elements}->[0]->[0]
4212                               unless defined $foster_parent_element;
4213            $foster_parent_element->insert_before ($last_node->[0], $next_sibling);
4214            $open_tables->[-1]->[1] = 1; # tainted
4215          } else {
4216            !!!cp ('t65.3');
4217            $common_ancestor_node->[0]->append_child ($last_node->[0]);
4218          }
4219                
4220        ## Step 9        ## Step 9
4221        my $clone = [$formatting_element->[0]->clone_node (0),        my $clone = [$formatting_element->[0]->clone_node (0),
# Line 2425  sub _tree_construction_main ($) { Line 4232  sub _tree_construction_main ($) {
4232        my $i;        my $i;
4233        AFE: for (reverse 0..$#$active_formatting_elements) {        AFE: for (reverse 0..$#$active_formatting_elements) {
4234          if ($active_formatting_elements->[$_]->[0] eq $formatting_element->[0]) {          if ($active_formatting_elements->[$_]->[0] eq $formatting_element->[0]) {
4235              !!!cp ('t66');
4236            splice @$active_formatting_elements, $_, 1;            splice @$active_formatting_elements, $_, 1;
4237            $i-- and last AFE if defined $i;            $i-- and last AFE if defined $i;
4238          } elsif ($active_formatting_elements->[$_]->[0] eq $bookmark_prev_el) {          } elsif ($active_formatting_elements->[$_]->[0] eq $bookmark_prev_el) {
4239              !!!cp ('t67');
4240            $i = $_;            $i = $_;
4241          }          }
4242        } # AFE        } # AFE
# Line 2437  sub _tree_construction_main ($) { Line 4246  sub _tree_construction_main ($) {
4246        undef $i;        undef $i;
4247        OE: for (reverse 0..$#{$self->{open_elements}}) {        OE: for (reverse 0..$#{$self->{open_elements}}) {
4248          if ($self->{open_elements}->[$_]->[0] eq $formatting_element->[0]) {          if ($self->{open_elements}->[$_]->[0] eq $formatting_element->[0]) {
4249              !!!cp ('t68');
4250            splice @{$self->{open_elements}}, $_, 1;            splice @{$self->{open_elements}}, $_, 1;
4251            $i-- and last OE if defined $i;            $i-- and last OE if defined $i;
4252          } elsif ($self->{open_elements}->[$_]->[0] eq $furthest_block->[0]) {          } elsif ($self->{open_elements}->[$_]->[0] eq $furthest_block->[0]) {
4253              !!!cp ('t69');
4254            $i = $_;            $i = $_;
4255          }          }
4256        } # OE        } # OE
4257        splice @{$self->{open_elements}}, $i + 1, 1, $clone;        splice @{$self->{open_elements}}, $i + 1, 0, $clone;
4258                
4259        ## Step 14        ## Step 14
4260        redo FET;        redo FET;
4261      } # FET      } # FET
4262    }; # $formatting_end_tag    }; # $formatting_end_tag
4263    
4264    my $insert_to_current = sub {    $insert = my $insert_to_current = sub {
4265      $self->{open_elements}->[-1]->[0]->append_child ($_[0]);      $self->{open_elements}->[-1]->[0]->append_child ($_[0]);
4266    }; # $insert_to_current    }; # $insert_to_current
4267    
4268    my $insert_to_foster = sub {    my $insert_to_foster = sub {
4269                         my $child = shift;      my $child = shift;
4270                         if ({      if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {
4271                              table => 1, tbody => 1, tfoot => 1,        # MUST
4272                              thead => 1, tr => 1,        my $foster_parent_element;
4273                             }->{$self->{open_elements}->[-1]->[1]}) {        my $next_sibling;
4274                           # MUST        OE: for (reverse 0..$#{$self->{open_elements}}) {
4275                           my $foster_parent_element;          if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {
                          my $next_sibling;  
                          OE: for (reverse 0..$#{$self->{open_elements}}) {  
                            if ($self->{open_elements}->[$_]->[1] eq 'table') {  
4276                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
4277                               if (defined $parent and $parent->node_type == 1) {                               if (defined $parent and $parent->node_type == 1) {
4278                                   !!!cp ('t70');
4279                                 $foster_parent_element = $parent;                                 $foster_parent_element = $parent;
4280                                 $next_sibling = $self->{open_elements}->[$_]->[0];                                 $next_sibling = $self->{open_elements}->[$_]->[0];
4281                               } else {                               } else {
4282                                   !!!cp ('t71');
4283                                 $foster_parent_element                                 $foster_parent_element
4284                                   = $self->{open_elements}->[$_ - 1]->[0];                                   = $self->{open_elements}->[$_ - 1]->[0];
4285                               }                               }
# Line 2480  sub _tree_construction_main ($) { Line 4290  sub _tree_construction_main ($) {
4290                             unless defined $foster_parent_element;                             unless defined $foster_parent_element;
4291                           $foster_parent_element->insert_before                           $foster_parent_element->insert_before
4292                             ($child, $next_sibling);                             ($child, $next_sibling);
4293                         } else {        $open_tables->[-1]->[1] = 1; # tainted
4294                           $self->{open_elements}->[-1]->[0]->append_child ($child);      } else {
4295                         }        !!!cp ('t72');
4296          $self->{open_elements}->[-1]->[0]->append_child ($child);
4297        }
4298    }; # $insert_to_foster    }; # $insert_to_foster
4299    
4300    my $in_body = sub {    ## NOTE: Insert a character (MUST): When a character is inserted, if
4301      my $insert = shift;    ## the last node that was inserted by the parser is a Text node and
4302      if ($token->{type} eq 'start tag') {    ## the character has to be inserted after that node, then the
4303        if ($token->{tag_name} eq 'script') {    ## character is appended to the Text node.  However, if any other
4304          ## NOTE: This is an "as if in head" code clone    ## node is inserted by the parser, then a new Text node is created
4305          $script_start_tag->($insert);    ## and the character is appended as that Text node.  If I'm not
4306          return;    ## wrong, for a parser with scripting disabled, there are only two
4307        } elsif ($token->{tag_name} eq 'style') {    ## cases where this occurs.  One is the case where an element node
4308          ## NOTE: This is an "as if in head" code clone    ## is inserted to the |head| element.  This is covered by using the
4309          $parse_rcdata->('CDATA', $insert);    ## |$self->{head_element_inserted}| flag.  Another is the case where
4310          return;    ## an element or comment is inserted into the |table| subtree while
4311        } elsif ({    ## foster parenting happens.  This is covered by using the [2] flag
4312                  base => 1, link => 1,    ## of the |$open_tables| structure.  All other cases are handled
4313                 }->{$token->{tag_name}}) {    ## simply by calling |manakai_append_text| method.
4314          ## NOTE: This is an "as if in head" code clone, only "-t" differs  
4315          !!!insert-element-t ($token->{tag_name}, $token->{attributes});    ## TODO: |<body><script>document.write("a<br>");
4316          pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.    ## document.body.removeChild (document.body.lastChild);
4317          !!!next-token;    ## document.write ("b")</script>|
4318          return;  
4319        } elsif ($token->{tag_name} eq 'meta') {    B: while (1) {
4320          ## NOTE: This is an "as if in head" code clone, only "-t" differs      if ($token->{type} == DOCTYPE_TOKEN) {
4321          !!!insert-element-t ($token->{tag_name}, $token->{attributes});        !!!cp ('t73');
4322          pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.        !!!parse-error (type => 'in html:#DOCTYPE', token => $token);
4323          ## Ignore the token
4324          ## Stay in the phase
4325          !!!next-token;
4326          next B;
4327        } elsif ($token->{type} == START_TAG_TOKEN and
4328                 $token->{tag_name} eq 'html') {
4329          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
4330            !!!cp ('t79');
4331            !!!parse-error (type => 'after html', text => 'html', token => $token);
4332            $self->{insertion_mode} = AFTER_BODY_IM;
4333          } elsif ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {
4334            !!!cp ('t80');
4335            !!!parse-error (type => 'after html', text => 'html', token => $token);
4336            $self->{insertion_mode} = AFTER_FRAMESET_IM;
4337          } else {
4338            !!!cp ('t81');
4339          }
4340    
4341          unless ($self->{confident}) {        !!!cp ('t82');
4342            my $charset;        !!!parse-error (type => 'not first start tag', token => $token);
4343            if ($token->{attributes}->{charset}) { ## TODO: And if supported        my $top_el = $self->{open_elements}->[0]->[0];
4344              $charset = $token->{attributes}->{charset}->{value};        for my $attr_name (keys %{$token->{attributes}}) {
4345            }          unless ($top_el->has_attribute_ns (undef, $attr_name)) {
4346            if ($token->{attributes}->{'http-equiv'}) {            !!!cp ('t84');
4347              ## ISSUE: Algorithm name in the spec was incorrect so that not linked to the definition.            $top_el->set_attribute_ns
4348              if ($token->{attributes}->{'http-equiv'}->{value}              (undef, [undef, $attr_name],
4349                  =~ /\A[^;]*;[\x09-\x0D\x20]*charset[\x09-\x0D\x20]*=               $token->{attributes}->{$attr_name}->{value});
                     [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|  
                     ([^"'\x09-\x0D\x20][^\x09-\x0D\x20]*))/x) {  
               $charset = defined $1 ? $1 : defined $2 ? $2 : $3;  
             } ## TODO: And if supported  
           }  
           ## TODO: Change the encoding  
4350          }          }
4351          }
4352          !!!next-token;        !!!nack ('t84.1');
4353          return;        !!!next-token;
4354        } elsif ($token->{tag_name} eq 'title') {        next B;
4355          !!!parse-error (type => 'in body:title');      } elsif ($token->{type} == COMMENT_TOKEN) {
4356          ## NOTE: This is an "as if in head" code clone        my $comment = $self->{document}->create_comment ($token->{data});
4357          $parse_rcdata->('RCDATA', sub {        if ($self->{insertion_mode} & AFTER_HTML_IMS) {
4358            if (defined $self->{head_element}) {          !!!cp ('t85');
4359              $self->{head_element}->append_child ($_[0]);          $self->{document}->append_child ($comment);
4360            } else {        } elsif ($self->{insertion_mode} == AFTER_BODY_IM) {
4361              $insert->($_[0]);          !!!cp ('t86');
4362            }          $self->{open_elements}->[0]->[0]->append_child ($comment);
4363          });        } else {
4364          return;          !!!cp ('t87');
4365        } elsif ($token->{tag_name} eq 'body') {          $self->{open_elements}->[-1]->[0]->append_child ($comment);
4366          !!!parse-error (type => 'in body:body');          $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
4367                        }
4368          if (@{$self->{open_elements}} == 1 or        !!!next-token;
4369              $self->{open_elements}->[1]->[1] ne 'body') {        next B;
4370            ## Ignore the token      } elsif ($self->{insertion_mode} & IN_CDATA_RCDATA_IM) {
4371          if ($token->{type} == CHARACTER_TOKEN) {
4372            $token->{data} =~ s/^\x0A// if $self->{ignore_newline};
4373            delete $self->{ignore_newline};
4374    
4375            if (length $token->{data}) {
4376              !!!cp ('t43');
4377              $self->{open_elements}->[-1]->[0]->manakai_append_text
4378                  ($token->{data});
4379          } else {          } else {
4380            my $body_el = $self->{open_elements}->[1]->[0];            !!!cp ('t43.1');
           for my $attr_name (keys %{$token->{attributes}}) {  
             unless ($body_el->has_attribute_ns (undef, $attr_name)) {  
               $body_el->set_attribute_ns  
                 (undef, [undef, $attr_name],  
                  $token->{attributes}->{$attr_name}->{value});  
             }  
           }  
4381          }          }
4382          !!!next-token;          !!!next-token;
4383          return;          next B;
4384        } elsif ({        } elsif ($token->{type} == END_TAG_TOKEN) {
4385                  address => 1, blockquote => 1, center => 1, dir => 1,          delete $self->{ignore_newline};
4386                  div => 1, dl => 1, fieldset => 1, listing => 1,  
4387                  menu => 1, ol => 1, p => 1, ul => 1,          if ($token->{tag_name} eq 'script') {
4388                  pre => 1,            !!!cp ('t50');
                }->{$token->{tag_name}}) {  
         ## has a p element in scope  
         INSCOPE: for (reverse @{$self->{open_elements}}) {  
           if ($_->[1] eq 'p') {  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'p'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$_->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
4389                        
4390          !!!insert-element-t ($token->{tag_name}, $token->{attributes});            ## Para 1-2
4391          if ($token->{tag_name} eq 'pre') {            my $script = pop @{$self->{open_elements}};
4392            !!!next-token;            
4393            if ($token->{type} eq 'character') {            ## Para 3
4394              $token->{data} =~ s/^\x0A//;            $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
4395              unless (length $token->{data}) {  
4396                !!!next-token;            ## Para 4
4397              }            ## TODO: $old_insertion_point = $current_insertion_point;
4398            }            ## TODO: $current_insertion_point = just before $self->{nc};
4399          } else {  
4400            !!!next-token;            ## Para 5
4401          }            ## TODO: Run the $script->[0].
4402          return;  
4403        } elsif ($token->{tag_name} eq 'form') {            ## Para 6
4404          if (defined $self->{form_element}) {            ## TODO: $current_insertion_point = $old_insertion_point;
4405            !!!parse-error (type => 'in form:form');  
4406            ## Ignore the token            ## Para 7
4407              ## TODO: if ($pending_external_script) {
4408                ## TODO: ...
4409              ## TODO: }
4410    
4411            !!!next-token;            !!!next-token;
4412            return;            next B;
4413          } else {          } else {
4414            ## has a p element in scope            !!!cp ('t42');
4415            INSCOPE: for (reverse @{$self->{open_elements}}) {  
4416              if ($_->[1] eq 'p') {            pop @{$self->{open_elements}};
4417                !!!back-token;  
4418                $token = {type => 'end tag', tag_name => 'p'};            $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
               return;  
             } elsif ({  
                       table => 1, caption => 1, td => 1, th => 1,  
                       button => 1, marquee => 1, object => 1, html => 1,  
                      }->{$_->[1]}) {  
               last INSCOPE;  
             }  
           } # INSCOPE  
               
           !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
           $self->{form_element} = $self->{open_elements}->[-1]->[0];  
4419            !!!next-token;            !!!next-token;
4420            return;            next B;
4421          }          }
4422        } elsif ($token->{tag_name} eq 'li') {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
4423          ## has a p element in scope          delete $self->{ignore_newline};
         INSCOPE: for (reverse @{$self->{open_elements}}) {  
           if ($_->[1] eq 'p') {  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'p'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$_->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         ## Step 1  
         my $i = -1;  
         my $node = $self->{open_elements}->[$i];  
         LI: {  
           ## Step 2  
           if ($node->[1] eq 'li') {  
             if ($i != -1) {  
               !!!parse-error (type => 'end tag missing:'.  
                               $self->{open_elements}->[-1]->[1]);  
             }  
             splice @{$self->{open_elements}}, $i;  
             last LI;  
           }  
             
           ## Step 3  
           if (not $formatting_category->{$node->[1]} and  
               #not $phrasing_category->{$node->[1]} and  
               ($special_category->{$node->[1]} or  
                $scoping_category->{$node->[1]}) and  
               $node->[1] ne 'address' and $node->[1] ne 'div') {  
             last LI;  
           }  
             
           ## Step 4  
           $i--;  
           $node = $self->{open_elements}->[$i];  
           redo LI;  
         } # LI  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'dd' or $token->{tag_name} eq 'dt') {  
         ## has a p element in scope  
         INSCOPE: for (reverse @{$self->{open_elements}}) {  
           if ($_->[1] eq 'p') {  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'p'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$_->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         ## Step 1  
         my $i = -1;  
         my $node = $self->{open_elements}->[$i];  
         LI: {  
           ## Step 2  
           if ($node->[1] eq 'dt' or $node->[1] eq 'dd') {  
             if ($i != -1) {  
               !!!parse-error (type => 'end tag missing:'.  
                               $self->{open_elements}->[-1]->[1]);  
             }  
             splice @{$self->{open_elements}}, $i;  
             last LI;  
           }  
             
           ## Step 3  
           if (not $formatting_category->{$node->[1]} and  
               #not $phrasing_category->{$node->[1]} and  
               ($special_category->{$node->[1]} or  
                $scoping_category->{$node->[1]}) and  
               $node->[1] ne 'address' and $node->[1] ne 'div') {  
             last LI;  
           }  
             
           ## Step 4  
           $i--;  
           $node = $self->{open_elements}->[$i];  
           redo LI;  
         } # LI  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'plaintext') {  
         ## has a p element in scope  
         INSCOPE: for (reverse @{$self->{open_elements}}) {  
           if ($_->[1] eq 'p') {  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'p'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$_->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
             
         $self->{content_model_flag} = 'PLAINTEXT';  
             
         !!!next-token;  
         return;  
       } elsif ({  
                 h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,  
                }->{$token->{tag_name}}) {  
         ## has a p element in scope  
         INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
           my $node = $self->{open_elements}->[$_];  
           if ($node->[1] eq 'p') {  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'p'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         ## NOTE: See <http://html5.org/tools/web-apps-tracker?from=925&to=926>  
         ## has an element in scope  
         #my $i;  
         #INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
         #  my $node = $self->{open_elements}->[$_];  
         #  if ({  
         #       h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,  
         #      }->{$node->[1]}) {  
         #    $i = $_;  
         #    last INSCOPE;  
         #  } elsif ({  
         #            table => 1, caption => 1, td => 1, th => 1,  
         #            button => 1, marquee => 1, object => 1, html => 1,  
         #           }->{$node->[1]}) {  
         #    last INSCOPE;  
         #  }  
         #} # INSCOPE  
         #    
         #if (defined $i) {  
         #  !!! parse-error (type => 'in hn:hn');  
         #  splice @{$self->{open_elements}}, $i;  
         #}  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
             
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'a') {  
         AFE: for my $i (reverse 0..$#$active_formatting_elements) {  
           my $node = $active_formatting_elements->[$i];  
           if ($node->[1] eq 'a') {  
             !!!parse-error (type => 'in a:a');  
               
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'a'};  
             $formatting_end_tag->($token->{tag_name});  
               
             AFE2: for (reverse 0..$#$active_formatting_elements) {  
               if ($active_formatting_elements->[$_]->[0] eq $node->[0]) {  
                 splice @$active_formatting_elements, $_, 1;  
                 last AFE2;  
               }  
             } # AFE2  
             OE: for (reverse 0..$#{$self->{open_elements}}) {  
               if ($self->{open_elements}->[$_]->[0] eq $node->[0]) {  
                 splice @{$self->{open_elements}}, $_, 1;  
                 last OE;  
               }  
             } # OE  
             last AFE;  
           } elsif ($node->[0] eq '#marker') {  
             last AFE;  
           }  
         } # AFE  
             
         $reconstruct_active_formatting_elements->($insert_to_current);  
4424    
4425          !!!insert-element-t ($token->{tag_name}, $token->{attributes});          !!!cp ('t44');
4426          push @$active_formatting_elements, $self->{open_elements}->[-1];          !!!parse-error (type => 'not closed',
4427                            text => $self->{open_elements}->[-1]->[0]
4428                                ->manakai_local_name,
4429                            token => $token);
4430    
4431          !!!next-token;          #if ($self->{open_elements}->[-1]->[1] & SCRIPT_EL) {
4432          return;          #  ## TODO: Mark as "already executed"
4433        } elsif ({          #}
                 b => 1, big => 1, em => 1, font => 1, i => 1,  
                 s => 1, small => 1, strile => 1,  
                 strong => 1, tt => 1, u => 1,  
                }->{$token->{tag_name}}) {  
         $reconstruct_active_formatting_elements->($insert_to_current);  
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         push @$active_formatting_elements, $self->{open_elements}->[-1];  
           
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'nobr') {  
         $reconstruct_active_formatting_elements->($insert_to_current);  
4434    
4435          ## has a |nobr| element in scope          pop @{$self->{open_elements}};
         INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
           my $node = $self->{open_elements}->[$_];  
           if ($node->[1] eq 'nobr') {  
             !!!parse-error (type => 'not closed:nobr');  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'nobr'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         push @$active_formatting_elements, $self->{open_elements}->[-1];  
           
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'button') {  
         ## has a button element in scope  
         INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
           my $node = $self->{open_elements}->[$_];  
           if ($node->[1] eq 'button') {  
             !!!parse-error (type => 'in button:button');  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'button'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         $reconstruct_active_formatting_elements->($insert_to_current);  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         push @$active_formatting_elements, ['#marker', ''];  
4436    
4437            $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
4438            ## Reprocess.
4439            next B;
4440          } else {
4441            die "$0: $token->{type}: In CDATA/RCDATA: Unknown token type";        
4442          }
4443        } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {
4444          if ($token->{type} == CHARACTER_TOKEN) {
4445            !!!cp ('t87.1');
4446            $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
4447          !!!next-token;          !!!next-token;
4448          return;          next B;
4449        } elsif ($token->{tag_name} eq 'marquee' or        } elsif ($token->{type} == START_TAG_TOKEN) {
4450                 $token->{tag_name} eq 'object') {          if ((not {mglyph => 1, malignmark => 1}->{$token->{tag_name}} and
4451          $reconstruct_active_formatting_elements->($insert_to_current);               $self->{open_elements}->[-1]->[1] & FOREIGN_FLOW_CONTENT_EL) or
4452                        not ($self->{open_elements}->[-1]->[1] & FOREIGN_EL) or
4453          !!!insert-element-t ($token->{tag_name}, $token->{attributes});              ($token->{tag_name} eq 'svg' and
4454          push @$active_formatting_elements, ['#marker', ''];               $self->{open_elements}->[-1]->[1] & MML_AXML_EL)) {
4455                      ## NOTE: "using the rules for secondary insertion mode"then"continue"
4456          !!!next-token;            !!!cp ('t87.2');
4457          return;            #
4458        } elsif ($token->{tag_name} eq 'xmp') {          } elsif ({
4459          $reconstruct_active_formatting_elements->($insert_to_current);                    b => 1, big => 1, blockquote => 1, body => 1, br => 1,
4460          $parse_rcdata->('CDATA', $insert);                    center => 1, code => 1, dd => 1, div => 1, dl => 1, dt => 1,
4461          return;                    em => 1, embed => 1, font => 1, h1 => 1, h2 => 1, h3 => 1,
4462        } elsif ($token->{tag_name} eq 'table') {                    h4 => 1, h5 => 1, h6 => 1, head => 1, hr => 1, i => 1,
4463          ## has a p element in scope                    img => 1, li => 1, listing => 1, menu => 1, meta => 1,
4464          INSCOPE: for (reverse @{$self->{open_elements}}) {                    nobr => 1, ol => 1, p => 1, pre => 1, ruby => 1, s => 1,
4465            if ($_->[1] eq 'p') {                    small => 1, span => 1, strong => 1, strike => 1, sub => 1,
4466              !!!back-token;                    sup => 1, table => 1, tt => 1, u => 1, ul => 1, var => 1,
4467              $token = {type => 'end tag', tag_name => 'p'};                   }->{$token->{tag_name}}) {
4468              return;            !!!cp ('t87.2');
4469            } elsif ({            !!!parse-error (type => 'not closed',
4470                      table => 1, caption => 1, td => 1, th => 1,                            text => $self->{open_elements}->[-1]->[0]
4471                      button => 1, marquee => 1, object => 1, html => 1,                                ->manakai_local_name,
4472                     }->{$_->[1]}) {                            token => $token);
4473              last INSCOPE;  
4474              pop @{$self->{open_elements}}
4475                  while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;
4476    
4477              $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
4478              ## Reprocess.
4479              next B;
4480            } else {
4481              my $nsuri = $self->{open_elements}->[-1]->[0]->namespace_uri;
4482              my $tag_name = $token->{tag_name};
4483              if ($nsuri eq $SVG_NS) {
4484                $tag_name = {
4485                   altglyph => 'altGlyph',
4486                   altglyphdef => 'altGlyphDef',
4487                   altglyphitem => 'altGlyphItem',
4488                   animatecolor => 'animateColor',
4489                   animatemotion => 'animateMotion',
4490                   animatetransform => 'animateTransform',
4491                   clippath => 'clipPath',
4492                   feblend => 'feBlend',
4493                   fecolormatrix => 'feColorMatrix',
4494                   fecomponenttransfer => 'feComponentTransfer',
4495                   fecomposite => 'feComposite',
4496                   feconvolvematrix => 'feConvolveMatrix',
4497                   fediffuselighting => 'feDiffuseLighting',
4498                   fedisplacementmap => 'feDisplacementMap',
4499                   fedistantlight => 'feDistantLight',
4500                   feflood => 'feFlood',
4501                   fefunca => 'feFuncA',
4502                   fefuncb => 'feFuncB',
4503                   fefuncg => 'feFuncG',
4504                   fefuncr => 'feFuncR',
4505                   fegaussianblur => 'feGaussianBlur',
4506                   feimage => 'feImage',
4507                   femerge => 'feMerge',
4508                   femergenode => 'feMergeNode',
4509                   femorphology => 'feMorphology',
4510                   feoffset => 'feOffset',
4511                   fepointlight => 'fePointLight',
4512                   fespecularlighting => 'feSpecularLighting',
4513                   fespotlight => 'feSpotLight',
4514                   fetile => 'feTile',
4515                   feturbulence => 'feTurbulence',
4516                   foreignobject => 'foreignObject',
4517                   glyphref => 'glyphRef',
4518                   lineargradient => 'linearGradient',
4519                   radialgradient => 'radialGradient',
4520                   #solidcolor => 'solidColor', ## NOTE: Commented in spec (SVG1.2)
4521                   textpath => 'textPath',  
4522                }->{$tag_name} || $tag_name;
4523            }            }
         } # INSCOPE  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
             
         $self->{insertion_mode} = 'in table';  
             
         !!!next-token;  
         return;  
       } elsif ({  
                 area => 1, basefont => 1, bgsound => 1, br => 1,  
                 embed => 1, img => 1, param => 1, spacer => 1, wbr => 1,  
                 image => 1,  
                }->{$token->{tag_name}}) {  
         if ($token->{tag_name} eq 'image') {  
           !!!parse-error (type => 'image');  
           $token->{tag_name} = 'img';  
         }  
4524    
4525          ## NOTE: There is an "as if <br>" code clone.            ## "adjust SVG attributes" (SVG only) - done in insert-element-f
4526          $reconstruct_active_formatting_elements->($insert_to_current);  
4527                      ## "adjust foreign attributes" - done in insert-element-f
4528          !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
4529          pop @{$self->{open_elements}};            !!!insert-element-f ($nsuri, $tag_name, $token->{attributes}, $token);
4530            
4531          !!!next-token;            if ($self->{self_closing}) {
4532          return;              pop @{$self->{open_elements}};
4533        } elsif ($token->{tag_name} eq 'hr') {              !!!ack ('t87.3');
         ## has a p element in scope  
         INSCOPE: for (reverse @{$self->{open_elements}}) {  
           if ($_->[1] eq 'p') {  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'p'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$_->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         pop @{$self->{open_elements}};  
             
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'input') {  
         $reconstruct_active_formatting_elements->($insert_to_current);  
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         ## TODO: associate with $self->{form_element} if defined  
         pop @{$self->{open_elements}};  
           
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'isindex') {  
         !!!parse-error (type => 'isindex');  
           
         if (defined $self->{form_element}) {  
           ## Ignore the token  
           !!!next-token;  
           return;  
         } else {  
           my $at = $token->{attributes};  
           my $form_attrs;  
           $form_attrs->{action} = $at->{action} if $at->{action};  
           my $prompt_attr = $at->{prompt};  
           $at->{name} = {name => 'name', value => 'isindex'};  
           delete $at->{action};  
           delete $at->{prompt};  
           my @tokens = (  
                         {type => 'start tag', tag_name => 'form',  
                          attributes => $form_attrs},  
                         {type => 'start tag', tag_name => 'hr'},  
                         {type => 'start tag', tag_name => 'p'},  
                         {type => 'start tag', tag_name => 'label'},  
                        );  
           if ($prompt_attr) {  
             push @tokens, {type => 'character', data => $prompt_attr->{value}};  
4534            } else {            } else {
4535              push @tokens, {type => 'character',              !!!cp ('t87.4');
                            data => 'This is a searchable index. Insert your search keywords here: '}; # SHOULD  
             ## TODO: make this configurable  
4536            }            }
4537            push @tokens,  
                         {type => 'start tag', tag_name => 'input', attributes => $at},  
                         #{type => 'character', data => ''}, # SHOULD  
                         {type => 'end tag', tag_name => 'label'},  
                         {type => 'end tag', tag_name => 'p'},  
                         {type => 'start tag', tag_name => 'hr'},  
                         {type => 'end tag', tag_name => 'form'};  
           $token = shift @tokens;  
           !!!back-token (@tokens);  
           return;  
         }  
       } elsif ($token->{tag_name} eq 'textarea') {  
         my $tag_name = $token->{tag_name};  
         my $el;  
         !!!create-element ($el, $token->{tag_name}, $token->{attributes});  
           
         ## TODO: $self->{form_element} if defined  
         $self->{content_model_flag} = 'RCDATA';  
         delete $self->{escape}; # MUST  
           
         $insert->($el);  
           
         my $text = '';  
         !!!next-token;  
         if ($token->{type} eq 'character') {  
           $token->{data} =~ s/^\x0A//;  
           unless (length $token->{data}) {  
             !!!next-token;  
           }  
         }  
         while ($token->{type} eq 'character') {  
           $text .= $token->{data};  
4538            !!!next-token;            !!!next-token;
4539              next B;
4540          }          }
4541          if (length $text) {        } elsif ($token->{type} == END_TAG_TOKEN) {
4542            $el->manakai_append_text ($text);          ## NOTE: "using the rules for secondary insertion mode" then "continue"
4543          }          !!!cp ('t87.5');
4544                    #
4545          $self->{content_model_flag} = 'PCDATA';        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
4546                    !!!cp ('t87.6');
4547          if ($token->{type} eq 'end tag' and          !!!parse-error (type => 'not closed',
4548              $token->{tag_name} eq $tag_name) {                          text => $self->{open_elements}->[-1]->[0]
4549            ## Ignore the token                              ->manakai_local_name,
4550          } else {                          token => $token);
4551            !!!parse-error (type => 'in RCDATA:#'.$token->{type});  
4552          }          pop @{$self->{open_elements}}
4553          !!!next-token;              while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;
4554          return;  
4555        } elsif ({          ## NOTE: |<span><svg>| ... two parse errors, |<svg>| ... a parse error.
4556                  iframe => 1,  
4557                  noembed => 1,          $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
4558                  noframes => 1,          ## Reprocess.
4559                  noscript => 0, ## TODO: 1 if scripting is enabled          next B;
                }->{$token->{tag_name}}) {  
         $parse_rcdata->('CDATA', $insert);  
         return;  
       } elsif ($token->{tag_name} eq 'select') {  
         $reconstruct_active_formatting_elements->($insert_to_current);  
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
           
         $self->{insertion_mode} = 'in select';  
         !!!next-token;  
         return;  
       } elsif ({  
                 caption => 1, col => 1, colgroup => 1, frame => 1,  
                 frameset => 1, head => 1, option => 1, optgroup => 1,  
                 tbody => 1, td => 1, tfoot => 1, th => 1,  
                 thead => 1, tr => 1,  
                }->{$token->{tag_name}}) {  
         !!!parse-error (type => 'in body:'.$token->{tag_name});  
         ## Ignore the token  
         !!!next-token;  
         return;  
           
         ## ISSUE: An issue on HTML5 new elements in the spec.  
4560        } else {        } else {
4561          $reconstruct_active_formatting_elements->($insert_to_current);          die "$0: $token->{type}: Unknown token type";        
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
           
         !!!next-token;  
         return;  
4562        }        }
4563      } elsif ($token->{type} eq 'end tag') {      }
       if ($token->{tag_name} eq 'body') {  
         if (@{$self->{open_elements}} > 1 and  
             $self->{open_elements}->[1]->[1] eq 'body') {  
           for (@{$self->{open_elements}}) {  
             unless ({  
                        dd => 1, dt => 1, li => 1, p => 1, td => 1,  
                        th => 1, tr => 1, body => 1, html => 1,  
                      tbody => 1, tfoot => 1, thead => 1,  
                     }->{$_->[1]}) {  
               !!!parse-error (type => 'not closed:'.$_->[1]);  
             }  
           }  
4564    
4565            $self->{insertion_mode} = 'after body';      if ($self->{insertion_mode} & HEAD_IMS) {
4566            !!!next-token;        if ($token->{type} == CHARACTER_TOKEN) {
4567            return;          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
4568          } else {            unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4569            !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});              if ($self->{head_element_inserted}) {
4570            ## Ignore the token                !!!cp ('t88.3');
4571            !!!next-token;                $self->{open_elements}->[-1]->[0]->append_child
4572            return;                  ($self->{document}->create_text_node ($1));
4573          }                delete $self->{head_element_inserted};
4574        } elsif ($token->{tag_name} eq 'html') {                ## NOTE: |</head> <link> |
4575          if (@{$self->{open_elements}} > 1 and $self->{open_elements}->[1]->[1] eq 'body') {                #
4576            ## ISSUE: There is an issue in the spec.              } else {
4577            if ($self->{open_elements}->[-1]->[1] ne 'body') {                !!!cp ('t88.2');
4578              !!!parse-error (type => 'not closed:'.$self->{open_elements}->[1]->[1]);                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
4579            }                ## NOTE: |</head> &#x20;|
4580            $self->{insertion_mode} = 'after body';                #
           ## reprocess  
           return;  
         } else {  
           !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
           ## Ignore the token  
           !!!next-token;  
           return;  
         }  
       } elsif ({  
                 address => 1, blockquote => 1, center => 1, dir => 1,  
                 div => 1, dl => 1, fieldset => 1, listing => 1,  
                 menu => 1, ol => 1, pre => 1, ul => 1,  
                 p => 1,  
                 dd => 1, dt => 1, li => 1,  
                 button => 1, marquee => 1, object => 1,  
                }->{$token->{tag_name}}) {  
         ## has an element in scope  
         my $i;  
         INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
           my $node = $self->{open_elements}->[$_];  
           if ($node->[1] eq $token->{tag_name}) {  
             ## generate implied end tags  
             if ({  
                  dd => ($token->{tag_name} ne 'dd'),  
                  dt => ($token->{tag_name} ne 'dt'),  
                  li => ($token->{tag_name} ne 'li'),  
                  p => ($token->{tag_name} ne 'p'),  
                  td => 1, th => 1, tr => 1,  
                  tbody => 1, tfoot=> 1, thead => 1,  
                 }->{$self->{open_elements}->[-1]->[1]}) {  
               !!!back-token;  
               $token = {type => 'end tag',  
                         tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
               return;  
4581              }              }
             $i = $_;  
             last INSCOPE unless $token->{tag_name} eq 'p';  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
           
         if ($self->{open_elements}->[-1]->[1] ne $token->{tag_name}) {  
           if (defined $i) {  
             !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
4582            } else {            } else {
4583              !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});              !!!cp ('t88.1');
4584                ## Ignore the token.
4585                #
4586            }            }
4587          }            unless (length $token->{data}) {
4588                        !!!cp ('t88');
4589          if (defined $i) {              !!!next-token;
4590            splice @{$self->{open_elements}}, $i;              next B;
         } elsif ($token->{tag_name} eq 'p') {  
           ## As if <p>, then reprocess the current token  
           my $el;  
           !!!create-element ($el, 'p');  
           $insert->($el);  
         }  
         $clear_up_to_marker->()  
           if {  
             button => 1, marquee => 1, object => 1,  
           }->{$token->{tag_name}};  
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'form') {  
         ## has an element in scope  
         INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
           my $node = $self->{open_elements}->[$_];  
           if ($node->[1] eq $token->{tag_name}) {  
             ## generate implied end tags  
             if ({  
                  dd => 1, dt => 1, li => 1, p => 1,  
                  td => 1, th => 1, tr => 1,  
                  tbody => 1, tfoot=> 1, thead => 1,  
                 }->{$self->{open_elements}->[-1]->[1]}) {  
               !!!back-token;  
               $token = {type => 'end tag',  
                         tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
               return;  
             }  
             last INSCOPE;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
             last INSCOPE;  
4591            }            }
4592          } # INSCOPE  ## TODO: set $token->{column} appropriately
           
         if ($self->{open_elements}->[-1]->[1] eq $token->{tag_name}) {  
           pop @{$self->{open_elements}};  
         } else {  
           !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
4593          }          }
4594    
4595          undef $self->{form_element};          if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4596          !!!next-token;            !!!cp ('t89');
4597          return;            ## As if <head>
4598        } elsif ({            !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4599                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,            $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4600                 }->{$token->{tag_name}}) {            push @{$self->{open_elements}},
4601          ## has an element in scope                [$self->{head_element}, $el_category->{head}];
         my $i;  
         INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
           my $node = $self->{open_elements}->[$_];  
           if ({  
                h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,  
               }->{$node->[1]}) {  
             ## generate implied end tags  
             if ({  
                  dd => 1, dt => 1, li => 1, p => 1,  
                  td => 1, th => 1, tr => 1,  
                  tbody => 1, tfoot=> 1, thead => 1,  
                 }->{$self->{open_elements}->[-1]->[1]}) {  
               !!!back-token;  
               $token = {type => 'end tag',  
                         tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
               return;  
             }  
             $i = $_;  
             last INSCOPE;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
           
         if ($self->{open_elements}->[-1]->[1] ne $token->{tag_name}) {  
           !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
         }  
           
         splice @{$self->{open_elements}}, $i if defined $i;  
         !!!next-token;  
         return;  
       } elsif ({  
                 a => 1,  
                 b => 1, big => 1, em => 1, font => 1, i => 1,  
                 nobr => 1, s => 1, small => 1, strile => 1,  
                 strong => 1, tt => 1, u => 1,  
                }->{$token->{tag_name}}) {  
         $formatting_end_tag->($token->{tag_name});  
         return;  
       } elsif ($token->{tag_name} eq 'br') {  
         !!!parse-error (type => 'unmatched end tag:br');  
4602    
4603          ## As if <br>            ## Reprocess in the "in head" insertion mode...
4604          $reconstruct_active_formatting_elements->($insert_to_current);            pop @{$self->{open_elements}};
           
         my $el;  
         !!!create-element ($el, 'br');  
         $insert->($el);  
           
         ## Ignore the token.  
         !!!next-token;  
         return;  
       } elsif ({  
                 caption => 1, col => 1, colgroup => 1, frame => 1,  
                 frameset => 1, head => 1, option => 1, optgroup => 1,  
                 tbody => 1, td => 1, tfoot => 1, th => 1,  
                 thead => 1, tr => 1,  
                 area => 1, basefont => 1, bgsound => 1,  
                 embed => 1, hr => 1, iframe => 1, image => 1,  
                 img => 1, input => 1, isindex => 1, noembed => 1,  
                 noframes => 1, param => 1, select => 1, spacer => 1,  
                 table => 1, textarea => 1, wbr => 1,  
                 noscript => 0, ## TODO: if scripting is enabled  
                }->{$token->{tag_name}}) {  
         !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
         ## Ignore the token  
         !!!next-token;  
         return;  
           
         ## ISSUE: Issue on HTML5 new elements in spec  
           
       } else {  
         ## Step 1  
         my $node_i = -1;  
         my $node = $self->{open_elements}->[$node_i];  
4605    
4606          ## Step 2            ## Reprocess in the "after head" insertion mode...
4607          S2: {          } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4608            if ($node->[1] eq $token->{tag_name}) {            !!!cp ('t90');
4609              ## Step 1            ## As if </noscript>
4610              ## generate implied end tags            pop @{$self->{open_elements}};
4611              if ({            !!!parse-error (type => 'in noscript:#text', token => $token);
4612                   dd => 1, dt => 1, li => 1, p => 1,            
4613                   td => 1, th => 1, tr => 1,            ## Reprocess in the "in head" insertion mode...
4614                   tbody => 1, tfoot=> 1, thead => 1,            ## As if </head>
4615                  }->{$self->{open_elements}->[-1]->[1]}) {            pop @{$self->{open_elements}};
               !!!back-token;  
               $token = {type => 'end tag',  
                         tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
               return;  
             }  
           
             ## Step 2  
             if ($token->{tag_name} ne $self->{open_elements}->[-1]->[1]) {  
               !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
             }  
               
             ## Step 3  
             splice @{$self->{open_elements}}, $node_i;  
4616    
4617              ## Reprocess in the "after head" insertion mode...
4618            } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4619              !!!cp ('t91');
4620              pop @{$self->{open_elements}};
4621    
4622              ## Reprocess in the "after head" insertion mode...
4623            } else {
4624              !!!cp ('t92');
4625            }
4626    
4627            ## "after head" insertion mode
4628            ## As if <body>
4629            !!!insert-element ('body',, $token);
4630            $self->{insertion_mode} = IN_BODY_IM;
4631            ## reprocess
4632            next B;
4633          } elsif ($token->{type} == START_TAG_TOKEN) {
4634            if ($token->{tag_name} eq 'head') {
4635              if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4636                !!!cp ('t93');
4637                !!!create-element ($self->{head_element}, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);
4638                $self->{open_elements}->[-1]->[0]->append_child
4639                    ($self->{head_element});
4640                push @{$self->{open_elements}},
4641                    [$self->{head_element}, $el_category->{head}];
4642                $self->{insertion_mode} = IN_HEAD_IM;
4643                !!!nack ('t93.1');
4644              !!!next-token;              !!!next-token;
4645              last S2;              next B;
4646              } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4647                !!!cp ('t93.2');
4648                !!!parse-error (type => 'after head', text => 'head',
4649                                token => $token);
4650                ## Ignore the token
4651                !!!nack ('t93.3');
4652                !!!next-token;
4653                next B;
4654            } else {            } else {
4655              ## Step 3              !!!cp ('t95');
4656              if (not $formatting_category->{$node->[1]} and              !!!parse-error (type => 'in head:head',
4657                  #not $phrasing_category->{$node->[1]} and                              token => $token); # or in head noscript
4658                  ($special_category->{$node->[1]} or              ## Ignore the token
4659                   $scoping_category->{$node->[1]})) {              !!!nack ('t95.1');
4660                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});              !!!next-token;
4661                ## Ignore the token              next B;
               !!!next-token;  
               last S2;  
             }  
4662            }            }
4663                      } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4664            ## Step 4            !!!cp ('t96');
4665            $node_i--;            ## As if <head>
4666            $node = $self->{open_elements}->[$node_i];            !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4667                        $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4668            ## Step 5;            push @{$self->{open_elements}},
4669            redo S2;                [$self->{head_element}, $el_category->{head}];
         } # S2  
         return;  
       }  
     }  
   }; # $in_body  
4670    
4671    B: {            $self->{insertion_mode} = IN_HEAD_IM;
4672      if ($self->{insertion_mode} ne 'trailing end') {            ## Reprocess in the "in head" insertion mode...
4673        if ($token->{type} eq 'DOCTYPE') {          } else {
4674          !!!parse-error (type => 'in html:#DOCTYPE');            !!!cp ('t97');
         ## Ignore the token  
         ## Stay in the phase  
         !!!next-token;  
         redo B;  
       } elsif ($token->{type} eq 'start tag' and  
                $token->{tag_name} eq 'html') {  
 ## ISSUE: "aa<html>" is not a parse error.  
 ## ISSUE: "<html>" in fragment is not a parse error.  
         unless ($token->{first_start_tag}) {  
           !!!parse-error (type => 'not first start tag');  
         }  
         my $top_el = $self->{open_elements}->[0]->[0];  
         for my $attr_name (keys %{$token->{attributes}}) {  
           unless ($top_el->has_attribute_ns (undef, $attr_name)) {  
             $top_el->set_attribute_ns  
               (undef, [undef, $attr_name],  
                $token->{attributes}->{$attr_name}->{value});  
           }  
         }  
         !!!next-token;  
         redo B;  
       } elsif ($token->{type} eq 'end-of-file') {  
         ## Generate implied end tags  
         if ({  
              dd => 1, dt => 1, li => 1, p => 1, td => 1, th => 1, tr => 1,  
              tbody => 1, tfoot=> 1, thead => 1,  
             }->{$self->{open_elements}->[-1]->[1]}) {  
           !!!back-token;  
           $token = {type => 'end tag', tag_name => $self->{open_elements}->[-1]->[1]};  
           redo B;  
         }  
           
         if (@{$self->{open_elements}} > 2 or  
             (@{$self->{open_elements}} == 2 and $self->{open_elements}->[1]->[1] ne 'body')) {  
           !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
         } elsif (defined $self->{inner_html_node} and  
                  @{$self->{open_elements}} > 1 and  
                  $self->{open_elements}->[1]->[1] ne 'body') {  
           !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
4675          }          }
4676    
4677          ## Stop parsing          if ($token->{tag_name} eq 'base') {
4678          last B;            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4679                !!!cp ('t98');
4680                ## As if </noscript>
4681                pop @{$self->{open_elements}};
4682                !!!parse-error (type => 'in noscript', text => 'base',
4683                                token => $token);
4684              
4685                $self->{insertion_mode} = IN_HEAD_IM;
4686                ## Reprocess in the "in head" insertion mode...
4687              } else {
4688                !!!cp ('t99');
4689              }
4690    
4691          ## ISSUE: There is an issue in the spec.            ## NOTE: There is a "as if in head" code clone.
4692        } else {            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4693          if ($self->{insertion_mode} eq 'before head') {              !!!cp ('t100');
4694            if ($token->{type} eq 'character') {              !!!parse-error (type => 'after head',
4695              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {                              text => $token->{tag_name}, token => $token);
4696                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);              push @{$self->{open_elements}},
4697                unless (length $token->{data}) {                  [$self->{head_element}, $el_category->{head}];
4698                  !!!next-token;              $self->{head_element_inserted} = 1;
                 redo B;  
               }  
             }  
             ## As if <head>  
             !!!create-element ($self->{head_element}, 'head');  
             $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});  
             push @{$self->{open_elements}}, [$self->{head_element}, 'head'];  
             $self->{insertion_mode} = 'in head';  
             ## reprocess  
             redo B;  
           } elsif ($token->{type} eq 'comment') {  
             my $comment = $self->{document}->create_comment ($token->{data});  
             $self->{open_elements}->[-1]->[0]->append_child ($comment);  
             !!!next-token;  
             redo B;  
           } elsif ($token->{type} eq 'start tag') {  
             my $attr = $token->{tag_name} eq 'head' ? $token->{attributes} : {};  
             !!!create-element ($self->{head_element}, 'head', $attr);  
             $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});  
             push @{$self->{open_elements}}, [$self->{head_element}, 'head'];  
             $self->{insertion_mode} = 'in head';  
             if ($token->{tag_name} eq 'head') {  
               !!!next-token;  
             #} elsif ({  
             #          base => 1, link => 1, meta => 1,  
             #          script => 1, style => 1, title => 1,  
             #         }->{$token->{tag_name}}) {  
             #  ## reprocess  
             } else {  
               ## reprocess  
             }  
             redo B;  
           } elsif ($token->{type} eq 'end tag') {  
             if ({  
                  head => 1, body => 1, html => 1,  
                  p => 1, br => 1,  
                 }->{$token->{tag_name}}) {  
               ## As if <head>  
               !!!create-element ($self->{head_element}, 'head');  
               $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});  
               push @{$self->{open_elements}}, [$self->{head_element}, 'head'];  
               $self->{insertion_mode} = 'in head';  
               ## reprocess  
               redo B;  
             } else {  
               !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
               ## Ignore the token ## ISSUE: An issue in the spec.  
               !!!next-token;  
               redo B;  
             }  
4699            } else {            } else {
4700              die "$0: $token->{type}: Unknown type";              !!!cp ('t101');
4701            }            }
4702          } elsif ($self->{insertion_mode} eq 'in head' or            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4703                   $self->{insertion_mode} eq 'in head noscript' or            pop @{$self->{open_elements}};
4704                   $self->{insertion_mode} eq 'after head') {            pop @{$self->{open_elements}} # <head>
4705            if ($token->{type} eq 'character') {                if $self->{insertion_mode} == AFTER_HEAD_IM;
4706              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {            !!!nack ('t101.1');
4707                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);            !!!next-token;
4708                unless (length $token->{data}) {            next B;
4709                  !!!next-token;          } elsif ($token->{tag_name} eq 'link') {
4710                  redo B;            ## NOTE: There is a "as if in head" code clone.
4711                }            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4712              }              !!!cp ('t102');
4713                            !!!parse-error (type => 'after head',
4714              #                              text => $token->{tag_name}, token => $token);
4715            } elsif ($token->{type} eq 'comment') {              push @{$self->{open_elements}},
4716              my $comment = $self->{document}->create_comment ($token->{data});                  [$self->{head_element}, $el_category->{head}];
4717              $self->{open_elements}->[-1]->[0]->append_child ($comment);              $self->{head_element_inserted} = 1;
4718              } else {
4719                !!!cp ('t103');
4720              }
4721              !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4722              pop @{$self->{open_elements}};
4723              pop @{$self->{open_elements}} # <head>
4724                  if $self->{insertion_mode} == AFTER_HEAD_IM;
4725              !!!ack ('t103.1');
4726              !!!next-token;
4727              next B;
4728            } elsif ($token->{tag_name} eq 'command' or
4729                     $token->{tag_name} eq 'eventsource') {
4730              if ($self->{insertion_mode} == IN_HEAD_IM) {
4731                ## NOTE: If the insertion mode at the time of the emission
4732                ## of the token was "before head", $self->{insertion_mode}
4733                ## is already changed to |IN_HEAD_IM|.
4734    
4735                ## NOTE: There is a "as if in head" code clone.
4736                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4737                pop @{$self->{open_elements}};
4738                pop @{$self->{open_elements}} # <head>
4739                    if $self->{insertion_mode} == AFTER_HEAD_IM;
4740                !!!ack ('t103.2');
4741              !!!next-token;              !!!next-token;
4742              redo B;              next B;
4743            } elsif ($token->{type} eq 'start tag') {            } else {
4744              if ({base => ($self->{insertion_mode} eq 'in head' or              ## NOTE: "in head noscript" or "after head" insertion mode
4745                            $self->{insertion_mode} eq 'after head'),              ## - in these cases, these tags are treated as same as
4746                   link => 1}->{$token->{tag_name}}) {              ## normal in-body tags.
4747                ## NOTE: There is a "as if in head" code clone.              !!!cp ('t103.3');
4748                if ($self->{insertion_mode} eq 'after head') {              #
4749                  !!!parse-error (type => 'after head:'.$token->{tag_name});            }
4750                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];          } elsif ($token->{tag_name} eq 'meta') {
4751                }            ## NOTE: There is a "as if in head" code clone.
4752                !!!insert-element ($token->{tag_name}, $token->{attributes});            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4753                pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.              !!!cp ('t104');
4754                pop @{$self->{open_elements}}              !!!parse-error (type => 'after head',
4755                    if $self->{insertion_mode} eq 'after head';                              text => $token->{tag_name}, token => $token);
4756                !!!next-token;              push @{$self->{open_elements}},
4757                redo B;                  [$self->{head_element}, $el_category->{head}];
4758              } elsif ($token->{tag_name} eq 'meta') {              $self->{head_element_inserted} = 1;
4759                ## NOTE: There is a "as if in head" code clone.            } else {
4760                if ($self->{insertion_mode} eq 'after head') {              !!!cp ('t105');
4761                  !!!parse-error (type => 'after head:'.$token->{tag_name});            }
4762                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4763                }            my $meta_el = pop @{$self->{open_elements}};
               !!!insert-element ($token->{tag_name}, $token->{attributes});  
               pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.  
4764    
4765                unless ($self->{confident}) {                unless ($self->{confident}) {
4766                  my $charset;                  if ($token->{attributes}->{charset}) {
4767                  if ($token->{attributes}->{charset}) { ## TODO: And if supported                    !!!cp ('t106');
4768                    $charset = $token->{attributes}->{charset}->{value};                    ## NOTE: Whether the encoding is supported or not is handled
4769                      ## in the {change_encoding} callback.
4770                      $self->{change_encoding}
4771                          ->($self, $token->{attributes}->{charset}->{value},
4772                             $token);
4773                      
4774                      $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
4775                          ->set_user_data (manakai_has_reference =>
4776                                               $token->{attributes}->{charset}
4777                                                   ->{has_reference});
4778                    } elsif ($token->{attributes}->{content}) {
4779                      if ($token->{attributes}->{content}->{value}
4780                          =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
4781                              [\x09\x0A\x0C\x0D\x20]*=
4782                              [\x09\x0A\x0C\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
4783                              ([^"'\x09\x0A\x0C\x0D\x20]
4784                               [^\x09\x0A\x0C\x0D\x20\x3B]*))/x) {
4785                        !!!cp ('t107');
4786                        ## NOTE: Whether the encoding is supported or not is handled
4787                        ## in the {change_encoding} callback.
4788                        $self->{change_encoding}
4789                            ->($self, defined $1 ? $1 : defined $2 ? $2 : $3,
4790                               $token);
4791                        $meta_el->[0]->get_attribute_node_ns (undef, 'content')
4792                            ->set_user_data (manakai_has_reference =>
4793                                                 $token->{attributes}->{content}
4794                                                       ->{has_reference});
4795                      } else {
4796                        !!!cp ('t108');
4797                      }
4798                    }
4799                  } else {
4800                    if ($token->{attributes}->{charset}) {
4801                      !!!cp ('t109');
4802                      $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
4803                          ->set_user_data (manakai_has_reference =>
4804                                               $token->{attributes}->{charset}
4805                                                   ->{has_reference});
4806                  }                  }
4807                  if ($token->{attributes}->{'http-equiv'}) {                  if ($token->{attributes}->{content}) {
4808                    ## ISSUE: Algorithm name in the spec was incorrect so that not linked to the definition.                    !!!cp ('t110');
4809                    if ($token->{attributes}->{'http-equiv'}->{value}                    $meta_el->[0]->get_attribute_node_ns (undef, 'content')
4810                        =~ /\A[^;]*;[\x09-\x0D\x20]*charset[\x09-\x0D\x20]*=                        ->set_user_data (manakai_has_reference =>
4811                            [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|                                             $token->{attributes}->{content}
4812                            ([^"'\x09-\x0D\x20][^\x09-\x0D\x20]*))/x) {                                                 ->{has_reference});
                     $charset = defined $1 ? $1 : defined $2 ? $2 : $3;  
                   } ## TODO: And if supported  
4813                  }                  }
                 ## TODO: Change the encoding  
4814                }                }
4815    
4816                ## TODO: Extracting |charset| from |meta|.                pop @{$self->{open_elements}} # <head>
4817                pop @{$self->{open_elements}}                    if $self->{insertion_mode} == AFTER_HEAD_IM;
4818                    if $self->{insertion_mode} eq 'after head';                !!!ack ('t110.1');
4819                !!!next-token;                !!!next-token;
4820                redo B;                next B;
4821              } elsif ($token->{tag_name} eq 'title' and          } elsif ($token->{tag_name} eq 'title') {
4822                       $self->{insertion_mode} eq 'in head') {            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4823                ## NOTE: There is a "as if in head" code clone.              !!!cp ('t111');
4824                if ($self->{insertion_mode} eq 'after head') {              ## As if </noscript>
4825                  !!!parse-error (type => 'after head:'.$token->{tag_name});              pop @{$self->{open_elements}};
4826                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];              !!!parse-error (type => 'in noscript', text => 'title',
4827                }                              token => $token);
4828                my $parent = defined $self->{head_element} ? $self->{head_element}            
4829                    : $self->{open_elements}->[-1]->[0];              $self->{insertion_mode} = IN_HEAD_IM;
4830                $parse_rcdata->('RCDATA', sub { $parent->append_child ($_[0]) });              ## Reprocess in the "in head" insertion mode...
4831                pop @{$self->{open_elements}}            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4832                    if $self->{insertion_mode} eq 'after head';              !!!cp ('t112');
4833                redo B;              !!!parse-error (type => 'after head',
4834              } elsif ($token->{tag_name} eq 'style') {                              text => $token->{tag_name}, token => $token);
4835                ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and              push @{$self->{open_elements}},
4836                ## insertion mode 'in head')                  [$self->{head_element}, $el_category->{head}];
4837                ## NOTE: There is a "as if in head" code clone.              $self->{head_element_inserted} = 1;
4838                if ($self->{insertion_mode} eq 'after head') {            } else {
4839                  !!!parse-error (type => 'after head:'.$token->{tag_name});              !!!cp ('t113');
4840                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];            }
4841                }  
4842                $parse_rcdata->('CDATA', $insert_to_current);            ## NOTE: There is a "as if in head" code clone.
4843                pop @{$self->{open_elements}}            $parse_rcdata->(RCDATA_CONTENT_MODEL);
4844                    if $self->{insertion_mode} eq 'after head';            ## ISSUE: A spec bug [Bug 6038]
4845                redo B;            splice @{$self->{open_elements}}, -2, 1, () # <head>
4846              } elsif ($token->{tag_name} eq 'noscript') {                if ($self->{insertion_mode} & AFTER_HEAD_IM) == AFTER_HEAD_IM;
4847                if ($self->{insertion_mode} eq 'in head') {            next B;
4848            } elsif ($token->{tag_name} eq 'style' or
4849                     $token->{tag_name} eq 'noframes') {
4850              ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and
4851              ## insertion mode IN_HEAD_IM)
4852              ## NOTE: There is a "as if in head" code clone.
4853              if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4854                !!!cp ('t114');
4855                !!!parse-error (type => 'after head',
4856                                text => $token->{tag_name}, token => $token);
4857                push @{$self->{open_elements}},
4858                    [$self->{head_element}, $el_category->{head}];
4859                $self->{head_element_inserted} = 1;
4860              } else {
4861                !!!cp ('t115');
4862              }
4863              $parse_rcdata->(CDATA_CONTENT_MODEL);
4864              ## ISSUE: A spec bug [Bug 6038]
4865              splice @{$self->{open_elements}}, -2, 1, () # <head>
4866                  if ($self->{insertion_mode} & AFTER_HEAD_IM) == AFTER_HEAD_IM;
4867              next B;
4868            } elsif ($token->{tag_name} eq 'noscript') {
4869                  if ($self->{insertion_mode} == IN_HEAD_IM) {
4870                    !!!cp ('t116');
4871                  ## NOTE: and scripting is disalbed                  ## NOTE: and scripting is disalbed
4872                  !!!insert-element ($token->{tag_name}, $token->{attributes});                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4873                  $self->{insertion_mode} = 'in head noscript';                  $self->{insertion_mode} = IN_HEAD_NOSCRIPT_IM;
4874                    !!!nack ('t116.1');
4875                  !!!next-token;                  !!!next-token;
4876                  redo B;                  next B;
4877                } elsif ($self->{insertion_mode} eq 'in head noscript') {                } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4878                  !!!parse-error (type => 'in noscript:noscript');                  !!!cp ('t117');
4879                    !!!parse-error (type => 'in noscript', text => 'noscript',
4880                                    token => $token);
4881                  ## Ignore the token                  ## Ignore the token
4882                  redo B;                  !!!nack ('t117.1');
4883                    !!!next-token;
4884                    next B;
4885                } else {                } else {
4886                    !!!cp ('t118');
4887                  #                  #
4888                }                }
4889              } elsif ($token->{tag_name} eq 'head' and          } elsif ($token->{tag_name} eq 'script') {
4890                       $self->{insertion_mode} ne 'after head') {            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4891                !!!parse-error (type => 'in head:head'); # or in head noscript              !!!cp ('t119');
4892                ## Ignore the token              ## As if </noscript>
4893                !!!next-token;              pop @{$self->{open_elements}};
4894                redo B;              !!!parse-error (type => 'in noscript', text => 'script',
4895              } elsif ($self->{insertion_mode} ne 'in head noscript' and                              token => $token);
4896                       $token->{tag_name} eq 'script') {            
4897                if ($self->{insertion_mode} eq 'after head') {              $self->{insertion_mode} = IN_HEAD_IM;
4898                  !!!parse-error (type => 'after head:'.$token->{tag_name});              ## Reprocess in the "in head" insertion mode...
4899                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4900                !!!cp ('t120');
4901                !!!parse-error (type => 'after head',
4902                                text => $token->{tag_name}, token => $token);
4903                push @{$self->{open_elements}},
4904                    [$self->{head_element}, $el_category->{head}];
4905                $self->{head_element_inserted} = 1;
4906              } else {
4907                !!!cp ('t121');
4908              }
4909    
4910              ## NOTE: There is a "as if in head" code clone.
4911              $script_start_tag->();
4912              ## ISSUE: A spec bug  [Bug 6038]
4913              splice @{$self->{open_elements}}, -2, 1 # <head>
4914                  if ($self->{insertion_mode} & AFTER_HEAD_IM) == AFTER_HEAD_IM;
4915              next B;
4916            } elsif ($token->{tag_name} eq 'body' or
4917                     $token->{tag_name} eq 'frameset') {
4918                  if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4919                    !!!cp ('t122');
4920                    ## As if </noscript>
4921                    pop @{$self->{open_elements}};
4922                    !!!parse-error (type => 'in noscript',
4923                                    text => $token->{tag_name}, token => $token);
4924                    
4925                    ## Reprocess in the "in head" insertion mode...
4926                    ## As if </head>
4927                    pop @{$self->{open_elements}};
4928                    
4929                    ## Reprocess in the "after head" insertion mode...
4930                  } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4931                    !!!cp ('t124');
4932                    pop @{$self->{open_elements}};
4933                    
4934                    ## Reprocess in the "after head" insertion mode...
4935                  } else {
4936                    !!!cp ('t125');
4937                }                }
4938                ## NOTE: There is a "as if in head" code clone.  
4939                $script_start_tag->($insert_to_current);                ## "after head" insertion mode
4940                pop @{$self->{open_elements}}                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4941                    if $self->{insertion_mode} eq 'after head';                if ($token->{tag_name} eq 'body') {
4942                redo B;                  !!!cp ('t126');
4943              } elsif ($self->{insertion_mode} eq 'after head' and                  $self->{insertion_mode} = IN_BODY_IM;
4944                       $token->{tag_name} eq 'body') {                } elsif ($token->{tag_name} eq 'frameset') {
4945                !!!insert-element ('body', $token->{attributes});                  !!!cp ('t127');
4946                $self->{insertion_mode} = 'in body';                  $self->{insertion_mode} = IN_FRAMESET_IM;
4947                !!!next-token;                } else {
4948                redo B;                  die "$0: tag name: $self->{tag_name}";
4949              } elsif ($self->{insertion_mode} eq 'after head' and                }
4950                       $token->{tag_name} eq 'frameset') {                !!!nack ('t127.1');
               !!!insert-element ('frameset', $token->{attributes});  
               $self->{insertion_mode} = 'in frameset';  
4951                !!!next-token;                !!!next-token;
4952                redo B;                next B;
4953              } else {              } else {
4954                  !!!cp ('t128');
4955                #                #
4956              }              }
4957            } elsif ($token->{type} eq 'end tag') {  
4958              if ($self->{insertion_mode} eq 'in head' and              if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4959                  $token->{tag_name} eq 'head') {                !!!cp ('t129');
4960                  ## As if </noscript>
4961                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
4962                $self->{insertion_mode} = 'after head';                !!!parse-error (type => 'in noscript:/',
4963                !!!next-token;                                text => $token->{tag_name}, token => $token);
4964                redo B;                
4965              } elsif ($self->{insertion_mode} eq 'in head noscript' and                ## Reprocess in the "in head" insertion mode...
4966                  $token->{tag_name} eq 'noscript') {                ## As if </head>
4967                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
               $self->{insertion_mode} = 'in head';  
               !!!next-token;  
               redo B;  
             } elsif ($self->{insertion_mode} eq 'in head' and  
                      {  
                       body => 1, html => 1,  
                       p => 1, br => 1,  
                      }->{$token->{tag_name}}) {  
               #  
             } elsif ($self->{insertion_mode} eq 'in head noscript' and  
                      {  
                       p => 1, br => 1,  
                      }->{$token->{tag_name}}) {  
               #  
             } elsif ($self->{insertion_mode} ne 'after head') {  
               !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
               ## Ignore the token  
               !!!next-token;  
               redo B;  
             } else {  
               #  
             }  
           } else {  
             #  
           }  
4968    
4969            ## As if </head> or </noscript> or <body>                ## Reprocess in the "after head" insertion mode...
4970            if ($self->{insertion_mode} eq 'in head') {              } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4971              pop @{$self->{open_elements}};                !!!cp ('t130');
4972              $self->{insertion_mode} = 'after head';                ## As if </head>
4973            } elsif ($self->{insertion_mode} eq 'in head noscript') {                pop @{$self->{open_elements}};
             pop @{$self->{open_elements}};  
             !!!parse-error (type => 'in noscript:'.(defined $token->{tag_name} ? ($token->{type} eq 'end tag' ? '/' : '') . $token->{tag_name} : '#' . $token->{type}));  
             $self->{insertion_mode} = 'in head';  
           } else { # 'after head'  
             !!!insert-element ('body');  
             $self->{insertion_mode} = 'in body';  
           }  
           ## reprocess  
           redo B;  
   
           ## ISSUE: An issue in the spec.  
         } elsif ($self->{insertion_mode} eq 'in body') {  
           if ($token->{type} eq 'character') {  
             ## NOTE: There is a code clone of "character in body".  
             $reconstruct_active_formatting_elements->($insert_to_current);  
               
             $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});  
4974    
4975              !!!next-token;                ## Reprocess in the "after head" insertion mode...
4976              redo B;              } else {
4977            } elsif ($token->{type} eq 'comment') {                !!!cp ('t131');
             ## NOTE: There is a code clone of "comment in body".  
             my $comment = $self->{document}->create_comment ($token->{data});  
             $self->{open_elements}->[-1]->[0]->append_child ($comment);  
             !!!next-token;  
             redo B;  
           } else {  
             $in_body->($insert_to_current);  
             redo B;  
           }  
         } elsif ($self->{insertion_mode} eq 'in table') {  
           if ($token->{type} eq 'character') {  
             ## NOTE: There are "character in table" code clones.  
             if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {  
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);  
                 
               unless (length $token->{data}) {  
                 !!!next-token;  
                 redo B;  
               }  
4978              }              }
4979    
4980              !!!parse-error (type => 'in table:#character');              ## "after head" insertion mode
4981                ## As if <body>
4982                !!!insert-element ('body',, $token);
4983                $self->{insertion_mode} = IN_BODY_IM;
4984                ## reprocess
4985                !!!ack-later;
4986                next B;
4987              } elsif ($token->{type} == END_TAG_TOKEN) {
4988                if ($token->{tag_name} eq 'head') {
4989                  if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4990                    !!!cp ('t132');
4991                    ## As if <head>
4992                    !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4993                    $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4994                    push @{$self->{open_elements}},
4995                        [$self->{head_element}, $el_category->{head}];
4996    
4997              ## As if in body, but insert into foster parent element                  ## Reprocess in the "in head" insertion mode...
4998              ## ISSUE: Spec says that "whenever a node would be inserted                  pop @{$self->{open_elements}};
4999              ## into the current node" while characters might not be                  $self->{insertion_mode} = AFTER_HEAD_IM;
5000              ## result in a new Text node.                  !!!next-token;
5001              $reconstruct_active_formatting_elements->($insert_to_foster);                  next B;
5002                              } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5003              if ({                  !!!cp ('t133');
5004                   table => 1, tbody => 1, tfoot => 1,                  ## As if </noscript>
5005                   thead => 1, tr => 1,                  pop @{$self->{open_elements}};
5006                  }->{$self->{open_elements}->[-1]->[1]}) {                  !!!parse-error (type => 'in noscript:/',
5007                # MUST                                  text => 'head', token => $token);
5008                my $foster_parent_element;                  
5009                my $next_sibling;                  ## Reprocess in the "in head" insertion mode...
5010                my $prev_sibling;                  pop @{$self->{open_elements}};
5011                OE: for (reverse 0..$#{$self->{open_elements}}) {                  $self->{insertion_mode} = AFTER_HEAD_IM;
5012                  if ($self->{open_elements}->[$_]->[1] eq 'table') {                  !!!next-token;
5013                    my $parent = $self->{open_elements}->[$_]->[0]->parent_node;                  next B;
5014                    if (defined $parent and $parent->node_type == 1) {                } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
5015                      $foster_parent_element = $parent;                  !!!cp ('t134');
5016                      $next_sibling = $self->{open_elements}->[$_]->[0];                  pop @{$self->{open_elements}};
5017                      $prev_sibling = $next_sibling->previous_sibling;                  $self->{insertion_mode} = AFTER_HEAD_IM;
5018                    } else {                  !!!next-token;
5019                      $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];                  next B;
5020                      $prev_sibling = $foster_parent_element->last_child;                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
5021                    }                  !!!cp ('t134.1');
5022                    last OE;                  !!!parse-error (type => 'unmatched end tag', text => 'head',
5023                  }                                  token => $token);
5024                } # OE                  ## Ignore the token
5025                $foster_parent_element = $self->{open_elements}->[0]->[0] and                  !!!next-token;
5026                $prev_sibling = $foster_parent_element->last_child                  next B;
                 unless defined $foster_parent_element;  
               if (defined $prev_sibling and  
                   $prev_sibling->node_type == 3) {  
                 $prev_sibling->manakai_append_text ($token->{data});  
5027                } else {                } else {
5028                  $foster_parent_element->insert_before                  die "$0: $self->{insertion_mode}: Unknown insertion mode";
                   ($self->{document}->create_text_node ($token->{data}),  
                    $next_sibling);  
5029                }                }
5030              } else {              } elsif ($token->{tag_name} eq 'noscript') {
5031                $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5032              }                  !!!cp ('t136');
               
             !!!next-token;  
             redo B;  
           } elsif ($token->{type} eq 'comment') {  
             my $comment = $self->{document}->create_comment ($token->{data});  
             $self->{open_elements}->[-1]->[0]->append_child ($comment);  
             !!!next-token;  
             redo B;  
           } elsif ($token->{type} eq 'start tag') {  
             if ({  
                  caption => 1,  
                  colgroup => 1,  
                  tbody => 1, tfoot => 1, thead => 1,  
                 }->{$token->{tag_name}}) {  
               ## Clear back to table context  
               while ($self->{open_elements}->[-1]->[1] ne 'table' and  
                      $self->{open_elements}->[-1]->[1] ne 'html') {  
                 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
5033                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
5034                    $self->{insertion_mode} = IN_HEAD_IM;
5035                    !!!next-token;
5036                    next B;
5037                  } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM or
5038                           $self->{insertion_mode} == AFTER_HEAD_IM) {
5039                    !!!cp ('t137');
5040                    !!!parse-error (type => 'unmatched end tag',
5041                                    text => 'noscript', token => $token);
5042                    ## Ignore the token ## ISSUE: An issue in the spec.
5043                    !!!next-token;
5044                    next B;
5045                  } else {
5046                    !!!cp ('t138');
5047                    #
5048                }                }
   
               push @$active_formatting_elements, ['#marker', '']  
                 if $token->{tag_name} eq 'caption';  
   
               !!!insert-element ($token->{tag_name}, $token->{attributes});  
               $self->{insertion_mode} = {  
                                  caption => 'in caption',  
                                  colgroup => 'in column group',  
                                  tbody => 'in table body',  
                                  tfoot => 'in table body',  
                                  thead => 'in table body',  
                                 }->{$token->{tag_name}};  
               !!!next-token;  
               redo B;  
5049              } elsif ({              } elsif ({
5050                        col => 1,                        body => 1, html => 1,
                       td => 1, th => 1, tr => 1,  
5051                       }->{$token->{tag_name}}) {                       }->{$token->{tag_name}}) {
5052                ## Clear back to table context                ## TODO: This branch is entirely redundant.
5053                while ($self->{open_elements}->[-1]->[1] ne 'table' and                if ($self->{insertion_mode} == BEFORE_HEAD_IM or
5054                       $self->{open_elements}->[-1]->[1] ne 'html') {                    $self->{insertion_mode} == IN_HEAD_IM or
5055                  !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                    $self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5056                  pop @{$self->{open_elements}};                  !!!cp ('t140');
5057                    !!!parse-error (type => 'unmatched end tag',
5058                                    text => $token->{tag_name}, token => $token);
5059                    ## Ignore the token
5060                    !!!next-token;
5061                    next B;
5062                  } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
5063                    !!!cp ('t140.1');
5064                    !!!parse-error (type => 'unmatched end tag',
5065                                    text => $token->{tag_name}, token => $token);
5066                    ## Ignore the token
5067                    !!!next-token;
5068                    next B;
5069                  } else {
5070                    die "$0: $self->{insertion_mode}: Unknown insertion mode";
5071                }                }
5072                } elsif ($token->{tag_name} eq 'p') {
5073                  !!!cp ('t142');
5074                  !!!parse-error (type => 'unmatched end tag',
5075                                  text => $token->{tag_name}, token => $token);
5076                  ## Ignore the token
5077                  !!!next-token;
5078                  next B;
5079                } elsif ($token->{tag_name} eq 'br') {
5080                  if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
5081                    !!!cp ('t142.2');
5082                    ## (before head) as if <head>, (in head) as if </head>
5083                    !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
5084                    $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
5085                    $self->{insertion_mode} = AFTER_HEAD_IM;
5086      
5087                    ## Reprocess in the "after head" insertion mode...
5088                  } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
5089                    !!!cp ('t143.2');
5090                    ## As if </head>
5091                    pop @{$self->{open_elements}};
5092                    $self->{insertion_mode} = AFTER_HEAD_IM;
5093      
5094                    ## Reprocess in the "after head" insertion mode...
5095                  } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5096                    !!!cp ('t143.3');
5097                    ## ISSUE: Two parse errors for <head><noscript></br>
5098                    !!!parse-error (type => 'unmatched end tag',
5099                                    text => 'br', token => $token);
5100                    ## As if </noscript>
5101                    pop @{$self->{open_elements}};
5102                    $self->{insertion_mode} = IN_HEAD_IM;
5103    
5104                !!!insert-element ($token->{tag_name} eq 'col' ? 'colgroup' : 'tbody');                  ## Reprocess in the "in head" insertion mode...
5105                $self->{insertion_mode} = $token->{tag_name} eq 'col'                  ## As if </head>
5106                  ? 'in column group' : 'in table body';                  pop @{$self->{open_elements}};
5107                ## reprocess                  $self->{insertion_mode} = AFTER_HEAD_IM;
               redo B;  
             } elsif ($token->{tag_name} eq 'table') {  
               ## NOTE: There are code clones for this "table in table"  
               !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
5108    
5109                ## As if </table>                  ## Reprocess in the "after head" insertion mode...
5110                ## have a table element in table scope                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
5111                my $i;                  !!!cp ('t143.4');
5112                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  #
5113                  my $node = $self->{open_elements}->[$_];                } else {
5114                  if ($node->[1] eq 'table') {                  die "$0: $self->{insertion_mode}: Unknown insertion mode";
                   $i = $_;  
                   last INSCOPE;  
                 } elsif ({  
                           table => 1, html => 1,  
                          }->{$node->[1]}) {  
                   last INSCOPE;  
                 }  
               } # INSCOPE  
               unless (defined $i) {  
                 !!!parse-error (type => 'unmatched end tag:table');  
                 ## Ignore tokens </table><table>  
                 !!!next-token;  
                 redo B;  
               }  
                 
               ## generate implied end tags  
               if ({  
                    dd => 1, dt => 1, li => 1, p => 1,  
                    td => 1, th => 1, tr => 1,  
                    tbody => 1, tfoot=> 1, thead => 1,  
                   }->{$self->{open_elements}->[-1]->[1]}) {  
                 !!!back-token; # <table>  
                 $token = {type => 'end tag', tag_name => 'table'};  
                 !!!back-token;  
                 $token = {type => 'end tag',  
                           tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
                 redo B;  
5115                }                }
5116    
5117                if ($self->{open_elements}->[-1]->[1] ne 'table') {                ## ISSUE: does not agree with IE7 - it doesn't ignore </br>.
5118                  !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                !!!parse-error (type => 'unmatched end tag',
5119                }                                text => 'br', token => $token);
5120                  ## Ignore the token
5121                  !!!next-token;
5122                  next B;
5123                } else {
5124                  !!!cp ('t145');
5125                  !!!parse-error (type => 'unmatched end tag',
5126                                  text => $token->{tag_name}, token => $token);
5127                  ## Ignore the token
5128                  !!!next-token;
5129                  next B;
5130                }
5131    
5132                splice @{$self->{open_elements}}, $i;              if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5133                  !!!cp ('t146');
5134                  ## As if </noscript>
5135                  pop @{$self->{open_elements}};
5136                  !!!parse-error (type => 'in noscript:/',
5137                                  text => $token->{tag_name}, token => $token);
5138                  
5139                  ## Reprocess in the "in head" insertion mode...
5140                  ## As if </head>
5141                  pop @{$self->{open_elements}};
5142    
5143                $self->_reset_insertion_mode;                ## Reprocess in the "after head" insertion mode...
5144                } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
5145                  !!!cp ('t147');
5146                  ## As if </head>
5147                  pop @{$self->{open_elements}};
5148    
5149                ## reprocess                ## Reprocess in the "after head" insertion mode...
5150                redo B;              } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {
5151    ## ISSUE: This case cannot be reached?
5152                  !!!cp ('t148');
5153                  !!!parse-error (type => 'unmatched end tag',
5154                                  text => $token->{tag_name}, token => $token);
5155                  ## Ignore the token ## ISSUE: An issue in the spec.
5156                  !!!next-token;
5157                  next B;
5158              } else {              } else {
5159                #                !!!cp ('t149');
5160              }              }
           } elsif ($token->{type} eq 'end tag') {  
             if ($token->{tag_name} eq 'table') {  
               ## have a table element in table scope  
               my $i;  
               INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
                 my $node = $self->{open_elements}->[$_];  
                 if ($node->[1] eq $token->{tag_name}) {  
                   $i = $_;  
                   last INSCOPE;  
                 } elsif ({  
                           table => 1, html => 1,  
                          }->{$node->[1]}) {  
                   last INSCOPE;  
                 }  
               } # INSCOPE  
               unless (defined $i) {  
                 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
                 ## Ignore the token  
                 !!!next-token;  
                 redo B;  
               }  
                 
               ## generate implied end tags  
               if ({  
                    dd => 1, dt => 1, li => 1, p => 1,  
                    td => 1, th => 1, tr => 1,  
                    tbody => 1, tfoot=> 1, thead => 1,  
                   }->{$self->{open_elements}->[-1]->[1]}) {  
                 !!!back-token;  
                 $token = {type => 'end tag',  
                           tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
                 redo B;  
               }  
5161    
5162                if ($self->{open_elements}->[-1]->[1] ne 'table') {              ## "after head" insertion mode
5163                  !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);              ## As if <body>
5164                }              !!!insert-element ('body',, $token);
5165                $self->{insertion_mode} = IN_BODY_IM;
5166                ## reprocess
5167                next B;
5168          } elsif ($token->{type} == END_OF_FILE_TOKEN) {
5169            if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
5170              !!!cp ('t149.1');
5171    
5172              ## NOTE: As if <head>
5173              !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
5174              $self->{open_elements}->[-1]->[0]->append_child
5175                  ($self->{head_element});
5176              #push @{$self->{open_elements}},
5177              #    [$self->{head_element}, $el_category->{head}];
5178              #$self->{insertion_mode} = IN_HEAD_IM;
5179              ## NOTE: Reprocess.
5180    
5181              ## NOTE: As if </head>
5182              #pop @{$self->{open_elements}};
5183              #$self->{insertion_mode} = IN_AFTER_HEAD_IM;
5184              ## NOTE: Reprocess.
5185              
5186              #
5187            } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
5188              !!!cp ('t149.2');
5189    
5190                splice @{$self->{open_elements}}, $i;            ## NOTE: As if </head>
5191              pop @{$self->{open_elements}};
5192              #$self->{insertion_mode} = IN_AFTER_HEAD_IM;
5193              ## NOTE: Reprocess.
5194    
5195                $self->_reset_insertion_mode;            #
5196            } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
5197              !!!cp ('t149.3');
5198    
5199                !!!next-token;            !!!parse-error (type => 'in noscript:#eof', token => $token);
5200                redo B;  
5201              } elsif ({            ## As if </noscript>
5202                        body => 1, caption => 1, col => 1, colgroup => 1,            pop @{$self->{open_elements}};
5203                        html => 1, tbody => 1, td => 1, tfoot => 1, th => 1,            #$self->{insertion_mode} = IN_HEAD_IM;
5204                        thead => 1, tr => 1,            ## NOTE: Reprocess.
5205                       }->{$token->{tag_name}}) {  
5206                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});            ## NOTE: As if </head>
5207                ## Ignore the token            pop @{$self->{open_elements}};
5208                !!!next-token;            #$self->{insertion_mode} = IN_AFTER_HEAD_IM;
5209                redo B;            ## NOTE: Reprocess.
5210              } else {  
5211                #            #
5212              }          } else {
5213            } else {            !!!cp ('t149.4');
5214              #            #
5215            }          }
5216    
5217            !!!parse-error (type => 'in table:'.$token->{tag_name});          ## NOTE: As if <body>
5218            $in_body->($insert_to_foster);          !!!insert-element ('body',, $token);
5219            redo B;          $self->{insertion_mode} = IN_BODY_IM;
5220          } elsif ($self->{insertion_mode} eq 'in caption') {          ## NOTE: Reprocess.
5221            if ($token->{type} eq 'character') {          next B;
5222              ## NOTE: This is a code clone of "character in body".        } else {
5223            die "$0: $token->{type}: Unknown token type";
5224          }
5225        } elsif ($self->{insertion_mode} & BODY_IMS) {
5226              if ($token->{type} == CHARACTER_TOKEN) {
5227                !!!cp ('t150');
5228                ## NOTE: There is a code clone of "character in body".
5229              $reconstruct_active_formatting_elements->($insert_to_current);              $reconstruct_active_formatting_elements->($insert_to_current);
5230                            
5231              $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});              $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
5232    
5233              !!!next-token;              !!!next-token;
5234              redo B;              next B;
5235            } elsif ($token->{type} eq 'comment') {            } elsif ($token->{type} == START_TAG_TOKEN) {
             ## NOTE: This is a code clone of "comment in body".  
             my $comment = $self->{document}->create_comment ($token->{data});  
             $self->{open_elements}->[-1]->[0]->append_child ($comment);  
             !!!next-token;  
             redo B;  
           } elsif ($token->{type} eq 'start tag') {  
5236              if ({              if ({
5237                   caption => 1, col => 1, colgroup => 1, tbody => 1,                   caption => 1, col => 1, colgroup => 1, tbody => 1,
5238                   td => 1, tfoot => 1, th => 1, thead => 1, tr => 1,                   td => 1, tfoot => 1, th => 1, thead => 1, tr => 1,
5239                  }->{$token->{tag_name}}) {                  }->{$token->{tag_name}}) {
5240                !!!parse-error (type => 'not closed:caption');                if ($self->{insertion_mode} == IN_CELL_IM) {
5241                    ## have an element in table scope
5242                ## As if </caption>                  for (reverse 0..$#{$self->{open_elements}}) {
5243                ## have a table element in table scope                    my $node = $self->{open_elements}->[$_];
5244                my $i;                    if ($node->[1] & TABLE_CELL_EL) {
5245                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                      !!!cp ('t151');
5246                  my $node = $self->{open_elements}->[$_];  
5247                  if ($node->[1] eq 'caption') {                      ## Close the cell
5248                    $i = $_;                      !!!back-token; # <x>
5249                    last INSCOPE;                      $token = {type => END_TAG_TOKEN,
5250                  } elsif ({                                tag_name => $node->[0]->manakai_local_name,
5251                            table => 1, html => 1,                                line => $token->{line},
5252                           }->{$node->[1]}) {                                column => $token->{column}};
5253                    last INSCOPE;                      next B;
5254                      } elsif ($node->[1] & TABLE_SCOPING_EL) {
5255                        !!!cp ('t152');
5256                        ## ISSUE: This case can never be reached, maybe.
5257                        last;
5258                      }
5259                  }                  }
5260                } # INSCOPE  
5261                unless (defined $i) {                  !!!cp ('t153');
5262                  !!!parse-error (type => 'unmatched end tag:caption');                  !!!parse-error (type => 'start tag not allowed',
5263                        text => $token->{tag_name}, token => $token);
5264                  ## Ignore the token                  ## Ignore the token
5265                    !!!nack ('t153.1');
5266                  !!!next-token;                  !!!next-token;
5267                  redo B;                  next B;
5268                }                } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {
5269                                  !!!parse-error (type => 'not closed', text => 'caption',
5270                ## generate implied end tags                                  token => $token);
5271                if ({                  
5272                     dd => 1, dt => 1, li => 1, p => 1,                  ## NOTE: As if </caption>.
5273                     td => 1, th => 1, tr => 1,                  ## have a table element in table scope
5274                     tbody => 1, tfoot=> 1, thead => 1,                  my $i;
5275                    }->{$self->{open_elements}->[-1]->[1]}) {                  INSCOPE: {
5276                  !!!back-token; # <?>                    for (reverse 0..$#{$self->{open_elements}}) {
5277                  $token = {type => 'end tag', tag_name => 'caption'};                      my $node = $self->{open_elements}->[$_];
5278                  !!!back-token;                      if ($node->[1] & CAPTION_EL) {
5279                  $token = {type => 'end tag',                        !!!cp ('t155');
5280                            tag_name => $self->{open_elements}->[-1]->[1]}; # MUST                        $i = $_;
5281                  redo B;                        last INSCOPE;
5282                }                      } elsif ($node->[1] & TABLE_SCOPING_EL) {
5283                          !!!cp ('t156');
5284                if ($self->{open_elements}->[-1]->[1] ne 'caption') {                        last;
5285                  !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                      }
5286                }                    }
   
               splice @{$self->{open_elements}}, $i;  
   
               $clear_up_to_marker->();  
5287    
5288                $self->{insertion_mode} = 'in table';                    !!!cp ('t157');
5289                      !!!parse-error (type => 'start tag not allowed',
5290                                      text => $token->{tag_name}, token => $token);
5291                      ## Ignore the token
5292                      !!!nack ('t157.1');
5293                      !!!next-token;
5294                      next B;
5295                    } # INSCOPE
5296                    
5297                    ## generate implied end tags
5298                    while ($self->{open_elements}->[-1]->[1]
5299                               & END_TAG_OPTIONAL_EL) {
5300                      !!!cp ('t158');
5301                      pop @{$self->{open_elements}};
5302                    }
5303    
5304                ## reprocess                  unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
5305                redo B;                    !!!cp ('t159');
5306                      !!!parse-error (type => 'not closed',
5307                                      text => $self->{open_elements}->[-1]->[0]
5308                                          ->manakai_local_name,
5309                                      token => $token);
5310                    } else {
5311                      !!!cp ('t160');
5312                    }
5313                    
5314                    splice @{$self->{open_elements}}, $i;
5315                    
5316                    $clear_up_to_marker->();
5317                    
5318                    $self->{insertion_mode} = IN_TABLE_IM;
5319                    
5320                    ## reprocess
5321                    !!!ack-later;
5322                    next B;
5323                  } else {
5324                    !!!cp ('t161');
5325                    #
5326                  }
5327              } else {              } else {
5328                  !!!cp ('t162');
5329                #                #
5330              }              }
5331            } elsif ($token->{type} eq 'end tag') {            } elsif ($token->{type} == END_TAG_TOKEN) {
5332              if ($token->{tag_name} eq 'caption') {              if ($token->{tag_name} eq 'td' or $token->{tag_name} eq 'th') {
5333                ## have a table element in table scope                if ($self->{insertion_mode} == IN_CELL_IM) {
5334                my $i;                  ## have an element in table scope
5335                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                  my $i;
5336                  my $node = $self->{open_elements}->[$_];                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5337                  if ($node->[1] eq $token->{tag_name}) {                    my $node = $self->{open_elements}->[$_];
5338                    $i = $_;                    if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
5339                    last INSCOPE;                      !!!cp ('t163');
5340                  } elsif ({                      $i = $_;
5341                            table => 1, html => 1,                      last INSCOPE;
5342                           }->{$node->[1]}) {                    } elsif ($node->[1] & TABLE_SCOPING_EL) {
5343                    last INSCOPE;                      !!!cp ('t164');
5344                        last INSCOPE;
5345                      }
5346                    } # INSCOPE
5347                      unless (defined $i) {
5348                        !!!cp ('t165');
5349                        !!!parse-error (type => 'unmatched end tag',
5350                                        text => $token->{tag_name},
5351                                        token => $token);
5352                        ## Ignore the token
5353                        !!!next-token;
5354                        next B;
5355                      }
5356                    
5357                    ## generate implied end tags
5358                    while ($self->{open_elements}->[-1]->[1]
5359                               & END_TAG_OPTIONAL_EL) {
5360                      !!!cp ('t166');
5361                      pop @{$self->{open_elements}};
5362                  }                  }
5363                } # INSCOPE  
5364                unless (defined $i) {                  if ($self->{open_elements}->[-1]->[0]->manakai_local_name
5365                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                          ne $token->{tag_name}) {
5366                      !!!cp ('t167');
5367                      !!!parse-error (type => 'not closed',
5368                                      text => $self->{open_elements}->[-1]->[0]
5369                                          ->manakai_local_name,
5370                                      token => $token);
5371                    } else {
5372                      !!!cp ('t168');
5373                    }
5374                    
5375                    splice @{$self->{open_elements}}, $i;
5376                    
5377                    $clear_up_to_marker->();
5378                    
5379                    $self->{insertion_mode} = IN_ROW_IM;
5380                    
5381                    !!!next-token;
5382                    next B;
5383                  } elsif ($self->{insertion_mode} == IN_CAPTION_IM) {
5384                    !!!cp ('t169');
5385                    !!!parse-error (type => 'unmatched end tag',
5386                                    text => $token->{tag_name}, token => $token);
5387                  ## Ignore the token                  ## Ignore the token
5388                  !!!next-token;                  !!!next-token;
5389                  redo B;                  next B;
5390                }                } else {
5391                                  !!!cp ('t170');
5392                ## generate implied end tags                  #
               if ({  
                    dd => 1, dt => 1, li => 1, p => 1,  
                    td => 1, th => 1, tr => 1,  
                    tbody => 1, tfoot=> 1, thead => 1,  
                   }->{$self->{open_elements}->[-1]->[1]}) {  
                 !!!back-token;  
                 $token = {type => 'end tag',  
                           tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
                 redo B;  
5393                }                }
5394                } elsif ($token->{tag_name} eq 'caption') {
5395                  if ($self->{insertion_mode} == IN_CAPTION_IM) {
5396                    ## have a table element in table scope
5397                    my $i;
5398                    INSCOPE: {
5399                      for (reverse 0..$#{$self->{open_elements}}) {
5400                        my $node = $self->{open_elements}->[$_];
5401                        if ($node->[1] & CAPTION_EL) {
5402                          !!!cp ('t171');
5403                          $i = $_;
5404                          last INSCOPE;
5405                        } elsif ($node->[1] & TABLE_SCOPING_EL) {
5406                          !!!cp ('t172');
5407                          last;
5408                        }
5409                      }
5410    
5411                if ($self->{open_elements}->[-1]->[1] ne 'caption') {                    !!!cp ('t173');
5412                  !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                    !!!parse-error (type => 'unmatched end tag',
5413                                      text => $token->{tag_name}, token => $token);
5414                      ## Ignore the token
5415                      !!!next-token;
5416                      next B;
5417                    } # INSCOPE
5418                    
5419                    ## generate implied end tags
5420                    while ($self->{open_elements}->[-1]->[1]
5421                               & END_TAG_OPTIONAL_EL) {
5422                      !!!cp ('t174');
5423                      pop @{$self->{open_elements}};
5424                    }
5425                    
5426                    unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
5427                      !!!cp ('t175');
5428                      !!!parse-error (type => 'not closed',
5429                                      text => $self->{open_elements}->[-1]->[0]
5430                                          ->manakai_local_name,
5431                                      token => $token);
5432                    } else {
5433                      !!!cp ('t176');
5434                    }
5435                    
5436                    splice @{$self->{open_elements}}, $i;
5437                    
5438                    $clear_up_to_marker->();
5439                    
5440                    $self->{insertion_mode} = IN_TABLE_IM;
5441                    
5442                    !!!next-token;
5443                    next B;
5444                  } elsif ($self->{insertion_mode} == IN_CELL_IM) {
5445                    !!!cp ('t177');
5446                    !!!parse-error (type => 'unmatched end tag',
5447                                    text => $token->{tag_name}, token => $token);
5448                    ## Ignore the token
5449                    !!!next-token;
5450                    next B;
5451                  } else {
5452                    !!!cp ('t178');
5453                    #
5454                }                }
5455                } elsif ({
5456                          table => 1, tbody => 1, tfoot => 1,
5457                          thead => 1, tr => 1,
5458                         }->{$token->{tag_name}} and
5459                         $self->{insertion_mode} == IN_CELL_IM) {
5460                  ## have an element in table scope
5461                  my $i;
5462                  my $tn;
5463                  INSCOPE: {
5464                    for (reverse 0..$#{$self->{open_elements}}) {
5465                      my $node = $self->{open_elements}->[$_];
5466                      if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
5467                        !!!cp ('t179');
5468                        $i = $_;
5469    
5470                        ## Close the cell
5471                        !!!back-token; # </x>
5472                        $token = {type => END_TAG_TOKEN, tag_name => $tn,
5473                                  line => $token->{line},
5474                                  column => $token->{column}};
5475                        next B;
5476                      } elsif ($node->[1] & TABLE_CELL_EL) {
5477                        !!!cp ('t180');
5478                        $tn = $node->[0]->manakai_local_name;
5479                        ## NOTE: There is exactly one |td| or |th| element
5480                        ## in scope in the stack of open elements by definition.
5481                      } elsif ($node->[1] & TABLE_SCOPING_EL) {
5482                        ## ISSUE: Can this be reached?
5483                        !!!cp ('t181');
5484                        last;
5485                      }
5486                    }
5487    
5488                splice @{$self->{open_elements}}, $i;                  !!!cp ('t182');
5489                    !!!parse-error (type => 'unmatched end tag',
5490                $clear_up_to_marker->();                      text => $token->{tag_name}, token => $token);
5491                    ## Ignore the token
5492                $self->{insertion_mode} = 'in table';                  !!!next-token;
5493                    next B;
5494                !!!next-token;                } # INSCOPE
5495                redo B;              } elsif ($token->{tag_name} eq 'table' and
5496              } elsif ($token->{tag_name} eq 'table') {                       $self->{insertion_mode} == IN_CAPTION_IM) {
5497                !!!parse-error (type => 'not closed:caption');                !!!parse-error (type => 'not closed', text => 'caption',
5498                                  token => $token);
5499    
5500                ## As if </caption>                ## As if </caption>
5501                ## have a table element in table scope                ## have a table element in table scope
5502                my $i;                my $i;
5503                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5504                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
5505                  if ($node->[1] eq 'caption') {                  if ($node->[1] & CAPTION_EL) {
5506                      !!!cp ('t184');
5507                    $i = $_;                    $i = $_;
5508                    last INSCOPE;                    last INSCOPE;
5509                  } elsif ({                  } elsif ($node->[1] & TABLE_SCOPING_EL) {
5510                            table => 1, html => 1,                    !!!cp ('t185');
                          }->{$node->[1]}) {  
5511                    last INSCOPE;                    last INSCOPE;
5512                  }                  }
5513                } # INSCOPE                } # INSCOPE
5514                unless (defined $i) {                unless (defined $i) {
5515                  !!!parse-error (type => 'unmatched end tag:caption');                  !!!cp ('t186');
5516                    !!!parse-error (type => 'unmatched end tag',
5517                                    text => 'caption', token => $token);
5518                  ## Ignore the token                  ## Ignore the token
5519                  !!!next-token;                  !!!next-token;
5520                  redo B;                  next B;
5521                }                }
5522                                
5523                ## generate implied end tags                ## generate implied end tags
5524                if ({                while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
5525                     dd => 1, dt => 1, li => 1, p => 1,                  !!!cp ('t187');
5526                     td => 1, th => 1, tr => 1,                  pop @{$self->{open_elements}};
                    tbody => 1, tfoot=> 1, thead => 1,  
                   }->{$self->{open_elements}->[-1]->[1]}) {  
                 !!!back-token; # </table>  
                 $token = {type => 'end tag', tag_name => 'caption'};  
                 !!!back-token;  
                 $token = {type => 'end tag',  
                           tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
                 redo B;  
5527                }                }
5528    
5529                if ($self->{open_elements}->[-1]->[1] ne 'caption') {                unless ($self->{open_elements}->[-1]->[1] & CAPTION_EL) {
5530                  !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                  !!!cp ('t188');
5531                    !!!parse-error (type => 'not closed',
5532                                    text => $self->{open_elements}->[-1]->[0]
5533                                        ->manakai_local_name,
5534                                    token => $token);
5535                  } else {
5536                    !!!cp ('t189');
5537                }                }
5538    
5539                splice @{$self->{open_elements}}, $i;                splice @{$self->{open_elements}}, $i;
5540    
5541                $clear_up_to_marker->();                $clear_up_to_marker->();
5542    
5543                $self->{insertion_mode} = 'in table';                $self->{insertion_mode} = IN_TABLE_IM;
5544    
5545                ## reprocess                ## reprocess
5546                redo B;                next B;
5547              } elsif ({              } elsif ({
5548                        body => 1, col => 1, colgroup => 1,                        body => 1, col => 1, colgroup => 1, html => 1,
                       html => 1, tbody => 1, td => 1, tfoot => 1,  
                       th => 1, thead => 1, tr => 1,  
5549                       }->{$token->{tag_name}}) {                       }->{$token->{tag_name}}) {
5550                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                if ($self->{insertion_mode} & BODY_TABLE_IMS) {
5551                ## Ignore the token                  !!!cp ('t190');
5552                redo B;                  !!!parse-error (type => 'unmatched end tag',
5553              } else {                                  text => $token->{tag_name}, token => $token);
               #  
             }  
           } else {  
             #  
           }  
                 
           $in_body->($insert_to_current);  
           redo B;  
         } elsif ($self->{insertion_mode} eq 'in column group') {  
           if ($token->{type} eq 'character') {  
             if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {  
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);  
               unless (length $token->{data}) {  
                 !!!next-token;  
                 redo B;  
               }  
             }  
               
             #  
           } elsif ($token->{type} eq 'comment') {  
             my $comment = $self->{document}->create_comment ($token->{data});  
             $self->{open_elements}->[-1]->[0]->append_child ($comment);  
             !!!next-token;  
             redo B;  
           } elsif ($token->{type} eq 'start tag') {  
             if ($token->{tag_name} eq 'col') {  
               !!!insert-element ($token->{tag_name}, $token->{attributes});  
               pop @{$self->{open_elements}};  
               !!!next-token;  
               redo B;  
             } else {  
               #  
             }  
           } elsif ($token->{type} eq 'end tag') {  
             if ($token->{tag_name} eq 'colgroup') {  
               if ($self->{open_elements}->[-1]->[1] eq 'html') {  
                 !!!parse-error (type => 'unmatched end tag:colgroup');  
5554                  ## Ignore the token                  ## Ignore the token
5555                  !!!next-token;                  !!!next-token;
5556                  redo B;                  next B;
5557                } else {                } else {
5558                  pop @{$self->{open_elements}}; # colgroup                  !!!cp ('t191');
5559                  $self->{insertion_mode} = 'in table';                  #
                 !!!next-token;  
                 redo B;              
5560                }                }
5561              } elsif ($token->{tag_name} eq 'col') {              } elsif ({
5562                !!!parse-error (type => 'unmatched end tag:col');                        tbody => 1, tfoot => 1,
5563                          thead => 1, tr => 1,
5564                         }->{$token->{tag_name}} and
5565                         $self->{insertion_mode} == IN_CAPTION_IM) {
5566                  !!!cp ('t192');
5567                  !!!parse-error (type => 'unmatched end tag',
5568                                  text => $token->{tag_name}, token => $token);
5569                ## Ignore the token                ## Ignore the token
5570                !!!next-token;                !!!next-token;
5571                redo B;                next B;
5572              } else {              } else {
5573                #                !!!cp ('t193');
5574                  #
5575              }              }
5576            } else {        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
5577              #          for my $entry (@{$self->{open_elements}}) {
5578              unless ($entry->[1] & ALL_END_TAG_OPTIONAL_EL) {
5579                !!!cp ('t75');
5580                !!!parse-error (type => 'in body:#eof', token => $token);
5581                last;
5582            }            }
5583            }
5584    
5585            ## As if </colgroup>          ## Stop parsing.
5586            if ($self->{open_elements}->[-1]->[1] eq 'html') {          last B;
5587              !!!parse-error (type => 'unmatched end tag:colgroup');        } else {
5588              ## Ignore the token          die "$0: $token->{type}: Unknown token type";
5589          }
5590    
5591          $insert = $insert_to_current;
5592          #
5593        } elsif ($self->{insertion_mode} & TABLE_IMS) {
5594          if ($token->{type} == CHARACTER_TOKEN) {
5595            if (not $open_tables->[-1]->[1] and # tainted
5596                $token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
5597              $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
5598                  
5599              unless (length $token->{data}) {
5600                !!!cp ('t194');
5601              !!!next-token;              !!!next-token;
5602              redo B;              next B;
5603            } else {            } else {
5604              pop @{$self->{open_elements}}; # colgroup              !!!cp ('t195');
             $self->{insertion_mode} = 'in table';  
             ## reprocess  
             redo B;  
5605            }            }
5606          } elsif ($self->{insertion_mode} eq 'in table body') {          }
           if ($token->{type} eq 'character') {  
             ## NOTE: This is a "character in table" code clone.  
             if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {  
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);  
                 
               unless (length $token->{data}) {  
                 !!!next-token;  
                 redo B;  
               }  
             }  
   
             !!!parse-error (type => 'in table:#character');  
5607    
5608              ## As if in body, but insert into foster parent element          !!!parse-error (type => 'in table:#text', token => $token);
             ## ISSUE: Spec says that "whenever a node would be inserted  
             ## into the current node" while characters might not be  
             ## result in a new Text node.  
             $reconstruct_active_formatting_elements->($insert_to_foster);  
5609    
5610              if ({          ## NOTE: As if in body, but insert into the foster parent element.
5611                   table => 1, tbody => 1, tfoot => 1,          $reconstruct_active_formatting_elements->($insert_to_foster);
5612                   thead => 1, tr => 1,              
5613                  }->{$self->{open_elements}->[-1]->[1]}) {          if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {
5614                # MUST            # MUST
5615                my $foster_parent_element;            my $foster_parent_element;
5616                my $next_sibling;            my $next_sibling;
5617                my $prev_sibling;            my $prev_sibling;
5618                OE: for (reverse 0..$#{$self->{open_elements}}) {            OE: for (reverse 0..$#{$self->{open_elements}}) {
5619                  if ($self->{open_elements}->[$_]->[1] eq 'table') {              if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {
5620                    my $parent = $self->{open_elements}->[$_]->[0]->parent_node;                my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
5621                    if (defined $parent and $parent->node_type == 1) {                if (defined $parent and $parent->node_type == 1) {
5622                      $foster_parent_element = $parent;                  $foster_parent_element = $parent;
5623                      $next_sibling = $self->{open_elements}->[$_]->[0];                  !!!cp ('t196');
5624                      $prev_sibling = $next_sibling->previous_sibling;                  $next_sibling = $self->{open_elements}->[$_]->[0];
5625                    } else {                  $prev_sibling = $next_sibling->previous_sibling;
5626                      $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];                  #
                     $prev_sibling = $foster_parent_element->last_child;  
                   }  
                   last OE;  
                 }  
               } # OE  
               $foster_parent_element = $self->{open_elements}->[0]->[0] and  
               $prev_sibling = $foster_parent_element->last_child  
                 unless defined $foster_parent_element;  
               if (defined $prev_sibling and  
                   $prev_sibling->node_type == 3) {  
                 $prev_sibling->manakai_append_text ($token->{data});  
5627                } else {                } else {
5628                  $foster_parent_element->insert_before                  !!!cp ('t197');
5629                    ($self->{document}->create_text_node ($token->{data}),                  $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];
5630                     $next_sibling);                  $prev_sibling = $foster_parent_element->last_child;
5631                    #
5632                }                }
5633              } else {                last OE;
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});  
5634              }              }
5635              } # OE
5636              $foster_parent_element = $self->{open_elements}->[0]->[0] and
5637              $prev_sibling = $foster_parent_element->last_child
5638                  unless defined $foster_parent_element;
5639              undef $prev_sibling unless $open_tables->[-1]->[2]; # ~node inserted
5640              if (defined $prev_sibling and
5641                  $prev_sibling->node_type == 3) {
5642                !!!cp ('t198');
5643                $prev_sibling->manakai_append_text ($token->{data});
5644              } else {
5645                !!!cp ('t199');
5646                $foster_parent_element->insert_before
5647                    ($self->{document}->create_text_node ($token->{data}),
5648                     $next_sibling);
5649              }
5650              $open_tables->[-1]->[1] = 1; # tainted
5651              $open_tables->[-1]->[2] = 1; # ~node inserted
5652            } else {
5653              ## NOTE: Fragment case or in a foster parent'ed element
5654              ## (e.g. |<table><span>a|).  In fragment case, whether the
5655              ## character is appended to existing node or a new node is
5656              ## created is irrelevant, since the foster parent'ed nodes
5657              ## are discarded and fragment parsing does not invoke any
5658              ## script.
5659              !!!cp ('t200');
5660              $self->{open_elements}->[-1]->[0]->manakai_append_text
5661                  ($token->{data});
5662            }
5663                            
5664              !!!next-token;          !!!next-token;
5665              redo B;          next B;
5666            } elsif ($token->{type} eq 'comment') {        } elsif ($token->{type} == START_TAG_TOKEN) {
5667              ## Copied from 'in table'          if ({
5668              my $comment = $self->{document}->create_comment ($token->{data});               tr => ($self->{insertion_mode} != IN_ROW_IM),
5669              $self->{open_elements}->[-1]->[0]->append_child ($comment);               th => 1, td => 1,
5670              !!!next-token;              }->{$token->{tag_name}}) {
5671              redo B;            if ($self->{insertion_mode} == IN_TABLE_IM) {
5672            } elsif ($token->{type} eq 'start tag') {              ## Clear back to table context
5673              if ({              while (not ($self->{open_elements}->[-1]->[1]
5674                   tr => 1,                              & TABLE_SCOPING_EL)) {
5675                   th => 1, td => 1,                !!!cp ('t201');
5676                  }->{$token->{tag_name}}) {                pop @{$self->{open_elements}};
5677                unless ($token->{tag_name} eq 'tr') {              }
5678                  !!!parse-error (type => 'missing start tag:tr');              
5679                }              !!!insert-element ('tbody',, $token);
5680                $self->{insertion_mode} = IN_TABLE_BODY_IM;
5681                ## reprocess in the "in table body" insertion mode...
5682              }
5683              
5684              if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {
5685                unless ($token->{tag_name} eq 'tr') {
5686                  !!!cp ('t202');
5687                  !!!parse-error (type => 'missing start tag:tr', token => $token);
5688                }
5689                    
5690                ## Clear back to table body context
5691                while (not ($self->{open_elements}->[-1]->[1]
5692                                & TABLE_ROWS_SCOPING_EL)) {
5693                  !!!cp ('t203');
5694                  ## ISSUE: Can this case be reached?
5695                  pop @{$self->{open_elements}};
5696                }
5697                    
5698                $self->{insertion_mode} = IN_ROW_IM;
5699                if ($token->{tag_name} eq 'tr') {
5700                  !!!cp ('t204');
5701                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5702                  $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5703                  !!!nack ('t204');
5704                  !!!next-token;
5705                  next B;
5706                } else {
5707                  !!!cp ('t205');
5708                  !!!insert-element ('tr',, $token);
5709                  ## reprocess in the "in row" insertion mode
5710                }
5711              } else {
5712                !!!cp ('t206');
5713              }
5714    
5715                ## Clear back to table body context                ## Clear back to table row context
5716                while (not {                while (not ($self->{open_elements}->[-1]->[1]
5717                  tbody => 1, tfoot => 1, thead => 1, html => 1,                                & TABLE_ROW_SCOPING_EL)) {
5718                }->{$self->{open_elements}->[-1]->[1]}) {                  !!!cp ('t207');
                 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
5719                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
5720                }                }
5721                                
5722                $self->{insertion_mode} = 'in row';            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5723                if ($token->{tag_name} eq 'tr') {            $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5724                  !!!insert-element ($token->{tag_name}, $token->{attributes});            $self->{insertion_mode} = IN_CELL_IM;
5725                  !!!next-token;  
5726                } else {            push @$active_formatting_elements, ['#marker', ''];
5727                  !!!insert-element ('tr');                
5728                  ## reprocess            !!!nack ('t207.1');
5729                }            !!!next-token;
5730                redo B;            next B;
5731              } elsif ({          } elsif ({
5732                        caption => 1, col => 1, colgroup => 1,                    caption => 1, col => 1, colgroup => 1,
5733                        tbody => 1, tfoot => 1, thead => 1,                    tbody => 1, tfoot => 1, thead => 1,
5734                       }->{$token->{tag_name}}) {                    tr => 1, # $self->{insertion_mode} == IN_ROW_IM
5735                ## have an element in table scope                   }->{$token->{tag_name}}) {
5736                my $i;            if ($self->{insertion_mode} == IN_ROW_IM) {
5737                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {              ## As if </tr>
5738                  my $node = $self->{open_elements}->[$_];              ## have an element in table scope
5739                  if ({              my $i;
5740                       tbody => 1, thead => 1, tfoot => 1,              INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5741                      }->{$node->[1]}) {                my $node = $self->{open_elements}->[$_];
5742                    $i = $_;                if ($node->[1] & TABLE_ROW_EL) {
5743                    last INSCOPE;                  !!!cp ('t208');
5744                  } elsif ({                  $i = $_;
5745                            table => 1, html => 1,                  last INSCOPE;
5746                           }->{$node->[1]}) {                } elsif ($node->[1] & TABLE_SCOPING_EL) {
5747                    last INSCOPE;                  !!!cp ('t209');
5748                    last INSCOPE;
5749                  }
5750                } # INSCOPE
5751                unless (defined $i) {
5752                  !!!cp ('t210');
5753                  ## TODO: This type is wrong.
5754                  !!!parse-error (type => 'unmacthed end tag',
5755                                  text => $token->{tag_name}, token => $token);
5756                  ## Ignore the token
5757                  !!!nack ('t210.1');
5758                  !!!next-token;
5759                  next B;
5760                }
5761                    
5762                    ## Clear back to table row context
5763                    while (not ($self->{open_elements}->[-1]->[1]
5764                                    & TABLE_ROW_SCOPING_EL)) {
5765                      !!!cp ('t211');
5766                      ## ISSUE: Can this case be reached?
5767                      pop @{$self->{open_elements}};
5768                    }
5769                    
5770                    pop @{$self->{open_elements}}; # tr
5771                    $self->{insertion_mode} = IN_TABLE_BODY_IM;
5772                    if ($token->{tag_name} eq 'tr') {
5773                      !!!cp ('t212');
5774                      ## reprocess
5775                      !!!ack-later;
5776                      next B;
5777                    } else {
5778                      !!!cp ('t213');
5779                      ## reprocess in the "in table body" insertion mode...
5780                  }                  }
               } # INSCOPE  
               unless (defined $i) {  
                 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
                 ## Ignore the token  
                 !!!next-token;  
                 redo B;  
5781                }                }
5782    
5783                ## Clear back to table body context                if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {
5784                while (not {                  ## have an element in table scope
5785                  tbody => 1, tfoot => 1, thead => 1, html => 1,                  my $i;
5786                }->{$self->{open_elements}->[-1]->[1]}) {                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5787                  !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                    my $node = $self->{open_elements}->[$_];
5788                      if ($node->[1] & TABLE_ROW_GROUP_EL) {
5789                        !!!cp ('t214');
5790                        $i = $_;
5791                        last INSCOPE;
5792                      } elsif ($node->[1] & TABLE_SCOPING_EL) {
5793                        !!!cp ('t215');
5794                        last INSCOPE;
5795                      }
5796                    } # INSCOPE
5797                    unless (defined $i) {
5798                      !!!cp ('t216');
5799    ## TODO: This erorr type is wrong.
5800                      !!!parse-error (type => 'unmatched end tag',
5801                                      text => $token->{tag_name}, token => $token);
5802                      ## Ignore the token
5803                      !!!nack ('t216.1');
5804                      !!!next-token;
5805                      next B;
5806                    }
5807    
5808                    ## Clear back to table body context
5809                    while (not ($self->{open_elements}->[-1]->[1]
5810                                    & TABLE_ROWS_SCOPING_EL)) {
5811                      !!!cp ('t217');
5812                      ## ISSUE: Can this state be reached?
5813                      pop @{$self->{open_elements}};
5814                    }
5815                    
5816                    ## As if <{current node}>
5817                    ## have an element in table scope
5818                    ## true by definition
5819                    
5820                    ## Clear back to table body context
5821                    ## nop by definition
5822                    
5823                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
5824                    $self->{insertion_mode} = IN_TABLE_IM;
5825                    ## reprocess in "in table" insertion mode...
5826                  } else {
5827                    !!!cp ('t218');
5828                }                }
5829    
5830                ## As if <{current node}>            if ($token->{tag_name} eq 'col') {
5831                ## have an element in table scope              ## Clear back to table context
5832                ## true by definition              while (not ($self->{open_elements}->[-1]->[1]
5833                                & TABLE_SCOPING_EL)) {
5834                ## Clear back to table body context                !!!cp ('t219');
5835                ## nop by definition                ## ISSUE: Can this state be reached?
   
5836                pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
5837                $self->{insertion_mode} = 'in table';              }
5838                ## reprocess              
5839                redo B;              !!!insert-element ('colgroup',, $token);
5840                $self->{insertion_mode} = IN_COLUMN_GROUP_IM;
5841                ## reprocess
5842                $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5843                !!!ack-later;
5844                next B;
5845              } elsif ({
5846                        caption => 1,
5847                        colgroup => 1,
5848                        tbody => 1, tfoot => 1, thead => 1,
5849                       }->{$token->{tag_name}}) {
5850                ## Clear back to table context
5851                    while (not ($self->{open_elements}->[-1]->[1]
5852                                    & TABLE_SCOPING_EL)) {
5853                      !!!cp ('t220');
5854                      ## ISSUE: Can this state be reached?
5855                      pop @{$self->{open_elements}};
5856                    }
5857                    
5858                push @$active_formatting_elements, ['#marker', '']
5859                    if $token->{tag_name} eq 'caption';
5860                    
5861                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5862                $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5863                $self->{insertion_mode} = {
5864                                           caption => IN_CAPTION_IM,
5865                                           colgroup => IN_COLUMN_GROUP_IM,
5866                                           tbody => IN_TABLE_BODY_IM,
5867                                           tfoot => IN_TABLE_BODY_IM,
5868                                           thead => IN_TABLE_BODY_IM,
5869                                          }->{$token->{tag_name}};
5870                !!!next-token;
5871                !!!nack ('t220.1');
5872                next B;
5873              } else {
5874                die "$0: in table: <>: $token->{tag_name}";
5875              }
5876              } elsif ($token->{tag_name} eq 'table') {              } elsif ($token->{tag_name} eq 'table') {
5877                ## NOTE: This is a code clone of "table in table"                !!!parse-error (type => 'not closed',
5878                !!!parse-error (type => 'not closed:table');                                text => $self->{open_elements}->[-1]->[0]
5879                                      ->manakai_local_name,
5880                                  token => $token);
5881    
5882                ## As if </table>                ## As if </table>
5883                ## have a table element in table scope                ## have a table element in table scope
5884                my $i;                my $i;
5885                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
5886                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
5887                  if ($node->[1] eq 'table') {                  if ($node->[1] & TABLE_EL) {
5888                      !!!cp ('t221');
5889                    $i = $_;                    $i = $_;
5890                    last INSCOPE;                    last INSCOPE;
5891                  } elsif ({                  } elsif ($node->[1] & TABLE_SCOPING_EL) {
5892                            table => 1, html => 1,                    !!!cp ('t222');
                          }->{$node->[1]}) {  
5893                    last INSCOPE;                    last INSCOPE;
5894                  }                  }
5895                } # INSCOPE                } # INSCOPE
5896                unless (defined $i) {                unless (defined $i) {
5897                  !!!parse-error (type => 'unmatched end tag:table');                  !!!cp ('t223');
5898    ## TODO: The following is wrong, maybe.
5899                    !!!parse-error (type => 'unmatched end tag', text => 'table',
5900                                    token => $token);
5901                  ## Ignore tokens </table><table>                  ## Ignore tokens </table><table>
5902                    !!!nack ('t223.1');
5903                  !!!next-token;                  !!!next-token;
5904                  redo B;                  next B;
5905                }                }
5906                                
5907    ## TODO: Followings are removed from the latest spec.
5908                ## generate implied end tags                ## generate implied end tags
5909                if ({                while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
5910                     dd => 1, dt => 1, li => 1, p => 1,                  !!!cp ('t224');
5911                     td => 1, th => 1, tr => 1,                  pop @{$self->{open_elements}};
                    tbody => 1, tfoot=> 1, thead => 1,  
                   }->{$self->{open_elements}->[-1]->[1]}) {  
                 !!!back-token; # <table>  
                 $token = {type => 'end tag', tag_name => 'table'};  
                 !!!back-token;  
                 $token = {type => 'end tag',  
                           tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
                 redo B;  
5912                }                }
5913    
5914                if ($self->{open_elements}->[-1]->[1] ne 'table') {                unless ($self->{open_elements}->[-1]->[1] & TABLE_EL) {
5915                  !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                  !!!cp ('t225');
5916                    ## NOTE: |<table><tr><table>|
5917                    !!!parse-error (type => 'not closed',
5918                                    text => $self->{open_elements}->[-1]->[0]
5919                                        ->manakai_local_name,
5920                                    token => $token);
5921                  } else {
5922                    !!!cp ('t226');
5923                }                }
5924    
5925                splice @{$self->{open_elements}}, $i;                splice @{$self->{open_elements}}, $i;
5926                  pop @{$open_tables};
5927    
5928                $self->_reset_insertion_mode;                $self->_reset_insertion_mode;
5929    
5930                ## reprocess            ## reprocess
5931                redo B;            !!!ack-later;
5932              } else {            next B;
5933                #          } elsif ($token->{tag_name} eq 'style') {
5934              }            if (not $open_tables->[-1]->[1]) { # tainted
5935            } elsif ($token->{type} eq 'end tag') {              !!!cp ('t227.8');
5936              if ({              ## NOTE: This is a "as if in head" code clone.
5937                   tbody => 1, tfoot => 1, thead => 1,              $parse_rcdata->(CDATA_CONTENT_MODEL);
5938                  }->{$token->{tag_name}}) {              $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5939                ## have an element in table scope              next B;
5940                my $i;            } else {
5941                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {              !!!cp ('t227.7');
5942                  my $node = $self->{open_elements}->[$_];              #
5943                  if ($node->[1] eq $token->{tag_name}) {            }
5944                    $i = $_;          } elsif ($token->{tag_name} eq 'script') {
5945                    last INSCOPE;            if (not $open_tables->[-1]->[1]) { # tainted
5946                  } elsif ({              !!!cp ('t227.6');
5947                            table => 1, html => 1,              ## NOTE: This is a "as if in head" code clone.
5948                           }->{$node->[1]}) {              $script_start_tag->();
5949                    last INSCOPE;              $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
5950                  }              next B;
5951                } # INSCOPE            } else {
5952                unless (defined $i) {              !!!cp ('t227.5');
5953                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});              #
5954                  ## Ignore the token            }
5955                  !!!next-token;          } elsif ($token->{tag_name} eq 'input') {
5956                  redo B;            if (not $open_tables->[-1]->[1]) { # tainted
5957                }              if ($token->{attributes}->{type}) { ## TODO: case
5958                  my $type = lc $token->{attributes}->{type}->{value};
5959                  if ($type eq 'hidden') {
5960                    !!!cp ('t227.3');
5961                    !!!parse-error (type => 'in table',
5962                                    text => $token->{tag_name}, token => $token);
5963    
5964                ## Clear back to table body context                  !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
5965                while (not {                  $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
                 tbody => 1, tfoot => 1, thead => 1, html => 1,  
               }->{$self->{open_elements}->[-1]->[1]}) {  
                 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
                 pop @{$self->{open_elements}};  
               }  
5966    
5967                pop @{$self->{open_elements}};                  ## TODO: form element pointer
               $self->{insertion_mode} = 'in table';  
               !!!next-token;  
               redo B;  
             } elsif ($token->{tag_name} eq 'table') {  
               ## have an element in table scope  
               my $i;  
               INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
                 my $node = $self->{open_elements}->[$_];  
                 if ({  
                      tbody => 1, thead => 1, tfoot => 1,  
                     }->{$node->[1]}) {  
                   $i = $_;  
                   last INSCOPE;  
                 } elsif ({  
                           table => 1, html => 1,  
                          }->{$node->[1]}) {  
                   last INSCOPE;  
                 }  
               } # INSCOPE  
               unless (defined $i) {  
                 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
                 ## Ignore the token  
                 !!!next-token;  
                 redo B;  
               }  
5968    
               ## Clear back to table body context  
               while (not {  
                 tbody => 1, tfoot => 1, thead => 1, html => 1,  
               }->{$self->{open_elements}->[-1]->[1]}) {  
                 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
5969                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
               }  
   
               ## As if <{current node}>  
               ## have an element in table scope  
               ## true by definition  
5970    
5971                ## Clear back to table body context                  !!!next-token;
5972                ## nop by definition                  !!!ack ('t227.2.1');
5973                    next B;
5974                pop @{$self->{open_elements}};                } else {
5975                $self->{insertion_mode} = 'in table';                  !!!cp ('t227.2');
5976                ## reprocess                  #
5977                redo B;                }
             } elsif ({  
                       body => 1, caption => 1, col => 1, colgroup => 1,  
                       html => 1, td => 1, th => 1, tr => 1,  
                      }->{$token->{tag_name}}) {  
               !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
               ## Ignore the token  
               !!!next-token;  
               redo B;  
5978              } else {              } else {
5979                  !!!cp ('t227.1');
5980                #                #
5981              }              }
5982            } else {            } else {
5983                !!!cp ('t227.4');
5984              #              #
5985            }            }
5986                      } else {
5987            ## As if in table            !!!cp ('t227');
5988            !!!parse-error (type => 'in table:'.$token->{tag_name});            #
5989            $in_body->($insert_to_foster);          }
           redo B;  
         } elsif ($self->{insertion_mode} eq 'in row') {  
           if ($token->{type} eq 'character') {  
             ## NOTE: This is a "character in table" code clone.  
             if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {  
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);  
                 
               unless (length $token->{data}) {  
                 !!!next-token;  
                 redo B;  
               }  
             }  
   
             !!!parse-error (type => 'in table:#character');  
5990    
5991              ## As if in body, but insert into foster parent element          !!!parse-error (type => 'in table', text => $token->{tag_name},
5992              ## ISSUE: Spec says that "whenever a node would be inserted                          token => $token);
             ## into the current node" while characters might not be  
             ## result in a new Text node.  
             $reconstruct_active_formatting_elements->($insert_to_foster);  
               
             if ({  
                  table => 1, tbody => 1, tfoot => 1,  
                  thead => 1, tr => 1,  
                 }->{$self->{open_elements}->[-1]->[1]}) {  
               # MUST  
               my $foster_parent_element;  
               my $next_sibling;  
               my $prev_sibling;  
               OE: for (reverse 0..$#{$self->{open_elements}}) {  
                 if ($self->{open_elements}->[$_]->[1] eq 'table') {  
                   my $parent = $self->{open_elements}->[$_]->[0]->parent_node;  
                   if (defined $parent and $parent->node_type == 1) {  
                     $foster_parent_element = $parent;  
                     $next_sibling = $self->{open_elements}->[$_]->[0];  
                     $prev_sibling = $next_sibling->previous_sibling;  
                   } else {  
                     $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];  
                     $prev_sibling = $foster_parent_element->last_child;  
                   }  
                   last OE;  
                 }  
               } # OE  
               $foster_parent_element = $self->{open_elements}->[0]->[0] and  
               $prev_sibling = $foster_parent_element->last_child  
                 unless defined $foster_parent_element;  
               if (defined $prev_sibling and  
                   $prev_sibling->node_type == 3) {  
                 $prev_sibling->manakai_append_text ($token->{data});  
               } else {  
                 $foster_parent_element->insert_before  
                   ($self->{document}->create_text_node ($token->{data}),  
                    $next_sibling);  
               }  
             } else {  
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});  
             }  
               
             !!!next-token;  
             redo B;  
           } elsif ($token->{type} eq 'comment') {  
             ## Copied from 'in table'  
             my $comment = $self->{document}->create_comment ($token->{data});  
             $self->{open_elements}->[-1]->[0]->append_child ($comment);  
             !!!next-token;  
             redo B;  
           } elsif ($token->{type} eq 'start tag') {  
             if ($token->{tag_name} eq 'th' or  
                 $token->{tag_name} eq 'td') {  
               ## Clear back to table row context  
               while (not {  
                 tr => 1, html => 1,  
               }->{$self->{open_elements}->[-1]->[1]}) {  
                 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
                 pop @{$self->{open_elements}};  
               }  
                 
               !!!insert-element ($token->{tag_name}, $token->{attributes});  
               $self->{insertion_mode} = 'in cell';  
5993    
5994                push @$active_formatting_elements, ['#marker', ''];          $insert = $insert_to_foster;
5995                          #
5996                !!!next-token;        } elsif ($token->{type} == END_TAG_TOKEN) {
5997                redo B;              if ($token->{tag_name} eq 'tr' and
5998              } elsif ({                  $self->{insertion_mode} == IN_ROW_IM) {
                       caption => 1, col => 1, colgroup => 1,  
                       tbody => 1, tfoot => 1, thead => 1, tr => 1,  
                      }->{$token->{tag_name}}) {  
               ## As if </tr>  
5999                ## have an element in table scope                ## have an element in table scope
6000                my $i;                my $i;
6001                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6002                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
6003                  if ($node->[1] eq 'tr') {                  if ($node->[1] & TABLE_ROW_EL) {
6004                      !!!cp ('t228');
6005                    $i = $_;                    $i = $_;
6006                    last INSCOPE;                    last INSCOPE;
6007                  } elsif ({                  } elsif ($node->[1] & TABLE_SCOPING_EL) {
6008                            table => 1, html => 1,                    !!!cp ('t229');
                          }->{$node->[1]}) {  
6009                    last INSCOPE;                    last INSCOPE;
6010                  }                  }
6011                } # INSCOPE                } # INSCOPE
6012                unless (defined $i) {                unless (defined $i) {
6013                  !!!parse-error (type => 'unmacthed end tag:'.$token->{tag_name});                  !!!cp ('t230');
6014                    !!!parse-error (type => 'unmatched end tag',
6015                                    text => $token->{tag_name}, token => $token);
6016                  ## Ignore the token                  ## Ignore the token
6017                    !!!nack ('t230.1');
6018                  !!!next-token;                  !!!next-token;
6019                  redo B;                  next B;
6020                  } else {
6021                    !!!cp ('t232');
6022                }                }
6023    
6024                ## Clear back to table row context                ## Clear back to table row context
6025                while (not {                while (not ($self->{open_elements}->[-1]->[1]
6026                  tr => 1, html => 1,                                & TABLE_ROW_SCOPING_EL)) {
6027                }->{$self->{open_elements}->[-1]->[1]}) {                  !!!cp ('t231');
6028                  !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  ## ISSUE: Can this state be reached?
6029                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
6030                }                }
6031    
6032                pop @{$self->{open_elements}}; # tr                pop @{$self->{open_elements}}; # tr
6033                $self->{insertion_mode} = 'in table body';                $self->{insertion_mode} = IN_TABLE_BODY_IM;
6034                ## reprocess                !!!next-token;
6035                redo B;                !!!nack ('t231.1');
6036                  next B;
6037              } elsif ($token->{tag_name} eq 'table') {              } elsif ($token->{tag_name} eq 'table') {
6038                ## NOTE: This is a code clone of "table in table"                if ($self->{insertion_mode} == IN_ROW_IM) {
6039                !!!parse-error (type => 'not closed:table');                  ## As if </tr>
6040                    ## have an element in table scope
6041                ## As if </table>                  my $i;
6042                ## have a table element in table scope                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6043                my $i;                    my $node = $self->{open_elements}->[$_];
6044                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                    if ($node->[1] & TABLE_ROW_EL) {
6045                  my $node = $self->{open_elements}->[$_];                      !!!cp ('t233');
6046                  if ($node->[1] eq 'table') {                      $i = $_;
6047                    $i = $_;                      last INSCOPE;
6048                    last INSCOPE;                    } elsif ($node->[1] & TABLE_SCOPING_EL) {
6049                  } elsif ({                      !!!cp ('t234');
6050                            table => 1, html => 1,                      last INSCOPE;
6051                           }->{$node->[1]}) {                    }
6052                    last INSCOPE;                  } # INSCOPE
6053                    unless (defined $i) {
6054                      !!!cp ('t235');
6055    ## TODO: The following is wrong.
6056                      !!!parse-error (type => 'unmatched end tag',
6057                                      text => $token->{type}, token => $token);
6058                      ## Ignore the token
6059                      !!!nack ('t236.1');
6060                      !!!next-token;
6061                      next B;
6062                  }                  }
6063                } # INSCOPE                  
6064                unless (defined $i) {                  ## Clear back to table row context
6065                  !!!parse-error (type => 'unmatched end tag:table');                  while (not ($self->{open_elements}->[-1]->[1]
6066                  ## Ignore tokens </table><table>                                  & TABLE_ROW_SCOPING_EL)) {
6067                  !!!next-token;                    !!!cp ('t236');
6068                  redo B;  ## ISSUE: Can this state be reached?
6069                }                    pop @{$self->{open_elements}};
6070                                  }
6071                ## generate implied end tags                  
6072                if ({                  pop @{$self->{open_elements}}; # tr
6073                     dd => 1, dt => 1, li => 1, p => 1,                  $self->{insertion_mode} = IN_TABLE_BODY_IM;
6074                     td => 1, th => 1, tr => 1,                  ## reprocess in the "in table body" insertion mode...
6075                     tbody => 1, tfoot=> 1, thead => 1,                }
6076                    }->{$self->{open_elements}->[-1]->[1]}) {  
6077                  !!!back-token; # <table>                if ($self->{insertion_mode} == IN_TABLE_BODY_IM) {
6078                  $token = {type => 'end tag', tag_name => 'table'};                  ## have an element in table scope
6079                  !!!back-token;                  my $i;
6080                  $token = {type => 'end tag',                  INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6081                            tag_name => $self->{open_elements}->[-1]->[1]}; # MUST                    my $node = $self->{open_elements}->[$_];
6082                  redo B;                    if ($node->[1] & TABLE_ROW_GROUP_EL) {
6083                        !!!cp ('t237');
6084                        $i = $_;
6085                        last INSCOPE;
6086                      } elsif ($node->[1] & TABLE_SCOPING_EL) {
6087                        !!!cp ('t238');
6088                        last INSCOPE;
6089                      }
6090                    } # INSCOPE
6091                    unless (defined $i) {
6092                      !!!cp ('t239');
6093                      !!!parse-error (type => 'unmatched end tag',
6094                                      text => $token->{tag_name}, token => $token);
6095                      ## Ignore the token
6096                      !!!nack ('t239.1');
6097                      !!!next-token;
6098                      next B;
6099                    }
6100                    
6101                    ## Clear back to table body context
6102                    while (not ($self->{open_elements}->[-1]->[1]
6103                                    & TABLE_ROWS_SCOPING_EL)) {
6104                      !!!cp ('t240');
6105                      pop @{$self->{open_elements}};
6106                    }
6107                    
6108                    ## As if <{current node}>
6109                    ## have an element in table scope
6110                    ## true by definition
6111                    
6112                    ## Clear back to table body context
6113                    ## nop by definition
6114                    
6115                    pop @{$self->{open_elements}};
6116                    $self->{insertion_mode} = IN_TABLE_IM;
6117                    ## reprocess in the "in table" insertion mode...
6118                }                }
6119    
6120                if ($self->{open_elements}->[-1]->[1] ne 'table') {                ## NOTE: </table> in the "in table" insertion mode.
6121                  !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                ## When you edit the code fragment below, please ensure that
6122                }                ## the code for <table> in the "in table" insertion mode
6123                  ## is synced with it.
6124    
6125                splice @{$self->{open_elements}}, $i;                ## have a table element in table scope
   
               $self->_reset_insertion_mode;  
   
               ## reprocess  
               redo B;  
             } else {  
               #  
             }  
           } elsif ($token->{type} eq 'end tag') {  
             if ($token->{tag_name} eq 'tr') {  
               ## have an element in table scope  
6126                my $i;                my $i;
6127                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6128                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
6129                  if ($node->[1] eq $token->{tag_name}) {                  if ($node->[1] & TABLE_EL) {
6130                      !!!cp ('t241');
6131                    $i = $_;                    $i = $_;
6132                    last INSCOPE;                    last INSCOPE;
6133                  } elsif ({                  } elsif ($node->[1] & TABLE_SCOPING_EL) {
6134                            table => 1, html => 1,                    !!!cp ('t242');
                          }->{$node->[1]}) {  
6135                    last INSCOPE;                    last INSCOPE;
6136                  }                  }
6137                } # INSCOPE                } # INSCOPE
6138                unless (defined $i) {                unless (defined $i) {
6139                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                  !!!cp ('t243');
6140                    !!!parse-error (type => 'unmatched end tag',
6141                                    text => $token->{tag_name}, token => $token);
6142                  ## Ignore the token                  ## Ignore the token
6143                    !!!nack ('t243.1');
6144                  !!!next-token;                  !!!next-token;
6145                  redo B;                  next B;
6146                }                }
6147                    
6148                ## Clear back to table row context                splice @{$self->{open_elements}}, $i;
6149                while (not {                pop @{$open_tables};
6150                  tr => 1, html => 1,                
6151                }->{$self->{open_elements}->[-1]->[1]}) {                $self->_reset_insertion_mode;
6152                  !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);                
6153                  pop @{$self->{open_elements}};                !!!next-token;
6154                  next B;
6155                } elsif ({
6156                          tbody => 1, tfoot => 1, thead => 1,
6157                         }->{$token->{tag_name}} and
6158                         $self->{insertion_mode} & ROW_IMS) {
6159                  if ($self->{insertion_mode} == IN_ROW_IM) {
6160                    ## have an element in table scope
6161                    my $i;
6162                    INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6163                      my $node = $self->{open_elements}->[$_];
6164                      if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
6165                        !!!cp ('t247');
6166                        $i = $_;
6167                        last INSCOPE;
6168                      } elsif ($node->[1] & TABLE_SCOPING_EL) {
6169                        !!!cp ('t248');
6170                        last INSCOPE;
6171                      }
6172                    } # INSCOPE
6173                      unless (defined $i) {
6174                        !!!cp ('t249');
6175                        !!!parse-error (type => 'unmatched end tag',
6176                                        text => $token->{tag_name}, token => $token);
6177                        ## Ignore the token
6178                        !!!nack ('t249.1');
6179                        !!!next-token;
6180                        next B;
6181                      }
6182                    
6183                    ## As if </tr>
6184                    ## have an element in table scope
6185                    my $i;
6186                    INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6187                      my $node = $self->{open_elements}->[$_];
6188                      if ($node->[1] & TABLE_ROW_EL) {
6189                        !!!cp ('t250');
6190                        $i = $_;
6191                        last INSCOPE;
6192                      } elsif ($node->[1] & TABLE_SCOPING_EL) {
6193                        !!!cp ('t251');
6194                        last INSCOPE;
6195                      }
6196                    } # INSCOPE
6197                      unless (defined $i) {
6198                        !!!cp ('t252');
6199                        !!!parse-error (type => 'unmatched end tag',
6200                                        text => 'tr', token => $token);
6201                        ## Ignore the token
6202                        !!!nack ('t252.1');
6203                        !!!next-token;
6204                        next B;
6205                      }
6206                    
6207                    ## Clear back to table row context
6208                    while (not ($self->{open_elements}->[-1]->[1]
6209                                    & TABLE_ROW_SCOPING_EL)) {
6210                      !!!cp ('t253');
6211    ## ISSUE: Can this case be reached?
6212                      pop @{$self->{open_elements}};
6213                    }
6214                    
6215                    pop @{$self->{open_elements}}; # tr
6216                    $self->{insertion_mode} = IN_TABLE_BODY_IM;
6217                    ## reprocess in the "in table body" insertion mode...
6218                }                }
6219    
               pop @{$self->{open_elements}}; # tr  
               $self->{insertion_mode} = 'in table body';  
               !!!next-token;  
               redo B;  
             } elsif ($token->{tag_name} eq 'table') {  
               ## As if </tr>  
6220                ## have an element in table scope                ## have an element in table scope
6221                my $i;                my $i;
6222                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6223                  my $node = $self->{open_elements}->[$_];                  my $node = $self->{open_elements}->[$_];
6224                  if ($node->[1] eq 'tr') {                  if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
6225                      !!!cp ('t254');
6226                    $i = $_;                    $i = $_;
6227                    last INSCOPE;                    last INSCOPE;
6228                  } elsif ({                  } elsif ($node->[1] & TABLE_SCOPING_EL) {
6229                            table => 1, html => 1,                    !!!cp ('t255');
                          }->{$node->[1]}) {  
6230                    last INSCOPE;                    last INSCOPE;
6231                  }                  }
6232                } # INSCOPE                } # INSCOPE
6233                unless (defined $i) {                unless (defined $i) {
6234                  !!!parse-error (type => 'unmatched end tag:'.$token->{type});                  !!!cp ('t256');
6235                    !!!parse-error (type => 'unmatched end tag',
6236                                    text => $token->{tag_name}, token => $token);
6237                  ## Ignore the token                  ## Ignore the token
6238                    !!!nack ('t256.1');
6239                  !!!next-token;                  !!!next-token;
6240                  redo B;                  next B;
6241                }                }
6242    
6243                ## Clear back to table row context                ## Clear back to table body context
6244                while (not {                while (not ($self->{open_elements}->[-1]->[1]
6245                  tr => 1, html => 1,                                & TABLE_ROWS_SCOPING_EL)) {
6246                }->{$self->{open_elements}->[-1]->[1]}) {                  !!!cp ('t257');
6247                  !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  ## ISSUE: Can this case be reached?
6248                  pop @{$self->{open_elements}};                  pop @{$self->{open_elements}};
6249                }                }
6250    
6251                pop @{$self->{open_elements}}; # tr                pop @{$self->{open_elements}};
6252                $self->{insertion_mode} = 'in table body';                $self->{insertion_mode} = IN_TABLE_IM;
6253                ## reprocess                !!!nack ('t257.1');
6254                redo B;                !!!next-token;
6255                  next B;
6256              } elsif ({              } elsif ({
6257                        tbody => 1, tfoot => 1, thead => 1,                        body => 1, caption => 1, col => 1, colgroup => 1,
6258                          html => 1, td => 1, th => 1,
6259                          tr => 1, # $self->{insertion_mode} == IN_ROW_IM
6260                          tbody => 1, tfoot => 1, thead => 1, # $self->{insertion_mode} == IN_TABLE_IM
6261                       }->{$token->{tag_name}}) {                       }->{$token->{tag_name}}) {
6262                ## have an element in table scope            !!!cp ('t258');
6263                my $i;            !!!parse-error (type => 'unmatched end tag',
6264                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {                            text => $token->{tag_name}, token => $token);
6265                  my $node = $self->{open_elements}->[$_];            ## Ignore the token
6266                  if ($node->[1] eq $token->{tag_name}) {            !!!nack ('t258.1');
6267                    $i = $_;             !!!next-token;
6268                    last INSCOPE;            next B;
6269                  } elsif ({          } else {
6270                            table => 1, html => 1,            !!!cp ('t259');
6271                           }->{$node->[1]}) {            !!!parse-error (type => 'in table:/',
6272                    last INSCOPE;                            text => $token->{tag_name}, token => $token);
6273                  }  
6274                } # INSCOPE            $insert = $insert_to_foster;
6275                unless (defined $i) {            #
6276                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});          }
6277                  ## Ignore the token        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6278            unless ($self->{open_elements}->[-1]->[1] & HTML_EL and
6279                    @{$self->{open_elements}} == 1) { # redundant, maybe
6280              !!!parse-error (type => 'in body:#eof', token => $token);
6281              !!!cp ('t259.1');
6282              #
6283            } else {
6284              !!!cp ('t259.2');
6285              #
6286            }
6287    
6288            ## Stop parsing
6289            last B;
6290          } else {
6291            die "$0: $token->{type}: Unknown token type";
6292          }
6293        } elsif ($self->{insertion_mode} == IN_COLUMN_GROUP_IM) {
6294              if ($token->{type} == CHARACTER_TOKEN) {
6295                if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6296                  $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6297                  unless (length $token->{data}) {
6298                    !!!cp ('t260');
6299                  !!!next-token;                  !!!next-token;
6300                  redo B;                  next B;
6301                }                }
6302                }
6303                ## As if </tr>              
6304                ## have an element in table scope              !!!cp ('t261');
6305                my $i;              #
6306                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {            } elsif ($token->{type} == START_TAG_TOKEN) {
6307                  my $node = $self->{open_elements}->[$_];              if ($token->{tag_name} eq 'col') {
6308                  if ($node->[1] eq 'tr') {                !!!cp ('t262');
6309                    $i = $_;                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6310                    last INSCOPE;                pop @{$self->{open_elements}};
6311                  } elsif ({                !!!ack ('t262.1');
6312                            table => 1, html => 1,                !!!next-token;
6313                           }->{$node->[1]}) {                next B;
6314                    last INSCOPE;              } else {
6315                  }                !!!cp ('t263');
6316                } # INSCOPE                #
6317                unless (defined $i) {              }
6318                  !!!parse-error (type => 'unmatched end tag:tr');            } elsif ($token->{type} == END_TAG_TOKEN) {
6319                if ($token->{tag_name} eq 'colgroup') {
6320                  if ($self->{open_elements}->[-1]->[1] & HTML_EL) {
6321                    !!!cp ('t264');
6322                    !!!parse-error (type => 'unmatched end tag',
6323                                    text => 'colgroup', token => $token);
6324                  ## Ignore the token                  ## Ignore the token
6325                  !!!next-token;                  !!!next-token;
6326                  redo B;                  next B;
6327                }                } else {
6328                    !!!cp ('t265');
6329                ## Clear back to table row context                  pop @{$self->{open_elements}}; # colgroup
6330                while (not {                  $self->{insertion_mode} = IN_TABLE_IM;
6331                  tr => 1, html => 1,                  !!!next-token;
6332                }->{$self->{open_elements}->[-1]->[1]}) {                  next B;            
                 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
                 pop @{$self->{open_elements}};  
6333                }                }
6334                } elsif ($token->{tag_name} eq 'col') {
6335                pop @{$self->{open_elements}}; # tr                !!!cp ('t266');
6336                $self->{insertion_mode} = 'in table body';                !!!parse-error (type => 'unmatched end tag',
6337                ## reprocess                                text => 'col', token => $token);
               redo B;  
             } elsif ({  
                       body => 1, caption => 1, col => 1,  
                       colgroup => 1, html => 1, td => 1, th => 1,  
                      }->{$token->{tag_name}}) {  
               !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
6338                ## Ignore the token                ## Ignore the token
6339                !!!next-token;                !!!next-token;
6340                redo B;                next B;
6341              } else {              } else {
6342                #                !!!cp ('t267');
6343                  #
6344              }              }
6345          } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6346            if ($self->{open_elements}->[-1]->[1] & HTML_EL and
6347                @{$self->{open_elements}} == 1) { # redundant, maybe
6348              !!!cp ('t270.2');
6349              ## Stop parsing.
6350              last B;
6351            } else {
6352              ## NOTE: As if </colgroup>.
6353              !!!cp ('t270.1');
6354              pop @{$self->{open_elements}}; # colgroup
6355              $self->{insertion_mode} = IN_TABLE_IM;
6356              ## Reprocess.
6357              next B;
6358            }
6359          } else {
6360            die "$0: $token->{type}: Unknown token type";
6361          }
6362    
6363              ## As if </colgroup>
6364              if ($self->{open_elements}->[-1]->[1] & HTML_EL) {
6365                !!!cp ('t269');
6366    ## TODO: Wrong error type?
6367                !!!parse-error (type => 'unmatched end tag',
6368                                text => 'colgroup', token => $token);
6369                ## Ignore the token
6370                !!!nack ('t269.1');
6371                !!!next-token;
6372                next B;
6373            } else {            } else {
6374              #              !!!cp ('t270');
6375                pop @{$self->{open_elements}}; # colgroup
6376                $self->{insertion_mode} = IN_TABLE_IM;
6377                !!!ack-later;
6378                ## reprocess
6379                next B;
6380              }
6381        } elsif ($self->{insertion_mode} & SELECT_IMS) {
6382          if ($token->{type} == CHARACTER_TOKEN) {
6383            !!!cp ('t271');
6384            $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
6385            !!!next-token;
6386            next B;
6387          } elsif ($token->{type} == START_TAG_TOKEN) {
6388            if ($token->{tag_name} eq 'option') {
6389              if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {
6390                !!!cp ('t272');
6391                ## As if </option>
6392                pop @{$self->{open_elements}};
6393              } else {
6394                !!!cp ('t273');
6395            }            }
6396    
6397            ## As if in table            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6398            !!!parse-error (type => 'in table:'.$token->{tag_name});            !!!nack ('t273.1');
6399            $in_body->($insert_to_foster);            !!!next-token;
6400            redo B;            next B;
6401          } elsif ($self->{insertion_mode} eq 'in cell') {          } elsif ($token->{tag_name} eq 'optgroup') {
6402            if ($token->{type} eq 'character') {            if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {
6403              ## NOTE: This is a code clone of "character in body".              !!!cp ('t274');
6404              $reconstruct_active_formatting_elements->($insert_to_current);              ## As if </option>
6405                            pop @{$self->{open_elements}};
6406              $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});            } else {
6407                !!!cp ('t275');
6408              }
6409    
6410              if ($self->{open_elements}->[-1]->[1] & OPTGROUP_EL) {
6411                !!!cp ('t276');
6412                ## As if </optgroup>
6413                pop @{$self->{open_elements}};
6414              } else {
6415                !!!cp ('t277');
6416              }
6417    
6418              !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6419              !!!nack ('t277.1');
6420              !!!next-token;
6421              next B;
6422            } elsif ({
6423                       select => 1, input => 1, textarea => 1,
6424                     }->{$token->{tag_name}} or
6425                     ($self->{insertion_mode} == IN_SELECT_IN_TABLE_IM and
6426                      {
6427                       caption => 1, table => 1,
6428                       tbody => 1, tfoot => 1, thead => 1,
6429                       tr => 1, td => 1, th => 1,
6430                      }->{$token->{tag_name}})) {
6431              ## TODO: The type below is not good - <select> is replaced by </select>
6432              !!!parse-error (type => 'not closed', text => 'select',
6433                              token => $token);
6434              ## NOTE: As if the token were </select> (<select> case) or
6435              ## as if there were </select> (otherwise).
6436              ## have an element in table scope
6437              my $i;
6438              INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6439                my $node = $self->{open_elements}->[$_];
6440                if ($node->[1] & SELECT_EL) {
6441                  !!!cp ('t278');
6442                  $i = $_;
6443                  last INSCOPE;
6444                } elsif ($node->[1] & TABLE_SCOPING_EL) {
6445                  !!!cp ('t279');
6446                  last INSCOPE;
6447                }
6448              } # INSCOPE
6449              unless (defined $i) {
6450                !!!cp ('t280');
6451                !!!parse-error (type => 'unmatched end tag',
6452                                text => 'select', token => $token);
6453                ## Ignore the token
6454                !!!nack ('t280.1');
6455              !!!next-token;              !!!next-token;
6456              redo B;              next B;
6457            } elsif ($token->{type} eq 'comment') {            }
6458              ## NOTE: This is a code clone of "comment in body".                
6459              my $comment = $self->{document}->create_comment ($token->{data});            !!!cp ('t281');
6460              $self->{open_elements}->[-1]->[0]->append_child ($comment);            splice @{$self->{open_elements}}, $i;
6461    
6462              $self->_reset_insertion_mode;
6463    
6464              if ($token->{tag_name} eq 'select') {
6465                !!!nack ('t281.2');
6466              !!!next-token;              !!!next-token;
6467              redo B;              next B;
6468            } elsif ($token->{type} eq 'start tag') {            } else {
6469              if ({              !!!cp ('t281.1');
6470                   caption => 1, col => 1, colgroup => 1,              !!!ack-later;
6471                   tbody => 1, td => 1, tfoot => 1, th => 1,              ## Reprocess the token.
6472                   thead => 1, tr => 1,              next B;
6473                  }->{$token->{tag_name}}) {            }
6474                ## have an element in table scope          } else {
6475                my $tn;            !!!cp ('t282');
6476                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {            !!!parse-error (type => 'in select',
6477                  my $node = $self->{open_elements}->[$_];                            text => $token->{tag_name}, token => $token);
6478                  if ($node->[1] eq 'td' or $node->[1] eq 'th') {            ## Ignore the token
6479                    $tn = $node->[1];            !!!nack ('t282.1');
6480                    last INSCOPE;            !!!next-token;
6481                  } elsif ({            next B;
6482                            table => 1, html => 1,          }
6483                           }->{$node->[1]}) {        } elsif ($token->{type} == END_TAG_TOKEN) {
6484                    last INSCOPE;          if ($token->{tag_name} eq 'optgroup') {
6485                  }            if ($self->{open_elements}->[-1]->[1] & OPTION_EL and
6486                } # INSCOPE                $self->{open_elements}->[-2]->[1] & OPTGROUP_EL) {
6487                unless (defined $tn) {              !!!cp ('t283');
6488                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});              ## As if </option>
6489                  ## Ignore the token              splice @{$self->{open_elements}}, -2;
6490                  !!!next-token;            } elsif ($self->{open_elements}->[-1]->[1] & OPTGROUP_EL) {
6491                  redo B;              !!!cp ('t284');
6492                }              pop @{$self->{open_elements}};
6493              } else {
6494                !!!cp ('t285');
6495                !!!parse-error (type => 'unmatched end tag',
6496                                text => $token->{tag_name}, token => $token);
6497                ## Ignore the token
6498              }
6499              !!!nack ('t285.1');
6500              !!!next-token;
6501              next B;
6502            } elsif ($token->{tag_name} eq 'option') {
6503              if ($self->{open_elements}->[-1]->[1] & OPTION_EL) {
6504                !!!cp ('t286');
6505                pop @{$self->{open_elements}};
6506              } else {
6507                !!!cp ('t287');
6508                !!!parse-error (type => 'unmatched end tag',
6509                                text => $token->{tag_name}, token => $token);
6510                ## Ignore the token
6511              }
6512              !!!nack ('t287.1');
6513              !!!next-token;
6514              next B;
6515            } elsif ($token->{tag_name} eq 'select') {
6516              ## have an element in table scope
6517              my $i;
6518              INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6519                my $node = $self->{open_elements}->[$_];
6520                if ($node->[1] & SELECT_EL) {
6521                  !!!cp ('t288');
6522                  $i = $_;
6523                  last INSCOPE;
6524                } elsif ($node->[1] & TABLE_SCOPING_EL) {
6525                  !!!cp ('t289');
6526                  last INSCOPE;
6527                }
6528              } # INSCOPE
6529              unless (defined $i) {
6530                !!!cp ('t290');
6531                !!!parse-error (type => 'unmatched end tag',
6532                                text => $token->{tag_name}, token => $token);
6533                ## Ignore the token
6534                !!!nack ('t290.1');
6535                !!!next-token;
6536                next B;
6537              }
6538                  
6539              !!!cp ('t291');
6540              splice @{$self->{open_elements}}, $i;
6541    
6542                ## Close the cell            $self->_reset_insertion_mode;
6543                !!!back-token; # <?>  
6544                $token = {type => 'end tag', tag_name => $tn};            !!!nack ('t291.1');
6545                redo B;            !!!next-token;
6546              } else {            next B;
6547                #          } elsif ($self->{insertion_mode} == IN_SELECT_IN_TABLE_IM and
6548                     {
6549                      caption => 1, table => 1, tbody => 1,
6550                      tfoot => 1, thead => 1, tr => 1, td => 1, th => 1,
6551                     }->{$token->{tag_name}}) {
6552    ## TODO: The following is wrong?
6553              !!!parse-error (type => 'unmatched end tag',
6554                              text => $token->{tag_name}, token => $token);
6555                  
6556              ## have an element in table scope
6557              my $i;
6558              INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6559                my $node = $self->{open_elements}->[$_];
6560                if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
6561                  !!!cp ('t292');
6562                  $i = $_;
6563                  last INSCOPE;
6564                } elsif ($node->[1] & TABLE_SCOPING_EL) {
6565                  !!!cp ('t293');
6566                  last INSCOPE;
6567              }              }
6568            } elsif ($token->{type} eq 'end tag') {            } # INSCOPE
6569              if ($token->{tag_name} eq 'td' or $token->{tag_name} eq 'th') {            unless (defined $i) {
6570                ## have an element in table scope              !!!cp ('t294');
6571                my $i;              ## Ignore the token
6572                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {              !!!nack ('t294.1');
6573                  my $node = $self->{open_elements}->[$_];              !!!next-token;
6574                  if ($node->[1] eq $token->{tag_name}) {              next B;
6575                    $i = $_;            }
                   last INSCOPE;  
                 } elsif ({  
                           table => 1, html => 1,  
                          }->{$node->[1]}) {  
                   last INSCOPE;  
                 }  
               } # INSCOPE  
               unless (defined $i) {  
                 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
                 ## Ignore the token  
                 !!!next-token;  
                 redo B;  
               }  
6576                                
6577                ## generate implied end tags            ## As if </select>
6578                if ({            ## have an element in table scope
6579                     dd => 1, dt => 1, li => 1, p => 1,            undef $i;
6580                     td => ($token->{tag_name} eq 'th'),            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
6581                     th => ($token->{tag_name} eq 'td'),              my $node = $self->{open_elements}->[$_];
6582                     tr => 1,              if ($node->[1] & SELECT_EL) {
6583                     tbody => 1, tfoot=> 1, thead => 1,                !!!cp ('t295');
6584                    }->{$self->{open_elements}->[-1]->[1]}) {                $i = $_;
6585                  !!!back-token;                last INSCOPE;
6586                  $token = {type => 'end tag',              } elsif ($node->[1] & TABLE_SCOPING_EL) {
6587                            tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  ## ISSUE: Can this state be reached?
6588                  redo B;                !!!cp ('t296');
6589                }                last INSCOPE;
6590                }
6591              } # INSCOPE
6592              unless (defined $i) {
6593                !!!cp ('t297');
6594    ## TODO: The following error type is correct?
6595                !!!parse-error (type => 'unmatched end tag',
6596                                text => 'select', token => $token);
6597                ## Ignore the </select> token
6598                !!!nack ('t297.1');
6599                !!!next-token; ## TODO: ok?
6600                next B;
6601              }
6602                  
6603              !!!cp ('t298');
6604              splice @{$self->{open_elements}}, $i;
6605    
6606                if ($self->{open_elements}->[-1]->[1] ne $token->{tag_name}) {            $self->_reset_insertion_mode;
                 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
               }  
6607    
6608                splice @{$self->{open_elements}}, $i;            !!!ack-later;
6609              ## reprocess
6610              next B;
6611            } else {
6612              !!!cp ('t299');
6613              !!!parse-error (type => 'in select:/',
6614                              text => $token->{tag_name}, token => $token);
6615              ## Ignore the token
6616              !!!nack ('t299.3');
6617              !!!next-token;
6618              next B;
6619            }
6620          } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6621            unless ($self->{open_elements}->[-1]->[1] & HTML_EL and
6622                    @{$self->{open_elements}} == 1) { # redundant, maybe
6623              !!!cp ('t299.1');
6624              !!!parse-error (type => 'in body:#eof', token => $token);
6625            } else {
6626              !!!cp ('t299.2');
6627            }
6628    
6629                $clear_up_to_marker->();          ## Stop parsing.
6630            last B;
6631          } else {
6632            die "$0: $token->{type}: Unknown token type";
6633          }
6634        } elsif ($self->{insertion_mode} & BODY_AFTER_IMS) {
6635          if ($token->{type} == CHARACTER_TOKEN) {
6636            if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6637              my $data = $1;
6638              ## As if in body
6639              $reconstruct_active_formatting_elements->($insert_to_current);
6640                  
6641              $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6642              
6643              unless (length $token->{data}) {
6644                !!!cp ('t300');
6645                !!!next-token;
6646                next B;
6647              }
6648            }
6649            
6650            if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6651              !!!cp ('t301');
6652              !!!parse-error (type => 'after html:#text', token => $token);
6653              #
6654            } else {
6655              !!!cp ('t302');
6656              ## "after body" insertion mode
6657              !!!parse-error (type => 'after body:#text', token => $token);
6658              #
6659            }
6660    
6661                $self->{insertion_mode} = 'in row';          $self->{insertion_mode} = IN_BODY_IM;
6662            ## reprocess
6663            next B;
6664          } elsif ($token->{type} == START_TAG_TOKEN) {
6665            if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6666              !!!cp ('t303');
6667              !!!parse-error (type => 'after html',
6668                              text => $token->{tag_name}, token => $token);
6669              #
6670            } else {
6671              !!!cp ('t304');
6672              ## "after body" insertion mode
6673              !!!parse-error (type => 'after body',
6674                              text => $token->{tag_name}, token => $token);
6675              #
6676            }
6677    
6678                !!!next-token;          $self->{insertion_mode} = IN_BODY_IM;
6679                redo B;          !!!ack-later;
6680              } elsif ({          ## reprocess
6681                        body => 1, caption => 1, col => 1,          next B;
6682                        colgroup => 1, html => 1,        } elsif ($token->{type} == END_TAG_TOKEN) {
6683                       }->{$token->{tag_name}}) {          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
6684                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});            !!!cp ('t305');
6685                ## Ignore the token            !!!parse-error (type => 'after html:/',
6686                !!!next-token;                            text => $token->{tag_name}, token => $token);
6687                redo B;            
6688              } elsif ({            $self->{insertion_mode} = IN_BODY_IM;
6689                        table => 1, tbody => 1, tfoot => 1,            ## Reprocess.
6690                        thead => 1, tr => 1,            next B;
6691                       }->{$token->{tag_name}}) {          } else {
6692                ## have an element in table scope            !!!cp ('t306');
6693                my $i;          }
               my $tn;  
               INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
                 my $node = $self->{open_elements}->[$_];  
                 if ($node->[1] eq $token->{tag_name}) {  
                   $i = $_;  
                   last INSCOPE;  
                 } elsif ($node->[1] eq 'td' or $node->[1] eq 'th') {  
                   $tn = $node->[1];  
                   ## NOTE: There is exactly one |td| or |th| element  
                   ## in scope in the stack of open elements by definition.  
                 } elsif ({  
                           table => 1, html => 1,  
                          }->{$node->[1]}) {  
                   last INSCOPE;  
                 }  
               } # INSCOPE  
               unless (defined $i) {  
                 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
                 ## Ignore the token  
                 !!!next-token;  
                 redo B;  
               }  
6694    
6695                ## Close the cell          ## "after body" insertion mode
6696                !!!back-token; # </?>          if ($token->{tag_name} eq 'html') {
6697                $token = {type => 'end tag', tag_name => $tn};            if (defined $self->{inner_html_node}) {
6698                redo B;              !!!cp ('t307');
6699              } else {              !!!parse-error (type => 'unmatched end tag',
6700                #                              text => 'html', token => $token);
6701              }              ## Ignore the token
6702                !!!next-token;
6703                next B;
6704            } else {            } else {
6705              #              !!!cp ('t308');
6706                $self->{insertion_mode} = AFTER_HTML_BODY_IM;
6707                !!!next-token;
6708                next B;
6709            }            }
6710            } else {
6711              !!!cp ('t309');
6712              !!!parse-error (type => 'after body:/',
6713                              text => $token->{tag_name}, token => $token);
6714    
6715              $self->{insertion_mode} = IN_BODY_IM;
6716              ## reprocess
6717              next B;
6718            }
6719          } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6720            !!!cp ('t309.2');
6721            ## Stop parsing
6722            last B;
6723          } else {
6724            die "$0: $token->{type}: Unknown token type";
6725          }
6726        } elsif ($self->{insertion_mode} & FRAME_IMS) {
6727          if ($token->{type} == CHARACTER_TOKEN) {
6728            if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
6729              $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
6730                        
6731            $in_body->($insert_to_current);            unless (length $token->{data}) {
6732            redo B;              !!!cp ('t310');
         } elsif ($self->{insertion_mode} eq 'in select') {  
           if ($token->{type} eq 'character') {  
             $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});  
6733              !!!next-token;              !!!next-token;
6734              redo B;              next B;
6735            } elsif ($token->{type} eq 'comment') {            }
6736              my $comment = $self->{document}->create_comment ($token->{data});          }
6737              $self->{open_elements}->[-1]->[0]->append_child ($comment);          
6738            if ($token->{data} =~ s/^[^\x09\x0A\x0C\x20]+//) {
6739              if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6740                !!!cp ('t311');
6741                !!!parse-error (type => 'in frameset:#text', token => $token);
6742              } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6743                !!!cp ('t312');
6744                !!!parse-error (type => 'after frameset:#text', token => $token);
6745              } else { # "after after frameset"
6746                !!!cp ('t313');
6747                !!!parse-error (type => 'after html:#text', token => $token);
6748              }
6749              
6750              ## Ignore the token.
6751              if (length $token->{data}) {
6752                !!!cp ('t314');
6753                ## reprocess the rest of characters
6754              } else {
6755                !!!cp ('t315');
6756              !!!next-token;              !!!next-token;
6757              redo B;            }
6758            } elsif ($token->{type} eq 'start tag') {            next B;
6759              if ($token->{tag_name} eq 'option') {          }
6760                if ($self->{open_elements}->[-1]->[1] eq 'option') {          
6761                  ## As if </option>          die qq[$0: Character "$token->{data}"];
6762                  pop @{$self->{open_elements}};        } elsif ($token->{type} == START_TAG_TOKEN) {
6763                }          if ($token->{tag_name} eq 'frameset' and
6764                $self->{insertion_mode} == IN_FRAMESET_IM) {
6765                !!!insert-element ($token->{tag_name}, $token->{attributes});            !!!cp ('t318');
6766                !!!next-token;            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6767                redo B;            !!!nack ('t318.1');
6768              } elsif ($token->{tag_name} eq 'optgroup') {            !!!next-token;
6769                if ($self->{open_elements}->[-1]->[1] eq 'option') {            next B;
6770                  ## As if </option>          } elsif ($token->{tag_name} eq 'frame' and
6771                  pop @{$self->{open_elements}};                   $self->{insertion_mode} == IN_FRAMESET_IM) {
6772                }            !!!cp ('t319');
6773              !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
6774              pop @{$self->{open_elements}};
6775              !!!ack ('t319.1');
6776              !!!next-token;
6777              next B;
6778            } elsif ($token->{tag_name} eq 'noframes') {
6779              !!!cp ('t320');
6780              ## NOTE: As if in head.
6781              $parse_rcdata->(CDATA_CONTENT_MODEL);
6782              next B;
6783    
6784                if ($self->{open_elements}->[-1]->[1] eq 'optgroup') {            ## NOTE: |<!DOCTYPE HTML><frameset></frameset></html><noframes></noframes>|
6785                  ## As if </optgroup>            ## has no parse error.
6786                  pop @{$self->{open_elements}};          } else {
6787                }            if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6788                !!!cp ('t321');
6789                !!!parse-error (type => 'in frameset',
6790                                text => $token->{tag_name}, token => $token);
6791              } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6792                !!!cp ('t322');
6793                !!!parse-error (type => 'after frameset',
6794                                text => $token->{tag_name}, token => $token);
6795              } else { # "after after frameset"
6796                !!!cp ('t322.2');
6797                !!!parse-error (type => 'after after frameset',
6798                                text => $token->{tag_name}, token => $token);
6799              }
6800              ## Ignore the token
6801              !!!nack ('t322.1');
6802              !!!next-token;
6803              next B;
6804            }
6805          } elsif ($token->{type} == END_TAG_TOKEN) {
6806            if ($token->{tag_name} eq 'frameset' and
6807                $self->{insertion_mode} == IN_FRAMESET_IM) {
6808              if ($self->{open_elements}->[-1]->[1] & HTML_EL and
6809                  @{$self->{open_elements}} == 1) {
6810                !!!cp ('t325');
6811                !!!parse-error (type => 'unmatched end tag',
6812                                text => $token->{tag_name}, token => $token);
6813                ## Ignore the token
6814                !!!next-token;
6815              } else {
6816                !!!cp ('t326');
6817                pop @{$self->{open_elements}};
6818                !!!next-token;
6819              }
6820    
6821                !!!insert-element ($token->{tag_name}, $token->{attributes});            if (not defined $self->{inner_html_node} and
6822                !!!next-token;                not ($self->{open_elements}->[-1]->[1] & FRAMESET_EL)) {
6823                redo B;              !!!cp ('t327');
6824              } elsif ($token->{tag_name} eq 'select') {              $self->{insertion_mode} = AFTER_FRAMESET_IM;
6825                !!!parse-error (type => 'not closed:select');            } else {
6826                ## As if </select> instead              !!!cp ('t328');
6827                ## have an element in table scope            }
6828                my $i;            next B;
6829                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          } elsif ($token->{tag_name} eq 'html' and
6830                  my $node = $self->{open_elements}->[$_];                   $self->{insertion_mode} == AFTER_FRAMESET_IM) {
6831                  if ($node->[1] eq $token->{tag_name}) {            !!!cp ('t329');
6832                    $i = $_;            $self->{insertion_mode} = AFTER_HTML_FRAMESET_IM;
6833                    last INSCOPE;            !!!next-token;
6834                  } elsif ({            next B;
6835                            table => 1, html => 1,          } else {
6836                           }->{$node->[1]}) {            if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6837                    last INSCOPE;              !!!cp ('t330');
6838                  }              !!!parse-error (type => 'in frameset:/',
6839                } # INSCOPE                              text => $token->{tag_name}, token => $token);
6840                unless (defined $i) {            } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6841                  !!!parse-error (type => 'unmatched end tag:select');              !!!cp ('t330.1');
6842                  ## Ignore the token              !!!parse-error (type => 'after frameset:/',
6843                  !!!next-token;                              text => $token->{tag_name}, token => $token);
6844                  redo B;            } else { # "after after html"
6845                }              !!!cp ('t331');
6846                              !!!parse-error (type => 'after after frameset:/',
6847                splice @{$self->{open_elements}}, $i;                              text => $token->{tag_name}, token => $token);
6848              }
6849              ## Ignore the token
6850              !!!next-token;
6851              next B;
6852            }
6853          } elsif ($token->{type} == END_OF_FILE_TOKEN) {
6854            unless ($self->{open_elements}->[-1]->[1] & HTML_EL and
6855                    @{$self->{open_elements}} == 1) { # redundant, maybe
6856              !!!cp ('t331.1');
6857              !!!parse-error (type => 'in body:#eof', token => $token);
6858            } else {
6859              !!!cp ('t331.2');
6860            }
6861            
6862            ## Stop parsing
6863            last B;
6864          } else {
6865            die "$0: $token->{type}: Unknown token type";
6866          }
6867        } else {
6868          die "$0: $self->{insertion_mode}: Unknown insertion mode";
6869        }
6870    
6871                $self->_reset_insertion_mode;      ## "in body" insertion mode
6872        if ($token->{type} == START_TAG_TOKEN) {
6873          if ($token->{tag_name} eq 'script') {
6874            !!!cp ('t332');
6875            ## NOTE: This is an "as if in head" code clone
6876            $script_start_tag->();
6877            next B;
6878          } elsif ($token->{tag_name} eq 'style') {
6879            !!!cp ('t333');
6880            ## NOTE: This is an "as if in head" code clone
6881            $parse_rcdata->(CDATA_CONTENT_MODEL);
6882            next B;
6883          } elsif ({
6884                    base => 1, command => 1, eventsource => 1, link => 1,
6885                   }->{$token->{tag_name}}) {
6886            !!!cp ('t334');
6887            ## NOTE: This is an "as if in head" code clone, only "-t" differs
6888            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6889            pop @{$self->{open_elements}};
6890            !!!ack ('t334.1');
6891            !!!next-token;
6892            next B;
6893          } elsif ($token->{tag_name} eq 'meta') {
6894            ## NOTE: This is an "as if in head" code clone, only "-t" differs
6895            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
6896            my $meta_el = pop @{$self->{open_elements}};
6897    
6898                !!!next-token;          unless ($self->{confident}) {
6899                redo B;            if ($token->{attributes}->{charset}) {
6900              } else {              !!!cp ('t335');
6901                #              ## NOTE: Whether the encoding is supported or not is handled
6902                ## in the {change_encoding} callback.
6903                $self->{change_encoding}
6904                    ->($self, $token->{attributes}->{charset}->{value}, $token);
6905                
6906                $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
6907                    ->set_user_data (manakai_has_reference =>
6908                                         $token->{attributes}->{charset}
6909                                             ->{has_reference});
6910              } elsif ($token->{attributes}->{content}) {
6911                if ($token->{attributes}->{content}->{value}
6912                    =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
6913                        [\x09\x0A\x0C\x0D\x20]*=
6914                        [\x09\x0A\x0C\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
6915                        ([^"'\x09\x0A\x0C\x0D\x20][^\x09\x0A\x0C\x0D\x20\x3B]*))
6916                       /x) {
6917                  !!!cp ('t336');
6918                  ## NOTE: Whether the encoding is supported or not is handled
6919                  ## in the {change_encoding} callback.
6920                  $self->{change_encoding}
6921                      ->($self, defined $1 ? $1 : defined $2 ? $2 : $3, $token);
6922                  $meta_el->[0]->get_attribute_node_ns (undef, 'content')
6923                      ->set_user_data (manakai_has_reference =>
6924                                           $token->{attributes}->{content}
6925                                                 ->{has_reference});
6926              }              }
6927            } elsif ($token->{type} eq 'end tag') {            }
6928              if ($token->{tag_name} eq 'optgroup') {          } else {
6929                if ($self->{open_elements}->[-1]->[1] eq 'option' and            if ($token->{attributes}->{charset}) {
6930                    $self->{open_elements}->[-2]->[1] eq 'optgroup') {              !!!cp ('t337');
6931                  ## As if </option>              $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
6932                  splice @{$self->{open_elements}}, -2;                  ->set_user_data (manakai_has_reference =>
6933                } elsif ($self->{open_elements}->[-1]->[1] eq 'optgroup') {                                       $token->{attributes}->{charset}
6934                  pop @{$self->{open_elements}};                                           ->{has_reference});
6935                } else {            }
6936                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});            if ($token->{attributes}->{content}) {
6937                  ## Ignore the token              !!!cp ('t338');
6938                }              $meta_el->[0]->get_attribute_node_ns (undef, 'content')
6939                !!!next-token;                  ->set_user_data (manakai_has_reference =>
6940                redo B;                                       $token->{attributes}->{content}
6941              } elsif ($token->{tag_name} eq 'option') {                                           ->{has_reference});
6942                if ($self->{open_elements}->[-1]->[1] eq 'option') {            }
6943                  pop @{$self->{open_elements}};          }
               } else {  
                 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
                 ## Ignore the token  
               }  
               !!!next-token;  
               redo B;  
             } elsif ($token->{tag_name} eq 'select') {  
               ## have an element in table scope  
               my $i;  
               INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
                 my $node = $self->{open_elements}->[$_];  
                 if ($node->[1] eq $token->{tag_name}) {  
                   $i = $_;  
                   last INSCOPE;  
                 } elsif ({  
                           table => 1, html => 1,  
                          }->{$node->[1]}) {  
                   last INSCOPE;  
                 }  
               } # INSCOPE  
               unless (defined $i) {  
                 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
                 ## Ignore the token  
                 !!!next-token;  
                 redo B;  
               }  
                 
               splice @{$self->{open_elements}}, $i;  
   
               $self->_reset_insertion_mode;  
6944    
6945                !!!next-token;          !!!ack ('t338.1');
6946                redo B;          !!!next-token;
6947              } elsif ({          next B;
6948                        caption => 1, table => 1, tbody => 1,        } elsif ($token->{tag_name} eq 'title') {
6949                        tfoot => 1, thead => 1, tr => 1, td => 1, th => 1,          !!!cp ('t341');
6950                       }->{$token->{tag_name}}) {          ## NOTE: This is an "as if in head" code clone
6951                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});          $parse_rcdata->(RCDATA_CONTENT_MODEL);
6952                          next B;
6953                ## have an element in table scope        } elsif ($token->{tag_name} eq 'body') {
6954                my $i;          !!!parse-error (type => 'in body', text => 'body', token => $token);
               INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
                 my $node = $self->{open_elements}->[$_];  
                 if ($node->[1] eq $token->{tag_name}) {  
                   $i = $_;  
                   last INSCOPE;  
                 } elsif ({  
                           table => 1, html => 1,  
                          }->{$node->[1]}) {  
                   last INSCOPE;  
                 }  
               } # INSCOPE  
               unless (defined $i) {  
                 ## Ignore the token  
                 !!!next-token;  
                 redo B;  
               }  
                 
               ## As if </select>  
               ## have an element in table scope  
               undef $i;  
               INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
                 my $node = $self->{open_elements}->[$_];  
                 if ($node->[1] eq 'select') {  
                   $i = $_;  
                   last INSCOPE;  
                 } elsif ({  
                           table => 1, html => 1,  
                          }->{$node->[1]}) {  
                   last INSCOPE;  
                 }  
               } # INSCOPE  
               unless (defined $i) {  
                 !!!parse-error (type => 'unmatched end tag:select');  
                 ## Ignore the </select> token  
                 !!!next-token; ## TODO: ok?  
                 redo B;  
               }  
6955                                
6956                splice @{$self->{open_elements}}, $i;          if (@{$self->{open_elements}} == 1 or
6957                not ($self->{open_elements}->[1]->[1] & BODY_EL)) {
6958              !!!cp ('t342');
6959              ## Ignore the token
6960            } else {
6961              my $body_el = $self->{open_elements}->[1]->[0];
6962              for my $attr_name (keys %{$token->{attributes}}) {
6963                unless ($body_el->has_attribute_ns (undef, $attr_name)) {
6964                  !!!cp ('t343');
6965                  $body_el->set_attribute_ns
6966                    (undef, [undef, $attr_name],
6967                     $token->{attributes}->{$attr_name}->{value});
6968                }
6969              }
6970            }
6971            !!!nack ('t343.1');
6972            !!!next-token;
6973            next B;
6974          } elsif ({
6975                    ## NOTE: Start tags for non-phrasing flow content elements
6976    
6977                $self->_reset_insertion_mode;                  ## NOTE: The normal one
6978                    address => 1, article => 1, aside => 1, blockquote => 1,
6979                    center => 1, datagrid => 1, details => 1, dialog => 1,
6980                    dir => 1, div => 1, dl => 1, fieldset => 1, figure => 1,
6981                    footer => 1, h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1,
6982                    h6 => 1, header => 1, menu => 1, nav => 1, ol => 1, p => 1,
6983                    section => 1, ul => 1,
6984                    ## NOTE: As normal, but drops leading newline
6985                    pre => 1, listing => 1,
6986                    ## NOTE: As normal, but interacts with the form element pointer
6987                    form => 1,
6988                    
6989                    table => 1,
6990                    hr => 1,
6991                   }->{$token->{tag_name}}) {
6992            if ($token->{tag_name} eq 'form' and defined $self->{form_element}) {
6993              !!!cp ('t350');
6994              !!!parse-error (type => 'in form:form', token => $token);
6995              ## Ignore the token
6996              !!!nack ('t350.1');
6997              !!!next-token;
6998              next B;
6999            }
7000    
7001                ## reprocess          ## has a p element in scope
7002                redo B;          INSCOPE: for (reverse @{$self->{open_elements}}) {
7003              if ($_->[1] & P_EL) {
7004                !!!cp ('t344');
7005                !!!back-token; # <form>
7006                $token = {type => END_TAG_TOKEN, tag_name => 'p',
7007                          line => $token->{line}, column => $token->{column}};
7008                next B;
7009              } elsif ($_->[1] & SCOPING_EL) {
7010                !!!cp ('t345');
7011                last INSCOPE;
7012              }
7013            } # INSCOPE
7014              
7015            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7016            if ($token->{tag_name} eq 'pre' or $token->{tag_name} eq 'listing') {
7017              !!!nack ('t346.1');
7018              !!!next-token;
7019              if ($token->{type} == CHARACTER_TOKEN) {
7020                $token->{data} =~ s/^\x0A//;
7021                unless (length $token->{data}) {
7022                  !!!cp ('t346');
7023                  !!!next-token;
7024              } else {              } else {
7025                #                !!!cp ('t349');
7026              }              }
7027            } else {            } else {
7028              #              !!!cp ('t348');
7029            }            }
7030            } elsif ($token->{tag_name} eq 'form') {
7031              !!!cp ('t347.1');
7032              $self->{form_element} = $self->{open_elements}->[-1]->[0];
7033    
7034            !!!parse-error (type => 'in select:'.$token->{tag_name});            !!!nack ('t347.2');
           ## Ignore the token  
7035            !!!next-token;            !!!next-token;
7036            redo B;          } elsif ($token->{tag_name} eq 'table') {
7037          } elsif ($self->{insertion_mode} eq 'after body') {            !!!cp ('t382');
7038            if ($token->{type} eq 'character') {            push @{$open_tables}, [$self->{open_elements}->[-1]->[0]];
7039              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {            
7040                my $data = $1;            $self->{insertion_mode} = IN_TABLE_IM;
               ## As if in body  
               $reconstruct_active_formatting_elements->($insert_to_current);  
                 
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);  
7041    
7042                unless (length $token->{data}) {            !!!nack ('t382.1');
7043                  !!!next-token;            !!!next-token;
7044                  redo B;          } elsif ($token->{tag_name} eq 'hr') {
7045                }            !!!cp ('t386');
7046              }            pop @{$self->{open_elements}};
7047                        
7048              #            !!!nack ('t386.1');
7049              !!!parse-error (type => 'after body:#'.$token->{type});            !!!next-token;
7050            } elsif ($token->{type} eq 'comment') {          } else {
7051              my $comment = $self->{document}->create_comment ($token->{data});            !!!nack ('t347.1');
7052              $self->{open_elements}->[0]->[0]->append_child ($comment);            !!!next-token;
7053              !!!next-token;          }
7054              redo B;          next B;
7055            } elsif ($token->{type} eq 'start tag') {        } elsif ($token->{tag_name} eq 'li') {
7056              !!!parse-error (type => 'after body:'.$token->{tag_name});          ## NOTE: As normal, but imply </li> when there's another <li> ...
7057              #  
7058            } elsif ($token->{type} eq 'end tag') {          ## NOTE: Special, Scope (<li><foo><li> == <li><foo><li/></foo></li>)
7059              if ($token->{tag_name} eq 'html') {            ## Interpreted as <li><foo/></li><li/> (non-conforming)
7060                if (defined $self->{inner_html_node}) {            ## blockquote (O9.27), center (O), dd (Fx3, O, S3.1.2, IE7),
7061                  !!!parse-error (type => 'unmatched end tag:html');            ## dt (Fx, O, S, IE), dl (O), fieldset (O, S, IE), form (Fx, O, S),
7062                  ## Ignore the token            ## hn (O), pre (O), applet (O, S), button (O, S), marquee (Fx, O, S),
7063                  !!!next-token;            ## object (Fx)
7064                  redo B;            ## Generate non-tree (non-conforming)
7065              ## basefont (IE7 (where basefont is non-void)), center (IE),
7066              ## form (IE), hn (IE)
7067            ## address, div, p (<li><foo><li> == <li><foo/></li><li/>)
7068              ## Interpreted as <li><foo><li/></foo></li> (non-conforming)
7069              ## div (Fx, S)
7070    
7071            my $non_optional;
7072            my $i = -1;
7073    
7074            ## 1.
7075            for my $node (reverse @{$self->{open_elements}}) {
7076              if ($node->[1] & LI_EL) {
7077                ## 2. (a) As if </li>
7078                {
7079                  ## If no </li> - not applied
7080                  #
7081    
7082                  ## Otherwise
7083    
7084                  ## 1. generate implied end tags, except for </li>
7085                  #
7086    
7087                  ## 2. If current node != "li", parse error
7088                  if ($non_optional) {
7089                    !!!parse-error (type => 'not closed',
7090                                    text => $non_optional->[0]->manakai_local_name,
7091                                    token => $token);
7092                    !!!cp ('t355');
7093                } else {                } else {
7094                  $previous_insertion_mode = $self->{insertion_mode};                  !!!cp ('t356');
                 $self->{insertion_mode} = 'trailing end';  
                 !!!next-token;  
                 redo B;  
7095                }                }
7096              } else {  
7097                !!!parse-error (type => 'after body:/'.$token->{tag_name});                ## 3. Pop
7098                  splice @{$self->{open_elements}}, $i;
7099              }              }
7100    
7101                last; ## 2. (b) goto 5.
7102              } elsif (
7103                       ## NOTE: not "formatting" and not "phrasing"
7104                       ($node->[1] & SPECIAL_EL or
7105                        $node->[1] & SCOPING_EL) and
7106                       ## NOTE: "li", "dt", and "dd" are in |SPECIAL_EL|.
7107    
7108                       (not $node->[1] & ADDRESS_EL) &
7109                       (not $node->[1] & DIV_EL) &
7110                       (not $node->[1] & P_EL)) {
7111                ## 3.
7112                !!!cp ('t357');
7113                last; ## goto 5.
7114              } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
7115                !!!cp ('t358');
7116                #
7117            } else {            } else {
7118              !!!parse-error (type => 'after body:#'.$token->{type});              !!!cp ('t359');
7119                $non_optional ||= $node;
7120                #
7121            }            }
7122              ## 4.
7123              ## goto 2.
7124              $i--;
7125            }
7126    
7127            $self->{insertion_mode} = 'in body';          ## 5. (a) has a |p| element in scope
7128            ## reprocess          INSCOPE: for (reverse @{$self->{open_elements}}) {
7129            redo B;            if ($_->[1] & P_EL) {
7130          } elsif ($self->{insertion_mode} eq 'in frameset') {              !!!cp ('t353');
           if ($token->{type} eq 'character') {  
             if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {  
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);  
7131    
7132                unless (length $token->{data}) {              ## NOTE: |<p><li>|, for example.
                 !!!next-token;  
                 redo B;  
               }  
             }  
7133    
7134              #              !!!back-token; # <x>
7135            } elsif ($token->{type} eq 'comment') {              $token = {type => END_TAG_TOKEN, tag_name => 'p',
7136              my $comment = $self->{document}->create_comment ($token->{data});                        line => $token->{line}, column => $token->{column}};
7137              $self->{open_elements}->[-1]->[0]->append_child ($comment);              next B;
7138              !!!next-token;            } elsif ($_->[1] & SCOPING_EL) {
7139              redo B;              !!!cp ('t354');
7140            } elsif ($token->{type} eq 'start tag') {              last INSCOPE;
7141              if ($token->{tag_name} eq 'frameset') {            }
7142                !!!insert-element ($token->{tag_name}, $token->{attributes});          } # INSCOPE
7143                !!!next-token;  
7144                redo B;          ## 5. (b) insert
7145              } elsif ($token->{tag_name} eq 'frame') {          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7146                !!!insert-element ($token->{tag_name}, $token->{attributes});          !!!nack ('t359.1');
7147                pop @{$self->{open_elements}};          !!!next-token;
7148                !!!next-token;          next B;
7149                redo B;        } elsif ($token->{tag_name} eq 'dt' or
7150              } elsif ($token->{tag_name} eq 'noframes') {                 $token->{tag_name} eq 'dd') {
7151                $in_body->($insert_to_current);          ## NOTE: As normal, but imply </dt> or </dd> when ...
7152                redo B;  
7153              } else {          my $non_optional;
7154            my $i = -1;
7155    
7156            ## 1.
7157            for my $node (reverse @{$self->{open_elements}}) {
7158              if ($node->[1] & DT_EL or $node->[1] & DD_EL) {
7159                ## 2. (a) As if </li>
7160                {
7161                  ## If no </li> - not applied
7162                #                #
7163              }  
7164            } elsif ($token->{type} eq 'end tag') {                ## Otherwise
7165              if ($token->{tag_name} eq 'frameset') {  
7166                if ($self->{open_elements}->[-1]->[1] eq 'html' and                ## 1. generate implied end tags, except for </dt> or </dd>
7167                    @{$self->{open_elements}} == 1) {                #
7168                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
7169                  ## Ignore the token                ## 2. If current node != "dt"|"dd", parse error
7170                  !!!next-token;                if ($non_optional) {
7171                    !!!parse-error (type => 'not closed',
7172                                    text => $non_optional->[0]->manakai_local_name,
7173                                    token => $token);
7174                    !!!cp ('t355.1');
7175                } else {                } else {
7176                  pop @{$self->{open_elements}};                  !!!cp ('t356.1');
                 !!!next-token;  
7177                }                }
7178                  
7179                ## if not inner_html and                ## 3. Pop
7180                if ($self->{open_elements}->[-1]->[1] ne 'frameset') {                splice @{$self->{open_elements}}, $i;
                 $self->{insertion_mode} = 'after frameset';  
               }  
               redo B;  
             } else {  
               #  
7181              }              }
7182    
7183                last; ## 2. (b) goto 5.
7184              } elsif (
7185                       ## NOTE: not "formatting" and not "phrasing"
7186                       ($node->[1] & SPECIAL_EL or
7187                        $node->[1] & SCOPING_EL) and
7188                       ## NOTE: "li", "dt", and "dd" are in |SPECIAL_EL|.
7189    
7190                       (not $node->[1] & ADDRESS_EL) &
7191                       (not $node->[1] & DIV_EL) &
7192                       (not $node->[1] & P_EL)) {
7193                ## 3.
7194                !!!cp ('t357.1');
7195                last; ## goto 5.
7196              } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
7197                !!!cp ('t358.1');
7198                #
7199            } else {            } else {
7200                !!!cp ('t359.1');
7201                $non_optional ||= $node;
7202              #              #
7203            }            }
7204              ## 4.
7205              ## goto 2.
7206              $i--;
7207            }
7208    
7209            ## 5. (a) has a |p| element in scope
7210            INSCOPE: for (reverse @{$self->{open_elements}}) {
7211              if ($_->[1] & P_EL) {
7212                !!!cp ('t353.1');
7213                !!!back-token; # <x>
7214                $token = {type => END_TAG_TOKEN, tag_name => 'p',
7215                          line => $token->{line}, column => $token->{column}};
7216                next B;
7217              } elsif ($_->[1] & SCOPING_EL) {
7218                !!!cp ('t354.1');
7219                last INSCOPE;
7220              }
7221            } # INSCOPE
7222    
7223            ## 5. (b) insert
7224            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7225            !!!nack ('t359.2');
7226            !!!next-token;
7227            next B;
7228          } elsif ($token->{tag_name} eq 'plaintext') {
7229            ## NOTE: As normal, but effectively ends parsing
7230    
7231            ## has a p element in scope
7232            INSCOPE: for (reverse @{$self->{open_elements}}) {
7233              if ($_->[1] & P_EL) {
7234                !!!cp ('t367');
7235                !!!back-token; # <plaintext>
7236                $token = {type => END_TAG_TOKEN, tag_name => 'p',
7237                          line => $token->{line}, column => $token->{column}};
7238                next B;
7239              } elsif ($_->[1] & SCOPING_EL) {
7240                !!!cp ('t368');
7241                last INSCOPE;
7242              }
7243            } # INSCOPE
7244                        
7245            if (defined $token->{tag_name}) {          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7246              !!!parse-error (type => 'in frameset:'.($token->{type} eq 'end tag' ? '/' : '').$token->{tag_name});            
7247            } else {          $self->{content_model} = PLAINTEXT_CONTENT_MODEL;
7248              !!!parse-error (type => 'in frameset:#'.$token->{type});            
7249            !!!nack ('t368.1');
7250            !!!next-token;
7251            next B;
7252          } elsif ($token->{tag_name} eq 'a') {
7253            AFE: for my $i (reverse 0..$#$active_formatting_elements) {
7254              my $node = $active_formatting_elements->[$i];
7255              if ($node->[1] & A_EL) {
7256                !!!cp ('t371');
7257                !!!parse-error (type => 'in a:a', token => $token);
7258                
7259                !!!back-token; # <a>
7260                $token = {type => END_TAG_TOKEN, tag_name => 'a',
7261                          line => $token->{line}, column => $token->{column}};
7262                $formatting_end_tag->($token);
7263                
7264                AFE2: for (reverse 0..$#$active_formatting_elements) {
7265                  if ($active_formatting_elements->[$_]->[0] eq $node->[0]) {
7266                    !!!cp ('t372');
7267                    splice @$active_formatting_elements, $_, 1;
7268                    last AFE2;
7269                  }
7270                } # AFE2
7271                OE: for (reverse 0..$#{$self->{open_elements}}) {
7272                  if ($self->{open_elements}->[$_]->[0] eq $node->[0]) {
7273                    !!!cp ('t373');
7274                    splice @{$self->{open_elements}}, $_, 1;
7275                    last OE;
7276                  }
7277                } # OE
7278                last AFE;
7279              } elsif ($node->[0] eq '#marker') {
7280                !!!cp ('t374');
7281                last AFE;
7282            }            }
7283            } # AFE
7284              
7285            $reconstruct_active_formatting_elements->($insert_to_current);
7286    
7287            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7288            push @$active_formatting_elements, $self->{open_elements}->[-1];
7289    
7290            !!!nack ('t374.1');
7291            !!!next-token;
7292            next B;
7293          } elsif ($token->{tag_name} eq 'nobr') {
7294            $reconstruct_active_formatting_elements->($insert_to_current);
7295    
7296            ## has a |nobr| element in scope
7297            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7298              my $node = $self->{open_elements}->[$_];
7299              if ($node->[1] & NOBR_EL) {
7300                !!!cp ('t376');
7301                !!!parse-error (type => 'in nobr:nobr', token => $token);
7302                !!!back-token; # <nobr>
7303                $token = {type => END_TAG_TOKEN, tag_name => 'nobr',
7304                          line => $token->{line}, column => $token->{column}};
7305                next B;
7306              } elsif ($node->[1] & SCOPING_EL) {
7307                !!!cp ('t377');
7308                last INSCOPE;
7309              }
7310            } # INSCOPE
7311            
7312            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7313            push @$active_formatting_elements, $self->{open_elements}->[-1];
7314            
7315            !!!nack ('t377.1');
7316            !!!next-token;
7317            next B;
7318          } elsif ($token->{tag_name} eq 'button') {
7319            ## has a button element in scope
7320            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7321              my $node = $self->{open_elements}->[$_];
7322              if ($node->[1] & BUTTON_EL) {
7323                !!!cp ('t378');
7324                !!!parse-error (type => 'in button:button', token => $token);
7325                !!!back-token; # <button>
7326                $token = {type => END_TAG_TOKEN, tag_name => 'button',
7327                          line => $token->{line}, column => $token->{column}};
7328                next B;
7329              } elsif ($node->[1] & SCOPING_EL) {
7330                !!!cp ('t379');
7331                last INSCOPE;
7332              }
7333            } # INSCOPE
7334              
7335            $reconstruct_active_formatting_elements->($insert_to_current);
7336              
7337            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7338    
7339            ## TODO: associate with $self->{form_element} if defined
7340    
7341            push @$active_formatting_elements, ['#marker', ''];
7342    
7343            !!!nack ('t379.1');
7344            !!!next-token;
7345            next B;
7346          } elsif ({
7347                    xmp => 1,
7348                    iframe => 1,
7349                    noembed => 1,
7350                    noframes => 1, ## NOTE: This is an "as if in head" code clone.
7351                    noscript => 0, ## TODO: 1 if scripting is enabled
7352                   }->{$token->{tag_name}}) {
7353            if ($token->{tag_name} eq 'xmp') {
7354              !!!cp ('t381');
7355              $reconstruct_active_formatting_elements->($insert_to_current);
7356            } else {
7357              !!!cp ('t399');
7358            }
7359            ## NOTE: There is an "as if in body" code clone.
7360            $parse_rcdata->(CDATA_CONTENT_MODEL);
7361            next B;
7362          } elsif ($token->{tag_name} eq 'isindex') {
7363            !!!parse-error (type => 'isindex', token => $token);
7364            
7365            if (defined $self->{form_element}) {
7366              !!!cp ('t389');
7367            ## Ignore the token            ## Ignore the token
7368              !!!nack ('t389'); ## NOTE: Not acknowledged.
7369            !!!next-token;            !!!next-token;
7370            redo B;            next B;
7371          } elsif ($self->{insertion_mode} eq 'after frameset') {          } else {
7372            if ($token->{type} eq 'character') {            !!!ack ('t391.1');
             if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {  
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);  
7373    
7374                unless (length $token->{data}) {            my $at = $token->{attributes};
7375                  !!!next-token;            my $form_attrs;
7376                  redo B;            $form_attrs->{action} = $at->{action} if $at->{action};
7377                }            my $prompt_attr = $at->{prompt};
7378              }            $at->{name} = {name => 'name', value => 'isindex'};
7379              delete $at->{action};
7380              delete $at->{prompt};
7381              my @tokens = (
7382                            {type => START_TAG_TOKEN, tag_name => 'form',
7383                             attributes => $form_attrs,
7384                             line => $token->{line}, column => $token->{column}},
7385                            {type => START_TAG_TOKEN, tag_name => 'hr',
7386                             line => $token->{line}, column => $token->{column}},
7387                            {type => START_TAG_TOKEN, tag_name => 'p',
7388                             line => $token->{line}, column => $token->{column}},
7389                            {type => START_TAG_TOKEN, tag_name => 'label',
7390                             line => $token->{line}, column => $token->{column}},
7391                           );
7392              if ($prompt_attr) {
7393                !!!cp ('t390');
7394                push @tokens, {type => CHARACTER_TOKEN, data => $prompt_attr->{value},
7395                               #line => $token->{line}, column => $token->{column},
7396                              };
7397              } else {
7398                !!!cp ('t391');
7399                push @tokens, {type => CHARACTER_TOKEN,
7400                               data => 'This is a searchable index. Insert your search keywords here: ',
7401                               #line => $token->{line}, column => $token->{column},
7402                              }; # SHOULD
7403                ## TODO: make this configurable
7404              }
7405              push @tokens,
7406                            {type => START_TAG_TOKEN, tag_name => 'input', attributes => $at,
7407                             line => $token->{line}, column => $token->{column}},
7408                            #{type => CHARACTER_TOKEN, data => ''}, # SHOULD
7409                            {type => END_TAG_TOKEN, tag_name => 'label',
7410                             line => $token->{line}, column => $token->{column}},
7411                            {type => END_TAG_TOKEN, tag_name => 'p',
7412                             line => $token->{line}, column => $token->{column}},
7413                            {type => START_TAG_TOKEN, tag_name => 'hr',
7414                             line => $token->{line}, column => $token->{column}},
7415                            {type => END_TAG_TOKEN, tag_name => 'form',
7416                             line => $token->{line}, column => $token->{column}};
7417              !!!back-token (@tokens);
7418              !!!next-token;
7419              next B;
7420            }
7421          } elsif ($token->{tag_name} eq 'textarea') {
7422            ## Step 1
7423            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7424            
7425            ## Step 2
7426            ## TODO: $self->{form_element} if defined
7427    
7428              if ($token->{data} =~ s/^[^\x09\x0A\x0B\x0C\x20]+//) {          ## Step 3
7429                !!!parse-error (type => 'after frameset:#character');          $self->{ignore_newline} = 1;
7430    
7431                ## Ignore the token.          ## Step 4
7432                if (length $token->{data}) {          ## ISSUE: This step is wrong. (r2302 enbugged)
7433                  ## reprocess the rest of characters  
7434                } else {          ## Step 5
7435                  !!!next-token;          $self->{content_model} = RCDATA_CONTENT_MODEL;
7436                }          delete $self->{escape}; # MUST
7437                redo B;  
7438            ## Step 6-7
7439            $self->{insertion_mode} |= IN_CDATA_RCDATA_IM;
7440    
7441            !!!nack ('t392.1');
7442            !!!next-token;
7443            next B;
7444          } elsif ($token->{tag_name} eq 'optgroup' or
7445                   $token->{tag_name} eq 'option') {
7446            ## has an |option| element in scope
7447            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7448              my $node = $self->{open_elements}->[$_];
7449              if ($node->[1] & OPTION_EL) {
7450                !!!cp ('t397.1');
7451                ## NOTE: As if </option>
7452                !!!back-token; # <option> or <optgroup>
7453                $token = {type => END_TAG_TOKEN, tag_name => 'option',
7454                          line => $token->{line}, column => $token->{column}};
7455                next B;
7456              } elsif ($node->[1] & SCOPING_EL) {
7457                !!!cp ('t397.2');
7458                last INSCOPE;
7459              }
7460            } # INSCOPE
7461    
7462            $reconstruct_active_formatting_elements->($insert_to_current);
7463    
7464            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7465    
7466            !!!nack ('t397.3');
7467            !!!next-token;
7468            redo B;
7469          } elsif ($token->{tag_name} eq 'rt' or
7470                   $token->{tag_name} eq 'rp') {
7471            ## has a |ruby| element in scope
7472            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7473              my $node = $self->{open_elements}->[$_];
7474              if ($node->[1] & RUBY_EL) {
7475                !!!cp ('t398.1');
7476                ## generate implied end tags
7477                while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7478                  !!!cp ('t398.2');
7479                  pop @{$self->{open_elements}};
7480              }              }
7481            } elsif ($token->{type} eq 'comment') {              unless ($self->{open_elements}->[-1]->[1] & RUBY_EL) {
7482              my $comment = $self->{document}->create_comment ($token->{data});                !!!cp ('t398.3');
7483              $self->{open_elements}->[-1]->[0]->append_child ($comment);                !!!parse-error (type => 'not closed',
7484              !!!next-token;                                text => $self->{open_elements}->[-1]->[0]
7485              redo B;                                    ->manakai_local_name,
7486            } elsif ($token->{type} eq 'start tag') {                                token => $token);
7487              if ($token->{tag_name} eq 'noframes') {                pop @{$self->{open_elements}}
7488                $in_body->($insert_to_current);                    while not $self->{open_elements}->[-1]->[1] & RUBY_EL;
               redo B;  
             } else {  
               #  
7489              }              }
7490            } elsif ($token->{type} eq 'end tag') {              last INSCOPE;
7491              if ($token->{tag_name} eq 'html') {            } elsif ($node->[1] & SCOPING_EL) {
7492                $previous_insertion_mode = $self->{insertion_mode};              !!!cp ('t398.4');
7493                $self->{insertion_mode} = 'trailing end';              last INSCOPE;
7494                !!!next-token;            }
7495                redo B;          } # INSCOPE
7496              } else {  
7497                #          !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7498    
7499            !!!nack ('t398.5');
7500            !!!next-token;
7501            redo B;
7502          } elsif ($token->{tag_name} eq 'math' or
7503                   $token->{tag_name} eq 'svg') {
7504            $reconstruct_active_formatting_elements->($insert_to_current);
7505    
7506            ## "Adjust MathML attributes" ('math' only) - done in insert-element-f
7507    
7508            ## "adjust SVG attributes" ('svg' only) - done in insert-element-f
7509    
7510            ## "adjust foreign attributes" - done in insert-element-f
7511            
7512            !!!insert-element-f ($token->{tag_name} eq 'math' ? $MML_NS : $SVG_NS, $token->{tag_name}, $token->{attributes}, $token);
7513            
7514            if ($self->{self_closing}) {
7515              pop @{$self->{open_elements}};
7516              !!!ack ('t398.6');
7517            } else {
7518              !!!cp ('t398.7');
7519              $self->{insertion_mode} |= IN_FOREIGN_CONTENT_IM;
7520              ## NOTE: |<body><math><mi><svg>| -> "in foreign content" insertion
7521              ## mode, "in body" (not "in foreign content") secondary insertion
7522              ## mode, maybe.
7523            }
7524    
7525            !!!next-token;
7526            next B;
7527          } elsif ({
7528                    caption => 1, col => 1, colgroup => 1, frame => 1,
7529                    frameset => 1, head => 1,
7530                    tbody => 1, td => 1, tfoot => 1, th => 1,
7531                    thead => 1, tr => 1,
7532                   }->{$token->{tag_name}}) {
7533            !!!cp ('t401');
7534            !!!parse-error (type => 'in body',
7535                            text => $token->{tag_name}, token => $token);
7536            ## Ignore the token
7537            !!!nack ('t401.1'); ## NOTE: |<col/>| or |<frame/>| here is an error.
7538            !!!next-token;
7539            next B;
7540          } elsif ($token->{tag_name} eq 'param' or
7541                   $token->{tag_name} eq 'source') {
7542            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7543            pop @{$self->{open_elements}};
7544    
7545            !!!ack ('t398.5');
7546            !!!next-token;
7547            redo B;
7548          } else {
7549            if ($token->{tag_name} eq 'image') {
7550              !!!cp ('t384');
7551              !!!parse-error (type => 'image', token => $token);
7552              $token->{tag_name} = 'img';
7553            } else {
7554              !!!cp ('t385');
7555            }
7556    
7557            ## NOTE: There is an "as if <br>" code clone.
7558            $reconstruct_active_formatting_elements->($insert_to_current);
7559            
7560            !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
7561    
7562            if ({
7563                 applet => 1, marquee => 1, object => 1,
7564                }->{$token->{tag_name}}) {
7565              !!!cp ('t380');
7566              push @$active_formatting_elements, ['#marker', ''];
7567              !!!nack ('t380.1');
7568            } elsif ({
7569                      b => 1, big => 1, em => 1, font => 1, i => 1,
7570                      s => 1, small => 1, strike => 1,
7571                      strong => 1, tt => 1, u => 1,
7572                     }->{$token->{tag_name}}) {
7573              !!!cp ('t375');
7574              push @$active_formatting_elements, $self->{open_elements}->[-1];
7575              !!!nack ('t375.1');
7576            } elsif ($token->{tag_name} eq 'input') {
7577              !!!cp ('t388');
7578              ## TODO: associate with $self->{form_element} if defined
7579              pop @{$self->{open_elements}};
7580              !!!ack ('t388.2');
7581            } elsif ({
7582                      area => 1, basefont => 1, bgsound => 1, br => 1,
7583                      embed => 1, img => 1, spacer => 1, wbr => 1,
7584                     }->{$token->{tag_name}}) {
7585              !!!cp ('t388.1');
7586              pop @{$self->{open_elements}};
7587              !!!ack ('t388.3');
7588            } elsif ($token->{tag_name} eq 'select') {
7589              ## TODO: associate with $self->{form_element} if defined
7590            
7591              if ($self->{insertion_mode} & TABLE_IMS or
7592                  $self->{insertion_mode} & BODY_TABLE_IMS or
7593                  $self->{insertion_mode} == IN_COLUMN_GROUP_IM) {
7594                !!!cp ('t400.1');
7595                $self->{insertion_mode} = IN_SELECT_IN_TABLE_IM;
7596              } else {
7597                !!!cp ('t400.2');
7598                $self->{insertion_mode} = IN_SELECT_IM;
7599              }
7600              !!!nack ('t400.3');
7601            } else {
7602              !!!nack ('t402');
7603            }
7604            
7605            !!!next-token;
7606            next B;
7607          }
7608        } elsif ($token->{type} == END_TAG_TOKEN) {
7609          if ($token->{tag_name} eq 'body') {
7610            ## has a |body| element in scope
7611            my $i;
7612            INSCOPE: {
7613              for (reverse @{$self->{open_elements}}) {
7614                if ($_->[1] & BODY_EL) {
7615                  !!!cp ('t405');
7616                  $i = $_;
7617                  last INSCOPE;
7618                } elsif ($_->[1] & SCOPING_EL) {
7619                  !!!cp ('t405.1');
7620                  last;
7621              }              }
7622              }
7623    
7624              ## NOTE: |<marquee></body>|, |<svg><foreignobject></body>|
7625    
7626              !!!parse-error (type => 'unmatched end tag',
7627                              text => $token->{tag_name}, token => $token);
7628              ## NOTE: Ignore the token.
7629              !!!next-token;
7630              next B;
7631            } # INSCOPE
7632    
7633            for (@{$self->{open_elements}}) {
7634              unless ($_->[1] & ALL_END_TAG_OPTIONAL_EL) {
7635                !!!cp ('t403');
7636                !!!parse-error (type => 'not closed',
7637                                text => $_->[0]->manakai_local_name,
7638                                token => $token);
7639                last;
7640            } else {            } else {
7641              die "$0: $token->{type}: Unknown token type";              !!!cp ('t404');
7642            }            }
7643                      }
7644            !!!parse-error (type => 'after frameset:'.($token->{tag_name} eq 'end tag' ? '/' : '').$token->{tag_name});  
7645            $self->{insertion_mode} = AFTER_BODY_IM;
7646            !!!next-token;
7647            next B;
7648          } elsif ($token->{tag_name} eq 'html') {
7649            ## TODO: Update this code.  It seems that the code below is not
7650            ## up-to-date, though it has same effect as speced.
7651            if (@{$self->{open_elements}} > 1 and
7652                $self->{open_elements}->[1]->[1] & BODY_EL) {
7653              unless ($self->{open_elements}->[-1]->[1] & BODY_EL) {
7654                !!!cp ('t406');
7655                !!!parse-error (type => 'not closed',
7656                                text => $self->{open_elements}->[1]->[0]
7657                                    ->manakai_local_name,
7658                                token => $token);
7659              } else {
7660                !!!cp ('t407');
7661              }
7662              $self->{insertion_mode} = AFTER_BODY_IM;
7663              ## reprocess
7664              next B;
7665            } else {
7666              !!!cp ('t408');
7667              !!!parse-error (type => 'unmatched end tag',
7668                              text => $token->{tag_name}, token => $token);
7669            ## Ignore the token            ## Ignore the token
7670            !!!next-token;            !!!next-token;
7671            redo B;            next B;
7672            }
7673          } elsif ({
7674                    ## NOTE: End tags for non-phrasing flow content elements
7675    
7676                    ## NOTE: The normal ones
7677                    address => 1, article => 1, aside => 1, blockquote => 1,
7678                    center => 1, datagrid => 1, details => 1, dialog => 1,
7679                    dir => 1, div => 1, dl => 1, fieldset => 1, figure => 1,
7680                    footer => 1, header => 1, listing => 1, menu => 1, nav => 1,
7681                    ol => 1, pre => 1, section => 1, ul => 1,
7682    
7683                    ## NOTE: As normal, but ... optional tags
7684                    dd => 1, dt => 1, li => 1,
7685    
7686                    applet => 1, button => 1, marquee => 1, object => 1,
7687                   }->{$token->{tag_name}}) {
7688            ## NOTE: Code for <li> start tags includes "as if </li>" code.
7689            ## Code for <dt> or <dd> start tags includes "as if </dt> or
7690            ## </dd>" code.
7691    
7692            ## ISSUE: An issue in spec there          ## has an element in scope
7693            my $i;
7694            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7695              my $node = $self->{open_elements}->[$_];
7696              if ($node->[0]->manakai_local_name eq $token->{tag_name}) {
7697                !!!cp ('t410');
7698                $i = $_;
7699                last INSCOPE;
7700              } elsif ($node->[1] & SCOPING_EL) {
7701                !!!cp ('t411');
7702                last INSCOPE;
7703              }
7704            } # INSCOPE
7705    
7706            unless (defined $i) { # has an element in scope
7707              !!!cp ('t413');
7708              !!!parse-error (type => 'unmatched end tag',
7709                              text => $token->{tag_name}, token => $token);
7710              ## NOTE: Ignore the token.
7711          } else {          } else {
7712            die "$0: $self->{insertion_mode}: Unknown insertion mode";            ## Step 1. generate implied end tags
7713              while ({
7714                      ## END_TAG_OPTIONAL_EL
7715                      dd => ($token->{tag_name} ne 'dd'),
7716                      dt => ($token->{tag_name} ne 'dt'),
7717                      li => ($token->{tag_name} ne 'li'),
7718                      option => 1,
7719                      optgroup => 1,
7720                      p => 1,
7721                      rt => 1,
7722                      rp => 1,
7723                     }->{$self->{open_elements}->[-1]->[0]->manakai_local_name}) {
7724                !!!cp ('t409');
7725                pop @{$self->{open_elements}};
7726              }
7727    
7728              ## Step 2.
7729              if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7730                      ne $token->{tag_name}) {
7731                !!!cp ('t412');
7732                !!!parse-error (type => 'not closed',
7733                                text => $self->{open_elements}->[-1]->[0]
7734                                    ->manakai_local_name,
7735                                token => $token);
7736              } else {
7737                !!!cp ('t414');
7738              }
7739    
7740              ## Step 3.
7741              splice @{$self->{open_elements}}, $i;
7742    
7743              ## Step 4.
7744              $clear_up_to_marker->()
7745                  if {
7746                    applet => 1, button => 1, marquee => 1, object => 1,
7747                  }->{$token->{tag_name}};
7748          }          }
       }  
     } elsif ($self->{insertion_mode} eq 'trailing end') {  
       ## states in the main stage is preserved yet # MUST  
         
       if ($token->{type} eq 'DOCTYPE') {  
         !!!parse-error (type => 'after html:#DOCTYPE');  
         ## Ignore the token  
7749          !!!next-token;          !!!next-token;
7750          redo B;          next B;
7751        } elsif ($token->{type} eq 'comment') {        } elsif ($token->{tag_name} eq 'form') {
7752          my $comment = $self->{document}->create_comment ($token->{data});          ## NOTE: As normal, but interacts with the form element pointer
7753          $self->{document}->append_child ($comment);  
7754          !!!next-token;          undef $self->{form_element};
7755          redo B;  
7756        } elsif ($token->{type} eq 'character') {          ## has an element in scope
7757          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          my $i;
7758            my $data = $1;          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7759            ## As if in the main phase.            my $node = $self->{open_elements}->[$_];
7760            ## NOTE: The insertion mode in the main phase            if ($node->[1] & FORM_EL) {
7761            ## just before the phase has been changed to the trailing              !!!cp ('t418');
7762            ## end phase is either "after body" or "after frameset".              $i = $_;
7763            $reconstruct_active_formatting_elements->($insert_to_current);              last INSCOPE;
7764              } elsif ($node->[1] & SCOPING_EL) {
7765                !!!cp ('t419');
7766                last INSCOPE;
7767              }
7768            } # INSCOPE
7769    
7770            unless (defined $i) { # has an element in scope
7771              !!!cp ('t421');
7772              !!!parse-error (type => 'unmatched end tag',
7773                              text => $token->{tag_name}, token => $token);
7774              ## NOTE: Ignore the token.
7775            } else {
7776              ## Step 1. generate implied end tags
7777              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7778                !!!cp ('t417');
7779                pop @{$self->{open_elements}};
7780              }
7781                        
7782            $self->{open_elements}->[-1]->[0]->manakai_append_text ($data);            ## Step 2.
7783              if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7784                      ne $token->{tag_name}) {
7785                !!!cp ('t417.1');
7786                !!!parse-error (type => 'not closed',
7787                                text => $self->{open_elements}->[-1]->[0]
7788                                    ->manakai_local_name,
7789                                token => $token);
7790              } else {
7791                !!!cp ('t420');
7792              }  
7793                        
7794            unless (length $token->{data}) {            ## Step 3.
7795              !!!next-token;            splice @{$self->{open_elements}}, $i;
7796              redo B;          }
7797    
7798            !!!next-token;
7799            next B;
7800          } elsif ({
7801                    ## NOTE: As normal, except acts as a closer for any ...
7802                    h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
7803                   }->{$token->{tag_name}}) {
7804            ## has an element in scope
7805            my $i;
7806            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7807              my $node = $self->{open_elements}->[$_];
7808              if ($node->[1] & HEADING_EL) {
7809                !!!cp ('t423');
7810                $i = $_;
7811                last INSCOPE;
7812              } elsif ($node->[1] & SCOPING_EL) {
7813                !!!cp ('t424');
7814                last INSCOPE;
7815              }
7816            } # INSCOPE
7817    
7818            unless (defined $i) { # has an element in scope
7819              !!!cp ('t425.1');
7820              !!!parse-error (type => 'unmatched end tag',
7821                              text => $token->{tag_name}, token => $token);
7822              ## NOTE: Ignore the token.
7823            } else {
7824              ## Step 1. generate implied end tags
7825              while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7826                !!!cp ('t422');
7827                pop @{$self->{open_elements}};
7828            }            }
7829              
7830              ## Step 2.
7831              if ($self->{open_elements}->[-1]->[0]->manakai_local_name
7832                      ne $token->{tag_name}) {
7833                !!!cp ('t425');
7834                !!!parse-error (type => 'unmatched end tag',
7835                                text => $token->{tag_name}, token => $token);
7836              } else {
7837                !!!cp ('t426');
7838              }
7839    
7840              ## Step 3.
7841              splice @{$self->{open_elements}}, $i;
7842          }          }
7843            
7844            !!!next-token;
7845            next B;
7846          } elsif ($token->{tag_name} eq 'p') {
7847            ## NOTE: As normal, except </p> implies <p> and ...
7848    
7849          !!!parse-error (type => 'after html:#character');          ## has an element in scope
7850          $self->{insertion_mode} = $previous_insertion_mode;          my $non_optional;
7851          ## reprocess          my $i;
7852          redo B;          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
7853        } elsif ($token->{type} eq 'start tag' or            my $node = $self->{open_elements}->[$_];
7854                 $token->{type} eq 'end tag') {            if ($node->[1] & P_EL) {
7855          !!!parse-error (type => 'after html:'.($token->{type} eq 'end tag' ? '/' : '').$token->{tag_name});              !!!cp ('t410.1');
7856          $self->{insertion_mode} = $previous_insertion_mode;              $i = $_;
7857          ## reprocess              last INSCOPE;
7858          redo B;            } elsif ($node->[1] & SCOPING_EL) {
7859        } elsif ($token->{type} eq 'end-of-file') {              !!!cp ('t411.1');
7860          ## Stop parsing              last INSCOPE;
7861          last B;            } elsif ($node->[1] & END_TAG_OPTIONAL_EL) {
7862                ## NOTE: |END_TAG_OPTIONAL_EL| includes "p"
7863                !!!cp ('t411.2');
7864                #
7865              } else {
7866                !!!cp ('t411.3');
7867                $non_optional ||= $node;
7868                #
7869              }
7870            } # INSCOPE
7871    
7872            if (defined $i) {
7873              ## 1. Generate implied end tags
7874              #
7875    
7876              ## 2. If current node != "p", parse error
7877              if ($non_optional) {
7878                !!!cp ('t412.1');
7879                !!!parse-error (type => 'not closed',
7880                                text => $non_optional->[0]->manakai_local_name,
7881                                token => $token);
7882              } else {
7883                !!!cp ('t414.1');
7884              }
7885    
7886              ## 3. Pop
7887              splice @{$self->{open_elements}}, $i;
7888            } else {
7889              !!!cp ('t413.1');
7890              !!!parse-error (type => 'unmatched end tag',
7891                              text => $token->{tag_name}, token => $token);
7892    
7893              !!!cp ('t415.1');
7894              ## As if <p>, then reprocess the current token
7895              my $el;
7896              !!!create-element ($el, $HTML_NS, 'p',, $token);
7897              $insert->($el);
7898              ## NOTE: Not inserted into |$self->{open_elements}|.
7899            }
7900    
7901            !!!next-token;
7902            next B;
7903          } elsif ({
7904                    a => 1,
7905                    b => 1, big => 1, em => 1, font => 1, i => 1,
7906                    nobr => 1, s => 1, small => 1, strike => 1,
7907                    strong => 1, tt => 1, u => 1,
7908                   }->{$token->{tag_name}}) {
7909            !!!cp ('t427');
7910            $formatting_end_tag->($token);
7911            next B;
7912          } elsif ($token->{tag_name} eq 'br') {
7913            !!!cp ('t428');
7914            !!!parse-error (type => 'unmatched end tag',
7915                            text => 'br', token => $token);
7916    
7917            ## As if <br>
7918            $reconstruct_active_formatting_elements->($insert_to_current);
7919            
7920            my $el;
7921            !!!create-element ($el, $HTML_NS, 'br',, $token);
7922            $insert->($el);
7923            
7924            ## Ignore the token.
7925            !!!next-token;
7926            next B;
7927        } else {        } else {
7928          die "$0: $token->{type}: Unknown token";          if ($token->{tag_name} eq 'sarcasm') {
7929              sleep 0.001; # take a deep breath
7930            }
7931    
7932            ## Step 1
7933            my $node_i = -1;
7934            my $node = $self->{open_elements}->[$node_i];
7935    
7936            ## Step 2
7937            S2: {
7938              my $node_tag_name = $node->[0]->manakai_local_name;
7939              $node_tag_name =~ tr/A-Z/a-z/; # for SVG camelCase tag names
7940              if ($node_tag_name eq $token->{tag_name}) {
7941                ## Step 1
7942                ## generate implied end tags
7943                while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
7944                  !!!cp ('t430');
7945                  ## NOTE: |<ruby><rt></ruby>|.
7946                  ## ISSUE: <ruby><rt></rt> will also take this code path,
7947                  ## which seems wrong.
7948                  pop @{$self->{open_elements}};
7949                  $node_i++;
7950                }
7951            
7952                ## Step 2
7953                my $current_tag_name
7954                    = $self->{open_elements}->[-1]->[0]->manakai_local_name;
7955                $current_tag_name =~ tr/A-Z/a-z/;
7956                if ($current_tag_name ne $token->{tag_name}) {
7957                  !!!cp ('t431');
7958                  ## NOTE: <x><y></x>
7959                  !!!parse-error (type => 'not closed',
7960                                  text => $self->{open_elements}->[-1]->[0]
7961                                      ->manakai_local_name,
7962                                  token => $token);
7963                } else {
7964                  !!!cp ('t432');
7965                }
7966                
7967                ## Step 3
7968                splice @{$self->{open_elements}}, $node_i if $node_i < 0;
7969    
7970                !!!next-token;
7971                last S2;
7972              } else {
7973                ## Step 3
7974                if (not ($node->[1] & FORMATTING_EL) and
7975                    #not $phrasing_category->{$node->[1]} and
7976                    ($node->[1] & SPECIAL_EL or
7977                     $node->[1] & SCOPING_EL)) {
7978                  !!!cp ('t433');
7979                  !!!parse-error (type => 'unmatched end tag',
7980                                  text => $token->{tag_name}, token => $token);
7981                  ## Ignore the token
7982                  !!!next-token;
7983                  last S2;
7984    
7985                  ## NOTE: |<span><dd></span>a|: In Safari 3.1.2 and Opera
7986                  ## 9.27, "a" is a child of <dd> (conforming).  In
7987                  ## Firefox 3.0.2, "a" is a child of <body>.  In WinIE 7,
7988                  ## "a" is a child of both <body> and <dd>.
7989                }
7990                
7991                !!!cp ('t434');
7992              }
7993              
7994              ## Step 4
7995              $node_i--;
7996              $node = $self->{open_elements}->[$node_i];
7997              
7998              ## Step 5;
7999              redo S2;
8000            } # S2
8001            next B;
8002        }        }
8003      }      }
8004        next B;
8005      } continue { # B
8006        if ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {
8007          ## NOTE: The code below is executed in cases where it does not have
8008          ## to be, but it it is harmless even in those cases.
8009          ## has an element in scope
8010          INSCOPE: {
8011            for (reverse 0..$#{$self->{open_elements}}) {
8012              my $node = $self->{open_elements}->[$_];
8013              if ($node->[1] & FOREIGN_EL) {
8014                last INSCOPE;
8015              } elsif ($node->[1] & SCOPING_EL) {
8016                last;
8017              }
8018            }
8019            
8020            ## NOTE: No foreign element in scope.
8021            $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
8022          } # INSCOPE
8023        }
8024    } # B    } # B
8025    
8026    ## Stop parsing # MUST    ## Stop parsing # MUST
# Line 5262  sub _tree_construction_main ($) { Line 8028  sub _tree_construction_main ($) {
8028    ## TODO: script stuffs    ## TODO: script stuffs
8029  } # _tree_construct_main  } # _tree_construct_main
8030    
8031  sub set_inner_html ($$$) {  sub set_inner_html ($$$$;$) {
8032    my $class = shift;    my $class = shift;
8033    my $node = shift;    my $node = shift;
8034    my $s = \$_[0];    #my $s = \$_[0];
8035    my $onerror = $_[1];    my $onerror = $_[1];
8036      my $get_wrapper = $_[2] || sub ($) { return $_[0] };
8037    
8038      ## ISSUE: Should {confident} be true?
8039    
8040    my $nt = $node->node_type;    my $nt = $node->node_type;
8041    if ($nt == 9) {    if ($nt == 9) {
# Line 5283  sub set_inner_html ($$$) { Line 8052  sub set_inner_html ($$$) {
8052      }      }
8053    
8054      ## Step 3, 4, 5 # MUST      ## Step 3, 4, 5 # MUST
8055      $class->parse_string ($$s => $node, $onerror);      $class->parse_char_string ($_[0] => $node, $onerror, $get_wrapper);
8056    } elsif ($nt == 1) {    } elsif ($nt == 1) {
8057      ## TODO: If non-html element      ## TODO: If non-html element
8058    
8059      ## NOTE: Most of this code is copied from |parse_string|      ## NOTE: Most of this code is copied from |parse_string|
8060    
8061    ## TODO: Support for $get_wrapper
8062    
8063      ## Step 1 # MUST      ## Step 1 # MUST
8064      my $this_doc = $node->owner_document;      my $this_doc = $node->owner_document;
8065      my $doc = $this_doc->implementation->create_document;      my $doc = $this_doc->implementation->create_document;
# Line 5296  sub set_inner_html ($$$) { Line 8067  sub set_inner_html ($$$) {
8067      my $p = $class->new;      my $p = $class->new;
8068      $p->{document} = $doc;      $p->{document} = $doc;
8069    
8070      ## Step 9 # MUST      ## Step 8 # MUST
8071      my $i = 0;      my $i = 0;
8072      my $line = 1;      $p->{line_prev} = $p->{line} = 1;
8073      my $column = 0;      $p->{column_prev} = $p->{column} = 0;
8074      $p->{set_next_input_character} = sub {      require Whatpm::Charset::DecodeHandle;
8075        my $input = Whatpm::Charset::DecodeHandle::CharString->new (\($_[0]));
8076        $input = $get_wrapper->($input);
8077        $p->{set_nc} = sub {
8078        my $self = shift;        my $self = shift;
8079    
8080        pop @{$self->{prev_input_character}};        my $char = '';
8081        unshift @{$self->{prev_input_character}}, $self->{next_input_character};        if (defined $self->{next_nc}) {
8082            $char = $self->{next_nc};
8083            delete $self->{next_nc};
8084            $self->{nc} = ord $char;
8085          } else {
8086            $self->{char_buffer} = '';
8087            $self->{char_buffer_pos} = 0;
8088            
8089            my $count = $input->manakai_read_until
8090                ($self->{char_buffer}, qr/[^\x00\x0A\x0D]/,
8091                 $self->{char_buffer_pos});
8092            if ($count) {
8093              $self->{line_prev} = $self->{line};
8094              $self->{column_prev} = $self->{column};
8095              $self->{column}++;
8096              $self->{nc}
8097                  = ord substr ($self->{char_buffer},
8098                                $self->{char_buffer_pos}++, 1);
8099              return;
8100            }
8101            
8102            if ($input->read ($char, 1)) {
8103              $self->{nc} = ord $char;
8104            } else {
8105              $self->{nc} = -1;
8106              return;
8107            }
8108          }
8109    
8110          ($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column});
8111          $p->{column}++;
8112    
8113        $self->{next_input_character} = -1 and return if $i >= length $$s;        if ($self->{nc} == 0x000A) { # LF
8114        $self->{next_input_character} = ord substr $$s, $i++, 1;          $p->{line}++;
8115        $column++;          $p->{column} = 0;
8116            !!!cp ('i1');
8117        if ($self->{next_input_character} == 0x000A) { # LF        } elsif ($self->{nc} == 0x000D) { # CR
8118          $line++;  ## TODO: support for abort/streaming
8119          $column = 0;          my $next = '';
8120        } elsif ($self->{next_input_character} == 0x000D) { # CR          if ($input->read ($next, 1) and $next ne "\x0A") {
8121          $i++ if substr ($$s, $i, 1) eq "\x0A";            $self->{next_nc} = $next;
8122          $self->{next_input_character} = 0x000A; # LF # MUST          }
8123          $line++;          $self->{nc} = 0x000A; # LF # MUST
8124          $column = 0;          $p->{line}++;
8125        } elsif ($self->{next_input_character} > 0x10FFFF) {          $p->{column} = 0;
8126          $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST          !!!cp ('i2');
8127        } elsif ($self->{next_input_character} == 0x0000) { # NULL        } elsif ($self->{nc} == 0x0000) { # NULL
8128            !!!cp ('i4');
8129          !!!parse-error (type => 'NULL');          !!!parse-error (type => 'NULL');
8130          $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST          $self->{nc} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
8131        }        }
8132      };      };
8133      $p->{prev_input_character} = [-1, -1, -1];  
8134      $p->{next_input_character} = -1;      $p->{read_until} = sub {
8135              #my ($scalar, $specials_range, $offset) = @_;
8136          return 0 if defined $p->{next_nc};
8137    
8138          my $pattern = qr/[^$_[1]\x00\x0A\x0D]/;
8139          my $offset = $_[2] || 0;
8140          
8141          if ($p->{char_buffer_pos} < length $p->{char_buffer}) {
8142            pos ($p->{char_buffer}) = $p->{char_buffer_pos};
8143            if ($p->{char_buffer} =~ /\G(?>$pattern)+/) {
8144              substr ($_[0], $offset)
8145                  = substr ($p->{char_buffer}, $-[0], $+[0] - $-[0]);
8146              my $count = $+[0] - $-[0];
8147              if ($count) {
8148                $p->{column} += $count;
8149                $p->{char_buffer_pos} += $count;
8150                $p->{line_prev} = $p->{line};
8151                $p->{column_prev} = $p->{column} - 1;
8152                $p->{nc} = -1;
8153              }
8154              return $count;
8155            } else {
8156              return 0;
8157            }
8158          } else {
8159            my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]);
8160            if ($count) {
8161              $p->{column} += $count;
8162              $p->{column_prev} += $count;
8163              $p->{nc} = -1;
8164            }
8165            return $count;
8166          }
8167        }; # $p->{read_until}
8168    
8169      my $ponerror = $onerror || sub {      my $ponerror = $onerror || sub {
8170        my (%opt) = @_;        my (%opt) = @_;
8171        warn "Parse error ($opt{type}) at line $opt{line} column $opt{column}\n";        my $line = $opt{line};
8172          my $column = $opt{column};
8173          if (defined $opt{token} and defined $opt{token}->{line}) {
8174            $line = $opt{token}->{line};
8175            $column = $opt{token}->{column};
8176          }
8177          warn "Parse error ($opt{type}) at line $line column $column\n";
8178      };      };
8179      $p->{parse_error} = sub {      $p->{parse_error} = sub {
8180        $ponerror->(@_, line => $line, column => $column);        $ponerror->(line => $p->{line}, column => $p->{column}, @_);
8181      };      };
8182            
8183        my $char_onerror = sub {
8184          my (undef, $type, %opt) = @_;
8185          $ponerror->(layer => 'encode',
8186                      line => $p->{line}, column => $p->{column} + 1,
8187                      %opt, type => $type);
8188        }; # $char_onerror
8189        $input->onerror ($char_onerror);
8190    
8191      $p->_initialize_tokenizer;      $p->_initialize_tokenizer;
8192      $p->_initialize_tree_constructor;      $p->_initialize_tree_constructor;
8193    
8194      ## Step 2      ## Step 2
8195      my $node_ln = $node->local_name;      my $node_ln = $node->manakai_local_name;
8196      $p->{content_model_flag} = {      $p->{content_model} = {
8197        title => 'RCDATA',        title => RCDATA_CONTENT_MODEL,
8198        textarea => 'RCDATA',        textarea => RCDATA_CONTENT_MODEL,
8199        style => 'CDATA',        style => CDATA_CONTENT_MODEL,
8200        script => 'CDATA',        script => CDATA_CONTENT_MODEL,
8201        xmp => 'CDATA',        xmp => CDATA_CONTENT_MODEL,
8202        iframe => 'CDATA',        iframe => CDATA_CONTENT_MODEL,
8203        noembed => 'CDATA',        noembed => CDATA_CONTENT_MODEL,
8204        noframes => 'CDATA',        noframes => CDATA_CONTENT_MODEL,
8205        noscript => 'CDATA',        noscript => CDATA_CONTENT_MODEL,
8206        plaintext => 'PLAINTEXT',        plaintext => PLAINTEXT_CONTENT_MODEL,
8207      }->{$node_ln} || 'PCDATA';      }->{$node_ln};
8208         ## ISSUE: What is "the name of the element"? local name?      $p->{content_model} = PCDATA_CONTENT_MODEL
8209            unless defined $p->{content_model};
8210            ## ISSUE: What is "the name of the element"? local name?
8211    
8212      $p->{inner_html_node} = [$node, $node_ln];      $p->{inner_html_node} = [$node, $el_category->{$node_ln}];
8213          ## TODO: Foreign element OK?
8214    
8215      ## Step 4      ## Step 3
8216      my $root = $doc->create_element_ns      my $root = $doc->create_element_ns
8217        ('http://www.w3.org/1999/xhtml', [undef, 'html']);        ('http://www.w3.org/1999/xhtml', [undef, 'html']);
8218    
8219      ## Step 5 # MUST      ## Step 4 # MUST
8220      $doc->append_child ($root);      $doc->append_child ($root);
8221    
8222      ## Step 6 # MUST      ## Step 5 # MUST
8223      push @{$p->{open_elements}}, [$root, 'html'];      push @{$p->{open_elements}}, [$root, $el_category->{html}];
8224    
8225      undef $p->{head_element};      undef $p->{head_element};
8226        undef $p->{head_element_inserted};
8227    
8228      ## Step 7 # MUST      ## Step 6 # MUST
8229      $p->_reset_insertion_mode;      $p->_reset_insertion_mode;
8230    
8231      ## Step 8 # MUST      ## Step 7 # MUST
8232      my $anode = $node;      my $anode = $node;
8233      AN: while (defined $anode) {      AN: while (defined $anode) {
8234        if ($anode->node_type == 1) {        if ($anode->node_type == 1) {
8235          my $nsuri = $anode->namespace_uri;          my $nsuri = $anode->namespace_uri;
8236          if (defined $nsuri and $nsuri eq 'http://www.w3.org/1999/xhtml') {          if (defined $nsuri and $nsuri eq 'http://www.w3.org/1999/xhtml') {
8237            if ($anode->local_name eq 'form') { ## TODO: case?            if ($anode->manakai_local_name eq 'form') {
8238                !!!cp ('i5');
8239              $p->{form_element} = $anode;              $p->{form_element} = $anode;
8240              last AN;              last AN;
8241            }            }
# Line 5387  sub set_inner_html ($$$) { Line 8244  sub set_inner_html ($$$) {
8244        $anode = $anode->parent_node;        $anode = $anode->parent_node;
8245      } # AN      } # AN
8246            
8247      ## Step 3 # MUST      ## Step 9 # MUST
     ## Step 10 # MUST  
8248      {      {
8249        my $self = $p;        my $self = $p;
8250        !!!next-token;        !!!next-token;
8251      }      }
8252      $p->_tree_construction_main;      $p->_tree_construction_main;
8253    
8254      ## Step 11 # MUST      ## Step 10 # MUST
8255      my @cn = @{$node->child_nodes};      my @cn = @{$node->child_nodes};
8256      for (@cn) {      for (@cn) {
8257        $node->remove_child ($_);        $node->remove_child ($_);
8258      }      }
8259      ## ISSUE: mutation events? read-only?      ## ISSUE: mutation events? read-only?
8260    
8261      ## Step 12 # MUST      ## Step 11 # MUST
8262      @cn = @{$root->child_nodes};      @cn = @{$root->child_nodes};
8263      for (@cn) {      for (@cn) {
8264        $this_doc->adopt_node ($_);        $this_doc->adopt_node ($_);
# Line 5411  sub set_inner_html ($$$) { Line 8267  sub set_inner_html ($$$) {
8267      ## ISSUE: mutation events?      ## ISSUE: mutation events?
8268    
8269      $p->_terminate_tree_constructor;      $p->_terminate_tree_constructor;
8270    
8271        delete $p->{parse_error}; # delete loop
8272    } else {    } else {
8273      die "$0: |set_inner_html| is not defined for node of type $nt";      die "$0: |set_inner_html| is not defined for node of type $nt";
8274    }    }
# Line 5418  sub set_inner_html ($$$) { Line 8276  sub set_inner_html ($$$) {
8276    
8277  } # tree construction stage  } # tree construction stage
8278    
8279  sub get_inner_html ($$$) {  package Whatpm::HTML::RestartParser;
8280    my (undef, $node, $on_error) = @_;  push our @ISA, 'Error';
   
   ## Step 1  
   my $s = '';  
   
   my $in_cdata;  
   my $parent = $node;  
   while (defined $parent) {  
     if ($parent->node_type == 1 and  
         $parent->namespace_uri eq 'http://www.w3.org/1999/xhtml' and  
         {  
           style => 1, script => 1, xmp => 1, iframe => 1,  
           noembed => 1, noframes => 1, noscript => 1,  
         }->{$parent->local_name}) { ## TODO: case thingy  
       $in_cdata = 1;  
     }  
     $parent = $parent->parent_node;  
   }  
   
   ## Step 2  
   my @node = @{$node->child_nodes};  
   C: while (@node) {  
     my $child = shift @node;  
     unless (ref $child) {  
       if ($child eq 'cdata-out') {  
         $in_cdata = 0;  
       } else {  
         $s .= $child; # end tag  
       }  
       next C;  
     }  
       
     my $nt = $child->node_type;  
     if ($nt == 1) { # Element  
       my $tag_name = $child->tag_name; ## TODO: manakai_tag_name  
       $s .= '<' . $tag_name;  
       ## NOTE: Non-HTML case:  
       ## <http://permalink.gmane.org/gmane.org.w3c.whatwg.discuss/11191>  
   
       my @attrs = @{$child->attributes}; # sort order MUST be stable  
       for my $attr (@attrs) { # order is implementation dependent  
         my $attr_name = $attr->name; ## TODO: manakai_name  
         $s .= ' ' . $attr_name . '="';  
         my $attr_value = $attr->value;  
         ## escape  
         $attr_value =~ s/&/&amp;/g;  
         $attr_value =~ s/</&lt;/g;  
         $attr_value =~ s/>/&gt;/g;  
         $attr_value =~ s/"/&quot;/g;  
         $s .= $attr_value . '"';  
       }  
       $s .= '>';  
         
       next C if {  
         area => 1, base => 1, basefont => 1, bgsound => 1,  
         br => 1, col => 1, embed => 1, frame => 1, hr => 1,  
         img => 1, input => 1, link => 1, meta => 1, param => 1,  
         spacer => 1, wbr => 1,  
       }->{$tag_name};  
   
       $s .= "\x0A" if $tag_name eq 'pre' or $tag_name eq 'textarea';  
   
       if (not $in_cdata and {  
         style => 1, script => 1, xmp => 1, iframe => 1,  
         noembed => 1, noframes => 1, noscript => 1,  
         plaintext => 1,  
       }->{$tag_name}) {  
         unshift @node, 'cdata-out';  
         $in_cdata = 1;  
       }  
   
       unshift @node, @{$child->child_nodes}, '</' . $tag_name . '>';  
     } elsif ($nt == 3 or $nt == 4) {  
       if ($in_cdata) {  
         $s .= $child->data;  
       } else {  
         my $value = $child->data;  
         $value =~ s/&/&amp;/g;  
         $value =~ s/</&lt;/g;  
         $value =~ s/>/&gt;/g;  
         $value =~ s/"/&quot;/g;  
         $s .= $value;  
       }  
     } elsif ($nt == 8) {  
       $s .= '<!--' . $child->data . '-->';  
     } elsif ($nt == 10) {  
       $s .= '<!DOCTYPE ' . $child->name . '>';  
     } elsif ($nt == 5) { # entrefs  
       push @node, @{$child->child_nodes};  
     } else {  
       $on_error->($child) if defined $on_error;  
     }  
     ## ISSUE: This code does not support PIs.  
   } # C  
     
   ## Step 3  
   return \$s;  
 } # get_inner_html  
8281    
8282  1;  1;
8283  # $Date$  # $Date$

Legend:
Removed from v.1.35  
changed lines
  Added in v.1.205

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24