/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.24 by wakaba, Sat Jun 23 16:42:43 2007 UTC revision 1.192 by wakaba, Thu Oct 2 10:59:04 2008 UTC
# Line 1  Line 1 
1  package Whatpm::HTML;  package Whatpm::HTML;
2  use strict;  use strict;
3  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};
4    use Error qw(:try);
5    
6    ## NOTE: This module don't check all HTML5 parse errors; character
7    ## encoding related parse errors are expected to be handled by relevant
8    ## modules.
9    ## Parse errors for control characters that are not allowed in HTML5
10    ## documents, for surrogate code points, and for noncharacter code
11    ## points, as well as U+FFFD substitions for characters whose code points
12    ## is higher than U+10FFFF may be detected by combining the parser with
13    ## the checker implemented by Whatpm::Charset::UnicodeChecker (for its
14    ## usage example, see |t/HTML-tree.t| in the Whatpm package or the
15    ## WebHACC::Language::HTML module in the WebHACC package).
16    
17  ## ISSUE:  ## ISSUE:
18  ## var doc = implementation.createDocument (null, null, null);  ## var doc = implementation.createDocument (null, null, null);
19  ## doc.write ('');  ## doc.write ('');
20  ## alert (doc.compatMode);  ## alert (doc.compatMode);
21    
22  my $permitted_slash_tag_name = {  require IO::Handle;
23    base => 1,  
24    link => 1,  my $HTML_NS = q<http://www.w3.org/1999/xhtml>;
25    meta => 1,  my $MML_NS = q<http://www.w3.org/1998/Math/MathML>;
26    hr => 1,  my $SVG_NS = q<http://www.w3.org/2000/svg>;
27    br => 1,  my $XLINK_NS = q<http://www.w3.org/1999/xlink>;
28    img=> 1,  my $XML_NS = q<http://www.w3.org/XML/1998/namespace>;
29    embed => 1,  my $XMLNS_NS = q<http://www.w3.org/2000/xmlns/>;
30    param => 1,  
31    area => 1,  sub A_EL () { 0b1 }
32    col => 1,  sub ADDRESS_EL () { 0b10 }
33    input => 1,  sub BODY_EL () { 0b100 }
34    sub BUTTON_EL () { 0b1000 }
35    sub CAPTION_EL () { 0b10000 }
36    sub DD_EL () { 0b100000 }
37    sub DIV_EL () { 0b1000000 }
38    sub DT_EL () { 0b10000000 }
39    sub FORM_EL () { 0b100000000 }
40    sub FORMATTING_EL () { 0b1000000000 }
41    sub FRAMESET_EL () { 0b10000000000 }
42    sub HEADING_EL () { 0b100000000000 }
43    sub HTML_EL () { 0b1000000000000 }
44    sub LI_EL () { 0b10000000000000 }
45    sub NOBR_EL () { 0b100000000000000 }
46    sub OPTION_EL () { 0b1000000000000000 }
47    sub OPTGROUP_EL () { 0b10000000000000000 }
48    sub P_EL () { 0b100000000000000000 }
49    sub SELECT_EL () { 0b1000000000000000000 }
50    sub TABLE_EL () { 0b10000000000000000000 }
51    sub TABLE_CELL_EL () { 0b100000000000000000000 }
52    sub TABLE_ROW_EL () { 0b1000000000000000000000 }
53    sub TABLE_ROW_GROUP_EL () { 0b10000000000000000000000 }
54    sub MISC_SCOPING_EL () { 0b100000000000000000000000 }
55    sub MISC_SPECIAL_EL () { 0b1000000000000000000000000 }
56    sub FOREIGN_EL () { 0b10000000000000000000000000 }
57    sub FOREIGN_FLOW_CONTENT_EL () { 0b100000000000000000000000000 }
58    sub MML_AXML_EL () { 0b1000000000000000000000000000 }
59    sub RUBY_EL () { 0b10000000000000000000000000000 }
60    sub RUBY_COMPONENT_EL () { 0b100000000000000000000000000000 }
61    
62    sub TABLE_ROWS_EL () {
63      TABLE_EL |
64      TABLE_ROW_EL |
65      TABLE_ROW_GROUP_EL
66    }
67    
68    ## NOTE: Used in "generate implied end tags" algorithm.
69    ## NOTE: There is a code where a modified version of END_TAG_OPTIONAL_EL
70    ## is used in "generate implied end tags" implementation (search for the
71    ## function mae).
72    sub END_TAG_OPTIONAL_EL () {
73      DD_EL |
74      DT_EL |
75      LI_EL |
76      P_EL |
77      RUBY_COMPONENT_EL
78    }
79    
80    ## NOTE: Used in </body> and EOF algorithms.
81    sub ALL_END_TAG_OPTIONAL_EL () {
82      DD_EL |
83      DT_EL |
84      LI_EL |
85      P_EL |
86    
87      BODY_EL |
88      HTML_EL |
89      TABLE_CELL_EL |
90      TABLE_ROW_EL |
91      TABLE_ROW_GROUP_EL
92    }
93    
94    sub SCOPING_EL () {
95      BUTTON_EL |
96      CAPTION_EL |
97      HTML_EL |
98      TABLE_EL |
99      TABLE_CELL_EL |
100      MISC_SCOPING_EL
101    }
102    
103    sub TABLE_SCOPING_EL () {
104      HTML_EL |
105      TABLE_EL
106    }
107    
108    sub TABLE_ROWS_SCOPING_EL () {
109      HTML_EL |
110      TABLE_ROW_GROUP_EL
111    }
112    
113    sub TABLE_ROW_SCOPING_EL () {
114      HTML_EL |
115      TABLE_ROW_EL
116    }
117    
118    sub SPECIAL_EL () {
119      ADDRESS_EL |
120      BODY_EL |
121      DIV_EL |
122    
123      DD_EL |
124      DT_EL |
125      LI_EL |
126      P_EL |
127    
128      FORM_EL |
129      FRAMESET_EL |
130      HEADING_EL |
131      OPTION_EL |
132      OPTGROUP_EL |
133      SELECT_EL |
134      TABLE_ROW_EL |
135      TABLE_ROW_GROUP_EL |
136      MISC_SPECIAL_EL
137    }
138    
139    my $el_category = {
140      a => A_EL | FORMATTING_EL,
141      address => ADDRESS_EL,
142      applet => MISC_SCOPING_EL,
143      area => MISC_SPECIAL_EL,
144      b => FORMATTING_EL,
145      base => MISC_SPECIAL_EL,
146      basefont => MISC_SPECIAL_EL,
147      bgsound => MISC_SPECIAL_EL,
148      big => FORMATTING_EL,
149      blockquote => MISC_SPECIAL_EL,
150      body => BODY_EL,
151      br => MISC_SPECIAL_EL,
152      button => BUTTON_EL,
153      caption => CAPTION_EL,
154      center => MISC_SPECIAL_EL,
155      col => MISC_SPECIAL_EL,
156      colgroup => MISC_SPECIAL_EL,
157      dd => DD_EL,
158      dir => MISC_SPECIAL_EL,
159      div => DIV_EL,
160      dl => MISC_SPECIAL_EL,
161      dt => DT_EL,
162      em => FORMATTING_EL,
163      embed => MISC_SPECIAL_EL,
164      fieldset => MISC_SPECIAL_EL,
165      font => FORMATTING_EL,
166      form => FORM_EL,
167      frame => MISC_SPECIAL_EL,
168      frameset => FRAMESET_EL,
169      h1 => HEADING_EL,
170      h2 => HEADING_EL,
171      h3 => HEADING_EL,
172      h4 => HEADING_EL,
173      h5 => HEADING_EL,
174      h6 => HEADING_EL,
175      head => MISC_SPECIAL_EL,
176      hr => MISC_SPECIAL_EL,
177      html => HTML_EL,
178      i => FORMATTING_EL,
179      iframe => MISC_SPECIAL_EL,
180      img => MISC_SPECIAL_EL,
181      input => MISC_SPECIAL_EL,
182      isindex => MISC_SPECIAL_EL,
183      li => LI_EL,
184      link => MISC_SPECIAL_EL,
185      listing => MISC_SPECIAL_EL,
186      marquee => MISC_SCOPING_EL,
187      menu => MISC_SPECIAL_EL,
188      meta => MISC_SPECIAL_EL,
189      nobr => NOBR_EL | FORMATTING_EL,
190      noembed => MISC_SPECIAL_EL,
191      noframes => MISC_SPECIAL_EL,
192      noscript => MISC_SPECIAL_EL,
193      object => MISC_SCOPING_EL,
194      ol => MISC_SPECIAL_EL,
195      optgroup => OPTGROUP_EL,
196      option => OPTION_EL,
197      p => P_EL,
198      param => MISC_SPECIAL_EL,
199      plaintext => MISC_SPECIAL_EL,
200      pre => MISC_SPECIAL_EL,
201      rp => RUBY_COMPONENT_EL,
202      rt => RUBY_COMPONENT_EL,
203      ruby => RUBY_EL,
204      s => FORMATTING_EL,
205      script => MISC_SPECIAL_EL,
206      select => SELECT_EL,
207      small => FORMATTING_EL,
208      spacer => MISC_SPECIAL_EL,
209      strike => FORMATTING_EL,
210      strong => FORMATTING_EL,
211      style => MISC_SPECIAL_EL,
212      table => TABLE_EL,
213      tbody => TABLE_ROW_GROUP_EL,
214      td => TABLE_CELL_EL,
215      textarea => MISC_SPECIAL_EL,
216      tfoot => TABLE_ROW_GROUP_EL,
217      th => TABLE_CELL_EL,
218      thead => TABLE_ROW_GROUP_EL,
219      title => MISC_SPECIAL_EL,
220      tr => TABLE_ROW_EL,
221      tt => FORMATTING_EL,
222      u => FORMATTING_EL,
223      ul => MISC_SPECIAL_EL,
224      wbr => MISC_SPECIAL_EL,
225    };
226    
227    my $el_category_f = {
228      $MML_NS => {
229        'annotation-xml' => MML_AXML_EL,
230        mi => FOREIGN_FLOW_CONTENT_EL,
231        mo => FOREIGN_FLOW_CONTENT_EL,
232        mn => FOREIGN_FLOW_CONTENT_EL,
233        ms => FOREIGN_FLOW_CONTENT_EL,
234        mtext => FOREIGN_FLOW_CONTENT_EL,
235      },
236      $SVG_NS => {
237        foreignObject => FOREIGN_FLOW_CONTENT_EL,
238        desc => FOREIGN_FLOW_CONTENT_EL,
239        title => FOREIGN_FLOW_CONTENT_EL,
240      },
241      ## NOTE: In addition, FOREIGN_EL is set to non-HTML elements.
242    };
243    
244    my $svg_attr_name = {
245      attributename => 'attributeName',
246      attributetype => 'attributeType',
247      basefrequency => 'baseFrequency',
248      baseprofile => 'baseProfile',
249      calcmode => 'calcMode',
250      clippathunits => 'clipPathUnits',
251      contentscripttype => 'contentScriptType',
252      contentstyletype => 'contentStyleType',
253      diffuseconstant => 'diffuseConstant',
254      edgemode => 'edgeMode',
255      externalresourcesrequired => 'externalResourcesRequired',
256      filterres => 'filterRes',
257      filterunits => 'filterUnits',
258      glyphref => 'glyphRef',
259      gradienttransform => 'gradientTransform',
260      gradientunits => 'gradientUnits',
261      kernelmatrix => 'kernelMatrix',
262      kernelunitlength => 'kernelUnitLength',
263      keypoints => 'keyPoints',
264      keysplines => 'keySplines',
265      keytimes => 'keyTimes',
266      lengthadjust => 'lengthAdjust',
267      limitingconeangle => 'limitingConeAngle',
268      markerheight => 'markerHeight',
269      markerunits => 'markerUnits',
270      markerwidth => 'markerWidth',
271      maskcontentunits => 'maskContentUnits',
272      maskunits => 'maskUnits',
273      numoctaves => 'numOctaves',
274      pathlength => 'pathLength',
275      patterncontentunits => 'patternContentUnits',
276      patterntransform => 'patternTransform',
277      patternunits => 'patternUnits',
278      pointsatx => 'pointsAtX',
279      pointsaty => 'pointsAtY',
280      pointsatz => 'pointsAtZ',
281      preservealpha => 'preserveAlpha',
282      preserveaspectratio => 'preserveAspectRatio',
283      primitiveunits => 'primitiveUnits',
284      refx => 'refX',
285      refy => 'refY',
286      repeatcount => 'repeatCount',
287      repeatdur => 'repeatDur',
288      requiredextensions => 'requiredExtensions',
289      requiredfeatures => 'requiredFeatures',
290      specularconstant => 'specularConstant',
291      specularexponent => 'specularExponent',
292      spreadmethod => 'spreadMethod',
293      startoffset => 'startOffset',
294      stddeviation => 'stdDeviation',
295      stitchtiles => 'stitchTiles',
296      surfacescale => 'surfaceScale',
297      systemlanguage => 'systemLanguage',
298      tablevalues => 'tableValues',
299      targetx => 'targetX',
300      targety => 'targetY',
301      textlength => 'textLength',
302      viewbox => 'viewBox',
303      viewtarget => 'viewTarget',
304      xchannelselector => 'xChannelSelector',
305      ychannelselector => 'yChannelSelector',
306      zoomandpan => 'zoomAndPan',
307  };  };
308    
309  my $c1_entity_char = {  my $foreign_attr_xname = {
310      'xlink:actuate' => [$XLINK_NS, ['xlink', 'actuate']],
311      'xlink:arcrole' => [$XLINK_NS, ['xlink', 'arcrole']],
312      'xlink:href' => [$XLINK_NS, ['xlink', 'href']],
313      'xlink:role' => [$XLINK_NS, ['xlink', 'role']],
314      'xlink:show' => [$XLINK_NS, ['xlink', 'show']],
315      'xlink:title' => [$XLINK_NS, ['xlink', 'title']],
316      'xlink:type' => [$XLINK_NS, ['xlink', 'type']],
317      'xml:base' => [$XML_NS, ['xml', 'base']],
318      'xml:lang' => [$XML_NS, ['xml', 'lang']],
319      'xml:space' => [$XML_NS, ['xml', 'space']],
320      'xmlns' => [$XMLNS_NS, [undef, 'xmlns']],
321      'xmlns:xlink' => [$XMLNS_NS, ['xmlns', 'xlink']],
322    };
323    
324    ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.
325    
326    my $charref_map = {
327      0x0D => 0x000A,
328    0x80 => 0x20AC,    0x80 => 0x20AC,
329    0x81 => 0xFFFD,    0x81 => 0xFFFD,
330    0x82 => 0x201A,    0x82 => 0x201A,
# Line 54  my $c1_entity_char = { Line 357  my $c1_entity_char = {
357    0x9D => 0xFFFD,    0x9D => 0xFFFD,
358    0x9E => 0x017E,    0x9E => 0x017E,
359    0x9F => 0x0178,    0x9F => 0x0178,
360  }; # $c1_entity_char  }; # $charref_map
361    $charref_map->{$_} = 0xFFFD
362        for 0x0000..0x0008, 0x000B, 0x000E..0x001F, 0x007F,
363            0xD800..0xDFFF, 0xFDD0..0xFDDF, ## ISSUE: 0xFDEF
364            0xFFFE, 0xFFFF, 0x1FFFE, 0x1FFFF, 0x2FFFE, 0x2FFFF, 0x3FFFE, 0x3FFFF,
365            0x4FFFE, 0x4FFFF, 0x5FFFE, 0x5FFFF, 0x6FFFE, 0x6FFFF, 0x7FFFE,
366            0x7FFFF, 0x8FFFE, 0x8FFFF, 0x9FFFE, 0x9FFFF, 0xAFFFE, 0xAFFFF,
367            0xBFFFE, 0xBFFFF, 0xCFFFE, 0xCFFFF, 0xDFFFE, 0xDFFFF, 0xEFFFE,
368            0xEFFFF, 0xFFFFE, 0xFFFFF, 0x10FFFE, 0x10FFFF;
369    
370  my $special_category = {  ## TODO: Invoke the reset algorithm when a resettable element is
371    address => 1, area => 1, base => 1, basefont => 1, bgsound => 1,  ## created (cf. HTML5 revision 2259).
372    blockquote => 1, body => 1, br => 1, center => 1, col => 1, colgroup => 1,  
373    dd => 1, dir => 1, div => 1, dl => 1, dt => 1, embed => 1, fieldset => 1,  sub parse_byte_string ($$$$;$) {
374    form => 1, frame => 1, frameset => 1, h1 => 1, h2 => 1, h3 => 1,    my $self = shift;
375    h4 => 1, h5 => 1, h6 => 1, head => 1, hr => 1, iframe => 1, image => 1,    my $charset_name = shift;
376    img => 1, input => 1, isindex => 1, li => 1, link => 1, listing => 1,    open my $input, '<', ref $_[0] ? $_[0] : \($_[0]);
377    menu => 1, meta => 1, noembed => 1, noframes => 1, noscript => 1,    return $self->parse_byte_stream ($charset_name, $input, @_[1..$#_]);
378    ol => 1, optgroup => 1, option => 1, p => 1, param => 1, plaintext => 1,  } # parse_byte_string
379    pre => 1, script => 1, select => 1, spacer => 1, style => 1, tbody => 1,  
380    textarea => 1, tfoot => 1, thead => 1, title => 1, tr => 1, ul => 1, wbr => 1,  sub parse_byte_stream ($$$$;$$) {
381  };    # my ($self, $charset_name, $byte_stream, $doc, $onerror, $get_wrapper) = @_;
382  my $scoping_category = {    my $self = ref $_[0] ? shift : shift->new;
383    button => 1, caption => 1, html => 1, marquee => 1, object => 1,    my $charset_name = shift;
384    table => 1, td => 1, th => 1,    my $byte_stream = $_[0];
385  };  
386  my $formatting_category = {    my $onerror = $_[2] || sub {
387    a => 1, b => 1, big => 1, em => 1, font => 1, i => 1, nobr => 1,      my (%opt) = @_;
388    s => 1, small => 1, strile => 1, strong => 1, tt => 1, u => 1,      warn "Parse error ($opt{type})\n";
389  };    };
390  # $phrasing_category: all other elements    $self->{parse_error} = $onerror; # updated later by parse_char_string
391    
392      my $get_wrapper = $_[3] || sub ($) {
393        return $_[0]; # $_[0] = byte stream handle, returned = arg to char handle
394      };
395    
396      ## HTML5 encoding sniffing algorithm
397      require Message::Charset::Info;
398      my $charset;
399      my $buffer;
400      my ($char_stream, $e_status);
401    
402      SNIFFING: {
403        ## NOTE: By setting |allow_fallback| option true when the
404        ## |get_decode_handle| method is invoked, we ignore what the HTML5
405        ## spec requires, i.e. unsupported encoding should be ignored.
406          ## TODO: We should not do this unless the parser is invoked
407          ## in the conformance checking mode, in which this behavior
408          ## would be useful.
409    
410        ## Step 1
411        if (defined $charset_name) {
412          $charset = Message::Charset::Info->get_by_html_name ($charset_name);
413              ## TODO: Is this ok?  Transfer protocol's parameter should be
414              ## interpreted in its semantics?
415    
416          ($char_stream, $e_status) = $charset->get_decode_handle
417              ($byte_stream, allow_error_reporting => 1,
418               allow_fallback => 1);
419          if ($char_stream) {
420            $self->{confident} = 1;
421            last SNIFFING;
422          } else {
423            !!!parse-error (type => 'charset:not supported',
424                            layer => 'encode',
425                            line => 1, column => 1,
426                            value => $charset_name,
427                            level => $self->{level}->{uncertain});
428          }
429        }
430    
431        ## Step 2
432        my $byte_buffer = '';
433        for (1..1024) {
434          my $char = $byte_stream->getc;
435          last unless defined $char;
436          $byte_buffer .= $char;
437        } ## TODO: timeout
438    
439        ## Step 3
440        if ($byte_buffer =~ /^\xFE\xFF/) {
441          $charset = Message::Charset::Info->get_by_html_name ('utf-16be');
442          ($char_stream, $e_status) = $charset->get_decode_handle
443              ($byte_stream, allow_error_reporting => 1,
444               allow_fallback => 1, byte_buffer => \$byte_buffer);
445          $self->{confident} = 1;
446          last SNIFFING;
447        } elsif ($byte_buffer =~ /^\xFF\xFE/) {
448          $charset = Message::Charset::Info->get_by_html_name ('utf-16le');
449          ($char_stream, $e_status) = $charset->get_decode_handle
450              ($byte_stream, allow_error_reporting => 1,
451               allow_fallback => 1, byte_buffer => \$byte_buffer);
452          $self->{confident} = 1;
453          last SNIFFING;
454        } elsif ($byte_buffer =~ /^\xEF\xBB\xBF/) {
455          $charset = Message::Charset::Info->get_by_html_name ('utf-8');
456          ($char_stream, $e_status) = $charset->get_decode_handle
457              ($byte_stream, allow_error_reporting => 1,
458               allow_fallback => 1, byte_buffer => \$byte_buffer);
459          $self->{confident} = 1;
460          last SNIFFING;
461        }
462    
463        ## Step 4
464        ## TODO: <meta charset>
465    
466        ## Step 5
467        ## TODO: from history
468    
469        ## Step 6
470        require Whatpm::Charset::UniversalCharDet;
471        $charset_name = Whatpm::Charset::UniversalCharDet->detect_byte_string
472            ($byte_buffer);
473        if (defined $charset_name) {
474          $charset = Message::Charset::Info->get_by_html_name ($charset_name);
475    
476          ## ISSUE: Unsupported encoding is not ignored according to the spec.
477          require Whatpm::Charset::DecodeHandle;
478          $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new
479              ($byte_stream);
480          ($char_stream, $e_status) = $charset->get_decode_handle
481              ($buffer, allow_error_reporting => 1,
482               allow_fallback => 1, byte_buffer => \$byte_buffer);
483          if ($char_stream) {
484            $buffer->{buffer} = $byte_buffer;
485            !!!parse-error (type => 'sniffing:chardet',
486                            text => $charset_name,
487                            level => $self->{level}->{info},
488                            layer => 'encode',
489                            line => 1, column => 1);
490            $self->{confident} = 0;
491            last SNIFFING;
492          }
493        }
494    
495        ## Step 7: default
496        ## TODO: Make this configurable.
497        $charset = Message::Charset::Info->get_by_html_name ('windows-1252');
498            ## NOTE: We choose |windows-1252| here, since |utf-8| should be
499            ## detectable in the step 6.
500        require Whatpm::Charset::DecodeHandle;
501        $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new
502            ($byte_stream);
503        ($char_stream, $e_status)
504            = $charset->get_decode_handle ($buffer,
505                                           allow_error_reporting => 1,
506                                           allow_fallback => 1,
507                                           byte_buffer => \$byte_buffer);
508        $buffer->{buffer} = $byte_buffer;
509        !!!parse-error (type => 'sniffing:default',
510                        text => 'windows-1252',
511                        level => $self->{level}->{info},
512                        line => 1, column => 1,
513                        layer => 'encode');
514        $self->{confident} = 0;
515      } # SNIFFING
516    
517      if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
518        $self->{input_encoding} = $charset->get_iana_name; ## TODO: Should we set actual charset decoder's encoding name?
519        !!!parse-error (type => 'chardecode:fallback',
520                        #text => $self->{input_encoding},
521                        level => $self->{level}->{uncertain},
522                        line => 1, column => 1,
523                        layer => 'encode');
524      } elsif (not ($e_status &
525                    Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) {
526        $self->{input_encoding} = $charset->get_iana_name;
527        !!!parse-error (type => 'chardecode:no error',
528                        text => $self->{input_encoding},
529                        level => $self->{level}->{uncertain},
530                        line => 1, column => 1,
531                        layer => 'encode');
532      } else {
533        $self->{input_encoding} = $charset->get_iana_name;
534      }
535    
536      $self->{change_encoding} = sub {
537        my $self = shift;
538        $charset_name = shift;
539        my $token = shift;
540    
541        $charset = Message::Charset::Info->get_by_html_name ($charset_name);
542        ($char_stream, $e_status) = $charset->get_decode_handle
543            ($byte_stream, allow_error_reporting => 1, allow_fallback => 1,
544             byte_buffer => \ $buffer->{buffer});
545        
546        if ($char_stream) { # if supported
547          ## "Change the encoding" algorithm:
548    
549          ## Step 1    
550          if ($charset->{category} &
551              Message::Charset::Info::CHARSET_CATEGORY_UTF16 ()) {
552            $charset = Message::Charset::Info->get_by_html_name ('utf-8');
553            ($char_stream, $e_status) = $charset->get_decode_handle
554                ($byte_stream,
555                 byte_buffer => \ $buffer->{buffer});
556          }
557          $charset_name = $charset->get_iana_name;
558          
559          ## Step 2
560          if (defined $self->{input_encoding} and
561              $self->{input_encoding} eq $charset_name) {
562            !!!parse-error (type => 'charset label:matching',
563                            text => $charset_name,
564                            level => $self->{level}->{info});
565            $self->{confident} = 1;
566            return;
567          }
568    
569          !!!parse-error (type => 'charset label detected',
570                          text => $self->{input_encoding},
571                          value => $charset_name,
572                          level => $self->{level}->{warn},
573                          token => $token);
574          
575          ## Step 3
576          # if (can) {
577            ## change the encoding on the fly.
578            #$self->{confident} = 1;
579            #return;
580          # }
581          
582          ## Step 4
583          throw Whatpm::HTML::RestartParser ();
584        }
585      }; # $self->{change_encoding}
586    
587      my $char_onerror = sub {
588        my (undef, $type, %opt) = @_;
589        !!!parse-error (layer => 'encode',
590                        line => $self->{line}, column => $self->{column} + 1,
591                        %opt, type => $type);
592        if ($opt{octets}) {
593          ${$opt{octets}} = "\x{FFFD}"; # relacement character
594        }
595      };
596    
597      my $wrapped_char_stream = $get_wrapper->($char_stream);
598      $wrapped_char_stream->onerror ($char_onerror);
599    
600  sub parse_string ($$$;$) {    my @args = ($_[1], $_[2]); # $doc, $onerror - $get_wrapper = undef;
601    my $self = shift->new;    my $return;
602    my $s = \$_[0];    try {
603        $return = $self->parse_char_stream ($wrapped_char_stream, @args);  
604      } catch Whatpm::HTML::RestartParser with {
605        ## NOTE: Invoked after {change_encoding}.
606    
607        if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
608          $self->{input_encoding} = $charset->get_iana_name; ## TODO: Should we set actual charset decoder's encoding name?
609          !!!parse-error (type => 'chardecode:fallback',
610                          level => $self->{level}->{uncertain},
611                          #text => $self->{input_encoding},
612                          line => 1, column => 1,
613                          layer => 'encode');
614        } elsif (not ($e_status &
615                      Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) {
616          $self->{input_encoding} = $charset->get_iana_name;
617          !!!parse-error (type => 'chardecode:no error',
618                          text => $self->{input_encoding},
619                          level => $self->{level}->{uncertain},
620                          line => 1, column => 1,
621                          layer => 'encode');
622        } else {
623          $self->{input_encoding} = $charset->get_iana_name;
624        }
625        $self->{confident} = 1;
626    
627        $wrapped_char_stream = $get_wrapper->($char_stream);
628        $wrapped_char_stream->onerror ($char_onerror);
629    
630        $return = $self->parse_char_stream ($wrapped_char_stream, @args);
631      };
632      return $return;
633    } # parse_byte_stream
634    
635    ## NOTE: HTML5 spec says that the encoding layer MUST NOT strip BOM
636    ## and the HTML layer MUST ignore it.  However, we does strip BOM in
637    ## the encoding layer and the HTML layer does not ignore any U+FEFF,
638    ## because the core part of our HTML parser expects a string of character,
639    ## not a string of bytes or code units or anything which might contain a BOM.
640    ## Therefore, any parser interface that accepts a string of bytes,
641    ## such as |parse_byte_string| in this module, must ensure that it does
642    ## strip the BOM and never strip any ZWNBSP.
643    
644    sub parse_char_string ($$$;$$) {
645      #my ($self, $s, $doc, $onerror, $get_wrapper) = @_;
646      my $self = shift;
647      my $s = ref $_[0] ? $_[0] : \($_[0]);
648      require Whatpm::Charset::DecodeHandle;
649      my $input = Whatpm::Charset::DecodeHandle::CharString->new ($s);
650      return $self->parse_char_stream ($input, @_[1..$#_]);
651    } # parse_char_string
652    *parse_string = \&parse_char_string; ## NOTE: Alias for backward compatibility.
653    
654    sub parse_char_stream ($$$;$$) {
655      my $self = ref $_[0] ? shift : shift->new;
656      my $input = $_[0];
657    $self->{document} = $_[1];    $self->{document} = $_[1];
658      @{$self->{document}->child_nodes} = ();
659    
660    ## NOTE: |set_inner_html| copies most of this method's code    ## NOTE: |set_inner_html| copies most of this method's code
661    
662    my $i = 0;    $self->{confident} = 1 unless exists $self->{confident};
663    my $line = 1;    $self->{document}->input_encoding ($self->{input_encoding})
664    my $column = 0;        if defined $self->{input_encoding};
665    $self->{set_next_input_character} = sub {  ## TODO: |{input_encoding}| is needless?
666    
667      $self->{line_prev} = $self->{line} = 1;
668      $self->{column_prev} = -1;
669      $self->{column} = 0;
670      $self->{set_nc} = sub {
671      my $self = shift;      my $self = shift;
672    
673      pop @{$self->{prev_input_character}};      my $char = '';
674      unshift @{$self->{prev_input_character}}, $self->{next_input_character};      if (defined $self->{next_nc}) {
675          $char = $self->{next_nc};
676          delete $self->{next_nc};
677          $self->{nc} = ord $char;
678        } else {
679          $self->{char_buffer} = '';
680          $self->{char_buffer_pos} = 0;
681    
682      $self->{next_input_character} = -1 and return if $i >= length $$s;        my $count = $input->manakai_read_until
683      $self->{next_input_character} = ord substr $$s, $i++, 1;           ($self->{char_buffer}, qr/[^\x00\x0A\x0D]/, $self->{char_buffer_pos});
684      $column++;        if ($count) {
685            $self->{line_prev} = $self->{line};
686            $self->{column_prev} = $self->{column};
687            $self->{column}++;
688            $self->{nc}
689                = ord substr ($self->{char_buffer}, $self->{char_buffer_pos}++, 1);
690            return;
691          }
692    
693          if ($input->read ($char, 1)) {
694            $self->{nc} = ord $char;
695          } else {
696            $self->{nc} = -1;
697            return;
698          }
699        }
700    
701        ($self->{line_prev}, $self->{column_prev})
702            = ($self->{line}, $self->{column});
703        $self->{column}++;
704            
705      if ($self->{next_input_character} == 0x000A) { # LF      if ($self->{nc} == 0x000A) { # LF
706        $line++;        !!!cp ('j1');
707        $column = 0;        $self->{line}++;
708      } elsif ($self->{next_input_character} == 0x000D) { # CR        $self->{column} = 0;
709        $i++ if substr ($$s, $i, 1) eq "\x0A";      } elsif ($self->{nc} == 0x000D) { # CR
710        $self->{next_input_character} = 0x000A; # LF # MUST        !!!cp ('j2');
711        $line++;  ## TODO: support for abort/streaming
712        $column = 0;        my $next = '';
713      } elsif ($self->{next_input_character} > 0x10FFFF) {        if ($input->read ($next, 1) and $next ne "\x0A") {
714        $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST          $self->{next_nc} = $next;
715      } elsif ($self->{next_input_character} == 0x0000) { # NULL        }
716          $self->{nc} = 0x000A; # LF # MUST
717          $self->{line}++;
718          $self->{column} = 0;
719        } elsif ($self->{nc} == 0x0000) { # NULL
720          !!!cp ('j4');
721        !!!parse-error (type => 'NULL');        !!!parse-error (type => 'NULL');
722        $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST        $self->{nc} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
723      }      }
724    };    };
725    $self->{prev_input_character} = [-1, -1, -1];  
726    $self->{next_input_character} = -1;    $self->{read_until} = sub {
727        #my ($scalar, $specials_range, $offset) = @_;
728        return 0 if defined $self->{next_nc};
729    
730        my $pattern = qr/[^$_[1]\x00\x0A\x0D]/;
731        my $offset = $_[2] || 0;
732    
733        if ($self->{char_buffer_pos} < length $self->{char_buffer}) {
734          pos ($self->{char_buffer}) = $self->{char_buffer_pos};
735          if ($self->{char_buffer} =~ /\G(?>$pattern)+/) {
736            substr ($_[0], $offset)
737                = substr ($self->{char_buffer}, $-[0], $+[0] - $-[0]);
738            my $count = $+[0] - $-[0];
739            if ($count) {
740              $self->{column} += $count;
741              $self->{char_buffer_pos} += $count;
742              $self->{line_prev} = $self->{line};
743              $self->{column_prev} = $self->{column} - 1;
744              $self->{nc} = -1;
745            }
746            return $count;
747          } else {
748            return 0;
749          }
750        } else {
751          my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]);
752          if ($count) {
753            $self->{column} += $count;
754            $self->{line_prev} = $self->{line};
755            $self->{column_prev} = $self->{column} - 1;
756            $self->{nc} = -1;
757          }
758          return $count;
759        }
760      }; # $self->{read_until}
761    
762    my $onerror = $_[2] || sub {    my $onerror = $_[2] || sub {
763      my (%opt) = @_;      my (%opt) = @_;
764      warn "Parse error ($opt{type}) at line $opt{line} column $opt{column}\n";      my $line = $opt{token} ? $opt{token}->{line} : $opt{line};
765        my $column = $opt{token} ? $opt{token}->{column} : $opt{column};
766        warn "Parse error ($opt{type}) at line $line column $column\n";
767    };    };
768    $self->{parse_error} = sub {    $self->{parse_error} = sub {
769      $onerror->(@_, line => $line, column => $column);      $onerror->(line => $self->{line}, column => $self->{column}, @_);
770    };    };
771    
772      my $char_onerror = sub {
773        my (undef, $type, %opt) = @_;
774        !!!parse-error (layer => 'encode',
775                        line => $self->{line}, column => $self->{column} + 1,
776                        %opt, type => $type);
777      }; # $char_onerror
778    
779      if ($_[3]) {
780        $input = $_[3]->($input);
781        $input->onerror ($char_onerror);
782      } else {
783        $input->onerror ($char_onerror) unless defined $input->onerror;
784      }
785    
786    $self->_initialize_tokenizer;    $self->_initialize_tokenizer;
787    $self->_initialize_tree_constructor;    $self->_initialize_tree_constructor;
788    $self->_construct_tree;    $self->_construct_tree;
789    $self->_terminate_tree_constructor;    $self->_terminate_tree_constructor;
790    
791      delete $self->{parse_error}; # remove loop
792    
793    return $self->{document};    return $self->{document};
794  } # parse_string  } # parse_char_stream
795    
796  sub new ($) {  sub new ($) {
797    my $class = shift;    my $class = shift;
798    my $self = bless {}, $class;    my $self = bless {
799    $self->{set_next_input_character} = sub {      level => {must => 'm',
800      $self->{next_input_character} = -1;                should => 's',
801                  warn => 'w',
802                  info => 'i',
803                  uncertain => 'u'},
804      }, $class;
805      $self->{set_nc} = sub {
806        $self->{nc} = -1;
807    };    };
808    $self->{parse_error} = sub {    $self->{parse_error} = sub {
809      #      #
810    };    };
811      $self->{change_encoding} = sub {
812        # if ($_[0] is a supported encoding) {
813        #   run "change the encoding" algorithm;
814        #   throw Whatpm::HTML::RestartParser (charset => $new_encoding);
815        # }
816      };
817      $self->{application_cache_selection} = sub {
818        #
819      };
820    return $self;    return $self;
821  } # new  } # new
822    
823    sub CM_ENTITY () { 0b001 } # & markup in data
824    sub CM_LIMITED_MARKUP () { 0b010 } # < markup in data (limited)
825    sub CM_FULL_MARKUP () { 0b100 } # < markup in data (any)
826    
827    sub PLAINTEXT_CONTENT_MODEL () { 0 }
828    sub CDATA_CONTENT_MODEL () { CM_LIMITED_MARKUP }
829    sub RCDATA_CONTENT_MODEL () { CM_ENTITY | CM_LIMITED_MARKUP }
830    sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP }
831    
832    sub DATA_STATE () { 0 }
833    #sub ENTITY_DATA_STATE () { 1 }
834    sub TAG_OPEN_STATE () { 2 }
835    sub CLOSE_TAG_OPEN_STATE () { 3 }
836    sub TAG_NAME_STATE () { 4 }
837    sub BEFORE_ATTRIBUTE_NAME_STATE () { 5 }
838    sub ATTRIBUTE_NAME_STATE () { 6 }
839    sub AFTER_ATTRIBUTE_NAME_STATE () { 7 }
840    sub BEFORE_ATTRIBUTE_VALUE_STATE () { 8 }
841    sub ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE () { 9 }
842    sub ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE () { 10 }
843    sub ATTRIBUTE_VALUE_UNQUOTED_STATE () { 11 }
844    #sub ENTITY_IN_ATTRIBUTE_VALUE_STATE () { 12 }
845    sub MARKUP_DECLARATION_OPEN_STATE () { 13 }
846    sub COMMENT_START_STATE () { 14 }
847    sub COMMENT_START_DASH_STATE () { 15 }
848    sub COMMENT_STATE () { 16 }
849    sub COMMENT_END_STATE () { 17 }
850    sub COMMENT_END_DASH_STATE () { 18 }
851    sub BOGUS_COMMENT_STATE () { 19 }
852    sub DOCTYPE_STATE () { 20 }
853    sub BEFORE_DOCTYPE_NAME_STATE () { 21 }
854    sub DOCTYPE_NAME_STATE () { 22 }
855    sub AFTER_DOCTYPE_NAME_STATE () { 23 }
856    sub BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE () { 24 }
857    sub DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE () { 25 }
858    sub DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE () { 26 }
859    sub AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE () { 27 }
860    sub BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 28 }
861    sub DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE () { 29 }
862    sub DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE () { 30 }
863    sub AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 31 }
864    sub BOGUS_DOCTYPE_STATE () { 32 }
865    sub AFTER_ATTRIBUTE_VALUE_QUOTED_STATE () { 33 }
866    sub SELF_CLOSING_START_TAG_STATE () { 34 }
867    sub CDATA_SECTION_STATE () { 35 }
868    sub MD_HYPHEN_STATE () { 36 } # "markup declaration open state" in the spec
869    sub MD_DOCTYPE_STATE () { 37 } # "markup declaration open state" in the spec
870    sub MD_CDATA_STATE () { 38 } # "markup declaration open state" in the spec
871    sub CDATA_RCDATA_CLOSE_TAG_STATE () { 39 } # "close tag open state" in the spec
872    sub CDATA_SECTION_MSE1_STATE () { 40 } # "CDATA section state" in the spec
873    sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec
874    sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec
875    sub SYSTEM_STATE () { 43 } # "after DOCTYPE name state" in the spec
876    ## NOTE: "Entity data state", "entity in attribute value state", and
877    ## "consume a character reference" algorithm are jointly implemented
878    ## using the following six states:
879    sub ENTITY_STATE () { 44 }
880    sub ENTITY_HASH_STATE () { 45 }
881    sub NCR_NUM_STATE () { 46 }
882    sub HEXREF_X_STATE () { 47 }
883    sub HEXREF_HEX_STATE () { 48 }
884    sub ENTITY_NAME_STATE () { 49 }
885    sub PCDATA_STATE () { 50 } # "data state" in the spec
886    
887    sub DOCTYPE_TOKEN () { 1 }
888    sub COMMENT_TOKEN () { 2 }
889    sub START_TAG_TOKEN () { 3 }
890    sub END_TAG_TOKEN () { 4 }
891    sub END_OF_FILE_TOKEN () { 5 }
892    sub CHARACTER_TOKEN () { 6 }
893    
894    sub AFTER_HTML_IMS () { 0b100 }
895    sub HEAD_IMS ()       { 0b1000 }
896    sub BODY_IMS ()       { 0b10000 }
897    sub BODY_TABLE_IMS () { 0b100000 }
898    sub TABLE_IMS ()      { 0b1000000 }
899    sub ROW_IMS ()        { 0b10000000 }
900    sub BODY_AFTER_IMS () { 0b100000000 }
901    sub FRAME_IMS ()      { 0b1000000000 }
902    sub SELECT_IMS ()     { 0b10000000000 }
903    sub IN_FOREIGN_CONTENT_IM () { 0b100000000000 }
904        ## NOTE: "in foreign content" insertion mode is special; it is combined
905        ## with the secondary insertion mode.  In this parser, they are stored
906        ## together in the bit-or'ed form.
907    
908    ## NOTE: "initial" and "before html" insertion modes have no constants.
909    
910    ## NOTE: "after after body" insertion mode.
911    sub AFTER_HTML_BODY_IM () { AFTER_HTML_IMS | BODY_AFTER_IMS }
912    
913    ## NOTE: "after after frameset" insertion mode.
914    sub AFTER_HTML_FRAMESET_IM () { AFTER_HTML_IMS | FRAME_IMS }
915    
916    sub IN_HEAD_IM () { HEAD_IMS | 0b00 }
917    sub IN_HEAD_NOSCRIPT_IM () { HEAD_IMS | 0b01 }
918    sub AFTER_HEAD_IM () { HEAD_IMS | 0b10 }
919    sub BEFORE_HEAD_IM () { HEAD_IMS | 0b11 }
920    sub IN_BODY_IM () { BODY_IMS }
921    sub IN_CELL_IM () { BODY_IMS | BODY_TABLE_IMS | 0b01 }
922    sub IN_CAPTION_IM () { BODY_IMS | BODY_TABLE_IMS | 0b10 }
923    sub IN_ROW_IM () { TABLE_IMS | ROW_IMS | 0b01 }
924    sub IN_TABLE_BODY_IM () { TABLE_IMS | ROW_IMS | 0b10 }
925    sub IN_TABLE_IM () { TABLE_IMS }
926    sub AFTER_BODY_IM () { BODY_AFTER_IMS }
927    sub IN_FRAMESET_IM () { FRAME_IMS | 0b01 }
928    sub AFTER_FRAMESET_IM () { FRAME_IMS | 0b10 }
929    sub IN_SELECT_IM () { SELECT_IMS | 0b01 }
930    sub IN_SELECT_IN_TABLE_IM () { SELECT_IMS | 0b10 }
931    sub IN_COLUMN_GROUP_IM () { 0b10 }
932    
933  ## Implementations MUST act as if state machine in the spec  ## Implementations MUST act as if state machine in the spec
934    
935  sub _initialize_tokenizer ($) {  sub _initialize_tokenizer ($) {
936    my $self = shift;    my $self = shift;
937    $self->{state} = 'data'; # MUST    $self->{state} = DATA_STATE; # MUST
938    $self->{content_model_flag} = 'PCDATA'; # be    #$self->{s_kwd}; # state keyword - initialized when used
939    undef $self->{current_token}; # start tag, end tag, comment, or DOCTYPE    #$self->{entity__value}; # initialized when used
940    undef $self->{current_attribute};    #$self->{entity__match}; # initialized when used
941    undef $self->{last_emitted_start_tag_name};    $self->{content_model} = PCDATA_CONTENT_MODEL; # be
942    undef $self->{last_attribute_value_state};    undef $self->{ct}; # current token
943    $self->{char} = [];    undef $self->{ca}; # current attribute
944    # $self->{next_input_character}    undef $self->{last_stag_name}; # last emitted start tag name
945      #$self->{prev_state}; # initialized when used
946      delete $self->{self_closing};
947      $self->{char_buffer} = '';
948      $self->{char_buffer_pos} = 0;
949      $self->{nc} = -1; # next input character
950      #$self->{next_nc}
951    !!!next-input-character;    !!!next-input-character;
952    $self->{token} = [];    $self->{token} = [];
953    # $self->{escape}    # $self->{escape}
954  } # _initialize_tokenizer  } # _initialize_tokenizer
955    
956  ## A token has:  ## A token has:
957  ##   ->{type} eq 'DOCTYPE', 'start tag', 'end tag', 'comment',  ##   ->{type} == DOCTYPE_TOKEN, START_TAG_TOKEN, END_TAG_TOKEN, COMMENT_TOKEN,
958  ##       'character', or 'end-of-file'  ##       CHARACTER_TOKEN, or END_OF_FILE_TOKEN
959  ##   ->{name} (DOCTYPE, start tag (tag name), end tag (tag name))  ##   ->{name} (DOCTYPE_TOKEN)
960  ##   ->{public_identifier} (DOCTYPE)  ##   ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)
961  ##   ->{system_identifier} (DOCTYPE)  ##   ->{pubid} (DOCTYPE_TOKEN)
962  ##   ->{correct} == 1 or 0 (DOCTYPE)  ##   ->{sysid} (DOCTYPE_TOKEN)
963  ##   ->{attributes} isa HASH (start tag, end tag)  ##   ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag
964  ##   ->{data} (comment, character)  ##   ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)
965    ##        ->{name}
966    ##        ->{value}
967    ##        ->{has_reference} == 1 or 0
968    ##   ->{data} (COMMENT_TOKEN, CHARACTER_TOKEN)
969    ## NOTE: The "self-closing flag" is hold as |$self->{self_closing}|.
970    ##     |->{self_closing}| is used to save the value of |$self->{self_closing}|
971    ##     while the token is pushed back to the stack.
972    
973  ## Emitted token MUST immediately be handled by the tree construction state.  ## Emitted token MUST immediately be handled by the tree construction state.
974    
# Line 179  sub _initialize_tokenizer ($) { Line 978  sub _initialize_tokenizer ($) {
978  ## has completed loading.  If one has, then it MUST be executed  ## has completed loading.  If one has, then it MUST be executed
979  ## and removed from the list.  ## and removed from the list.
980    
981    ## TODO: Polytheistic slash SHOULD NOT be used. (Applied only to atheists.)
982    ## (This requirement was dropped from HTML5 spec, unfortunately.)
983    
984    my $is_space = {
985      0x0009 => 1, # CHARACTER TABULATION (HT)
986      0x000A => 1, # LINE FEED (LF)
987      #0x000B => 0, # LINE TABULATION (VT)
988      0x000C => 1, # FORM FEED (FF)
989      #0x000D => 1, # CARRIAGE RETURN (CR)
990      0x0020 => 1, # SPACE (SP)
991    };
992    
993  sub _get_next_token ($) {  sub _get_next_token ($) {
994    my $self = shift;    my $self = shift;
995    
996      if ($self->{self_closing}) {
997        !!!parse-error (type => 'nestc', token => $self->{ct});
998        ## NOTE: The |self_closing| flag is only set by start tag token.
999        ## In addition, when a start tag token is emitted, it is always set to
1000        ## |ct|.
1001        delete $self->{self_closing};
1002      }
1003    
1004    if (@{$self->{token}}) {    if (@{$self->{token}}) {
1005        $self->{self_closing} = $self->{token}->[0]->{self_closing};
1006      return shift @{$self->{token}};      return shift @{$self->{token}};
1007    }    }
1008    
1009    A: {    A: {
1010      if ($self->{state} eq 'data') {      if ($self->{state} == PCDATA_STATE) {
1011        if ($self->{next_input_character} == 0x0026) { # &        ## NOTE: Same as |DATA_STATE|, but only for |PCDATA| content model.
1012          if ($self->{content_model_flag} eq 'PCDATA' or  
1013              $self->{content_model_flag} eq 'RCDATA') {        if ($self->{nc} == 0x0026) { # &
1014            $self->{state} = 'entity data';          !!!cp (0.1);
1015            ## NOTE: In the spec, the tokenizer is switched to the
1016            ## "entity data state".  In this implementation, the tokenizer
1017            ## is switched to the |ENTITY_STATE|, which is an implementation
1018            ## of the "consume a character reference" algorithm.
1019            $self->{entity_add} = -1;
1020            $self->{prev_state} = DATA_STATE;
1021            $self->{state} = ENTITY_STATE;
1022            !!!next-input-character;
1023            redo A;
1024          } elsif ($self->{nc} == 0x003C) { # <
1025            !!!cp (0.2);
1026            $self->{state} = TAG_OPEN_STATE;
1027            !!!next-input-character;
1028            redo A;
1029          } elsif ($self->{nc} == -1) {
1030            !!!cp (0.3);
1031            !!!emit ({type => END_OF_FILE_TOKEN,
1032                      line => $self->{line}, column => $self->{column}});
1033            last A; ## TODO: ok?
1034          } else {
1035            !!!cp (0.4);
1036            #
1037          }
1038    
1039          # Anything else
1040          my $token = {type => CHARACTER_TOKEN,
1041                       data => chr $self->{nc},
1042                       line => $self->{line}, column => $self->{column},
1043                      };
1044          $self->{read_until}->($token->{data}, q[<&], length $token->{data});
1045    
1046          ## Stay in the state.
1047          !!!next-input-character;
1048          !!!emit ($token);
1049          redo A;
1050        } elsif ($self->{state} == DATA_STATE) {
1051          $self->{s_kwd} = '' unless defined $self->{s_kwd};
1052          if ($self->{nc} == 0x0026) { # &
1053            $self->{s_kwd} = '';
1054            if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA
1055                not $self->{escape}) {
1056              !!!cp (1);
1057              ## NOTE: In the spec, the tokenizer is switched to the
1058              ## "entity data state".  In this implementation, the tokenizer
1059              ## is switched to the |ENTITY_STATE|, which is an implementation
1060              ## of the "consume a character reference" algorithm.
1061              $self->{entity_add} = -1;
1062              $self->{prev_state} = DATA_STATE;
1063              $self->{state} = ENTITY_STATE;
1064            !!!next-input-character;            !!!next-input-character;
1065            redo A;            redo A;
1066          } else {          } else {
1067              !!!cp (2);
1068            #            #
1069          }          }
1070        } elsif ($self->{next_input_character} == 0x002D) { # -        } elsif ($self->{nc} == 0x002D) { # -
1071          if ($self->{content_model_flag} eq 'RCDATA' or          if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1072              $self->{content_model_flag} eq 'CDATA') {            $self->{s_kwd} .= '-';
1073            unless ($self->{escape}) {            
1074              if ($self->{prev_input_character}->[0] == 0x002D and # -            if ($self->{s_kwd} eq '<!--') {
1075                  $self->{prev_input_character}->[1] == 0x0021 and # !              !!!cp (3);
1076                  $self->{prev_input_character}->[2] == 0x003C) { # <              $self->{escape} = 1; # unless $self->{escape};
1077                $self->{escape} = 1;              $self->{s_kwd} = '--';
1078              }              #
1079              } elsif ($self->{s_kwd} eq '---') {
1080                !!!cp (4);
1081                $self->{s_kwd} = '--';
1082                #
1083              } else {
1084                !!!cp (5);
1085                #
1086            }            }
1087          }          }
1088                    
1089          #          #
1090        } elsif ($self->{next_input_character} == 0x003C) { # <        } elsif ($self->{nc} == 0x0021) { # !
1091          if ($self->{content_model_flag} eq 'PCDATA' or          if (length $self->{s_kwd}) {
1092              (($self->{content_model_flag} eq 'CDATA' or            !!!cp (5.1);
1093                $self->{content_model_flag} eq 'RCDATA') and            $self->{s_kwd} .= '!';
1094              #
1095            } else {
1096              !!!cp (5.2);
1097              #$self->{s_kwd} = '';
1098              #
1099            }
1100            #
1101          } elsif ($self->{nc} == 0x003C) { # <
1102            if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA
1103                (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA
1104               not $self->{escape})) {               not $self->{escape})) {
1105            $self->{state} = 'tag open';            !!!cp (6);
1106              $self->{state} = TAG_OPEN_STATE;
1107            !!!next-input-character;            !!!next-input-character;
1108            redo A;            redo A;
1109          } else {          } else {
1110              !!!cp (7);
1111              $self->{s_kwd} = '';
1112            #            #
1113          }          }
1114        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1115          if ($self->{escape} and          if ($self->{escape} and
1116              ($self->{content_model_flag} eq 'RCDATA' or              ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA
1117               $self->{content_model_flag} eq 'CDATA')) {            if ($self->{s_kwd} eq '--') {
1118            if ($self->{prev_input_character}->[0] == 0x002D and # -              !!!cp (8);
               $self->{prev_input_character}->[1] == 0x002D) { # -  
1119              delete $self->{escape};              delete $self->{escape};
1120              } else {
1121                !!!cp (9);
1122            }            }
1123            } else {
1124              !!!cp (10);
1125          }          }
1126                    
1127            $self->{s_kwd} = '';
1128          #          #
1129        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1130          !!!emit ({type => 'end-of-file'});          !!!cp (11);
1131            $self->{s_kwd} = '';
1132            !!!emit ({type => END_OF_FILE_TOKEN,
1133                      line => $self->{line}, column => $self->{column}});
1134          last A; ## TODO: ok?          last A; ## TODO: ok?
1135          } else {
1136            !!!cp (12);
1137            $self->{s_kwd} = '';
1138            #
1139        }        }
       # Anything else  
       my $token = {type => 'character',  
                    data => chr $self->{next_input_character}};  
       ## Stay in the data state  
       !!!next-input-character;  
   
       !!!emit ($token);  
1140    
1141        redo A;        # Anything else
1142      } elsif ($self->{state} eq 'entity data') {        my $token = {type => CHARACTER_TOKEN,
1143        ## (cannot happen in CDATA state)                     data => chr $self->{nc},
1144                             line => $self->{line}, column => $self->{column},
1145        my $token = $self->_tokenize_attempt_to_consume_an_entity;                    };
1146          if ($self->{read_until}->($token->{data}, q[-!<>&],
1147        $self->{state} = 'data';                                  length $token->{data})) {
1148        # next-input-character is already done          $self->{s_kwd} = '';
1149          }
1150    
1151        unless (defined $token) {        ## Stay in the data state.
1152          !!!emit ({type => 'character', data => '&'});        if ($self->{content_model} == PCDATA_CONTENT_MODEL) {
1153            !!!cp (13);
1154            $self->{state} = PCDATA_STATE;
1155        } else {        } else {
1156          !!!emit ($token);          !!!cp (14);
1157            ## Stay in the state.
1158        }        }
1159          !!!next-input-character;
1160          !!!emit ($token);
1161        redo A;        redo A;
1162      } elsif ($self->{state} eq 'tag open') {      } elsif ($self->{state} == TAG_OPEN_STATE) {
1163        if ($self->{content_model_flag} eq 'RCDATA' or        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1164            $self->{content_model_flag} eq 'CDATA') {          if ($self->{nc} == 0x002F) { # /
1165          if ($self->{next_input_character} == 0x002F) { # /            !!!cp (15);
1166            !!!next-input-character;            !!!next-input-character;
1167            $self->{state} = 'close tag open';            $self->{state} = CLOSE_TAG_OPEN_STATE;
1168            redo A;            redo A;
1169            } elsif ($self->{nc} == 0x0021) { # !
1170              !!!cp (15.1);
1171              $self->{s_kwd} = '<' unless $self->{escape};
1172              #
1173          } else {          } else {
1174            ## reconsume            !!!cp (16);
1175            $self->{state} = 'data';            #
   
           !!!emit ({type => 'character', data => '<'});  
   
           redo A;  
1176          }          }
1177        } elsif ($self->{content_model_flag} eq 'PCDATA') {  
1178          if ($self->{next_input_character} == 0x0021) { # !          ## reconsume
1179            $self->{state} = 'markup declaration open';          $self->{state} = DATA_STATE;
1180            !!!emit ({type => CHARACTER_TOKEN, data => '<',
1181                      line => $self->{line_prev},
1182                      column => $self->{column_prev},
1183                     });
1184            redo A;
1185          } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA
1186            if ($self->{nc} == 0x0021) { # !
1187              !!!cp (17);
1188              $self->{state} = MARKUP_DECLARATION_OPEN_STATE;
1189            !!!next-input-character;            !!!next-input-character;
1190            redo A;            redo A;
1191          } elsif ($self->{next_input_character} == 0x002F) { # /          } elsif ($self->{nc} == 0x002F) { # /
1192            $self->{state} = 'close tag open';            !!!cp (18);
1193              $self->{state} = CLOSE_TAG_OPEN_STATE;
1194            !!!next-input-character;            !!!next-input-character;
1195            redo A;            redo A;
1196          } elsif (0x0041 <= $self->{next_input_character} and          } elsif (0x0041 <= $self->{nc} and
1197                   $self->{next_input_character} <= 0x005A) { # A..Z                   $self->{nc} <= 0x005A) { # A..Z
1198            $self->{current_token}            !!!cp (19);
1199              = {type => 'start tag',            $self->{ct}
1200                 tag_name => chr ($self->{next_input_character} + 0x0020)};              = {type => START_TAG_TOKEN,
1201            $self->{state} = 'tag name';                 tag_name => chr ($self->{nc} + 0x0020),
1202                   line => $self->{line_prev},
1203                   column => $self->{column_prev}};
1204              $self->{state} = TAG_NAME_STATE;
1205            !!!next-input-character;            !!!next-input-character;
1206            redo A;            redo A;
1207          } elsif (0x0061 <= $self->{next_input_character} and          } elsif (0x0061 <= $self->{nc} and
1208                   $self->{next_input_character} <= 0x007A) { # a..z                   $self->{nc} <= 0x007A) { # a..z
1209            $self->{current_token} = {type => 'start tag',            !!!cp (20);
1210                              tag_name => chr ($self->{next_input_character})};            $self->{ct} = {type => START_TAG_TOKEN,
1211            $self->{state} = 'tag name';                                      tag_name => chr ($self->{nc}),
1212                                        line => $self->{line_prev},
1213                                        column => $self->{column_prev}};
1214              $self->{state} = TAG_NAME_STATE;
1215            !!!next-input-character;            !!!next-input-character;
1216            redo A;            redo A;
1217          } elsif ($self->{next_input_character} == 0x003E) { # >          } elsif ($self->{nc} == 0x003E) { # >
1218            !!!parse-error (type => 'empty start tag');            !!!cp (21);
1219            $self->{state} = 'data';            !!!parse-error (type => 'empty start tag',
1220                              line => $self->{line_prev},
1221                              column => $self->{column_prev});
1222              $self->{state} = DATA_STATE;
1223            !!!next-input-character;            !!!next-input-character;
1224    
1225            !!!emit ({type => 'character', data => '<>'});            !!!emit ({type => CHARACTER_TOKEN, data => '<>',
1226                        line => $self->{line_prev},
1227                        column => $self->{column_prev},
1228                       });
1229    
1230            redo A;            redo A;
1231          } elsif ($self->{next_input_character} == 0x003F) { # ?          } elsif ($self->{nc} == 0x003F) { # ?
1232            !!!parse-error (type => 'pio');            !!!cp (22);
1233            $self->{state} = 'bogus comment';            !!!parse-error (type => 'pio',
1234            ## $self->{next_input_character} is intentionally left as is                            line => $self->{line_prev},
1235                              column => $self->{column_prev});
1236              $self->{state} = BOGUS_COMMENT_STATE;
1237              $self->{ct} = {type => COMMENT_TOKEN, data => '',
1238                                        line => $self->{line_prev},
1239                                        column => $self->{column_prev},
1240                                       };
1241              ## $self->{nc} is intentionally left as is
1242            redo A;            redo A;
1243          } else {          } else {
1244            !!!parse-error (type => 'bare stago');            !!!cp (23);
1245            $self->{state} = 'data';            !!!parse-error (type => 'bare stago',
1246                              line => $self->{line_prev},
1247                              column => $self->{column_prev});
1248              $self->{state} = DATA_STATE;
1249            ## reconsume            ## reconsume
1250    
1251            !!!emit ({type => 'character', data => '<'});            !!!emit ({type => CHARACTER_TOKEN, data => '<',
1252                        line => $self->{line_prev},
1253                        column => $self->{column_prev},
1254                       });
1255    
1256            redo A;            redo A;
1257          }          }
1258        } else {        } else {
1259          die "$0: $self->{content_model_flag}: Unknown content model flag";          die "$0: $self->{content_model} in tag open";
1260        }        }
1261      } elsif ($self->{state} eq 'close tag open') {      } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {
1262        if ($self->{content_model_flag} eq 'RCDATA' or        ## NOTE: The "close tag open state" in the spec is implemented as
1263            $self->{content_model_flag} eq 'CDATA') {        ## |CLOSE_TAG_OPEN_STATE| and |CDATA_RCDATA_CLOSE_TAG_STATE|.
1264          if (defined $self->{last_emitted_start_tag_name}) {  
1265            my @next_char;        my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"
1266            TAGNAME: for (my $i = 0; $i < length $self->{last_emitted_start_tag_name}; $i++) {        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1267              push @next_char, $self->{next_input_character};          if (defined $self->{last_stag_name}) {
1268              my $c = ord substr ($self->{last_emitted_start_tag_name}, $i, 1);            $self->{state} = CDATA_RCDATA_CLOSE_TAG_STATE;
1269              my $C = 0x0061 <= $c && $c <= 0x007A ? $c - 0x0020 : $c;            $self->{s_kwd} = '';
1270              if ($self->{next_input_character} == $c or $self->{next_input_character} == $C) {            ## Reconsume.
1271                !!!next-input-character;            redo A;
               next TAGNAME;  
             } else {  
               $self->{next_input_character} = shift @next_char; # reconsume  
               !!!back-next-input-character (@next_char);  
               $self->{state} = 'data';  
   
               !!!emit ({type => 'character', data => '</'});  
     
               redo A;  
             }  
           }  
           push @next_char, $self->{next_input_character};  
         
           unless ($self->{next_input_character} == 0x0009 or # HT  
                   $self->{next_input_character} == 0x000A or # LF  
                   $self->{next_input_character} == 0x000B or # VT  
                   $self->{next_input_character} == 0x000C or # FF  
                   $self->{next_input_character} == 0x0020 or # SP  
                   $self->{next_input_character} == 0x003E or # >  
                   $self->{next_input_character} == 0x002F or # /  
                   $self->{next_input_character} == -1) {  
             $self->{next_input_character} = shift @next_char; # reconsume  
             !!!back-next-input-character (@next_char);  
             $self->{state} = 'data';  
             !!!emit ({type => 'character', data => '</'});  
             redo A;  
           } else {  
             $self->{next_input_character} = shift @next_char;  
             !!!back-next-input-character (@next_char);  
             # and consume...  
           }  
1272          } else {          } else {
1273            ## No start tag token has ever been emitted            ## No start tag token has ever been emitted
1274            # next-input-character is already done            ## NOTE: See <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>.
1275            $self->{state} = 'data';            !!!cp (28);
1276            !!!emit ({type => 'character', data => '</'});            $self->{state} = DATA_STATE;
1277              ## Reconsume.
1278              !!!emit ({type => CHARACTER_TOKEN, data => '</',
1279                        line => $l, column => $c,
1280                       });
1281            redo A;            redo A;
1282          }          }
1283        }        }
1284          
1285        if (0x0041 <= $self->{next_input_character} and        if (0x0041 <= $self->{nc} and
1286            $self->{next_input_character} <= 0x005A) { # A..Z            $self->{nc} <= 0x005A) { # A..Z
1287          $self->{current_token} = {type => 'end tag',          !!!cp (29);
1288                            tag_name => chr ($self->{next_input_character} + 0x0020)};          $self->{ct}
1289          $self->{state} = 'tag name';              = {type => END_TAG_TOKEN,
1290          !!!next-input-character;                 tag_name => chr ($self->{nc} + 0x0020),
1291          redo A;                 line => $l, column => $c};
1292        } elsif (0x0061 <= $self->{next_input_character} and          $self->{state} = TAG_NAME_STATE;
1293                 $self->{next_input_character} <= 0x007A) { # a..z          !!!next-input-character;
1294          $self->{current_token} = {type => 'end tag',          redo A;
1295                            tag_name => chr ($self->{next_input_character})};        } elsif (0x0061 <= $self->{nc} and
1296          $self->{state} = 'tag name';                 $self->{nc} <= 0x007A) { # a..z
1297          !!!next-input-character;          !!!cp (30);
1298          redo A;          $self->{ct} = {type => END_TAG_TOKEN,
1299        } elsif ($self->{next_input_character} == 0x003E) { # >                                    tag_name => chr ($self->{nc}),
1300          !!!parse-error (type => 'empty end tag');                                    line => $l, column => $c};
1301          $self->{state} = 'data';          $self->{state} = TAG_NAME_STATE;
1302            !!!next-input-character;
1303            redo A;
1304          } elsif ($self->{nc} == 0x003E) { # >
1305            !!!cp (31);
1306            !!!parse-error (type => 'empty end tag',
1307                            line => $self->{line_prev}, ## "<" in "</>"
1308                            column => $self->{column_prev} - 1);
1309            $self->{state} = DATA_STATE;
1310          !!!next-input-character;          !!!next-input-character;
1311          redo A;          redo A;
1312        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1313            !!!cp (32);
1314          !!!parse-error (type => 'bare etago');          !!!parse-error (type => 'bare etago');
1315          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1316          # reconsume          # reconsume
1317    
1318          !!!emit ({type => 'character', data => '</'});          !!!emit ({type => CHARACTER_TOKEN, data => '</',
1319                      line => $l, column => $c,
1320                     });
1321    
1322          redo A;          redo A;
1323        } else {        } else {
1324            !!!cp (33);
1325          !!!parse-error (type => 'bogus end tag');          !!!parse-error (type => 'bogus end tag');
1326          $self->{state} = 'bogus comment';          $self->{state} = BOGUS_COMMENT_STATE;
1327          ## $self->{next_input_character} is intentionally left as is          $self->{ct} = {type => COMMENT_TOKEN, data => '',
1328          redo A;                                    line => $self->{line_prev}, # "<" of "</"
1329        }                                    column => $self->{column_prev} - 1,
1330      } elsif ($self->{state} eq 'tag name') {                                   };
1331        if ($self->{next_input_character} == 0x0009 or # HT          ## NOTE: $self->{nc} is intentionally left as is.
1332            $self->{next_input_character} == 0x000A or # LF          ## Although the "anything else" case of the spec not explicitly
1333            $self->{next_input_character} == 0x000B or # VT          ## states that the next input character is to be reconsumed,
1334            $self->{next_input_character} == 0x000C or # FF          ## it will be included to the |data| of the comment token
1335            $self->{next_input_character} == 0x0020) { # SP          ## generated from the bogus end tag, as defined in the
1336          $self->{state} = 'before attribute name';          ## "bogus comment state" entry.
1337          !!!next-input-character;          redo A;
1338          redo A;        }
1339        } elsif ($self->{next_input_character} == 0x003E) { # >      } elsif ($self->{state} == CDATA_RCDATA_CLOSE_TAG_STATE) {
1340          if ($self->{current_token}->{type} eq 'start tag') {        my $ch = substr $self->{last_stag_name}, length $self->{s_kwd}, 1;
1341            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};        if (length $ch) {
1342          } elsif ($self->{current_token}->{type} eq 'end tag') {          my $CH = $ch;
1343            $self->{content_model_flag} = 'PCDATA'; # MUST          $ch =~ tr/a-z/A-Z/;
1344            if ($self->{current_token}->{attributes}) {          my $nch = chr $self->{nc};
1345              !!!parse-error (type => 'end tag attribute');          if ($nch eq $ch or $nch eq $CH) {
1346            }            !!!cp (24);
1347              ## Stay in the state.
1348              $self->{s_kwd} .= $nch;
1349              !!!next-input-character;
1350              redo A;
1351          } else {          } else {
1352            die "$0: $self->{current_token}->{type}: Unknown token type";            !!!cp (25);
1353              $self->{state} = DATA_STATE;
1354              ## Reconsume.
1355              !!!emit ({type => CHARACTER_TOKEN,
1356                        data => '</' . $self->{s_kwd},
1357                        line => $self->{line_prev},
1358                        column => $self->{column_prev} - 1 - length $self->{s_kwd},
1359                       });
1360              redo A;
1361          }          }
1362          $self->{state} = 'data';        } else { # after "<{tag-name}"
1363          !!!next-input-character;          unless ($is_space->{$self->{nc}} or
1364                    {
1365          !!!emit ($self->{current_token}); # start tag or end tag                   0x003E => 1, # >
1366                     0x002F => 1, # /
1367          redo A;                   -1 => 1, # EOF
1368        } elsif (0x0041 <= $self->{next_input_character} and                  }->{$self->{nc}}) {
1369                 $self->{next_input_character} <= 0x005A) { # A..Z            !!!cp (26);
1370          $self->{current_token}->{tag_name} .= chr ($self->{next_input_character} + 0x0020);            ## Reconsume.
1371              $self->{state} = DATA_STATE;
1372              !!!emit ({type => CHARACTER_TOKEN,
1373                        data => '</' . $self->{s_kwd},
1374                        line => $self->{line_prev},
1375                        column => $self->{column_prev} - 1 - length $self->{s_kwd},
1376                       });
1377              redo A;
1378            } else {
1379              !!!cp (27);
1380              $self->{ct}
1381                  = {type => END_TAG_TOKEN,
1382                     tag_name => $self->{last_stag_name},
1383                     line => $self->{line_prev},
1384                     column => $self->{column_prev} - 1 - length $self->{s_kwd}};
1385              $self->{state} = TAG_NAME_STATE;
1386              ## Reconsume.
1387              redo A;
1388            }
1389          }
1390        } elsif ($self->{state} == TAG_NAME_STATE) {
1391          if ($is_space->{$self->{nc}}) {
1392            !!!cp (34);
1393            $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1394            !!!next-input-character;
1395            redo A;
1396          } elsif ($self->{nc} == 0x003E) { # >
1397            if ($self->{ct}->{type} == START_TAG_TOKEN) {
1398              !!!cp (35);
1399              $self->{last_stag_name} = $self->{ct}->{tag_name};
1400            } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1401              $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1402              #if ($self->{ct}->{attributes}) {
1403              #  ## NOTE: This should never be reached.
1404              #  !!! cp (36);
1405              #  !!! parse-error (type => 'end tag attribute');
1406              #} else {
1407                !!!cp (37);
1408              #}
1409            } else {
1410              die "$0: $self->{ct}->{type}: Unknown token type";
1411            }
1412            $self->{state} = DATA_STATE;
1413            !!!next-input-character;
1414    
1415            !!!emit ($self->{ct}); # start tag or end tag
1416    
1417            redo A;
1418          } elsif (0x0041 <= $self->{nc} and
1419                   $self->{nc} <= 0x005A) { # A..Z
1420            !!!cp (38);
1421            $self->{ct}->{tag_name} .= chr ($self->{nc} + 0x0020);
1422            # start tag or end tag            # start tag or end tag
1423          ## Stay in this state          ## Stay in this state
1424          !!!next-input-character;          !!!next-input-character;
1425          redo A;          redo A;
1426        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1427          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1428          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1429            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            !!!cp (39);
1430          } elsif ($self->{current_token}->{type} eq 'end tag') {            $self->{last_stag_name} = $self->{ct}->{tag_name};
1431            $self->{content_model_flag} = 'PCDATA'; # MUST          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1432            if ($self->{current_token}->{attributes}) {            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1433              !!!parse-error (type => 'end tag attribute');            #if ($self->{ct}->{attributes}) {
1434            }            #  ## NOTE: This state should never be reached.
1435              #  !!! cp (40);
1436              #  !!! parse-error (type => 'end tag attribute');
1437              #} else {
1438                !!!cp (41);
1439              #}
1440          } else {          } else {
1441            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1442          }          }
1443          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1444          # reconsume          # reconsume
1445    
1446          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1447    
1448          redo A;          redo A;
1449        } elsif ($self->{next_input_character} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1450            !!!cp (42);
1451            $self->{state} = SELF_CLOSING_START_TAG_STATE;
1452          !!!next-input-character;          !!!next-input-character;
         if ($self->{next_input_character} == 0x003E and # >  
             $self->{current_token}->{type} eq 'start tag' and  
             $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {  
           # permitted slash  
           #  
         } else {  
           !!!parse-error (type => 'nestc');  
         }  
         $self->{state} = 'before attribute name';  
         # next-input-character is already done  
1453          redo A;          redo A;
1454        } else {        } else {
1455          $self->{current_token}->{tag_name} .= chr $self->{next_input_character};          !!!cp (44);
1456            $self->{ct}->{tag_name} .= chr $self->{nc};
1457            # start tag or end tag            # start tag or end tag
1458          ## Stay in the state          ## Stay in the state
1459          !!!next-input-character;          !!!next-input-character;
1460          redo A;          redo A;
1461        }        }
1462      } elsif ($self->{state} eq 'before attribute name') {      } elsif ($self->{state} == BEFORE_ATTRIBUTE_NAME_STATE) {
1463        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
1464            $self->{next_input_character} == 0x000A or # LF          !!!cp (45);
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
1465          ## Stay in the state          ## Stay in the state
1466          !!!next-input-character;          !!!next-input-character;
1467          redo A;          redo A;
1468        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1469          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1470            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            !!!cp (46);
1471          } elsif ($self->{current_token}->{type} eq 'end tag') {            $self->{last_stag_name} = $self->{ct}->{tag_name};
1472            $self->{content_model_flag} = 'PCDATA'; # MUST          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1473            if ($self->{current_token}->{attributes}) {            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1474              if ($self->{ct}->{attributes}) {
1475                !!!cp (47);
1476              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1477              } else {
1478                !!!cp (48);
1479            }            }
1480          } else {          } else {
1481            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1482          }          }
1483          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1484          !!!next-input-character;          !!!next-input-character;
1485    
1486          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1487    
1488          redo A;          redo A;
1489        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{nc} and
1490                 $self->{next_input_character} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1491          $self->{current_attribute} = {name => chr ($self->{next_input_character} + 0x0020),          !!!cp (49);
1492                                value => ''};          $self->{ca}
1493          $self->{state} = 'attribute name';              = {name => chr ($self->{nc} + 0x0020),
1494                   value => '',
1495                   line => $self->{line}, column => $self->{column}};
1496            $self->{state} = ATTRIBUTE_NAME_STATE;
1497          !!!next-input-character;          !!!next-input-character;
1498          redo A;          redo A;
1499        } elsif ($self->{next_input_character} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1500            !!!cp (50);
1501            $self->{state} = SELF_CLOSING_START_TAG_STATE;
1502          !!!next-input-character;          !!!next-input-character;
         if ($self->{next_input_character} == 0x003E and # >  
             $self->{current_token}->{type} eq 'start tag' and  
             $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {  
           # permitted slash  
           #  
         } else {  
           !!!parse-error (type => 'nestc');  
         }  
         ## Stay in the state  
         # next-input-character is already done  
1503          redo A;          redo A;
1504        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1505          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1506          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1507            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            !!!cp (52);
1508          } elsif ($self->{current_token}->{type} eq 'end tag') {            $self->{last_stag_name} = $self->{ct}->{tag_name};
1509            $self->{content_model_flag} = 'PCDATA'; # MUST          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1510            if ($self->{current_token}->{attributes}) {            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1511              if ($self->{ct}->{attributes}) {
1512                !!!cp (53);
1513              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1514              } else {
1515                !!!cp (54);
1516            }            }
1517          } else {          } else {
1518            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1519          }          }
1520          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1521          # reconsume          # reconsume
1522    
1523          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1524    
1525          redo A;          redo A;
1526        } else {        } else {
1527          $self->{current_attribute} = {name => chr ($self->{next_input_character}),          if ({
1528                                value => ''};               0x0022 => 1, # "
1529          $self->{state} = 'attribute name';               0x0027 => 1, # '
1530                 0x003D => 1, # =
1531                }->{$self->{nc}}) {
1532              !!!cp (55);
1533              !!!parse-error (type => 'bad attribute name');
1534            } else {
1535              !!!cp (56);
1536            }
1537            $self->{ca}
1538                = {name => chr ($self->{nc}),
1539                   value => '',
1540                   line => $self->{line}, column => $self->{column}};
1541            $self->{state} = ATTRIBUTE_NAME_STATE;
1542          !!!next-input-character;          !!!next-input-character;
1543          redo A;          redo A;
1544        }        }
1545      } elsif ($self->{state} eq 'attribute name') {      } elsif ($self->{state} == ATTRIBUTE_NAME_STATE) {
1546        my $before_leave = sub {        my $before_leave = sub {
1547          if (exists $self->{current_token}->{attributes} # start tag or end tag          if (exists $self->{ct}->{attributes} # start tag or end tag
1548              ->{$self->{current_attribute}->{name}}) { # MUST              ->{$self->{ca}->{name}}) { # MUST
1549            !!!parse-error (type => 'dupulicate attribute');            !!!cp (57);
1550            ## Discard $self->{current_attribute} # MUST            !!!parse-error (type => 'duplicate attribute', text => $self->{ca}->{name}, line => $self->{ca}->{line}, column => $self->{ca}->{column});
1551          } else {            ## Discard $self->{ca} # MUST
1552            $self->{current_token}->{attributes}->{$self->{current_attribute}->{name}}          } else {
1553              = $self->{current_attribute};            !!!cp (58);
1554              $self->{ct}->{attributes}->{$self->{ca}->{name}}
1555                = $self->{ca};
1556          }          }
1557        }; # $before_leave        }; # $before_leave
1558    
1559        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
1560            $self->{next_input_character} == 0x000A or # LF          !!!cp (59);
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
1561          $before_leave->();          $before_leave->();
1562          $self->{state} = 'after attribute name';          $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;
1563          !!!next-input-character;          !!!next-input-character;
1564          redo A;          redo A;
1565        } elsif ($self->{next_input_character} == 0x003D) { # =        } elsif ($self->{nc} == 0x003D) { # =
1566            !!!cp (60);
1567          $before_leave->();          $before_leave->();
1568          $self->{state} = 'before attribute value';          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1569          !!!next-input-character;          !!!next-input-character;
1570          redo A;          redo A;
1571        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1572          $before_leave->();          $before_leave->();
1573          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1574            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            !!!cp (61);
1575          } elsif ($self->{current_token}->{type} eq 'end tag') {            $self->{last_stag_name} = $self->{ct}->{tag_name};
1576            $self->{content_model_flag} = 'PCDATA'; # MUST          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1577            if ($self->{current_token}->{attributes}) {            !!!cp (62);
1578              $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1579              if ($self->{ct}->{attributes}) {
1580              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1581            }            }
1582          } else {          } else {
1583            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1584          }          }
1585          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1586          !!!next-input-character;          !!!next-input-character;
1587    
1588          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1589    
1590          redo A;          redo A;
1591        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{nc} and
1592                 $self->{next_input_character} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1593          $self->{current_attribute}->{name} .= chr ($self->{next_input_character} + 0x0020);          !!!cp (63);
1594            $self->{ca}->{name} .= chr ($self->{nc} + 0x0020);
1595          ## Stay in the state          ## Stay in the state
1596          !!!next-input-character;          !!!next-input-character;
1597          redo A;          redo A;
1598        } elsif ($self->{next_input_character} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1599            !!!cp (64);
1600          $before_leave->();          $before_leave->();
1601            $self->{state} = SELF_CLOSING_START_TAG_STATE;
1602          !!!next-input-character;          !!!next-input-character;
         if ($self->{next_input_character} == 0x003E and # >  
             $self->{current_token}->{type} eq 'start tag' and  
             $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {  
           # permitted slash  
           #  
         } else {  
           !!!parse-error (type => 'nestc');  
         }  
         $self->{state} = 'before attribute name';  
         # next-input-character is already done  
1603          redo A;          redo A;
1604        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1605          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1606          $before_leave->();          $before_leave->();
1607          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1608            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            !!!cp (66);
1609          } elsif ($self->{current_token}->{type} eq 'end tag') {            $self->{last_stag_name} = $self->{ct}->{tag_name};
1610            $self->{content_model_flag} = 'PCDATA'; # MUST          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1611            if ($self->{current_token}->{attributes}) {            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1612              if ($self->{ct}->{attributes}) {
1613                !!!cp (67);
1614              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1615              } else {
1616                ## NOTE: This state should never be reached.
1617                !!!cp (68);
1618            }            }
1619          } else {          } else {
1620            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1621          }          }
1622          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1623          # reconsume          # reconsume
1624    
1625          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1626    
1627          redo A;          redo A;
1628        } else {        } else {
1629          $self->{current_attribute}->{name} .= chr ($self->{next_input_character});          if ($self->{nc} == 0x0022 or # "
1630                $self->{nc} == 0x0027) { # '
1631              !!!cp (69);
1632              !!!parse-error (type => 'bad attribute name');
1633            } else {
1634              !!!cp (70);
1635            }
1636            $self->{ca}->{name} .= chr ($self->{nc});
1637          ## Stay in the state          ## Stay in the state
1638          !!!next-input-character;          !!!next-input-character;
1639          redo A;          redo A;
1640        }        }
1641      } elsif ($self->{state} eq 'after attribute name') {      } elsif ($self->{state} == AFTER_ATTRIBUTE_NAME_STATE) {
1642        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
1643            $self->{next_input_character} == 0x000A or # LF          !!!cp (71);
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
1644          ## Stay in the state          ## Stay in the state
1645          !!!next-input-character;          !!!next-input-character;
1646          redo A;          redo A;
1647        } elsif ($self->{next_input_character} == 0x003D) { # =        } elsif ($self->{nc} == 0x003D) { # =
1648          $self->{state} = 'before attribute value';          !!!cp (72);
1649            $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
1650          !!!next-input-character;          !!!next-input-character;
1651          redo A;          redo A;
1652        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1653          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1654            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            !!!cp (73);
1655          } elsif ($self->{current_token}->{type} eq 'end tag') {            $self->{last_stag_name} = $self->{ct}->{tag_name};
1656            $self->{content_model_flag} = 'PCDATA'; # MUST          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1657            if ($self->{current_token}->{attributes}) {            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1658              if ($self->{ct}->{attributes}) {
1659                !!!cp (74);
1660              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1661              } else {
1662                ## NOTE: This state should never be reached.
1663                !!!cp (75);
1664            }            }
1665          } else {          } else {
1666            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1667          }          }
1668          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1669          !!!next-input-character;          !!!next-input-character;
1670    
1671          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1672    
1673          redo A;          redo A;
1674        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{nc} and
1675                 $self->{next_input_character} <= 0x005A) { # A..Z                 $self->{nc} <= 0x005A) { # A..Z
1676          $self->{current_attribute} = {name => chr ($self->{next_input_character} + 0x0020),          !!!cp (76);
1677                                value => ''};          $self->{ca}
1678          $self->{state} = 'attribute name';              = {name => chr ($self->{nc} + 0x0020),
1679                   value => '',
1680                   line => $self->{line}, column => $self->{column}};
1681            $self->{state} = ATTRIBUTE_NAME_STATE;
1682          !!!next-input-character;          !!!next-input-character;
1683          redo A;          redo A;
1684        } elsif ($self->{next_input_character} == 0x002F) { # /        } elsif ($self->{nc} == 0x002F) { # /
1685            !!!cp (77);
1686            $self->{state} = SELF_CLOSING_START_TAG_STATE;
1687          !!!next-input-character;          !!!next-input-character;
         if ($self->{next_input_character} == 0x003E and # >  
             $self->{current_token}->{type} eq 'start tag' and  
             $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {  
           # permitted slash  
           #  
         } else {  
           !!!parse-error (type => 'nestc');  
         }  
         $self->{state} = 'before attribute name';  
         # next-input-character is already done  
1688          redo A;          redo A;
1689        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1690          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1691          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1692            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            !!!cp (79);
1693          } elsif ($self->{current_token}->{type} eq 'end tag') {            $self->{last_stag_name} = $self->{ct}->{tag_name};
1694            $self->{content_model_flag} = 'PCDATA'; # MUST          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1695            if ($self->{current_token}->{attributes}) {            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1696              if ($self->{ct}->{attributes}) {
1697                !!!cp (80);
1698              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1699              } else {
1700                ## NOTE: This state should never be reached.
1701                !!!cp (81);
1702            }            }
1703          } else {          } else {
1704            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1705          }          }
1706          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1707          # reconsume          # reconsume
1708    
1709          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1710    
1711          redo A;          redo A;
1712        } else {        } else {
1713          $self->{current_attribute} = {name => chr ($self->{next_input_character}),          if ($self->{nc} == 0x0022 or # "
1714                                value => ''};              $self->{nc} == 0x0027) { # '
1715          $self->{state} = 'attribute name';            !!!cp (78);
1716              !!!parse-error (type => 'bad attribute name');
1717            } else {
1718              !!!cp (82);
1719            }
1720            $self->{ca}
1721                = {name => chr ($self->{nc}),
1722                   value => '',
1723                   line => $self->{line}, column => $self->{column}};
1724            $self->{state} = ATTRIBUTE_NAME_STATE;
1725          !!!next-input-character;          !!!next-input-character;
1726          redo A;                  redo A;        
1727        }        }
1728      } elsif ($self->{state} eq 'before attribute value') {      } elsif ($self->{state} == BEFORE_ATTRIBUTE_VALUE_STATE) {
1729        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
1730            $self->{next_input_character} == 0x000A or # LF          !!!cp (83);
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP        
1731          ## Stay in the state          ## Stay in the state
1732          !!!next-input-character;          !!!next-input-character;
1733          redo A;          redo A;
1734        } elsif ($self->{next_input_character} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
1735          $self->{state} = 'attribute value (double-quoted)';          !!!cp (84);
1736            $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;
1737          !!!next-input-character;          !!!next-input-character;
1738          redo A;          redo A;
1739        } elsif ($self->{next_input_character} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1740          $self->{state} = 'attribute value (unquoted)';          !!!cp (85);
1741            $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1742          ## reconsume          ## reconsume
1743          redo A;          redo A;
1744        } elsif ($self->{next_input_character} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
1745          $self->{state} = 'attribute value (single-quoted)';          !!!cp (86);
1746            $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;
1747          !!!next-input-character;          !!!next-input-character;
1748          redo A;          redo A;
1749        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1750          if ($self->{current_token}->{type} eq 'start tag') {          !!!parse-error (type => 'empty unquoted attribute value');
1751            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1752          } elsif ($self->{current_token}->{type} eq 'end tag') {            !!!cp (87);
1753            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{last_stag_name} = $self->{ct}->{tag_name};
1754            if ($self->{current_token}->{attributes}) {          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1755              $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1756              if ($self->{ct}->{attributes}) {
1757                !!!cp (88);
1758              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1759              } else {
1760                ## NOTE: This state should never be reached.
1761                !!!cp (89);
1762            }            }
1763          } else {          } else {
1764            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1765          }          }
1766          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1767          !!!next-input-character;          !!!next-input-character;
1768    
1769          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1770    
1771          redo A;          redo A;
1772        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1773          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1774          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1775            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            !!!cp (90);
1776          } elsif ($self->{current_token}->{type} eq 'end tag') {            $self->{last_stag_name} = $self->{ct}->{tag_name};
1777            $self->{content_model_flag} = 'PCDATA'; # MUST          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1778            if ($self->{current_token}->{attributes}) {            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1779              if ($self->{ct}->{attributes}) {
1780                !!!cp (91);
1781              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1782              } else {
1783                ## NOTE: This state should never be reached.
1784                !!!cp (92);
1785            }            }
1786          } else {          } else {
1787            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1788          }          }
1789          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1790          ## reconsume          ## reconsume
1791    
1792          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1793    
1794          redo A;          redo A;
1795        } else {        } else {
1796          $self->{current_attribute}->{value} .= chr ($self->{next_input_character});          if ($self->{nc} == 0x003D) { # =
1797          $self->{state} = 'attribute value (unquoted)';            !!!cp (93);
1798              !!!parse-error (type => 'bad attribute value');
1799            } else {
1800              !!!cp (94);
1801            }
1802            $self->{ca}->{value} .= chr ($self->{nc});
1803            $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1804          !!!next-input-character;          !!!next-input-character;
1805          redo A;          redo A;
1806        }        }
1807      } elsif ($self->{state} eq 'attribute value (double-quoted)') {      } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {
1808        if ($self->{next_input_character} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
1809          $self->{state} = 'before attribute name';          !!!cp (95);
1810            $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1811          !!!next-input-character;          !!!next-input-character;
1812          redo A;          redo A;
1813        } elsif ($self->{next_input_character} == 0x0026) { # &        } elsif ($self->{nc} == 0x0026) { # &
1814          $self->{last_attribute_value_state} = 'attribute value (double-quoted)';          !!!cp (96);
1815          $self->{state} = 'entity in attribute value';          ## NOTE: In the spec, the tokenizer is switched to the
1816            ## "entity in attribute value state".  In this implementation, the
1817            ## tokenizer is switched to the |ENTITY_STATE|, which is an
1818            ## implementation of the "consume a character reference" algorithm.
1819            $self->{prev_state} = $self->{state};
1820            $self->{entity_add} = 0x0022; # "
1821            $self->{state} = ENTITY_STATE;
1822          !!!next-input-character;          !!!next-input-character;
1823          redo A;          redo A;
1824        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1825          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
1826          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1827            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            !!!cp (97);
1828          } elsif ($self->{current_token}->{type} eq 'end tag') {            $self->{last_stag_name} = $self->{ct}->{tag_name};
1829            $self->{content_model_flag} = 'PCDATA'; # MUST          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1830            if ($self->{current_token}->{attributes}) {            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1831              if ($self->{ct}->{attributes}) {
1832                !!!cp (98);
1833              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1834              } else {
1835                ## NOTE: This state should never be reached.
1836                !!!cp (99);
1837            }            }
1838          } else {          } else {
1839            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1840          }          }
1841          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1842          ## reconsume          ## reconsume
1843    
1844          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1845    
1846          redo A;          redo A;
1847        } else {        } else {
1848          $self->{current_attribute}->{value} .= chr ($self->{next_input_character});          !!!cp (100);
1849            $self->{ca}->{value} .= chr ($self->{nc});
1850            $self->{read_until}->($self->{ca}->{value},
1851                                  q["&],
1852                                  length $self->{ca}->{value});
1853    
1854          ## Stay in the state          ## Stay in the state
1855          !!!next-input-character;          !!!next-input-character;
1856          redo A;          redo A;
1857        }        }
1858      } elsif ($self->{state} eq 'attribute value (single-quoted)') {      } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {
1859        if ($self->{next_input_character} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
1860          $self->{state} = 'before attribute name';          !!!cp (101);
1861          !!!next-input-character;          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1862          redo A;          !!!next-input-character;
1863        } elsif ($self->{next_input_character} == 0x0026) { # &          redo A;
1864          $self->{last_attribute_value_state} = 'attribute value (single-quoted)';        } elsif ($self->{nc} == 0x0026) { # &
1865          $self->{state} = 'entity in attribute value';          !!!cp (102);
1866            ## NOTE: In the spec, the tokenizer is switched to the
1867            ## "entity in attribute value state".  In this implementation, the
1868            ## tokenizer is switched to the |ENTITY_STATE|, which is an
1869            ## implementation of the "consume a character reference" algorithm.
1870            $self->{entity_add} = 0x0027; # '
1871            $self->{prev_state} = $self->{state};
1872            $self->{state} = ENTITY_STATE;
1873          !!!next-input-character;          !!!next-input-character;
1874          redo A;          redo A;
1875        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1876          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
1877          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1878            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            !!!cp (103);
1879          } elsif ($self->{current_token}->{type} eq 'end tag') {            $self->{last_stag_name} = $self->{ct}->{tag_name};
1880            $self->{content_model_flag} = 'PCDATA'; # MUST          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1881            if ($self->{current_token}->{attributes}) {            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1882              if ($self->{ct}->{attributes}) {
1883                !!!cp (104);
1884              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1885              } else {
1886                ## NOTE: This state should never be reached.
1887                !!!cp (105);
1888            }            }
1889          } else {          } else {
1890            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1891          }          }
1892          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1893          ## reconsume          ## reconsume
1894    
1895          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1896    
1897          redo A;          redo A;
1898        } else {        } else {
1899          $self->{current_attribute}->{value} .= chr ($self->{next_input_character});          !!!cp (106);
1900            $self->{ca}->{value} .= chr ($self->{nc});
1901            $self->{read_until}->($self->{ca}->{value},
1902                                  q['&],
1903                                  length $self->{ca}->{value});
1904    
1905          ## Stay in the state          ## Stay in the state
1906          !!!next-input-character;          !!!next-input-character;
1907          redo A;          redo A;
1908        }        }
1909      } elsif ($self->{state} eq 'attribute value (unquoted)') {      } elsif ($self->{state} == ATTRIBUTE_VALUE_UNQUOTED_STATE) {
1910        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
1911            $self->{next_input_character} == 0x000A or # LF          !!!cp (107);
1912            $self->{next_input_character} == 0x000B or # HT          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1913            $self->{next_input_character} == 0x000C or # FF          !!!next-input-character;
1914            $self->{next_input_character} == 0x0020) { # SP          redo A;
1915          $self->{state} = 'before attribute name';        } elsif ($self->{nc} == 0x0026) { # &
1916          !!!next-input-character;          !!!cp (108);
1917          redo A;          ## NOTE: In the spec, the tokenizer is switched to the
1918        } elsif ($self->{next_input_character} == 0x0026) { # &          ## "entity in attribute value state".  In this implementation, the
1919          $self->{last_attribute_value_state} = 'attribute value (unquoted)';          ## tokenizer is switched to the |ENTITY_STATE|, which is an
1920          $self->{state} = 'entity in attribute value';          ## implementation of the "consume a character reference" algorithm.
1921          !!!next-input-character;          $self->{entity_add} = -1;
1922          redo A;          $self->{prev_state} = $self->{state};
1923        } elsif ($self->{next_input_character} == 0x003E) { # >          $self->{state} = ENTITY_STATE;
1924          if ($self->{current_token}->{type} eq 'start tag') {          !!!next-input-character;
1925            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};          redo A;
1926          } elsif ($self->{current_token}->{type} eq 'end tag') {        } elsif ($self->{nc} == 0x003E) { # >
1927            $self->{content_model_flag} = 'PCDATA'; # MUST          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1928            if ($self->{current_token}->{attributes}) {            !!!cp (109);
1929              $self->{last_stag_name} = $self->{ct}->{tag_name};
1930            } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1931              $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1932              if ($self->{ct}->{attributes}) {
1933                !!!cp (110);
1934              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1935              } else {
1936                ## NOTE: This state should never be reached.
1937                !!!cp (111);
1938            }            }
1939          } else {          } else {
1940            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1941          }          }
1942          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1943          !!!next-input-character;          !!!next-input-character;
1944    
1945          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1946    
1947          redo A;          redo A;
1948        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
1949          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1950          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{ct}->{type} == START_TAG_TOKEN) {
1951            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            !!!cp (112);
1952          } elsif ($self->{current_token}->{type} eq 'end tag') {            $self->{last_stag_name} = $self->{ct}->{tag_name};
1953            $self->{content_model_flag} = 'PCDATA'; # MUST          } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
1954            if ($self->{current_token}->{attributes}) {            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1955              if ($self->{ct}->{attributes}) {
1956                !!!cp (113);
1957              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1958              } else {
1959                ## NOTE: This state should never be reached.
1960                !!!cp (114);
1961            }            }
1962          } else {          } else {
1963            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{ct}->{type}: Unknown token type";
1964          }          }
1965          $self->{state} = 'data';          $self->{state} = DATA_STATE;
1966          ## reconsume          ## reconsume
1967    
1968          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{ct}); # start tag or end tag
1969    
1970          redo A;          redo A;
1971        } else {        } else {
1972          $self->{current_attribute}->{value} .= chr ($self->{next_input_character});          if ({
1973                 0x0022 => 1, # "
1974                 0x0027 => 1, # '
1975                 0x003D => 1, # =
1976                }->{$self->{nc}}) {
1977              !!!cp (115);
1978              !!!parse-error (type => 'bad attribute value');
1979            } else {
1980              !!!cp (116);
1981            }
1982            $self->{ca}->{value} .= chr ($self->{nc});
1983            $self->{read_until}->($self->{ca}->{value},
1984                                  q["'=& >],
1985                                  length $self->{ca}->{value});
1986    
1987          ## Stay in the state          ## Stay in the state
1988          !!!next-input-character;          !!!next-input-character;
1989          redo A;          redo A;
1990        }        }
1991      } elsif ($self->{state} eq 'entity in attribute value') {      } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {
1992        my $token = $self->_tokenize_attempt_to_consume_an_entity;        if ($is_space->{$self->{nc}}) {
1993            !!!cp (118);
1994            $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1995            !!!next-input-character;
1996            redo A;
1997          } elsif ($self->{nc} == 0x003E) { # >
1998            if ($self->{ct}->{type} == START_TAG_TOKEN) {
1999              !!!cp (119);
2000              $self->{last_stag_name} = $self->{ct}->{tag_name};
2001            } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2002              $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
2003              if ($self->{ct}->{attributes}) {
2004                !!!cp (120);
2005                !!!parse-error (type => 'end tag attribute');
2006              } else {
2007                ## NOTE: This state should never be reached.
2008                !!!cp (121);
2009              }
2010            } else {
2011              die "$0: $self->{ct}->{type}: Unknown token type";
2012            }
2013            $self->{state} = DATA_STATE;
2014            !!!next-input-character;
2015    
2016        unless (defined $token) {          !!!emit ($self->{ct}); # start tag or end tag
2017          $self->{current_attribute}->{value} .= '&';  
2018            redo A;
2019          } elsif ($self->{nc} == 0x002F) { # /
2020            !!!cp (122);
2021            $self->{state} = SELF_CLOSING_START_TAG_STATE;
2022            !!!next-input-character;
2023            redo A;
2024          } elsif ($self->{nc} == -1) {
2025            !!!parse-error (type => 'unclosed tag');
2026            if ($self->{ct}->{type} == START_TAG_TOKEN) {
2027              !!!cp (122.3);
2028              $self->{last_stag_name} = $self->{ct}->{tag_name};
2029            } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2030              if ($self->{ct}->{attributes}) {
2031                !!!cp (122.1);
2032                !!!parse-error (type => 'end tag attribute');
2033              } else {
2034                ## NOTE: This state should never be reached.
2035                !!!cp (122.2);
2036              }
2037            } else {
2038              die "$0: $self->{ct}->{type}: Unknown token type";
2039            }
2040            $self->{state} = DATA_STATE;
2041            ## Reconsume.
2042            !!!emit ($self->{ct}); # start tag or end tag
2043            redo A;
2044        } else {        } else {
2045          $self->{current_attribute}->{value} .= $token->{data};          !!!cp ('124.1');
2046          ## ISSUE: spec says "append the returned character token to the current attribute's value"          !!!parse-error (type => 'no space between attributes');
2047            $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
2048            ## reconsume
2049            redo A;
2050        }        }
2051        } elsif ($self->{state} == SELF_CLOSING_START_TAG_STATE) {
2052          if ($self->{nc} == 0x003E) { # >
2053            if ($self->{ct}->{type} == END_TAG_TOKEN) {
2054              !!!cp ('124.2');
2055              !!!parse-error (type => 'nestc', token => $self->{ct});
2056              ## TODO: Different type than slash in start tag
2057              $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
2058              if ($self->{ct}->{attributes}) {
2059                !!!cp ('124.4');
2060                !!!parse-error (type => 'end tag attribute');
2061              } else {
2062                !!!cp ('124.5');
2063              }
2064              ## TODO: Test |<title></title/>|
2065            } else {
2066              !!!cp ('124.3');
2067              $self->{self_closing} = 1;
2068            }
2069    
2070        $self->{state} = $self->{last_attribute_value_state};          $self->{state} = DATA_STATE;
2071        # next-input-character is already done          !!!next-input-character;
       redo A;  
     } elsif ($self->{state} eq 'bogus comment') {  
       ## (only happen if PCDATA state)  
         
       my $token = {type => 'comment', data => ''};  
   
       BC: {  
         if ($self->{next_input_character} == 0x003E) { # >  
           $self->{state} = 'data';  
           !!!next-input-character;  
   
           !!!emit ($token);  
   
           redo A;  
         } elsif ($self->{next_input_character} == -1) {  
           $self->{state} = 'data';  
           ## reconsume  
2072    
2073            !!!emit ($token);          !!!emit ($self->{ct}); # start tag or end tag
2074    
2075            redo A;          redo A;
2076          } elsif ($self->{nc} == -1) {
2077            !!!parse-error (type => 'unclosed tag');
2078            if ($self->{ct}->{type} == START_TAG_TOKEN) {
2079              !!!cp (124.7);
2080              $self->{last_stag_name} = $self->{ct}->{tag_name};
2081            } elsif ($self->{ct}->{type} == END_TAG_TOKEN) {
2082              if ($self->{ct}->{attributes}) {
2083                !!!cp (124.5);
2084                !!!parse-error (type => 'end tag attribute');
2085              } else {
2086                ## NOTE: This state should never be reached.
2087                !!!cp (124.6);
2088              }
2089          } else {          } else {
2090            $token->{data} .= chr ($self->{next_input_character});            die "$0: $self->{ct}->{type}: Unknown token type";
           !!!next-input-character;  
           redo BC;  
2091          }          }
2092        } # BC          $self->{state} = DATA_STATE;
2093      } elsif ($self->{state} eq 'markup declaration open') {          ## Reconsume.
2094            !!!emit ($self->{ct}); # start tag or end tag
2095            redo A;
2096          } else {
2097            !!!cp ('124.4');
2098            !!!parse-error (type => 'nestc');
2099            ## TODO: This error type is wrong.
2100            $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
2101            ## Reconsume.
2102            redo A;
2103          }
2104        } elsif ($self->{state} == BOGUS_COMMENT_STATE) {
2105        ## (only happen if PCDATA state)        ## (only happen if PCDATA state)
2106    
2107        my @next_char;        ## NOTE: Unlike spec's "bogus comment state", this implementation
2108        push @next_char, $self->{next_input_character};        ## consumes characters one-by-one basis.
2109                
2110        if ($self->{next_input_character} == 0x002D) { # -        if ($self->{nc} == 0x003E) { # >
2111            !!!cp (124);
2112            $self->{state} = DATA_STATE;
2113          !!!next-input-character;          !!!next-input-character;
2114          push @next_char, $self->{next_input_character};  
2115          if ($self->{next_input_character} == 0x002D) { # -          !!!emit ($self->{ct}); # comment
2116            $self->{current_token} = {type => 'comment', data => ''};          redo A;
2117            $self->{state} = 'comment start';        } elsif ($self->{nc} == -1) {
2118            !!!next-input-character;          !!!cp (125);
2119            redo A;          $self->{state} = DATA_STATE;
2120          }          ## reconsume
2121        } elsif ($self->{next_input_character} == 0x0044 or # D  
2122                 $self->{next_input_character} == 0x0064) { # d          !!!emit ($self->{ct}); # comment
2123            redo A;
2124          } else {
2125            !!!cp (126);
2126            $self->{ct}->{data} .= chr ($self->{nc}); # comment
2127            $self->{read_until}->($self->{ct}->{data},
2128                                  q[>],
2129                                  length $self->{ct}->{data});
2130    
2131            ## Stay in the state.
2132          !!!next-input-character;          !!!next-input-character;
2133          push @next_char, $self->{next_input_character};          redo A;
2134          if ($self->{next_input_character} == 0x004F or # O        }
2135              $self->{next_input_character} == 0x006F) { # o      } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {
2136            !!!next-input-character;        ## (only happen if PCDATA state)
2137            push @next_char, $self->{next_input_character};        
2138            if ($self->{next_input_character} == 0x0043 or # C        if ($self->{nc} == 0x002D) { # -
2139                $self->{next_input_character} == 0x0063) { # c          !!!cp (133);
2140              !!!next-input-character;          $self->{state} = MD_HYPHEN_STATE;
2141              push @next_char, $self->{next_input_character};          !!!next-input-character;
2142              if ($self->{next_input_character} == 0x0054 or # T          redo A;
2143                  $self->{next_input_character} == 0x0074) { # t        } elsif ($self->{nc} == 0x0044 or # D
2144                !!!next-input-character;                 $self->{nc} == 0x0064) { # d
2145                push @next_char, $self->{next_input_character};          ## ASCII case-insensitive.
2146                if ($self->{next_input_character} == 0x0059 or # Y          !!!cp (130);
2147                    $self->{next_input_character} == 0x0079) { # y          $self->{state} = MD_DOCTYPE_STATE;
2148                  !!!next-input-character;          $self->{s_kwd} = chr $self->{nc};
2149                  push @next_char, $self->{next_input_character};          !!!next-input-character;
2150                  if ($self->{next_input_character} == 0x0050 or # P          redo A;
2151                      $self->{next_input_character} == 0x0070) { # p        } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and
2152                    !!!next-input-character;                 $self->{open_elements}->[-1]->[1] & FOREIGN_EL and
2153                    push @next_char, $self->{next_input_character};                 $self->{nc} == 0x005B) { # [
2154                    if ($self->{next_input_character} == 0x0045 or # E          !!!cp (135.4);                
2155                        $self->{next_input_character} == 0x0065) { # e          $self->{state} = MD_CDATA_STATE;
2156                      ## ISSUE: What a stupid code this is!          $self->{s_kwd} = '[';
2157                      $self->{state} = 'DOCTYPE';          !!!next-input-character;
2158                      !!!next-input-character;          redo A;
2159                      redo A;        } else {
2160                    }          !!!cp (136);
                 }  
               }  
             }  
           }  
         }  
2161        }        }
2162    
2163        !!!parse-error (type => 'bogus comment open');        !!!parse-error (type => 'bogus comment',
2164        $self->{next_input_character} = shift @next_char;                        line => $self->{line_prev},
2165        !!!back-next-input-character (@next_char);                        column => $self->{column_prev} - 1);
2166        $self->{state} = 'bogus comment';        ## Reconsume.
2167          $self->{state} = BOGUS_COMMENT_STATE;
2168          $self->{ct} = {type => COMMENT_TOKEN, data => '',
2169                                    line => $self->{line_prev},
2170                                    column => $self->{column_prev} - 1,
2171                                   };
2172        redo A;        redo A;
2173              } elsif ($self->{state} == MD_HYPHEN_STATE) {
2174        ## ISSUE: typos in spec: chacacters, is is a parse error        if ($self->{nc} == 0x002D) { # -
2175        ## ISSUE: spec is somewhat unclear on "is the first character that will be in the comment"; what is "that will be in the comment" is what the algorithm defines, isn't it?          !!!cp (127);
2176      } elsif ($self->{state} eq 'comment start') {          $self->{ct} = {type => COMMENT_TOKEN, data => '',
2177        if ($self->{next_input_character} == 0x002D) { # -                                    line => $self->{line_prev},
2178          $self->{state} = 'comment start dash';                                    column => $self->{column_prev} - 2,
2179                                     };
2180            $self->{state} = COMMENT_START_STATE;
2181          !!!next-input-character;          !!!next-input-character;
2182          redo A;          redo A;
2183        } elsif ($self->{next_input_character} == 0x003E) { # >        } else {
2184            !!!cp (128);
2185            !!!parse-error (type => 'bogus comment',
2186                            line => $self->{line_prev},
2187                            column => $self->{column_prev} - 2);
2188            $self->{state} = BOGUS_COMMENT_STATE;
2189            ## Reconsume.
2190            $self->{ct} = {type => COMMENT_TOKEN,
2191                                      data => '-',
2192                                      line => $self->{line_prev},
2193                                      column => $self->{column_prev} - 2,
2194                                     };
2195            redo A;
2196          }
2197        } elsif ($self->{state} == MD_DOCTYPE_STATE) {
2198          ## ASCII case-insensitive.
2199          if ($self->{nc} == [
2200                undef,
2201                0x004F, # O
2202                0x0043, # C
2203                0x0054, # T
2204                0x0059, # Y
2205                0x0050, # P
2206              ]->[length $self->{s_kwd}] or
2207              $self->{nc} == [
2208                undef,
2209                0x006F, # o
2210                0x0063, # c
2211                0x0074, # t
2212                0x0079, # y
2213                0x0070, # p
2214              ]->[length $self->{s_kwd}]) {
2215            !!!cp (131);
2216            ## Stay in the state.
2217            $self->{s_kwd} .= chr $self->{nc};
2218            !!!next-input-character;
2219            redo A;
2220          } elsif ((length $self->{s_kwd}) == 6 and
2221                   ($self->{nc} == 0x0045 or # E
2222                    $self->{nc} == 0x0065)) { # e
2223            !!!cp (129);
2224            $self->{state} = DOCTYPE_STATE;
2225            $self->{ct} = {type => DOCTYPE_TOKEN,
2226                                      quirks => 1,
2227                                      line => $self->{line_prev},
2228                                      column => $self->{column_prev} - 7,
2229                                     };
2230            !!!next-input-character;
2231            redo A;
2232          } else {
2233            !!!cp (132);        
2234            !!!parse-error (type => 'bogus comment',
2235                            line => $self->{line_prev},
2236                            column => $self->{column_prev} - 1 - length $self->{s_kwd});
2237            $self->{state} = BOGUS_COMMENT_STATE;
2238            ## Reconsume.
2239            $self->{ct} = {type => COMMENT_TOKEN,
2240                                      data => $self->{s_kwd},
2241                                      line => $self->{line_prev},
2242                                      column => $self->{column_prev} - 1 - length $self->{s_kwd},
2243                                     };
2244            redo A;
2245          }
2246        } elsif ($self->{state} == MD_CDATA_STATE) {
2247          if ($self->{nc} == {
2248                '[' => 0x0043, # C
2249                '[C' => 0x0044, # D
2250                '[CD' => 0x0041, # A
2251                '[CDA' => 0x0054, # T
2252                '[CDAT' => 0x0041, # A
2253              }->{$self->{s_kwd}}) {
2254            !!!cp (135.1);
2255            ## Stay in the state.
2256            $self->{s_kwd} .= chr $self->{nc};
2257            !!!next-input-character;
2258            redo A;
2259          } elsif ($self->{s_kwd} eq '[CDATA' and
2260                   $self->{nc} == 0x005B) { # [
2261            !!!cp (135.2);
2262            $self->{ct} = {type => CHARACTER_TOKEN,
2263                                      data => '',
2264                                      line => $self->{line_prev},
2265                                      column => $self->{column_prev} - 7};
2266            $self->{state} = CDATA_SECTION_STATE;
2267            !!!next-input-character;
2268            redo A;
2269          } else {
2270            !!!cp (135.3);
2271            !!!parse-error (type => 'bogus comment',
2272                            line => $self->{line_prev},
2273                            column => $self->{column_prev} - 1 - length $self->{s_kwd});
2274            $self->{state} = BOGUS_COMMENT_STATE;
2275            ## Reconsume.
2276            $self->{ct} = {type => COMMENT_TOKEN,
2277                                      data => $self->{s_kwd},
2278                                      line => $self->{line_prev},
2279                                      column => $self->{column_prev} - 1 - length $self->{s_kwd},
2280                                     };
2281            redo A;
2282          }
2283        } elsif ($self->{state} == COMMENT_START_STATE) {
2284          if ($self->{nc} == 0x002D) { # -
2285            !!!cp (137);
2286            $self->{state} = COMMENT_START_DASH_STATE;
2287            !!!next-input-character;
2288            redo A;
2289          } elsif ($self->{nc} == 0x003E) { # >
2290            !!!cp (138);
2291          !!!parse-error (type => 'bogus comment');          !!!parse-error (type => 'bogus comment');
2292          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2293          !!!next-input-character;          !!!next-input-character;
2294    
2295          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2296    
2297          redo A;          redo A;
2298        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2299            !!!cp (139);
2300          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2301          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2302          ## reconsume          ## reconsume
2303    
2304          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2305    
2306          redo A;          redo A;
2307        } else {        } else {
2308          $self->{current_token}->{data} # comment          !!!cp (140);
2309              .= chr ($self->{next_input_character});          $self->{ct}->{data} # comment
2310          $self->{state} = 'comment';              .= chr ($self->{nc});
2311            $self->{state} = COMMENT_STATE;
2312          !!!next-input-character;          !!!next-input-character;
2313          redo A;          redo A;
2314        }        }
2315      } elsif ($self->{state} eq 'comment start dash') {      } elsif ($self->{state} == COMMENT_START_DASH_STATE) {
2316        if ($self->{next_input_character} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2317          $self->{state} = 'comment end';          !!!cp (141);
2318            $self->{state} = COMMENT_END_STATE;
2319          !!!next-input-character;          !!!next-input-character;
2320          redo A;          redo A;
2321        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2322            !!!cp (142);
2323          !!!parse-error (type => 'bogus comment');          !!!parse-error (type => 'bogus comment');
2324          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2325          !!!next-input-character;          !!!next-input-character;
2326    
2327          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2328    
2329          redo A;          redo A;
2330        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2331            !!!cp (143);
2332          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2333          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2334          ## reconsume          ## reconsume
2335    
2336          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2337    
2338          redo A;          redo A;
2339        } else {        } else {
2340          $self->{current_token}->{data} # comment          !!!cp (144);
2341              .= chr ($self->{next_input_character});          $self->{ct}->{data} # comment
2342          $self->{state} = 'comment';              .= '-' . chr ($self->{nc});
2343            $self->{state} = COMMENT_STATE;
2344          !!!next-input-character;          !!!next-input-character;
2345          redo A;          redo A;
2346        }        }
2347      } elsif ($self->{state} eq 'comment') {      } elsif ($self->{state} == COMMENT_STATE) {
2348        if ($self->{next_input_character} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2349          $self->{state} = 'comment end dash';          !!!cp (145);
2350            $self->{state} = COMMENT_END_DASH_STATE;
2351          !!!next-input-character;          !!!next-input-character;
2352          redo A;          redo A;
2353        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2354            !!!cp (146);
2355          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2356          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2357          ## reconsume          ## reconsume
2358    
2359          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2360    
2361          redo A;          redo A;
2362        } else {        } else {
2363          $self->{current_token}->{data} .= chr ($self->{next_input_character}); # comment          !!!cp (147);
2364            $self->{ct}->{data} .= chr ($self->{nc}); # comment
2365            $self->{read_until}->($self->{ct}->{data},
2366                                  q[-],
2367                                  length $self->{ct}->{data});
2368    
2369          ## Stay in the state          ## Stay in the state
2370          !!!next-input-character;          !!!next-input-character;
2371          redo A;          redo A;
2372        }        }
2373      } elsif ($self->{state} eq 'comment end dash') {      } elsif ($self->{state} == COMMENT_END_DASH_STATE) {
2374        if ($self->{next_input_character} == 0x002D) { # -        if ($self->{nc} == 0x002D) { # -
2375          $self->{state} = 'comment end';          !!!cp (148);
2376            $self->{state} = COMMENT_END_STATE;
2377          !!!next-input-character;          !!!next-input-character;
2378          redo A;          redo A;
2379        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2380            !!!cp (149);
2381          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2382          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2383          ## reconsume          ## reconsume
2384    
2385          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2386    
2387          redo A;          redo A;
2388        } else {        } else {
2389          $self->{current_token}->{data} .= '-' . chr ($self->{next_input_character}); # comment          !!!cp (150);
2390          $self->{state} = 'comment';          $self->{ct}->{data} .= '-' . chr ($self->{nc}); # comment
2391            $self->{state} = COMMENT_STATE;
2392          !!!next-input-character;          !!!next-input-character;
2393          redo A;          redo A;
2394        }        }
2395      } elsif ($self->{state} eq 'comment end') {      } elsif ($self->{state} == COMMENT_END_STATE) {
2396        if ($self->{next_input_character} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2397          $self->{state} = 'data';          !!!cp (151);
2398            $self->{state} = DATA_STATE;
2399          !!!next-input-character;          !!!next-input-character;
2400    
2401          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2402    
2403          redo A;          redo A;
2404        } elsif ($self->{next_input_character} == 0x002D) { # -        } elsif ($self->{nc} == 0x002D) { # -
2405          !!!parse-error (type => 'dash in comment');          !!!cp (152);
2406          $self->{current_token}->{data} .= '-'; # comment          !!!parse-error (type => 'dash in comment',
2407                            line => $self->{line_prev},
2408                            column => $self->{column_prev});
2409            $self->{ct}->{data} .= '-'; # comment
2410          ## Stay in the state          ## Stay in the state
2411          !!!next-input-character;          !!!next-input-character;
2412          redo A;          redo A;
2413        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2414            !!!cp (153);
2415          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
2416          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2417          ## reconsume          ## reconsume
2418    
2419          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{ct}); # comment
2420    
2421          redo A;          redo A;
2422        } else {        } else {
2423          !!!parse-error (type => 'dash in comment');          !!!cp (154);
2424          $self->{current_token}->{data} .= '--' . chr ($self->{next_input_character}); # comment          !!!parse-error (type => 'dash in comment',
2425          $self->{state} = 'comment';                          line => $self->{line_prev},
2426                            column => $self->{column_prev});
2427            $self->{ct}->{data} .= '--' . chr ($self->{nc}); # comment
2428            $self->{state} = COMMENT_STATE;
2429          !!!next-input-character;          !!!next-input-character;
2430          redo A;          redo A;
2431        }        }
2432      } elsif ($self->{state} eq 'DOCTYPE') {      } elsif ($self->{state} == DOCTYPE_STATE) {
2433        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
2434            $self->{next_input_character} == 0x000A or # LF          !!!cp (155);
2435            $self->{next_input_character} == 0x000B or # VT          $self->{state} = BEFORE_DOCTYPE_NAME_STATE;
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
         $self->{state} = 'before DOCTYPE name';  
2436          !!!next-input-character;          !!!next-input-character;
2437          redo A;          redo A;
2438        } else {        } else {
2439            !!!cp (156);
2440          !!!parse-error (type => 'no space before DOCTYPE name');          !!!parse-error (type => 'no space before DOCTYPE name');
2441          $self->{state} = 'before DOCTYPE name';          $self->{state} = BEFORE_DOCTYPE_NAME_STATE;
2442          ## reconsume          ## reconsume
2443          redo A;          redo A;
2444        }        }
2445      } elsif ($self->{state} eq 'before DOCTYPE name') {      } elsif ($self->{state} == BEFORE_DOCTYPE_NAME_STATE) {
2446        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
2447            $self->{next_input_character} == 0x000A or # LF          !!!cp (157);
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
2448          ## Stay in the state          ## Stay in the state
2449          !!!next-input-character;          !!!next-input-character;
2450          redo A;          redo A;
2451        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2452            !!!cp (158);
2453          !!!parse-error (type => 'no DOCTYPE name');          !!!parse-error (type => 'no DOCTYPE name');
2454          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2455          !!!next-input-character;          !!!next-input-character;
2456    
2457          !!!emit ({type => 'DOCTYPE'}); # incorrect          !!!emit ($self->{ct}); # DOCTYPE (quirks)
2458    
2459          redo A;          redo A;
2460        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2461            !!!cp (159);
2462          !!!parse-error (type => 'no DOCTYPE name');          !!!parse-error (type => 'no DOCTYPE name');
2463          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2464          ## reconsume          ## reconsume
2465    
2466          !!!emit ({type => 'DOCTYPE'}); # incorrect          !!!emit ($self->{ct}); # DOCTYPE (quirks)
2467    
2468          redo A;          redo A;
2469        } else {        } else {
2470          $self->{current_token}          !!!cp (160);
2471              = {type => 'DOCTYPE',          $self->{ct}->{name} = chr $self->{nc};
2472                 name => chr ($self->{next_input_character}),          delete $self->{ct}->{quirks};
                correct => 1};  
2473  ## ISSUE: "Set the token's name name to the" in the spec  ## ISSUE: "Set the token's name name to the" in the spec
2474          $self->{state} = 'DOCTYPE name';          $self->{state} = DOCTYPE_NAME_STATE;
2475          !!!next-input-character;          !!!next-input-character;
2476          redo A;          redo A;
2477        }        }
2478      } elsif ($self->{state} eq 'DOCTYPE name') {      } elsif ($self->{state} == DOCTYPE_NAME_STATE) {
2479  ## ISSUE: Redundant "First," in the spec.  ## ISSUE: Redundant "First," in the spec.
2480        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
2481            $self->{next_input_character} == 0x000A or # LF          !!!cp (161);
2482            $self->{next_input_character} == 0x000B or # VT          $self->{state} = AFTER_DOCTYPE_NAME_STATE;
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
         $self->{state} = 'after DOCTYPE name';  
2483          !!!next-input-character;          !!!next-input-character;
2484          redo A;          redo A;
2485        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2486          $self->{state} = 'data';          !!!cp (162);
2487            $self->{state} = DATA_STATE;
2488          !!!next-input-character;          !!!next-input-character;
2489    
2490          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2491    
2492          redo A;          redo A;
2493        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2494            !!!cp (163);
2495          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2496          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2497          ## reconsume          ## reconsume
2498    
2499          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2500          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2501    
2502          redo A;          redo A;
2503        } else {        } else {
2504          $self->{current_token}->{name}          !!!cp (164);
2505            .= chr ($self->{next_input_character}); # DOCTYPE          $self->{ct}->{name}
2506              .= chr ($self->{nc}); # DOCTYPE
2507          ## Stay in the state          ## Stay in the state
2508          !!!next-input-character;          !!!next-input-character;
2509          redo A;          redo A;
2510        }        }
2511      } elsif ($self->{state} eq 'after DOCTYPE name') {      } elsif ($self->{state} == AFTER_DOCTYPE_NAME_STATE) {
2512        if ($self->{next_input_character} == 0x0009 or # HT        if ($is_space->{$self->{nc}}) {
2513            $self->{next_input_character} == 0x000A or # LF          !!!cp (165);
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
2514          ## Stay in the state          ## Stay in the state
2515          !!!next-input-character;          !!!next-input-character;
2516          redo A;          redo A;
2517        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2518          $self->{state} = 'data';          !!!cp (166);
2519            $self->{state} = DATA_STATE;
2520          !!!next-input-character;          !!!next-input-character;
2521    
2522          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2523    
2524          redo A;          redo A;
2525        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2526            !!!cp (167);
2527          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2528          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2529          ## reconsume          ## reconsume
2530    
2531          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2532          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2533    
2534          redo A;          redo A;
2535        } elsif ($self->{next_input_character} == 0x0050 or # P        } elsif ($self->{nc} == 0x0050 or # P
2536                 $self->{next_input_character} == 0x0070) { # p                 $self->{nc} == 0x0070) { # p
2537            $self->{state} = PUBLIC_STATE;
2538            $self->{s_kwd} = chr $self->{nc};
2539          !!!next-input-character;          !!!next-input-character;
2540          if ($self->{next_input_character} == 0x0055 or # U          redo A;
2541              $self->{next_input_character} == 0x0075) { # u        } elsif ($self->{nc} == 0x0053 or # S
2542            !!!next-input-character;                 $self->{nc} == 0x0073) { # s
2543            if ($self->{next_input_character} == 0x0042 or # B          $self->{state} = SYSTEM_STATE;
2544                $self->{next_input_character} == 0x0062) { # b          $self->{s_kwd} = chr $self->{nc};
             !!!next-input-character;  
             if ($self->{next_input_character} == 0x004C or # L  
                 $self->{next_input_character} == 0x006C) { # l  
               !!!next-input-character;  
               if ($self->{next_input_character} == 0x0049 or # I  
                   $self->{next_input_character} == 0x0069) { # i  
                 !!!next-input-character;  
                 if ($self->{next_input_character} == 0x0043 or # C  
                     $self->{next_input_character} == 0x0063) { # c  
                   $self->{state} = 'before DOCTYPE public identifier';  
                   !!!next-input-character;  
                   redo A;  
                 }  
               }  
             }  
           }  
         }  
   
         #  
       } elsif ($self->{next_input_character} == 0x0053 or # S  
                $self->{next_input_character} == 0x0073) { # s  
2545          !!!next-input-character;          !!!next-input-character;
2546          if ($self->{next_input_character} == 0x0059 or # Y          redo A;
             $self->{next_input_character} == 0x0079) { # y  
           !!!next-input-character;  
           if ($self->{next_input_character} == 0x0053 or # S  
               $self->{next_input_character} == 0x0073) { # s  
             !!!next-input-character;  
             if ($self->{next_input_character} == 0x0054 or # T  
                 $self->{next_input_character} == 0x0074) { # t  
               !!!next-input-character;  
               if ($self->{next_input_character} == 0x0045 or # E  
                   $self->{next_input_character} == 0x0065) { # e  
                 !!!next-input-character;  
                 if ($self->{next_input_character} == 0x004D or # M  
                     $self->{next_input_character} == 0x006D) { # m  
                   $self->{state} = 'before DOCTYPE system identifier';  
                   !!!next-input-character;  
                   redo A;  
                 }  
               }  
             }  
           }  
         }  
   
         #  
2547        } else {        } else {
2548            !!!cp (180);
2549            !!!parse-error (type => 'string after DOCTYPE name');
2550            $self->{ct}->{quirks} = 1;
2551    
2552            $self->{state} = BOGUS_DOCTYPE_STATE;
2553          !!!next-input-character;          !!!next-input-character;
2554          #          redo A;
2555        }        }
2556        } elsif ($self->{state} == PUBLIC_STATE) {
2557        !!!parse-error (type => 'string after DOCTYPE name');        ## ASCII case-insensitive
2558        $self->{state} = 'bogus DOCTYPE';        if ($self->{nc} == [
2559        # next-input-character is already done              undef,
2560        redo A;              0x0055, # U
2561      } elsif ($self->{state} eq 'before DOCTYPE public identifier') {              0x0042, # B
2562        if ({              0x004C, # L
2563              0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,              0x0049, # I
2564              #0x000D => 1, # HT, LF, VT, FF, SP, CR            ]->[length $self->{s_kwd}] or
2565            }->{$self->{next_input_character}}) {            $self->{nc} == [
2566                undef,
2567                0x0075, # u
2568                0x0062, # b
2569                0x006C, # l
2570                0x0069, # i
2571              ]->[length $self->{s_kwd}]) {
2572            !!!cp (175);
2573            ## Stay in the state.
2574            $self->{s_kwd} .= chr $self->{nc};
2575            !!!next-input-character;
2576            redo A;
2577          } elsif ((length $self->{s_kwd}) == 5 and
2578                   ($self->{nc} == 0x0043 or # C
2579                    $self->{nc} == 0x0063)) { # c
2580            !!!cp (168);
2581            $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2582            !!!next-input-character;
2583            redo A;
2584          } else {
2585            !!!cp (169);
2586            !!!parse-error (type => 'string after DOCTYPE name',
2587                            line => $self->{line_prev},
2588                            column => $self->{column_prev} + 1 - length $self->{s_kwd});
2589            $self->{ct}->{quirks} = 1;
2590    
2591            $self->{state} = BOGUS_DOCTYPE_STATE;
2592            ## Reconsume.
2593            redo A;
2594          }
2595        } elsif ($self->{state} == SYSTEM_STATE) {
2596          ## ASCII case-insensitive
2597          if ($self->{nc} == [
2598                undef,
2599                0x0059, # Y
2600                0x0053, # S
2601                0x0054, # T
2602                0x0045, # E
2603              ]->[length $self->{s_kwd}] or
2604              $self->{nc} == [
2605                undef,
2606                0x0079, # y
2607                0x0073, # s
2608                0x0074, # t
2609                0x0065, # e
2610              ]->[length $self->{s_kwd}]) {
2611            !!!cp (170);
2612            ## Stay in the state.
2613            $self->{s_kwd} .= chr $self->{nc};
2614            !!!next-input-character;
2615            redo A;
2616          } elsif ((length $self->{s_kwd}) == 5 and
2617                   ($self->{nc} == 0x004D or # M
2618                    $self->{nc} == 0x006D)) { # m
2619            !!!cp (171);
2620            $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2621            !!!next-input-character;
2622            redo A;
2623          } else {
2624            !!!cp (172);
2625            !!!parse-error (type => 'string after DOCTYPE name',
2626                            line => $self->{line_prev},
2627                            column => $self->{column_prev} + 1 - length $self->{s_kwd});
2628            $self->{ct}->{quirks} = 1;
2629    
2630            $self->{state} = BOGUS_DOCTYPE_STATE;
2631            ## Reconsume.
2632            redo A;
2633          }
2634        } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2635          if ($is_space->{$self->{nc}}) {
2636            !!!cp (181);
2637          ## Stay in the state          ## Stay in the state
2638          !!!next-input-character;          !!!next-input-character;
2639          redo A;          redo A;
2640        } elsif ($self->{next_input_character} eq 0x0022) { # "        } elsif ($self->{nc} eq 0x0022) { # "
2641          $self->{current_token}->{public_identifier} = ''; # DOCTYPE          !!!cp (182);
2642          $self->{state} = 'DOCTYPE public identifier (double-quoted)';          $self->{ct}->{pubid} = ''; # DOCTYPE
2643            $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;
2644          !!!next-input-character;          !!!next-input-character;
2645          redo A;          redo A;
2646        } elsif ($self->{next_input_character} eq 0x0027) { # '        } elsif ($self->{nc} eq 0x0027) { # '
2647          $self->{current_token}->{public_identifier} = ''; # DOCTYPE          !!!cp (183);
2648          $self->{state} = 'DOCTYPE public identifier (single-quoted)';          $self->{ct}->{pubid} = ''; # DOCTYPE
2649            $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;
2650          !!!next-input-character;          !!!next-input-character;
2651          redo A;          redo A;
2652        } elsif ($self->{next_input_character} eq 0x003E) { # >        } elsif ($self->{nc} eq 0x003E) { # >
2653            !!!cp (184);
2654          !!!parse-error (type => 'no PUBLIC literal');          !!!parse-error (type => 'no PUBLIC literal');
2655    
2656          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2657          !!!next-input-character;          !!!next-input-character;
2658    
2659          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2660          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2661    
2662          redo A;          redo A;
2663        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2664            !!!cp (185);
2665          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2666    
2667          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2668          ## reconsume          ## reconsume
2669    
2670          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2671          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2672    
2673          redo A;          redo A;
2674        } else {        } else {
2675            !!!cp (186);
2676          !!!parse-error (type => 'string after PUBLIC');          !!!parse-error (type => 'string after PUBLIC');
2677          $self->{state} = 'bogus DOCTYPE';          $self->{ct}->{quirks} = 1;
2678    
2679            $self->{state} = BOGUS_DOCTYPE_STATE;
2680          !!!next-input-character;          !!!next-input-character;
2681          redo A;          redo A;
2682        }        }
2683      } elsif ($self->{state} eq 'DOCTYPE public identifier (double-quoted)') {      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2684        if ($self->{next_input_character} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
2685          $self->{state} = 'after DOCTYPE public identifier';          !!!cp (187);
2686            $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2687          !!!next-input-character;          !!!next-input-character;
2688          redo A;          redo A;
2689        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == 0x003E) { # >
2690            !!!cp (188);
2691          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2692    
2693          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2694            !!!next-input-character;
2695    
2696            $self->{ct}->{quirks} = 1;
2697            !!!emit ($self->{ct}); # DOCTYPE
2698    
2699            redo A;
2700          } elsif ($self->{nc} == -1) {
2701            !!!cp (189);
2702            !!!parse-error (type => 'unclosed PUBLIC literal');
2703    
2704            $self->{state} = DATA_STATE;
2705          ## reconsume          ## reconsume
2706    
2707          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2708          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2709    
2710          redo A;          redo A;
2711        } else {        } else {
2712          $self->{current_token}->{public_identifier} # DOCTYPE          !!!cp (190);
2713              .= chr $self->{next_input_character};          $self->{ct}->{pubid} # DOCTYPE
2714                .= chr $self->{nc};
2715            $self->{read_until}->($self->{ct}->{pubid}, q[">],
2716                                  length $self->{ct}->{pubid});
2717    
2718          ## Stay in the state          ## Stay in the state
2719          !!!next-input-character;          !!!next-input-character;
2720          redo A;          redo A;
2721        }        }
2722      } elsif ($self->{state} eq 'DOCTYPE public identifier (single-quoted)') {      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {
2723        if ($self->{next_input_character} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
2724          $self->{state} = 'after DOCTYPE public identifier';          !!!cp (191);
2725            $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
2726            !!!next-input-character;
2727            redo A;
2728          } elsif ($self->{nc} == 0x003E) { # >
2729            !!!cp (192);
2730            !!!parse-error (type => 'unclosed PUBLIC literal');
2731    
2732            $self->{state} = DATA_STATE;
2733          !!!next-input-character;          !!!next-input-character;
2734    
2735            $self->{ct}->{quirks} = 1;
2736            !!!emit ($self->{ct}); # DOCTYPE
2737    
2738          redo A;          redo A;
2739        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2740            !!!cp (193);
2741          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2742    
2743          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2744          ## reconsume          ## reconsume
2745    
2746          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2747          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2748    
2749          redo A;          redo A;
2750        } else {        } else {
2751          $self->{current_token}->{public_identifier} # DOCTYPE          !!!cp (194);
2752              .= chr $self->{next_input_character};          $self->{ct}->{pubid} # DOCTYPE
2753                .= chr $self->{nc};
2754            $self->{read_until}->($self->{ct}->{pubid}, q['>],
2755                                  length $self->{ct}->{pubid});
2756    
2757          ## Stay in the state          ## Stay in the state
2758          !!!next-input-character;          !!!next-input-character;
2759          redo A;          redo A;
2760        }        }
2761      } elsif ($self->{state} eq 'after DOCTYPE public identifier') {      } elsif ($self->{state} == AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
2762        if ({        if ($is_space->{$self->{nc}}) {
2763              0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,          !!!cp (195);
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_input_character}}) {  
2764          ## Stay in the state          ## Stay in the state
2765          !!!next-input-character;          !!!next-input-character;
2766          redo A;          redo A;
2767        } elsif ($self->{next_input_character} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
2768          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          !!!cp (196);
2769          $self->{state} = 'DOCTYPE system identifier (double-quoted)';          $self->{ct}->{sysid} = ''; # DOCTYPE
2770            $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2771          !!!next-input-character;          !!!next-input-character;
2772          redo A;          redo A;
2773        } elsif ($self->{next_input_character} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
2774          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          !!!cp (197);
2775          $self->{state} = 'DOCTYPE system identifier (single-quoted)';          $self->{ct}->{sysid} = ''; # DOCTYPE
2776            $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2777          !!!next-input-character;          !!!next-input-character;
2778          redo A;          redo A;
2779        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2780          $self->{state} = 'data';          !!!cp (198);
2781            $self->{state} = DATA_STATE;
2782          !!!next-input-character;          !!!next-input-character;
2783    
2784          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2785    
2786          redo A;          redo A;
2787        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2788            !!!cp (199);
2789          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2790    
2791          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2792          ## recomsume          ## reconsume
2793    
2794          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2795          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2796    
2797          redo A;          redo A;
2798        } else {        } else {
2799            !!!cp (200);
2800          !!!parse-error (type => 'string after PUBLIC literal');          !!!parse-error (type => 'string after PUBLIC literal');
2801          $self->{state} = 'bogus DOCTYPE';          $self->{ct}->{quirks} = 1;
2802    
2803            $self->{state} = BOGUS_DOCTYPE_STATE;
2804          !!!next-input-character;          !!!next-input-character;
2805          redo A;          redo A;
2806        }        }
2807      } elsif ($self->{state} eq 'before DOCTYPE system identifier') {      } elsif ($self->{state} == BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2808        if ({        if ($is_space->{$self->{nc}}) {
2809              0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,          !!!cp (201);
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_input_character}}) {  
2810          ## Stay in the state          ## Stay in the state
2811          !!!next-input-character;          !!!next-input-character;
2812          redo A;          redo A;
2813        } elsif ($self->{next_input_character} == 0x0022) { # "        } elsif ($self->{nc} == 0x0022) { # "
2814          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          !!!cp (202);
2815          $self->{state} = 'DOCTYPE system identifier (double-quoted)';          $self->{ct}->{sysid} = ''; # DOCTYPE
2816            $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2817          !!!next-input-character;          !!!next-input-character;
2818          redo A;          redo A;
2819        } elsif ($self->{next_input_character} == 0x0027) { # '        } elsif ($self->{nc} == 0x0027) { # '
2820          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          !!!cp (203);
2821          $self->{state} = 'DOCTYPE system identifier (single-quoted)';          $self->{ct}->{sysid} = ''; # DOCTYPE
2822            $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2823          !!!next-input-character;          !!!next-input-character;
2824          redo A;          redo A;
2825        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2826            !!!cp (204);
2827          !!!parse-error (type => 'no SYSTEM literal');          !!!parse-error (type => 'no SYSTEM literal');
2828          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2829          !!!next-input-character;          !!!next-input-character;
2830    
2831          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2832          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2833    
2834          redo A;          redo A;
2835        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2836            !!!cp (205);
2837          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2838    
2839          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2840          ## recomsume          ## reconsume
2841    
2842          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2843          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2844    
2845          redo A;          redo A;
2846        } else {        } else {
2847          !!!parse-error (type => 'string after PUBLIC literal');          !!!cp (206);
2848          $self->{state} = 'bogus DOCTYPE';          !!!parse-error (type => 'string after SYSTEM');
2849            $self->{ct}->{quirks} = 1;
2850    
2851            $self->{state} = BOGUS_DOCTYPE_STATE;
2852          !!!next-input-character;          !!!next-input-character;
2853          redo A;          redo A;
2854        }        }
2855      } elsif ($self->{state} eq 'DOCTYPE system identifier (double-quoted)') {      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2856        if ($self->{next_input_character} == 0x0022) { # "        if ($self->{nc} == 0x0022) { # "
2857          $self->{state} = 'after DOCTYPE system identifier';          !!!cp (207);
2858            $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2859            !!!next-input-character;
2860            redo A;
2861          } elsif ($self->{nc} == 0x003E) { # >
2862            !!!cp (208);
2863            !!!parse-error (type => 'unclosed SYSTEM literal');
2864    
2865            $self->{state} = DATA_STATE;
2866          !!!next-input-character;          !!!next-input-character;
2867    
2868            $self->{ct}->{quirks} = 1;
2869            !!!emit ($self->{ct}); # DOCTYPE
2870    
2871          redo A;          redo A;
2872        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2873            !!!cp (209);
2874          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2875    
2876          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2877          ## reconsume          ## reconsume
2878    
2879          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2880          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2881    
2882          redo A;          redo A;
2883        } else {        } else {
2884          $self->{current_token}->{system_identifier} # DOCTYPE          !!!cp (210);
2885              .= chr $self->{next_input_character};          $self->{ct}->{sysid} # DOCTYPE
2886                .= chr $self->{nc};
2887            $self->{read_until}->($self->{ct}->{sysid}, q[">],
2888                                  length $self->{ct}->{sysid});
2889    
2890          ## Stay in the state          ## Stay in the state
2891          !!!next-input-character;          !!!next-input-character;
2892          redo A;          redo A;
2893        }        }
2894      } elsif ($self->{state} eq 'DOCTYPE system identifier (single-quoted)') {      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {
2895        if ($self->{next_input_character} == 0x0027) { # '        if ($self->{nc} == 0x0027) { # '
2896          $self->{state} = 'after DOCTYPE system identifier';          !!!cp (211);
2897            $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2898          !!!next-input-character;          !!!next-input-character;
2899          redo A;          redo A;
2900        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == 0x003E) { # >
2901            !!!cp (212);
2902          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2903    
2904          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2905            !!!next-input-character;
2906    
2907            $self->{ct}->{quirks} = 1;
2908            !!!emit ($self->{ct}); # DOCTYPE
2909    
2910            redo A;
2911          } elsif ($self->{nc} == -1) {
2912            !!!cp (213);
2913            !!!parse-error (type => 'unclosed SYSTEM literal');
2914    
2915            $self->{state} = DATA_STATE;
2916          ## reconsume          ## reconsume
2917    
2918          delete $self->{current_token}->{correct};          $self->{ct}->{quirks} = 1;
2919          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2920    
2921          redo A;          redo A;
2922        } else {        } else {
2923          $self->{current_token}->{system_identifier} # DOCTYPE          !!!cp (214);
2924              .= chr $self->{next_input_character};          $self->{ct}->{sysid} # DOCTYPE
2925                .= chr $self->{nc};
2926            $self->{read_until}->($self->{ct}->{sysid}, q['>],
2927                                  length $self->{ct}->{sysid});
2928    
2929          ## Stay in the state          ## Stay in the state
2930          !!!next-input-character;          !!!next-input-character;
2931          redo A;          redo A;
2932        }        }
2933      } elsif ($self->{state} eq 'after DOCTYPE system identifier') {      } elsif ($self->{state} == AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE) {
2934        if ({        if ($is_space->{$self->{nc}}) {
2935              0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,          !!!cp (215);
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_input_character}}) {  
2936          ## Stay in the state          ## Stay in the state
2937          !!!next-input-character;          !!!next-input-character;
2938          redo A;          redo A;
2939        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
2940          $self->{state} = 'data';          !!!cp (216);
2941            $self->{state} = DATA_STATE;
2942          !!!next-input-character;          !!!next-input-character;
2943    
2944          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{ct}); # DOCTYPE
2945    
2946          redo A;          redo A;
2947        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2948            !!!cp (217);
2949          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2950            $self->{state} = DATA_STATE;
2951            ## reconsume
2952    
2953          $self->{state} = 'data';          $self->{ct}->{quirks} = 1;
2954          ## recomsume          !!!emit ($self->{ct}); # DOCTYPE
   
         delete $self->{current_token}->{correct};  
         !!!emit ($self->{current_token}); # DOCTYPE  
2955    
2956          redo A;          redo A;
2957        } else {        } else {
2958            !!!cp (218);
2959          !!!parse-error (type => 'string after SYSTEM literal');          !!!parse-error (type => 'string after SYSTEM literal');
2960          $self->{state} = 'bogus DOCTYPE';          #$self->{ct}->{quirks} = 1;
2961    
2962            $self->{state} = BOGUS_DOCTYPE_STATE;
2963          !!!next-input-character;          !!!next-input-character;
2964          redo A;          redo A;
2965        }        }
2966      } elsif ($self->{state} eq 'bogus DOCTYPE') {      } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {
2967        if ($self->{next_input_character} == 0x003E) { # >        if ($self->{nc} == 0x003E) { # >
2968          $self->{state} = 'data';          !!!cp (219);
2969            $self->{state} = DATA_STATE;
2970          !!!next-input-character;          !!!next-input-character;
2971    
2972          delete $self->{current_token}->{correct};          !!!emit ($self->{ct}); # DOCTYPE
         !!!emit ($self->{current_token}); # DOCTYPE  
2973    
2974          redo A;          redo A;
2975        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{nc} == -1) {
2976            !!!cp (220);
2977          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2978          $self->{state} = 'data';          $self->{state} = DATA_STATE;
2979          ## reconsume          ## reconsume
2980    
2981          delete $self->{current_token}->{correct};          !!!emit ($self->{ct}); # DOCTYPE
         !!!emit ($self->{current_token}); # DOCTYPE  
2982    
2983          redo A;          redo A;
2984        } else {        } else {
2985            !!!cp (221);
2986            my $s = '';
2987            $self->{read_until}->($s, q[>], 0);
2988    
2989          ## Stay in the state          ## Stay in the state
2990          !!!next-input-character;          !!!next-input-character;
2991          redo A;          redo A;
2992        }        }
2993      } else {      } elsif ($self->{state} == CDATA_SECTION_STATE) {
2994        die "$0: $self->{state}: Unknown state";        ## NOTE: "CDATA section state" in the state is jointly implemented
2995      }        ## by three states, |CDATA_SECTION_STATE|, |CDATA_SECTION_MSE1_STATE|,
2996    } # A          ## and |CDATA_SECTION_MSE2_STATE|.
2997          
2998    die "$0: _get_next_token: unexpected case";        if ($self->{nc} == 0x005D) { # ]
2999  } # _get_next_token          !!!cp (221.1);
3000            $self->{state} = CDATA_SECTION_MSE1_STATE;
3001            !!!next-input-character;
3002            redo A;
3003          } elsif ($self->{nc} == -1) {
3004            $self->{state} = DATA_STATE;
3005            !!!next-input-character;
3006            if (length $self->{ct}->{data}) { # character
3007              !!!cp (221.2);
3008              !!!emit ($self->{ct}); # character
3009            } else {
3010              !!!cp (221.3);
3011              ## No token to emit. $self->{ct} is discarded.
3012            }        
3013            redo A;
3014          } else {
3015            !!!cp (221.4);
3016            $self->{ct}->{data} .= chr $self->{nc};
3017            $self->{read_until}->($self->{ct}->{data},
3018                                  q<]>,
3019                                  length $self->{ct}->{data});
3020    
3021  sub _tokenize_attempt_to_consume_an_entity ($) {          ## Stay in the state.
3022    my $self = shift;          !!!next-input-character;
3023            redo A;
3024          }
3025    
3026    if ({        ## ISSUE: "text tokens" in spec.
3027         0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, # HT, LF, VT, FF,      } elsif ($self->{state} == CDATA_SECTION_MSE1_STATE) {
3028         0x0020 => 1, 0x003C => 1, 0x0026 => 1, -1 => 1, # SP, <, & # 0x000D # CR        if ($self->{nc} == 0x005D) { # ]
3029        }->{$self->{next_input_character}}) {          !!!cp (221.5);
3030      ## Don't consume          $self->{state} = CDATA_SECTION_MSE2_STATE;
3031      ## No error          !!!next-input-character;
3032      return undef;          redo A;
3033    } elsif ($self->{next_input_character} == 0x0023) { # #        } else {
3034      !!!next-input-character;          !!!cp (221.6);
3035      if ($self->{next_input_character} == 0x0078 or # x          $self->{ct}->{data} .= ']';
3036          $self->{next_input_character} == 0x0058) { # X          $self->{state} = CDATA_SECTION_STATE;
3037        my $num;          ## Reconsume.
3038        X: {          redo A;
3039          my $x_char = $self->{next_input_character};        }
3040          !!!next-input-character;      } elsif ($self->{state} == CDATA_SECTION_MSE2_STATE) {
3041          if (0x0030 <= $self->{next_input_character} and        if ($self->{nc} == 0x003E) { # >
3042              $self->{next_input_character} <= 0x0039) { # 0..9          $self->{state} = DATA_STATE;
3043            $num ||= 0;          !!!next-input-character;
3044            $num *= 0x10;          if (length $self->{ct}->{data}) { # character
3045            $num += $self->{next_input_character} - 0x0030;            !!!cp (221.7);
3046            redo X;            !!!emit ($self->{ct}); # character
         } elsif (0x0061 <= $self->{next_input_character} and  
                  $self->{next_input_character} <= 0x0066) { # a..f  
           ## ISSUE: the spec says U+0078, which is apparently incorrect  
           $num ||= 0;  
           $num *= 0x10;  
           $num += $self->{next_input_character} - 0x0060 + 9;  
           redo X;  
         } elsif (0x0041 <= $self->{next_input_character} and  
                  $self->{next_input_character} <= 0x0046) { # A..F  
           ## ISSUE: the spec says U+0058, which is apparently incorrect  
           $num ||= 0;  
           $num *= 0x10;  
           $num += $self->{next_input_character} - 0x0040 + 9;  
           redo X;  
         } elsif (not defined $num) { # no hexadecimal digit  
           !!!parse-error (type => 'bare hcro');  
           $self->{next_input_character} = 0x0023; # #  
           !!!back-next-input-character ($x_char);  
           return undef;  
         } elsif ($self->{next_input_character} == 0x003B) { # ;  
           !!!next-input-character;  
3047          } else {          } else {
3048            !!!parse-error (type => 'no refc');            !!!cp (221.8);
3049              ## No token to emit. $self->{ct} is discarded.
3050          }          }
3051            redo A;
3052          ## TODO: check the definition for |a valid Unicode character|.        } elsif ($self->{nc} == 0x005D) { # ]
3053          ## <http://lists.whatwg.org/pipermail/whatwg-whatwg.org/2006-December/thread.html#8189>          !!!cp (221.9); # character
3054          if ($num > 1114111 or $num == 0) {          $self->{ct}->{data} .= ']'; ## Add first "]" of "]]]".
3055            $num = 0xFFFD; # REPLACEMENT CHARACTER          ## Stay in the state.
3056            ## ISSUE: Why this is not an error?          !!!next-input-character;
3057          } elsif (0x80 <= $num and $num <= 0x9F) {          redo A;
3058            !!!parse-error (type => sprintf 'c1 entity:U+%04X', $num);        } else {
3059            $num = $c1_entity_char->{$num};          !!!cp (221.11);
3060          }          $self->{ct}->{data} .= ']]'; # character
3061            $self->{state} = CDATA_SECTION_STATE;
3062          return {type => 'character', data => chr $num};          ## Reconsume.
3063        } # X          redo A;
3064      } elsif (0x0030 <= $self->{next_input_character} and        }
3065               $self->{next_input_character} <= 0x0039) { # 0..9      } elsif ($self->{state} == ENTITY_STATE) {
3066        my $code = $self->{next_input_character} - 0x0030;        if ($is_space->{$self->{nc}} or
3067        !!!next-input-character;            {
3068                      0x003C => 1, 0x0026 => 1, -1 => 1, # <, &
3069        while (0x0030 <= $self->{next_input_character} and              $self->{entity_add} => 1,
3070                  $self->{next_input_character} <= 0x0039) { # 0..9            }->{$self->{nc}}) {
3071          $code *= 10;          !!!cp (1001);
3072          $code += $self->{next_input_character} - 0x0030;          ## Don't consume
3073                    ## No error
3074            ## Return nothing.
3075            #
3076          } elsif ($self->{nc} == 0x0023) { # #
3077            !!!cp (999);
3078            $self->{state} = ENTITY_HASH_STATE;
3079            $self->{s_kwd} = '#';
3080            !!!next-input-character;
3081            redo A;
3082          } elsif ((0x0041 <= $self->{nc} and
3083                    $self->{nc} <= 0x005A) or # A..Z
3084                   (0x0061 <= $self->{nc} and
3085                    $self->{nc} <= 0x007A)) { # a..z
3086            !!!cp (998);
3087            require Whatpm::_NamedEntityList;
3088            $self->{state} = ENTITY_NAME_STATE;
3089            $self->{s_kwd} = chr $self->{nc};
3090            $self->{entity__value} = $self->{s_kwd};
3091            $self->{entity__match} = 0;
3092          !!!next-input-character;          !!!next-input-character;
3093            redo A;
3094          } else {
3095            !!!cp (1027);
3096            !!!parse-error (type => 'bare ero');
3097            ## Return nothing.
3098            #
3099        }        }
3100    
3101        if ($self->{next_input_character} == 0x003B) { # ;        ## NOTE: No character is consumed by the "consume a character
3102          ## reference" algorithm.  In other word, there is an "&" character
3103          ## that does not introduce a character reference, which would be
3104          ## appended to the parent element or the attribute value in later
3105          ## process of the tokenizer.
3106    
3107          if ($self->{prev_state} == DATA_STATE) {
3108            !!!cp (997);
3109            $self->{state} = $self->{prev_state};
3110            ## Reconsume.
3111            !!!emit ({type => CHARACTER_TOKEN, data => '&',
3112                      line => $self->{line_prev},
3113                      column => $self->{column_prev},
3114                     });
3115            redo A;
3116          } else {
3117            !!!cp (996);
3118            $self->{ca}->{value} .= '&';
3119            $self->{state} = $self->{prev_state};
3120            ## Reconsume.
3121            redo A;
3122          }
3123        } elsif ($self->{state} == ENTITY_HASH_STATE) {
3124          if ($self->{nc} == 0x0078 or # x
3125              $self->{nc} == 0x0058) { # X
3126            !!!cp (995);
3127            $self->{state} = HEXREF_X_STATE;
3128            $self->{s_kwd} .= chr $self->{nc};
3129            !!!next-input-character;
3130            redo A;
3131          } elsif (0x0030 <= $self->{nc} and
3132                   $self->{nc} <= 0x0039) { # 0..9
3133            !!!cp (994);
3134            $self->{state} = NCR_NUM_STATE;
3135            $self->{s_kwd} = $self->{nc} - 0x0030;
3136          !!!next-input-character;          !!!next-input-character;
3137            redo A;
3138        } else {        } else {
3139            !!!parse-error (type => 'bare nero',
3140                            line => $self->{line_prev},
3141                            column => $self->{column_prev} - 1);
3142    
3143            ## NOTE: According to the spec algorithm, nothing is returned,
3144            ## and then "&#" is appended to the parent element or the attribute
3145            ## value in the later processing.
3146    
3147            if ($self->{prev_state} == DATA_STATE) {
3148              !!!cp (1019);
3149              $self->{state} = $self->{prev_state};
3150              ## Reconsume.
3151              !!!emit ({type => CHARACTER_TOKEN,
3152                        data => '&#',
3153                        line => $self->{line_prev},
3154                        column => $self->{column_prev} - 1,
3155                       });
3156              redo A;
3157            } else {
3158              !!!cp (993);
3159              $self->{ca}->{value} .= '&#';
3160              $self->{state} = $self->{prev_state};
3161              ## Reconsume.
3162              redo A;
3163            }
3164          }
3165        } elsif ($self->{state} == NCR_NUM_STATE) {
3166          if (0x0030 <= $self->{nc} and
3167              $self->{nc} <= 0x0039) { # 0..9
3168            !!!cp (1012);
3169            $self->{s_kwd} *= 10;
3170            $self->{s_kwd} += $self->{nc} - 0x0030;
3171            
3172            ## Stay in the state.
3173            !!!next-input-character;
3174            redo A;
3175          } elsif ($self->{nc} == 0x003B) { # ;
3176            !!!cp (1013);
3177            !!!next-input-character;
3178            #
3179          } else {
3180            !!!cp (1014);
3181          !!!parse-error (type => 'no refc');          !!!parse-error (type => 'no refc');
3182            ## Reconsume.
3183            #
3184        }        }
3185    
3186        ## TODO: check the definition for |a valid Unicode character|.        my $code = $self->{s_kwd};
3187        if ($code > 1114111 or $code == 0) {        my $l = $self->{line_prev};
3188          $code = 0xFFFD; # REPLACEMENT CHARACTER        my $c = $self->{column_prev};
3189          ## ISSUE: Why this is not an error?        if ($charref_map->{$code}) {
3190        } elsif (0x80 <= $code and $code <= 0x9F) {          !!!cp (1015);
3191          !!!parse-error (type => sprintf 'c1 entity:U+%04X', $code);          !!!parse-error (type => 'invalid character reference',
3192          $code = $c1_entity_char->{$code};                          text => (sprintf 'U+%04X', $code),
3193                            line => $l, column => $c);
3194            $code = $charref_map->{$code};
3195          } elsif ($code > 0x10FFFF) {
3196            !!!cp (1016);
3197            !!!parse-error (type => 'invalid character reference',
3198                            text => (sprintf 'U-%08X', $code),
3199                            line => $l, column => $c);
3200            $code = 0xFFFD;
3201          }
3202    
3203          if ($self->{prev_state} == DATA_STATE) {
3204            !!!cp (992);
3205            $self->{state} = $self->{prev_state};
3206            ## Reconsume.
3207            !!!emit ({type => CHARACTER_TOKEN, data => chr $code,
3208                      line => $l, column => $c,
3209                     });
3210            redo A;
3211          } else {
3212            !!!cp (991);
3213            $self->{ca}->{value} .= chr $code;
3214            $self->{ca}->{has_reference} = 1;
3215            $self->{state} = $self->{prev_state};
3216            ## Reconsume.
3217            redo A;
3218          }
3219        } elsif ($self->{state} == HEXREF_X_STATE) {
3220          if ((0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) or
3221              (0x0041 <= $self->{nc} and $self->{nc} <= 0x0046) or
3222              (0x0061 <= $self->{nc} and $self->{nc} <= 0x0066)) {
3223            # 0..9, A..F, a..f
3224            !!!cp (990);
3225            $self->{state} = HEXREF_HEX_STATE;
3226            $self->{s_kwd} = 0;
3227            ## Reconsume.
3228            redo A;
3229          } else {
3230            !!!parse-error (type => 'bare hcro',
3231                            line => $self->{line_prev},
3232                            column => $self->{column_prev} - 2);
3233    
3234            ## NOTE: According to the spec algorithm, nothing is returned,
3235            ## and then "&#" followed by "X" or "x" is appended to the parent
3236            ## element or the attribute value in the later processing.
3237    
3238            if ($self->{prev_state} == DATA_STATE) {
3239              !!!cp (1005);
3240              $self->{state} = $self->{prev_state};
3241              ## Reconsume.
3242              !!!emit ({type => CHARACTER_TOKEN,
3243                        data => '&' . $self->{s_kwd},
3244                        line => $self->{line_prev},
3245                        column => $self->{column_prev} - length $self->{s_kwd},
3246                       });
3247              redo A;
3248            } else {
3249              !!!cp (989);
3250              $self->{ca}->{value} .= '&' . $self->{s_kwd};
3251              $self->{state} = $self->{prev_state};
3252              ## Reconsume.
3253              redo A;
3254            }
3255        }        }
3256              } elsif ($self->{state} == HEXREF_HEX_STATE) {
3257        return {type => 'character', data => chr $code};        if (0x0030 <= $self->{nc} and $self->{nc} <= 0x0039) {
3258      } else {          # 0..9
3259        !!!parse-error (type => 'bare nero');          !!!cp (1002);
3260        !!!back-next-input-character ($self->{next_input_character});          $self->{s_kwd} *= 0x10;
3261        $self->{next_input_character} = 0x0023; # #          $self->{s_kwd} += $self->{nc} - 0x0030;
3262        return undef;          ## Stay in the state.
3263      }          !!!next-input-character;
3264    } elsif ((0x0041 <= $self->{next_input_character} and          redo A;
3265              $self->{next_input_character} <= 0x005A) or        } elsif (0x0061 <= $self->{nc} and
3266             (0x0061 <= $self->{next_input_character} and                 $self->{nc} <= 0x0066) { # a..f
3267              $self->{next_input_character} <= 0x007A)) {          !!!cp (1003);
3268      my $entity_name = chr $self->{next_input_character};          $self->{s_kwd} *= 0x10;
3269      !!!next-input-character;          $self->{s_kwd} += $self->{nc} - 0x0060 + 9;
3270            ## Stay in the state.
3271      my $value = $entity_name;          !!!next-input-character;
3272      my $match;          redo A;
3273      require Whatpm::_NamedEntityList;        } elsif (0x0041 <= $self->{nc} and
3274      our $EntityChar;                 $self->{nc} <= 0x0046) { # A..F
3275            !!!cp (1004);
3276      while (length $entity_name < 10 and          $self->{s_kwd} *= 0x10;
3277             ## NOTE: Some number greater than the maximum length of entity name          $self->{s_kwd} += $self->{nc} - 0x0040 + 9;
3278             ((0x0041 <= $self->{next_input_character} and # a          ## Stay in the state.
3279               $self->{next_input_character} <= 0x005A) or # x          !!!next-input-character;
3280              (0x0061 <= $self->{next_input_character} and # a          redo A;
3281               $self->{next_input_character} <= 0x007A) or # z        } elsif ($self->{nc} == 0x003B) { # ;
3282              (0x0030 <= $self->{next_input_character} and # 0          !!!cp (1006);
3283               $self->{next_input_character} <= 0x0039) or # 9          !!!next-input-character;
3284              $self->{next_input_character} == 0x003B)) { # ;          #
3285        $entity_name .= chr $self->{next_input_character};        } else {
3286        if (defined $EntityChar->{$entity_name}) {          !!!cp (1007);
3287          $value = $EntityChar->{$entity_name};          !!!parse-error (type => 'no refc',
3288          if ($self->{next_input_character} == 0x003B) { # ;                          line => $self->{line},
3289            $match = 1;                          column => $self->{column});
3290            ## Reconsume.
3291            #
3292          }
3293    
3294          my $code = $self->{s_kwd};
3295          my $l = $self->{line_prev};
3296          my $c = $self->{column_prev};
3297          if ($charref_map->{$code}) {
3298            !!!cp (1008);
3299            !!!parse-error (type => 'invalid character reference',
3300                            text => (sprintf 'U+%04X', $code),
3301                            line => $l, column => $c);
3302            $code = $charref_map->{$code};
3303          } elsif ($code > 0x10FFFF) {
3304            !!!cp (1009);
3305            !!!parse-error (type => 'invalid character reference',
3306                            text => (sprintf 'U-%08X', $code),
3307                            line => $l, column => $c);
3308            $code = 0xFFFD;
3309          }
3310    
3311          if ($self->{prev_state} == DATA_STATE) {
3312            !!!cp (988);
3313            $self->{state} = $self->{prev_state};
3314            ## Reconsume.
3315            !!!emit ({type => CHARACTER_TOKEN, data => chr $code,
3316                      line => $l, column => $c,
3317                     });
3318            redo A;
3319          } else {
3320            !!!cp (987);
3321            $self->{ca}->{value} .= chr $code;
3322            $self->{ca}->{has_reference} = 1;
3323            $self->{state} = $self->{prev_state};
3324            ## Reconsume.
3325            redo A;
3326          }
3327        } elsif ($self->{state} == ENTITY_NAME_STATE) {
3328          if (length $self->{s_kwd} < 30 and
3329              ## NOTE: Some number greater than the maximum length of entity name
3330              ((0x0041 <= $self->{nc} and # a
3331                $self->{nc} <= 0x005A) or # x
3332               (0x0061 <= $self->{nc} and # a
3333                $self->{nc} <= 0x007A) or # z
3334               (0x0030 <= $self->{nc} and # 0
3335                $self->{nc} <= 0x0039) or # 9
3336               $self->{nc} == 0x003B)) { # ;
3337            our $EntityChar;
3338            $self->{s_kwd} .= chr $self->{nc};
3339            if (defined $EntityChar->{$self->{s_kwd}}) {
3340              if ($self->{nc} == 0x003B) { # ;
3341                !!!cp (1020);
3342                $self->{entity__value} = $EntityChar->{$self->{s_kwd}};
3343                $self->{entity__match} = 1;
3344                !!!next-input-character;
3345                #
3346              } else {
3347                !!!cp (1021);
3348                $self->{entity__value} = $EntityChar->{$self->{s_kwd}};
3349                $self->{entity__match} = -1;
3350                ## Stay in the state.
3351                !!!next-input-character;
3352                redo A;
3353              }
3354            } else {
3355              !!!cp (1022);
3356              $self->{entity__value} .= chr $self->{nc};
3357              $self->{entity__match} *= 2;
3358              ## Stay in the state.
3359            !!!next-input-character;            !!!next-input-character;
3360            last;            redo A;
3361            }
3362          }
3363    
3364          my $data;
3365          my $has_ref;
3366          if ($self->{entity__match} > 0) {
3367            !!!cp (1023);
3368            $data = $self->{entity__value};
3369            $has_ref = 1;
3370            #
3371          } elsif ($self->{entity__match} < 0) {
3372            !!!parse-error (type => 'no refc');
3373            if ($self->{prev_state} != DATA_STATE and # in attribute
3374                $self->{entity__match} < -1) {
3375              !!!cp (1024);
3376              $data = '&' . $self->{s_kwd};
3377              #
3378          } else {          } else {
3379            $match = -1;            !!!cp (1025);
3380              $data = $self->{entity__value};
3381              $has_ref = 1;
3382              #
3383          }          }
3384        } else {        } else {
3385          $value .= chr $self->{next_input_character};          !!!cp (1026);
3386            !!!parse-error (type => 'bare ero',
3387                            line => $self->{line_prev},
3388                            column => $self->{column_prev} - length $self->{s_kwd});
3389            $data = '&' . $self->{s_kwd};
3390            #
3391          }
3392      
3393          ## NOTE: In these cases, when a character reference is found,
3394          ## it is consumed and a character token is returned, or, otherwise,
3395          ## nothing is consumed and returned, according to the spec algorithm.
3396          ## In this implementation, anything that has been examined by the
3397          ## tokenizer is appended to the parent element or the attribute value
3398          ## as string, either literal string when no character reference or
3399          ## entity-replaced string otherwise, in this stage, since any characters
3400          ## that would not be consumed are appended in the data state or in an
3401          ## appropriate attribute value state anyway.
3402    
3403          if ($self->{prev_state} == DATA_STATE) {
3404            !!!cp (986);
3405            $self->{state} = $self->{prev_state};
3406            ## Reconsume.
3407            !!!emit ({type => CHARACTER_TOKEN,
3408                      data => $data,
3409                      line => $self->{line_prev},
3410                      column => $self->{column_prev} + 1 - length $self->{s_kwd},
3411                     });
3412            redo A;
3413          } else {
3414            !!!cp (985);
3415            $self->{ca}->{value} .= $data;
3416            $self->{ca}->{has_reference} = 1 if $has_ref;
3417            $self->{state} = $self->{prev_state};
3418            ## Reconsume.
3419            redo A;
3420        }        }
       !!!next-input-character;  
     }  
       
     if ($match > 0) {  
       return {type => 'character', data => $value};  
     } elsif ($match < 0) {  
       !!!parse-error (type => 'refc');  
       return {type => 'character', data => $value};  
3421      } else {      } else {
3422        !!!parse-error (type => 'bare ero');        die "$0: $self->{state}: Unknown state";
       ## NOTE: No characters are consumed in the spec.  
       !!!back-token ({type => 'character', data => $value});  
       return undef;  
3423      }      }
3424    } else {    } # A  
3425      ## no characters are consumed  
3426      !!!parse-error (type => 'bare ero');    die "$0: _get_next_token: unexpected case";
3427      return undef;  } # _get_next_token
   }  
 } # _tokenize_attempt_to_consume_an_entity  
3428    
3429  sub _initialize_tree_constructor ($) {  sub _initialize_tree_constructor ($) {
3430    my $self = shift;    my $self = shift;
# Line 1728  sub _initialize_tree_constructor ($) { Line 3433  sub _initialize_tree_constructor ($) {
3433    ## TODO: Turn mutation events off # MUST    ## TODO: Turn mutation events off # MUST
3434    ## TODO: Turn loose Document option (manakai extension) on    ## TODO: Turn loose Document option (manakai extension) on
3435    $self->{document}->manakai_is_html (1); # MUST    $self->{document}->manakai_is_html (1); # MUST
3436      $self->{document}->set_user_data (manakai_source_line => 1);
3437      $self->{document}->set_user_data (manakai_source_column => 1);
3438  } # _initialize_tree_constructor  } # _initialize_tree_constructor
3439    
3440  sub _terminate_tree_constructor ($) {  sub _terminate_tree_constructor ($) {
# Line 1754  sub _construct_tree ($) { Line 3461  sub _construct_tree ($) {
3461        
3462    !!!next-token;    !!!next-token;
3463    
   $self->{insertion_mode} = 'before head';  
3464    undef $self->{form_element};    undef $self->{form_element};
3465    undef $self->{head_element};    undef $self->{head_element};
3466    $self->{open_elements} = [];    $self->{open_elements} = [];
3467    undef $self->{inner_html_node};    undef $self->{inner_html_node};
3468    
3469      ## NOTE: The "initial" insertion mode.
3470    $self->_tree_construction_initial; # MUST    $self->_tree_construction_initial; # MUST
3471    
3472      ## NOTE: The "before html" insertion mode.
3473    $self->_tree_construction_root_element;    $self->_tree_construction_root_element;
3474      $self->{insertion_mode} = BEFORE_HEAD_IM;
3475    
3476      ## NOTE: The "before head" insertion mode and so on.
3477    $self->_tree_construction_main;    $self->_tree_construction_main;
3478  } # _construct_tree  } # _construct_tree
3479    
3480  sub _tree_construction_initial ($) {  sub _tree_construction_initial ($) {
3481    my $self = shift;    my $self = shift;
3482    
3483      ## NOTE: "initial" insertion mode
3484    
3485    INITIAL: {    INITIAL: {
3486      if ($token->{type} eq 'DOCTYPE') {      if ($token->{type} == DOCTYPE_TOKEN) {
3487        ## NOTE: Conformance checkers MAY, instead of reporting "not HTML5"        ## NOTE: Conformance checkers MAY, instead of reporting "not HTML5"
3488        ## error, switch to a conformance checking mode for another        ## error, switch to a conformance checking mode for another
3489        ## language.        ## language.
3490        my $doctype_name = $token->{name};        my $doctype_name = $token->{name};
3491        $doctype_name = '' unless defined $doctype_name;        $doctype_name = '' unless defined $doctype_name;
3492        $doctype_name =~ tr/a-z/A-Z/;        $doctype_name =~ tr/a-z/A-Z/; # ASCII case-insensitive
3493        if (not defined $token->{name} or # <!DOCTYPE>        if (not defined $token->{name} or # <!DOCTYPE>
3494            defined $token->{public_identifier} or            defined $token->{sysid}) {
3495            defined $token->{system_identifier}) {          !!!cp ('t1');
3496          !!!parse-error (type => 'not HTML5');          !!!parse-error (type => 'not HTML5', token => $token);
3497        } elsif ($doctype_name ne 'HTML') {        } elsif ($doctype_name ne 'HTML') {
3498          ## ISSUE: ASCII case-insensitive? (in fact it does not matter)          !!!cp ('t2');
3499          !!!parse-error (type => 'not HTML5');          !!!parse-error (type => 'not HTML5', token => $token);
3500          } elsif (defined $token->{pubid}) {
3501            if ($token->{pubid} eq 'XSLT-compat') {
3502              !!!cp ('t1.2');
3503              !!!parse-error (type => 'XSLT-compat', token => $token,
3504                              level => $self->{level}->{should});
3505            } else {
3506              !!!parse-error (type => 'not HTML5', token => $token);
3507            }
3508          } else {
3509            !!!cp ('t3');
3510            #
3511        }        }
3512                
3513        my $doctype = $self->{document}->create_document_type_definition        my $doctype = $self->{document}->create_document_type_definition
3514          ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?          ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?
3515        $doctype->public_id ($token->{public_identifier})        ## NOTE: Default value for both |public_id| and |system_id| attributes
3516            if defined $token->{public_identifier};        ## are empty strings, so that we don't set any value in missing cases.
3517        $doctype->system_id ($token->{system_identifier})        $doctype->public_id ($token->{pubid}) if defined $token->{pubid};
3518            if defined $token->{system_identifier};        $doctype->system_id ($token->{sysid}) if defined $token->{sysid};
3519        ## NOTE: Other DocumentType attributes are null or empty lists.        ## NOTE: Other DocumentType attributes are null or empty lists.
3520        ## ISSUE: internalSubset = null??        ## ISSUE: internalSubset = null??
3521        $self->{document}->append_child ($doctype);        $self->{document}->append_child ($doctype);
3522                
3523        if (not $token->{correct} or $doctype_name ne 'HTML') {        if ($token->{quirks} or $doctype_name ne 'HTML') {
3524            !!!cp ('t4');
3525          $self->{document}->manakai_compat_mode ('quirks');          $self->{document}->manakai_compat_mode ('quirks');
3526        } elsif (defined $token->{public_identifier}) {        } elsif (defined $token->{pubid}) {
3527          my $pubid = $token->{public_identifier};          my $pubid = $token->{pubid};
3528          $pubid =~ tr/a-z/A-z/;          $pubid =~ tr/a-z/A-z/;
3529          if ({          my $prefix = [
3530            "+//SILMARIL//DTD HTML PRO V0R11 19970101//EN" => 1,            "+//SILMARIL//DTD HTML PRO V0R11 19970101//",
3531            "-//ADVASOFT LTD//DTD HTML 3.0 ASWEDIT + EXTENSIONS//EN" => 1,            "-//ADVASOFT LTD//DTD HTML 3.0 ASWEDIT + EXTENSIONS//",
3532            "-//AS//DTD HTML 3.0 ASWEDIT + EXTENSIONS//EN" => 1,            "-//AS//DTD HTML 3.0 ASWEDIT + EXTENSIONS//",
3533            "-//IETF//DTD HTML 2.0 LEVEL 1//EN" => 1,            "-//IETF//DTD HTML 2.0 LEVEL 1//",
3534            "-//IETF//DTD HTML 2.0 LEVEL 2//EN" => 1,            "-//IETF//DTD HTML 2.0 LEVEL 2//",
3535            "-//IETF//DTD HTML 2.0 STRICT LEVEL 1//EN" => 1,            "-//IETF//DTD HTML 2.0 STRICT LEVEL 1//",
3536            "-//IETF//DTD HTML 2.0 STRICT LEVEL 2//EN" => 1,            "-//IETF//DTD HTML 2.0 STRICT LEVEL 2//",
3537            "-//IETF//DTD HTML 2.0 STRICT//EN" => 1,            "-//IETF//DTD HTML 2.0 STRICT//",
3538            "-//IETF//DTD HTML 2.0//EN" => 1,            "-//IETF//DTD HTML 2.0//",
3539            "-//IETF//DTD HTML 2.1E//EN" => 1,            "-//IETF//DTD HTML 2.1E//",
3540            "-//IETF//DTD HTML 3.0//EN" => 1,            "-//IETF//DTD HTML 3.0//",
3541            "-//IETF//DTD HTML 3.0//EN//" => 1,            "-//IETF//DTD HTML 3.2 FINAL//",
3542            "-//IETF//DTD HTML 3.2 FINAL//EN" => 1,            "-//IETF//DTD HTML 3.2//",
3543            "-//IETF//DTD HTML 3.2//EN" => 1,            "-//IETF//DTD HTML 3//",
3544            "-//IETF//DTD HTML 3//EN" => 1,            "-//IETF//DTD HTML LEVEL 0//",
3545            "-//IETF//DTD HTML LEVEL 0//EN" => 1,            "-//IETF//DTD HTML LEVEL 1//",
3546            "-//IETF//DTD HTML LEVEL 0//EN//2.0" => 1,            "-//IETF//DTD HTML LEVEL 2//",
3547            "-//IETF//DTD HTML LEVEL 1//EN" => 1,            "-//IETF//DTD HTML LEVEL 3//",
3548            "-//IETF//DTD HTML LEVEL 1//EN//2.0" => 1,            "-//IETF//DTD HTML STRICT LEVEL 0//",
3549            "-//IETF//DTD HTML LEVEL 2//EN" => 1,            "-//IETF//DTD HTML STRICT LEVEL 1//",
3550            "-//IETF//DTD HTML LEVEL 2//EN//2.0" => 1,            "-//IETF//DTD HTML STRICT LEVEL 2//",
3551            "-//IETF//DTD HTML LEVEL 3//EN" => 1,            "-//IETF//DTD HTML STRICT LEVEL 3//",
3552            "-//IETF//DTD HTML LEVEL 3//EN//3.0" => 1,            "-//IETF//DTD HTML STRICT//",
3553            "-//IETF//DTD HTML STRICT LEVEL 0//EN" => 1,            "-//IETF//DTD HTML//",
3554            "-//IETF//DTD HTML STRICT LEVEL 0//EN//2.0" => 1,            "-//METRIUS//DTD METRIUS PRESENTATIONAL//",
3555            "-//IETF//DTD HTML STRICT LEVEL 1//EN" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML STRICT//",
3556            "-//IETF//DTD HTML STRICT LEVEL 1//EN//2.0" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML//",
3557            "-//IETF//DTD HTML STRICT LEVEL 2//EN" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 TABLES//",
3558            "-//IETF//DTD HTML STRICT LEVEL 2//EN//2.0" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML STRICT//",
3559            "-//IETF//DTD HTML STRICT LEVEL 3//EN" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML//",
3560            "-//IETF//DTD HTML STRICT LEVEL 3//EN//3.0" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 TABLES//",
3561            "-//IETF//DTD HTML STRICT//EN" => 1,            "-//NETSCAPE COMM. CORP.//DTD HTML//",
3562            "-//IETF//DTD HTML STRICT//EN//2.0" => 1,            "-//NETSCAPE COMM. CORP.//DTD STRICT HTML//",
3563            "-//IETF//DTD HTML STRICT//EN//3.0" => 1,            "-//O'REILLY AND ASSOCIATES//DTD HTML 2.0//",
3564            "-//IETF//DTD HTML//EN" => 1,            "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED 1.0//",
3565            "-//IETF//DTD HTML//EN//2.0" => 1,            "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED RELAXED 1.0//",
3566            "-//IETF//DTD HTML//EN//3.0" => 1,            "-//SOFTQUAD SOFTWARE//DTD HOTMETAL PRO 6.0::19990601::EXTENSIONS TO HTML 4.0//",
3567            "-//METRIUS//DTD METRIUS PRESENTATIONAL//EN" => 1,            "-//SOFTQUAD//DTD HOTMETAL PRO 4.0::19971010::EXTENSIONS TO HTML 4.0//",
3568            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML STRICT//EN" => 1,            "-//SPYGLASS//DTD HTML 2.0 EXTENDED//",
3569            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML//EN" => 1,            "-//SQ//DTD HTML 2.0 HOTMETAL + EXTENSIONS//",
3570            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 TABLES//EN" => 1,            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA HTML//",
3571            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML STRICT//EN" => 1,            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA STRICT HTML//",
3572            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML//EN" => 1,            "-//W3C//DTD HTML 3 1995-03-24//",
3573            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 TABLES//EN" => 1,            "-//W3C//DTD HTML 3.2 DRAFT//",
3574            "-//NETSCAPE COMM. CORP.//DTD HTML//EN" => 1,            "-//W3C//DTD HTML 3.2 FINAL//",
3575            "-//NETSCAPE COMM. CORP.//DTD STRICT HTML//EN" => 1,            "-//W3C//DTD HTML 3.2//",
3576            "-//O'REILLY AND ASSOCIATES//DTD HTML 2.0//EN" => 1,            "-//W3C//DTD HTML 3.2S DRAFT//",
3577            "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED 1.0//EN" => 1,            "-//W3C//DTD HTML 4.0 FRAMESET//",
3578            "-//SPYGLASS//DTD HTML 2.0 EXTENDED//EN" => 1,            "-//W3C//DTD HTML 4.0 TRANSITIONAL//",
3579            "-//SQ//DTD HTML 2.0 HOTMETAL + EXTENSIONS//EN" => 1,            "-//W3C//DTD HTML EXPERIMETNAL 19960712//",
3580            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA HTML//EN" => 1,            "-//W3C//DTD HTML EXPERIMENTAL 970421//",
3581            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA STRICT HTML//EN" => 1,            "-//W3C//DTD W3 HTML//",
3582            "-//W3C//DTD HTML 3 1995-03-24//EN" => 1,            "-//W3O//DTD W3 HTML 3.0//",
3583            "-//W3C//DTD HTML 3.2 DRAFT//EN" => 1,            "-//WEBTECHS//DTD MOZILLA HTML 2.0//",
3584            "-//W3C//DTD HTML 3.2 FINAL//EN" => 1,            "-//WEBTECHS//DTD MOZILLA HTML//",
3585            "-//W3C//DTD HTML 3.2//EN" => 1,          ]; # $prefix
3586            "-//W3C//DTD HTML 3.2S DRAFT//EN" => 1,          my $match;
3587            "-//W3C//DTD HTML 4.0 FRAMESET//EN" => 1,          for (@$prefix) {
3588            "-//W3C//DTD HTML 4.0 TRANSITIONAL//EN" => 1,            if (substr ($prefix, 0, length $_) eq $_) {
3589            "-//W3C//DTD HTML EXPERIMETNAL 19960712//EN" => 1,              $match = 1;
3590            "-//W3C//DTD HTML EXPERIMENTAL 970421//EN" => 1,              last;
3591            "-//W3C//DTD W3 HTML//EN" => 1,            }
3592            "-//W3O//DTD W3 HTML 3.0//EN" => 1,          }
3593            "-//W3O//DTD W3 HTML 3.0//EN//" => 1,          if ($match or
3594            "-//W3O//DTD W3 HTML STRICT 3.0//EN//" => 1,              $pubid eq "-//W3O//DTD W3 HTML STRICT 3.0//EN//" or
3595            "-//WEBTECHS//DTD MOZILLA HTML 2.0//EN" => 1,              $pubid eq "-/W3C/DTD HTML 4.0 TRANSITIONAL/EN" or
3596            "-//WEBTECHS//DTD MOZILLA HTML//EN" => 1,              $pubid eq "HTML") {
3597            "-/W3C/DTD HTML 4.0 TRANSITIONAL/EN" => 1,            !!!cp ('t5');
           "HTML" => 1,  
         }->{$pubid}) {  
3598            $self->{document}->manakai_compat_mode ('quirks');            $self->{document}->manakai_compat_mode ('quirks');
3599          } elsif ($pubid eq "-//W3C//DTD HTML 4.01 FRAMESET//EN" or          } elsif ($pubid =~ m[^-//W3C//DTD HTML 4.01 FRAMESET//] or
3600                   $pubid eq "-//W3C//DTD HTML 4.01 TRANSITIONAL//EN") {                   $pubid =~ m[^-//W3C//DTD HTML 4.01 TRANSITIONAL//]) {
3601            if (defined $token->{system_identifier}) {            if (defined $token->{sysid}) {
3602                !!!cp ('t6');
3603              $self->{document}->manakai_compat_mode ('quirks');              $self->{document}->manakai_compat_mode ('quirks');
3604            } else {            } else {
3605                !!!cp ('t7');
3606              $self->{document}->manakai_compat_mode ('limited quirks');              $self->{document}->manakai_compat_mode ('limited quirks');
3607            }            }
3608          } elsif ($pubid eq "-//W3C//DTD XHTML 1.0 Frameset//EN" or          } elsif ($pubid =~ m[^-//W3C//DTD XHTML 1.0 FRAMESET//] or
3609                   $pubid eq "-//W3C//DTD XHTML 1.0 Transitional//EN") {                   $pubid =~ m[^-//W3C//DTD XHTML 1.0 TRANSITIONAL//]) {
3610              !!!cp ('t8');
3611            $self->{document}->manakai_compat_mode ('limited quirks');            $self->{document}->manakai_compat_mode ('limited quirks');
3612            } else {
3613              !!!cp ('t9');
3614          }          }
3615          } else {
3616            !!!cp ('t10');
3617        }        }
3618        if (defined $token->{system_identifier}) {        if (defined $token->{sysid}) {
3619          my $sysid = $token->{system_identifier};          my $sysid = $token->{sysid};
3620          $sysid =~ tr/A-Z/a-z/;          $sysid =~ tr/A-Z/a-z/;
3621          if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {          if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {
3622              ## NOTE: Ensure that |PUBLIC "(limited quirks)" "(quirks)"| is
3623              ## marked as quirks.
3624            $self->{document}->manakai_compat_mode ('quirks');            $self->{document}->manakai_compat_mode ('quirks');
3625              !!!cp ('t11');
3626            } else {
3627              !!!cp ('t12');
3628          }          }
3629          } else {
3630            !!!cp ('t13');
3631        }        }
3632                
3633        ## Go to the root element phase.        ## Go to the "before html" insertion mode.
3634        !!!next-token;        !!!next-token;
3635        return;        return;
3636      } elsif ({      } elsif ({
3637                'start tag' => 1,                START_TAG_TOKEN, 1,
3638                'end tag' => 1,                END_TAG_TOKEN, 1,
3639                'end-of-file' => 1,                END_OF_FILE_TOKEN, 1,
3640               }->{$token->{type}}) {               }->{$token->{type}}) {
3641        !!!parse-error (type => 'no DOCTYPE');        !!!cp ('t14');
3642          !!!parse-error (type => 'no DOCTYPE', token => $token);
3643        $self->{document}->manakai_compat_mode ('quirks');        $self->{document}->manakai_compat_mode ('quirks');
3644        ## Go to the root element phase        ## Go to the "before html" insertion mode.
3645        ## reprocess        ## reprocess
3646          !!!ack-later;
3647        return;        return;
3648      } elsif ($token->{type} eq 'character') {      } elsif ($token->{type} == CHARACTER_TOKEN) {
3649        if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D        if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3650          ## Ignore the token          ## Ignore the token
3651    
3652          unless (length $token->{data}) {          unless (length $token->{data}) {
3653            ## Stay in the phase            !!!cp ('t15');
3654              ## Stay in the insertion mode.
3655            !!!next-token;            !!!next-token;
3656            redo INITIAL;            redo INITIAL;
3657            } else {
3658              !!!cp ('t16');
3659          }          }
3660          } else {
3661            !!!cp ('t17');
3662        }        }
3663    
3664        !!!parse-error (type => 'no DOCTYPE');        !!!parse-error (type => 'no DOCTYPE', token => $token);
3665        $self->{document}->manakai_compat_mode ('quirks');        $self->{document}->manakai_compat_mode ('quirks');
3666        ## Go to the root element phase        ## Go to the "before html" insertion mode.
3667        ## reprocess        ## reprocess
3668        return;        return;
3669      } elsif ($token->{type} eq 'comment') {      } elsif ($token->{type} == COMMENT_TOKEN) {
3670          !!!cp ('t18');
3671        my $comment = $self->{document}->create_comment ($token->{data});        my $comment = $self->{document}->create_comment ($token->{data});
3672        $self->{document}->append_child ($comment);        $self->{document}->append_child ($comment);
3673                
3674        ## Stay in the phase.        ## Stay in the insertion mode.
3675        !!!next-token;        !!!next-token;
3676        redo INITIAL;        redo INITIAL;
3677      } else {      } else {
3678        die "$0: $token->{type}: Unknown token";        die "$0: $token->{type}: Unknown token type";
3679      }      }
3680    } # INITIAL    } # INITIAL
3681    
3682      die "$0: _tree_construction_initial: This should be never reached";
3683  } # _tree_construction_initial  } # _tree_construction_initial
3684    
3685  sub _tree_construction_root_element ($) {  sub _tree_construction_root_element ($) {
3686    my $self = shift;    my $self = shift;
3687    
3688      ## NOTE: "before html" insertion mode.
3689        
3690    B: {    B: {
3691        if ($token->{type} eq 'DOCTYPE') {        if ($token->{type} == DOCTYPE_TOKEN) {
3692          !!!parse-error (type => 'in html:#DOCTYPE');          !!!cp ('t19');
3693            !!!parse-error (type => 'in html:#DOCTYPE', token => $token);
3694          ## Ignore the token          ## Ignore the token
3695          ## Stay in the phase          ## Stay in the insertion mode.
3696          !!!next-token;          !!!next-token;
3697          redo B;          redo B;
3698        } elsif ($token->{type} eq 'comment') {        } elsif ($token->{type} == COMMENT_TOKEN) {
3699            !!!cp ('t20');
3700          my $comment = $self->{document}->create_comment ($token->{data});          my $comment = $self->{document}->create_comment ($token->{data});
3701          $self->{document}->append_child ($comment);          $self->{document}->append_child ($comment);
3702          ## Stay in the phase          ## Stay in the insertion mode.
3703          !!!next-token;          !!!next-token;
3704          redo B;          redo B;
3705        } elsif ($token->{type} eq 'character') {        } elsif ($token->{type} == CHARACTER_TOKEN) {
3706          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
3707            $self->{document}->manakai_append_text ($1);            ## Ignore the token.
3708            ## ISSUE: DOM3 Core does not allow Document > Text  
3709            unless (length $token->{data}) {            unless (length $token->{data}) {
3710              ## Stay in the phase              !!!cp ('t21');
3711                ## Stay in the insertion mode.
3712              !!!next-token;              !!!next-token;
3713              redo B;              redo B;
3714              } else {
3715                !!!cp ('t22');
3716            }            }
3717            } else {
3718              !!!cp ('t23');
3719          }          }
3720    
3721            $self->{application_cache_selection}->(undef);
3722    
3723          #          #
3724          } elsif ($token->{type} == START_TAG_TOKEN) {
3725            if ($token->{tag_name} eq 'html') {
3726              my $root_element;
3727              !!!create-element ($root_element, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);
3728              $self->{document}->append_child ($root_element);
3729              push @{$self->{open_elements}},
3730                  [$root_element, $el_category->{html}];
3731    
3732              if ($token->{attributes}->{manifest}) {
3733                !!!cp ('t24');
3734                $self->{application_cache_selection}
3735                    ->($token->{attributes}->{manifest}->{value});
3736                ## ISSUE: Spec is unclear on relative references.
3737                ## According to Hixie (#whatwg 2008-03-19), it should be
3738                ## resolved against the base URI of the document in HTML
3739                ## or xml:base of the element in XHTML.
3740              } else {
3741                !!!cp ('t25');
3742                $self->{application_cache_selection}->(undef);
3743              }
3744    
3745              !!!nack ('t25c');
3746    
3747              !!!next-token;
3748              return; ## Go to the "before head" insertion mode.
3749            } else {
3750              !!!cp ('t25.1');
3751              #
3752            }
3753        } elsif ({        } elsif ({
3754                  'start tag' => 1,                  END_TAG_TOKEN, 1,
3755                  'end tag' => 1,                  END_OF_FILE_TOKEN, 1,
                 'end-of-file' => 1,  
3756                 }->{$token->{type}}) {                 }->{$token->{type}}) {
3757          ## ISSUE: There is an issue in the spec          !!!cp ('t26');
3758          #          #
3759        } else {        } else {
3760          die "$0: $token->{type}: Unknown token";          die "$0: $token->{type}: Unknown token type";
3761        }        }
3762        my $root_element; !!!create-element ($root_element, 'html');  
3763        $self->{document}->append_child ($root_element);      my $root_element;
3764        push @{$self->{open_elements}}, [$root_element, 'html'];      !!!create-element ($root_element, $HTML_NS, 'html',, $token);
3765        #$phase = 'main';      $self->{document}->append_child ($root_element);
3766        ## reprocess      push @{$self->{open_elements}}, [$root_element, $el_category->{html}];
3767        #redo B;  
3768        return;      $self->{application_cache_selection}->(undef);
3769    
3770        ## NOTE: Reprocess the token.
3771        !!!ack-later;
3772        return; ## Go to the "before head" insertion mode.
3773    
3774        ## ISSUE: There is an issue in the spec
3775    } # B    } # B
3776    
3777      die "$0: _tree_construction_root_element: This should never be reached";
3778  } # _tree_construction_root_element  } # _tree_construction_root_element
3779    
3780  sub _reset_insertion_mode ($) {  sub _reset_insertion_mode ($) {
# Line 1991  sub _reset_insertion_mode ($) { Line 3789  sub _reset_insertion_mode ($) {
3789            
3790      ## Step 3      ## Step 3
3791      S3: {      S3: {
3792        $last = 1 if $self->{open_elements}->[0]->[0] eq $node->[0];        if ($self->{open_elements}->[0]->[0] eq $node->[0]) {
3793        if (defined $self->{inner_html_node}) {          $last = 1;
3794          if ($self->{inner_html_node}->[1] eq 'td' or          if (defined $self->{inner_html_node}) {
3795              $self->{inner_html_node}->[1] eq 'th') {            !!!cp ('t28');
3796              $node = $self->{inner_html_node};
3797            } else {
3798              die "_reset_insertion_mode: t27";
3799            }
3800          }
3801          
3802          ## Step 4..14
3803          my $new_mode;
3804          if ($node->[1] & FOREIGN_EL) {
3805            !!!cp ('t28.1');
3806            ## NOTE: Strictly spaking, the line below only applies to MathML and
3807            ## SVG elements.  Currently the HTML syntax supports only MathML and
3808            ## SVG elements as foreigners.
3809            $new_mode = IN_BODY_IM | IN_FOREIGN_CONTENT_IM;
3810          } elsif ($node->[1] & TABLE_CELL_EL) {
3811            if ($last) {
3812              !!!cp ('t28.2');
3813            #            #
3814          } else {          } else {
3815            $node = $self->{inner_html_node};            !!!cp ('t28.3');
3816              $new_mode = IN_CELL_IM;
3817          }          }
3818          } else {
3819            !!!cp ('t28.4');
3820            $new_mode = {
3821                          select => IN_SELECT_IM,
3822                          ## NOTE: |option| and |optgroup| do not set
3823                          ## insertion mode to "in select" by themselves.
3824                          tr => IN_ROW_IM,
3825                          tbody => IN_TABLE_BODY_IM,
3826                          thead => IN_TABLE_BODY_IM,
3827                          tfoot => IN_TABLE_BODY_IM,
3828                          caption => IN_CAPTION_IM,
3829                          colgroup => IN_COLUMN_GROUP_IM,
3830                          table => IN_TABLE_IM,
3831                          head => IN_BODY_IM, # not in head!
3832                          body => IN_BODY_IM,
3833                          frameset => IN_FRAMESET_IM,
3834                         }->{$node->[0]->manakai_local_name};
3835        }        }
       
       ## Step 4..13  
       my $new_mode = {  
                       select => 'in select',  
                       td => 'in cell',  
                       th => 'in cell',  
                       tr => 'in row',  
                       tbody => 'in table body',  
                       thead => 'in table head',  
                       tfoot => 'in table foot',  
                       caption => 'in caption',  
                       colgroup => 'in column group',  
                       table => 'in table',  
                       head => 'in body', # not in head!  
                       body => 'in body',  
                       frameset => 'in frameset',  
                      }->{$node->[1]};  
3836        $self->{insertion_mode} = $new_mode and return if defined $new_mode;        $self->{insertion_mode} = $new_mode and return if defined $new_mode;
3837                
3838        ## Step 14        ## Step 15
3839        if ($node->[1] eq 'html') {        if ($node->[1] & HTML_EL) {
3840          unless (defined $self->{head_element}) {          unless (defined $self->{head_element}) {
3841            $self->{insertion_mode} = 'before head';            !!!cp ('t29');
3842              $self->{insertion_mode} = BEFORE_HEAD_IM;
3843          } else {          } else {
3844            $self->{insertion_mode} = 'after head';            ## ISSUE: Can this state be reached?
3845              !!!cp ('t30');
3846              $self->{insertion_mode} = AFTER_HEAD_IM;
3847          }          }
3848          return;          return;
3849          } else {
3850            !!!cp ('t31');
3851        }        }
3852                
       ## Step 15  
       $self->{insertion_mode} = 'in body' and return if $last;  
         
3853        ## Step 16        ## Step 16
3854          $self->{insertion_mode} = IN_BODY_IM and return if $last;
3855          
3856          ## Step 17
3857        $i--;        $i--;
3858        $node = $self->{open_elements}->[$i];        $node = $self->{open_elements}->[$i];
3859                
3860        ## Step 17        ## Step 18
3861        redo S3;        redo S3;
3862      } # S3      } # S3
3863    
3864      die "$0: _reset_insertion_mode: This line should never be reached";
3865  } # _reset_insertion_mode  } # _reset_insertion_mode
3866    
3867  sub _tree_construction_main ($) {  sub _tree_construction_main ($) {
3868    my $self = shift;    my $self = shift;
3869    
   my $phase = 'main';  
   
3870    my $active_formatting_elements = [];    my $active_formatting_elements = [];
3871    
3872    my $reconstruct_active_formatting_elements = sub { # MUST    my $reconstruct_active_formatting_elements = sub { # MUST
# Line 2062  sub _tree_construction_main ($) { Line 3883  sub _tree_construction_main ($) {
3883      return if $entry->[0] eq '#marker';      return if $entry->[0] eq '#marker';
3884      for (@{$self->{open_elements}}) {      for (@{$self->{open_elements}}) {
3885        if ($entry->[0] eq $_->[0]) {        if ($entry->[0] eq $_->[0]) {
3886            !!!cp ('t32');
3887          return;          return;
3888        }        }
3889      }      }
# Line 2076  sub _tree_construction_main ($) { Line 3898  sub _tree_construction_main ($) {
3898    
3899        ## Step 6        ## Step 6
3900        if ($entry->[0] eq '#marker') {        if ($entry->[0] eq '#marker') {
3901            !!!cp ('t33_1');
3902          #          #
3903        } else {        } else {
3904          my $in_open_elements;          my $in_open_elements;
3905          OE: for (@{$self->{open_elements}}) {          OE: for (@{$self->{open_elements}}) {
3906            if ($entry->[0] eq $_->[0]) {            if ($entry->[0] eq $_->[0]) {
3907                !!!cp ('t33');
3908              $in_open_elements = 1;              $in_open_elements = 1;
3909              last OE;              last OE;
3910            }            }
3911          }          }
3912          if ($in_open_elements) {          if ($in_open_elements) {
3913              !!!cp ('t34');
3914            #            #
3915          } else {          } else {
3916              ## NOTE: <!DOCTYPE HTML><p><b><i><u></p> <p>X
3917              !!!cp ('t35');
3918            redo S4;            redo S4;
3919          }          }
3920        }        }
# Line 2110  sub _tree_construction_main ($) { Line 3937  sub _tree_construction_main ($) {
3937    
3938        ## Step 11        ## Step 11
3939        unless ($clone->[0] eq $active_formatting_elements->[-1]->[0]) {        unless ($clone->[0] eq $active_formatting_elements->[-1]->[0]) {
3940            !!!cp ('t36');
3941          ## Step 7'          ## Step 7'
3942          $i++;          $i++;
3943          $entry = $active_formatting_elements->[$i];          $entry = $active_formatting_elements->[$i];
3944                    
3945          redo S7;          redo S7;
3946        }        }
3947    
3948          !!!cp ('t37');
3949      } # S7      } # S7
3950    }; # $reconstruct_active_formatting_elements    }; # $reconstruct_active_formatting_elements
3951    
3952    my $clear_up_to_marker = sub {    my $clear_up_to_marker = sub {
3953      for (reverse 0..$#$active_formatting_elements) {      for (reverse 0..$#$active_formatting_elements) {
3954        if ($active_formatting_elements->[$_]->[0] eq '#marker') {        if ($active_formatting_elements->[$_]->[0] eq '#marker') {
3955            !!!cp ('t38');
3956          splice @$active_formatting_elements, $_;          splice @$active_formatting_elements, $_;
3957          return;          return;
3958        }        }
3959      }      }
3960    
3961        !!!cp ('t39');
3962    }; # $clear_up_to_marker    }; # $clear_up_to_marker
3963    
3964    my $style_start_tag = sub {    my $insert;
3965      my $style_el; !!!create-element ($style_el, 'style', $token->{attributes});  
3966      ## $self->{insertion_mode} eq 'in head' and ... (always true)    my $parse_rcdata = sub ($) {
3967      (($self->{insertion_mode} eq 'in head' and defined $self->{head_element})      my ($content_model_flag) = @_;
3968       ? $self->{head_element} : $self->{open_elements}->[-1]->[0])  
3969        ->append_child ($style_el);      ## Step 1
3970      $self->{content_model_flag} = 'CDATA';      my $start_tag_name = $token->{tag_name};
3971        my $el;
3972        !!!create-element ($el, $HTML_NS, $start_tag_name, $token->{attributes}, $token);
3973    
3974        ## Step 2
3975        $insert->($el);
3976    
3977        ## Step 3
3978        $self->{content_model} = $content_model_flag; # CDATA or RCDATA
3979      delete $self->{escape}; # MUST      delete $self->{escape}; # MUST
3980                  
3981        ## Step 4
3982      my $text = '';      my $text = '';
3983        !!!nack ('t40.1');
3984      !!!next-token;      !!!next-token;
3985      while ($token->{type} eq 'character') {      while ($token->{type} == CHARACTER_TOKEN) { # or until stop tokenizing
3986          !!!cp ('t40');
3987        $text .= $token->{data};        $text .= $token->{data};
3988        !!!next-token;        !!!next-token;
3989      } # stop if non-character token or tokenizer stops tokenising      }
3990    
3991        ## Step 5
3992      if (length $text) {      if (length $text) {
3993        $style_el->manakai_append_text ($text);        !!!cp ('t41');
3994          my $text = $self->{document}->create_text_node ($text);
3995          $el->append_child ($text);
3996      }      }
3997        
3998      $self->{content_model_flag} = 'PCDATA';      ## Step 6
3999                      $self->{content_model} = PCDATA_CONTENT_MODEL;
4000      if ($token->{type} eq 'end tag' and $token->{tag_name} eq 'style') {  
4001        ## Step 7
4002        if ($token->{type} == END_TAG_TOKEN and
4003            $token->{tag_name} eq $start_tag_name) {
4004          !!!cp ('t42');
4005        ## Ignore the token        ## Ignore the token
4006      } else {      } else {
4007        !!!parse-error (type => 'in CDATA:#'.$token->{type});        ## NOTE: An end-of-file token.
4008        ## ISSUE: And ignore?        if ($content_model_flag == CDATA_CONTENT_MODEL) {
4009            !!!cp ('t43');
4010            !!!parse-error (type => 'in CDATA:#eof', token => $token);
4011          } elsif ($content_model_flag == RCDATA_CONTENT_MODEL) {
4012            !!!cp ('t44');
4013            !!!parse-error (type => 'in RCDATA:#eof', token => $token);
4014          } else {
4015            die "$0: $content_model_flag in parse_rcdata";
4016          }
4017      }      }
4018      !!!next-token;      !!!next-token;
4019    }; # $style_start_tag    }; # $parse_rcdata
4020    
4021    my $script_start_tag = sub {    my $script_start_tag = sub () {
4022      my $script_el;      my $script_el;
4023      !!!create-element ($script_el, 'script', $token->{attributes});      !!!create-element ($script_el, $HTML_NS, 'script', $token->{attributes}, $token);
4024      ## TODO: mark as "parser-inserted"      ## TODO: mark as "parser-inserted"
4025    
4026      $self->{content_model_flag} = 'CDATA';      $self->{content_model} = CDATA_CONTENT_MODEL;
4027      delete $self->{escape}; # MUST      delete $self->{escape}; # MUST
4028            
4029      my $text = '';      my $text = '';
4030        !!!nack ('t45.1');
4031      !!!next-token;      !!!next-token;
4032      while ($token->{type} eq 'character') {      while ($token->{type} == CHARACTER_TOKEN) {
4033          !!!cp ('t45');
4034        $text .= $token->{data};        $text .= $token->{data};
4035        !!!next-token;        !!!next-token;
4036      } # stop if non-character token or tokenizer stops tokenising      } # stop if non-character token or tokenizer stops tokenising
4037      if (length $text) {      if (length $text) {
4038          !!!cp ('t46');
4039        $script_el->manakai_append_text ($text);        $script_el->manakai_append_text ($text);
4040      }      }
4041                                
4042      $self->{content_model_flag} = 'PCDATA';      $self->{content_model} = PCDATA_CONTENT_MODEL;
4043    
4044      if ($token->{type} eq 'end tag' and      if ($token->{type} == END_TAG_TOKEN and
4045          $token->{tag_name} eq 'script') {          $token->{tag_name} eq 'script') {
4046          !!!cp ('t47');
4047        ## Ignore the token        ## Ignore the token
4048      } else {      } else {
4049        !!!parse-error (type => 'in CDATA:#'.$token->{type});        !!!cp ('t48');
4050          !!!parse-error (type => 'in CDATA:#eof', token => $token);
4051        ## ISSUE: And ignore?        ## ISSUE: And ignore?
4052        ## TODO: mark as "already executed"        ## TODO: mark as "already executed"
4053      }      }
4054            
4055      if (defined $self->{inner_html_node}) {      if (defined $self->{inner_html_node}) {
4056          !!!cp ('t49');
4057        ## TODO: mark as "already executed"        ## TODO: mark as "already executed"
4058      } else {      } else {
4059          !!!cp ('t50');
4060        ## TODO: $old_insertion_point = current insertion point        ## TODO: $old_insertion_point = current insertion point
4061        ## TODO: insertion point = just before the next input character        ## TODO: insertion point = just before the next input character
4062          
4063        (($self->{insertion_mode} eq 'in head' and defined $self->{head_element})        $insert->($script_el);
        ? $self->{head_element} : $self->{open_elements}->[-1]->[0])->append_child ($script_el);  
4064                
4065        ## TODO: insertion point = $old_insertion_point (might be "undefined")        ## TODO: insertion point = $old_insertion_point (might be "undefined")
4066                
# Line 2204  sub _tree_construction_main ($) { Line 4070  sub _tree_construction_main ($) {
4070      !!!next-token;      !!!next-token;
4071    }; # $script_start_tag    }; # $script_start_tag
4072    
4073      ## NOTE: $open_tables->[-1]->[0] is the "current table" element node.
4074      ## NOTE: $open_tables->[-1]->[1] is the "tainted" flag.
4075      my $open_tables = [[$self->{open_elements}->[0]->[0]]];
4076    
4077    my $formatting_end_tag = sub {    my $formatting_end_tag = sub {
4078      my $tag_name = shift;      my $end_tag_token = shift;
4079        my $tag_name = $end_tag_token->{tag_name};
4080    
4081        ## NOTE: The adoption agency algorithm (AAA).
4082    
4083      FET: {      FET: {
4084        ## Step 1        ## Step 1
4085        my $formatting_element;        my $formatting_element;
4086        my $formatting_element_i_in_active;        my $formatting_element_i_in_active;
4087        AFE: for (reverse 0..$#$active_formatting_elements) {        AFE: for (reverse 0..$#$active_formatting_elements) {
4088          if ($active_formatting_elements->[$_]->[1] eq $tag_name) {          if ($active_formatting_elements->[$_]->[0] eq '#marker') {
4089              !!!cp ('t52');
4090              last AFE;
4091            } elsif ($active_formatting_elements->[$_]->[0]->manakai_local_name
4092                         eq $tag_name) {
4093              !!!cp ('t51');
4094            $formatting_element = $active_formatting_elements->[$_];            $formatting_element = $active_formatting_elements->[$_];
4095            $formatting_element_i_in_active = $_;            $formatting_element_i_in_active = $_;
4096            last AFE;            last AFE;
         } elsif ($active_formatting_elements->[$_]->[0] eq '#marker') {  
           last AFE;  
4097          }          }
4098        } # AFE        } # AFE
4099        unless (defined $formatting_element) {        unless (defined $formatting_element) {
4100          !!!parse-error (type => 'unmatched end tag:'.$tag_name);          !!!cp ('t53');
4101            !!!parse-error (type => 'unmatched end tag', text => $tag_name, token => $end_tag_token);
4102          ## Ignore the token          ## Ignore the token
4103          !!!next-token;          !!!next-token;
4104          return;          return;
# Line 2233  sub _tree_construction_main ($) { Line 4110  sub _tree_construction_main ($) {
4110          my $node = $self->{open_elements}->[$_];          my $node = $self->{open_elements}->[$_];
4111          if ($node->[0] eq $formatting_element->[0]) {          if ($node->[0] eq $formatting_element->[0]) {
4112            if ($in_scope) {            if ($in_scope) {
4113                !!!cp ('t54');
4114              $formatting_element_i_in_open = $_;              $formatting_element_i_in_open = $_;
4115              last INSCOPE;              last INSCOPE;
4116            } else { # in open elements but not in scope            } else { # in open elements but not in scope
4117              !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});              !!!cp ('t55');
4118                !!!parse-error (type => 'unmatched end tag',
4119                                text => $token->{tag_name},
4120                                token => $end_tag_token);
4121              ## Ignore the token              ## Ignore the token
4122              !!!next-token;              !!!next-token;
4123              return;              return;
4124            }            }
4125          } elsif ({          } elsif ($node->[1] & SCOPING_EL) {
4126                    table => 1, caption => 1, td => 1, th => 1,            !!!cp ('t56');
                   button => 1, marquee => 1, object => 1, html => 1,  
                  }->{$node->[1]}) {  
4127            $in_scope = 0;            $in_scope = 0;
4128          }          }
4129        } # INSCOPE        } # INSCOPE
4130        unless (defined $formatting_element_i_in_open) {        unless (defined $formatting_element_i_in_open) {
4131          !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});          !!!cp ('t57');
4132            !!!parse-error (type => 'unmatched end tag',
4133                            text => $token->{tag_name},
4134                            token => $end_tag_token);
4135          pop @$active_formatting_elements; # $formatting_element          pop @$active_formatting_elements; # $formatting_element
4136          !!!next-token; ## TODO: ok?          !!!next-token; ## TODO: ok?
4137          return;          return;
4138        }        }
4139        if (not $self->{open_elements}->[-1]->[0] eq $formatting_element->[0]) {        if (not $self->{open_elements}->[-1]->[0] eq $formatting_element->[0]) {
4140          !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);          !!!cp ('t58');
4141            !!!parse-error (type => 'not closed',
4142                            text => $self->{open_elements}->[-1]->[0]
4143                                ->manakai_local_name,
4144                            token => $end_tag_token);
4145        }        }
4146                
4147        ## Step 2        ## Step 2
# Line 2263  sub _tree_construction_main ($) { Line 4149  sub _tree_construction_main ($) {
4149        my $furthest_block_i_in_open;        my $furthest_block_i_in_open;
4150        OE: for (reverse 0..$#{$self->{open_elements}}) {        OE: for (reverse 0..$#{$self->{open_elements}}) {
4151          my $node = $self->{open_elements}->[$_];          my $node = $self->{open_elements}->[$_];
4152          if (not $formatting_category->{$node->[1]} and          if (not ($node->[1] & FORMATTING_EL) and
4153              #not $phrasing_category->{$node->[1]} and              #not $phrasing_category->{$node->[1]} and
4154              ($special_category->{$node->[1]} or              ($node->[1] & SPECIAL_EL or
4155               $scoping_category->{$node->[1]})) {               $node->[1] & SCOPING_EL)) { ## Scoping is redundant, maybe
4156              !!!cp ('t59');
4157            $furthest_block = $node;            $furthest_block = $node;
4158            $furthest_block_i_in_open = $_;            $furthest_block_i_in_open = $_;
4159          } elsif ($node->[0] eq $formatting_element->[0]) {          } elsif ($node->[0] eq $formatting_element->[0]) {
4160              !!!cp ('t60');
4161            last OE;            last OE;
4162          }          }
4163        } # OE        } # OE
4164                
4165        ## Step 3        ## Step 3
4166        unless (defined $furthest_block) { # MUST        unless (defined $furthest_block) { # MUST
4167            !!!cp ('t61');
4168          splice @{$self->{open_elements}}, $formatting_element_i_in_open;          splice @{$self->{open_elements}}, $formatting_element_i_in_open;
4169          splice @$active_formatting_elements, $formatting_element_i_in_active, 1;          splice @$active_formatting_elements, $formatting_element_i_in_active, 1;
4170          !!!next-token;          !!!next-token;
# Line 2288  sub _tree_construction_main ($) { Line 4177  sub _tree_construction_main ($) {
4177        ## Step 5        ## Step 5
4178        my $furthest_block_parent = $furthest_block->[0]->parent_node;        my $furthest_block_parent = $furthest_block->[0]->parent_node;
4179        if (defined $furthest_block_parent) {        if (defined $furthest_block_parent) {
4180            !!!cp ('t62');
4181          $furthest_block_parent->remove_child ($furthest_block->[0]);          $furthest_block_parent->remove_child ($furthest_block->[0]);
4182        }        }
4183                
# Line 2310  sub _tree_construction_main ($) { Line 4200  sub _tree_construction_main ($) {
4200          S7S2: {          S7S2: {
4201            for (reverse 0..$#$active_formatting_elements) {            for (reverse 0..$#$active_formatting_elements) {
4202              if ($active_formatting_elements->[$_]->[0] eq $node->[0]) {              if ($active_formatting_elements->[$_]->[0] eq $node->[0]) {
4203                  !!!cp ('t63');
4204                $node_i_in_active = $_;                $node_i_in_active = $_;
4205                last S7S2;                last S7S2;
4206              }              }
# Line 2323  sub _tree_construction_main ($) { Line 4214  sub _tree_construction_main ($) {
4214                    
4215          ## Step 4          ## Step 4
4216          if ($last_node->[0] eq $furthest_block->[0]) {          if ($last_node->[0] eq $furthest_block->[0]) {
4217              !!!cp ('t64');
4218            $bookmark_prev_el = $node->[0];            $bookmark_prev_el = $node->[0];
4219          }          }
4220                    
4221          ## Step 5          ## Step 5
4222          if ($node->[0]->has_child_nodes ()) {          if ($node->[0]->has_child_nodes ()) {
4223              !!!cp ('t65');
4224            my $clone = [$node->[0]->clone_node (0), $node->[1]];            my $clone = [$node->[0]->clone_node (0), $node->[1]];
4225            $active_formatting_elements->[$node_i_in_active] = $clone;            $active_formatting_elements->[$node_i_in_active] = $clone;
4226            $self->{open_elements}->[$node_i_in_open] = $clone;            $self->{open_elements}->[$node_i_in_open] = $clone;
# Line 2345  sub _tree_construction_main ($) { Line 4238  sub _tree_construction_main ($) {
4238        } # S7          } # S7  
4239                
4240        ## Step 8        ## Step 8
4241        $common_ancestor_node->[0]->append_child ($last_node->[0]);        if ($common_ancestor_node->[1] & TABLE_ROWS_EL) {
4242            my $foster_parent_element;
4243            my $next_sibling;
4244            OE: for (reverse 0..$#{$self->{open_elements}}) {
4245              if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {
4246                                 my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
4247                                 if (defined $parent and $parent->node_type == 1) {
4248                                   !!!cp ('t65.1');
4249                                   $foster_parent_element = $parent;
4250                                   $next_sibling = $self->{open_elements}->[$_]->[0];
4251                                 } else {
4252                                   !!!cp ('t65.2');
4253                                   $foster_parent_element
4254                                     = $self->{open_elements}->[$_ - 1]->[0];
4255                                 }
4256                                 last OE;
4257                               }
4258                             } # OE
4259                             $foster_parent_element = $self->{open_elements}->[0]->[0]
4260                               unless defined $foster_parent_element;
4261            $foster_parent_element->insert_before ($last_node->[0], $next_sibling);
4262            $open_tables->[-1]->[1] = 1; # tainted
4263          } else {
4264            !!!cp ('t65.3');
4265            $common_ancestor_node->[0]->append_child ($last_node->[0]);
4266          }
4267                
4268        ## Step 9        ## Step 9
4269        my $clone = [$formatting_element->[0]->clone_node (0),        my $clone = [$formatting_element->[0]->clone_node (0),
# Line 2362  sub _tree_construction_main ($) { Line 4280  sub _tree_construction_main ($) {
4280        my $i;        my $i;
4281        AFE: for (reverse 0..$#$active_formatting_elements) {        AFE: for (reverse 0..$#$active_formatting_elements) {
4282          if ($active_formatting_elements->[$_]->[0] eq $formatting_element->[0]) {          if ($active_formatting_elements->[$_]->[0] eq $formatting_element->[0]) {
4283              !!!cp ('t66');
4284            splice @$active_formatting_elements, $_, 1;            splice @$active_formatting_elements, $_, 1;
4285            $i-- and last AFE if defined $i;            $i-- and last AFE if defined $i;
4286          } elsif ($active_formatting_elements->[$_]->[0] eq $bookmark_prev_el) {          } elsif ($active_formatting_elements->[$_]->[0] eq $bookmark_prev_el) {
4287              !!!cp ('t67');
4288            $i = $_;            $i = $_;
4289          }          }
4290        } # AFE        } # AFE
# Line 2374  sub _tree_construction_main ($) { Line 4294  sub _tree_construction_main ($) {
4294        undef $i;        undef $i;
4295        OE: for (reverse 0..$#{$self->{open_elements}}) {        OE: for (reverse 0..$#{$self->{open_elements}}) {
4296          if ($self->{open_elements}->[$_]->[0] eq $formatting_element->[0]) {          if ($self->{open_elements}->[$_]->[0] eq $formatting_element->[0]) {
4297              !!!cp ('t68');
4298            splice @{$self->{open_elements}}, $_, 1;            splice @{$self->{open_elements}}, $_, 1;
4299            $i-- and last OE if defined $i;            $i-- and last OE if defined $i;
4300          } elsif ($self->{open_elements}->[$_]->[0] eq $furthest_block->[0]) {          } elsif ($self->{open_elements}->[$_]->[0] eq $furthest_block->[0]) {
4301              !!!cp ('t69');
4302            $i = $_;            $i = $_;
4303          }          }
4304        } # OE        } # OE
# Line 2387  sub _tree_construction_main ($) { Line 4309  sub _tree_construction_main ($) {
4309      } # FET      } # FET
4310    }; # $formatting_end_tag    }; # $formatting_end_tag
4311    
4312    my $insert_to_current = sub {    $insert = my $insert_to_current = sub {
4313      $self->{open_elements}->[-1]->[0]->append_child (shift);      $self->{open_elements}->[-1]->[0]->append_child ($_[0]);
4314    }; # $insert_to_current    }; # $insert_to_current
4315    
4316    my $insert_to_foster = sub {    my $insert_to_foster = sub {
4317                         my $child = shift;      my $child = shift;
4318                         if ({      if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {
4319                              table => 1, tbody => 1, tfoot => 1,        # MUST
4320                              thead => 1, tr => 1,        my $foster_parent_element;
4321                             }->{$self->{open_elements}->[-1]->[1]}) {        my $next_sibling;
4322                           # MUST        OE: for (reverse 0..$#{$self->{open_elements}}) {
4323                           my $foster_parent_element;          if ($self->{open_elements}->[$_]->[1] & TABLE_EL) {
                          my $next_sibling;  
                          OE: for (reverse 0..$#{$self->{open_elements}}) {  
                            if ($self->{open_elements}->[$_]->[1] eq 'table') {  
4324                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;
4325                               if (defined $parent and $parent->node_type == 1) {                               if (defined $parent and $parent->node_type == 1) {
4326                                   !!!cp ('t70');
4327                                 $foster_parent_element = $parent;                                 $foster_parent_element = $parent;
4328                                 $next_sibling = $self->{open_elements}->[$_]->[0];                                 $next_sibling = $self->{open_elements}->[$_]->[0];
4329                               } else {                               } else {
4330                                   !!!cp ('t71');
4331                                 $foster_parent_element                                 $foster_parent_element
4332                                   = $self->{open_elements}->[$_ - 1]->[0];                                   = $self->{open_elements}->[$_ - 1]->[0];
4333                               }                               }
# Line 2417  sub _tree_construction_main ($) { Line 4338  sub _tree_construction_main ($) {
4338                             unless defined $foster_parent_element;                             unless defined $foster_parent_element;
4339                           $foster_parent_element->insert_before                           $foster_parent_element->insert_before
4340                             ($child, $next_sibling);                             ($child, $next_sibling);
4341                         } else {        $open_tables->[-1]->[1] = 1; # tainted
4342                           $self->{open_elements}->[-1]->[0]->append_child ($child);      } else {
4343                         }        !!!cp ('t72');
4344          $self->{open_elements}->[-1]->[0]->append_child ($child);
4345        }
4346    }; # $insert_to_foster    }; # $insert_to_foster
4347    
4348    my $in_body = sub {    B: while (1) {
4349      my $insert = shift;      if ($token->{type} == DOCTYPE_TOKEN) {
4350      if ($token->{type} eq 'start tag') {        !!!cp ('t73');
4351        if ($token->{tag_name} eq 'script') {        !!!parse-error (type => 'in html:#DOCTYPE', token => $token);
4352          $script_start_tag->();        ## Ignore the token
4353          return;        ## Stay in the phase
4354        } elsif ($token->{tag_name} eq 'style') {        !!!next-token;
4355          $style_start_tag->();        next B;
4356          return;      } elsif ($token->{type} == START_TAG_TOKEN and
4357        } elsif ({               $token->{tag_name} eq 'html') {
4358                  base => 1, link => 1, meta => 1,        if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
4359                 }->{$token->{tag_name}}) {          !!!cp ('t79');
4360          ## NOTE: This is an "as if in head" code clone          !!!parse-error (type => 'after html', text => 'html', token => $token);
4361          my $el;          $self->{insertion_mode} = AFTER_BODY_IM;
4362          !!!create-element ($el, $token->{tag_name}, $token->{attributes});        } elsif ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {
4363          if ($self->{insertion_mode} eq 'in head' and          !!!cp ('t80');
4364              defined $self->{head_element}) {          !!!parse-error (type => 'after html', text => 'html', token => $token);
4365            $self->{head_element}->append_child ($el);          $self->{insertion_mode} = AFTER_FRAMESET_IM;
4366          } else {        } else {
4367            $insert->($el);          !!!cp ('t81');
4368          }        }
4369            
4370          !!!next-token;        !!!cp ('t82');
4371          return;        !!!parse-error (type => 'not first start tag', token => $token);
4372        } elsif ($token->{tag_name} eq 'title') {        my $top_el = $self->{open_elements}->[0]->[0];
4373          !!!parse-error (type => 'in body:title');        for my $attr_name (keys %{$token->{attributes}}) {
4374          ## NOTE: There is an "as if in head" code clone          unless ($top_el->has_attribute_ns (undef, $attr_name)) {
4375          my $title_el;            !!!cp ('t84');
4376          !!!create-element ($title_el, 'title', $token->{attributes});            $top_el->set_attribute_ns
4377          (defined $self->{head_element} ? $self->{head_element} : $self->{open_elements}->[-1]->[0])              (undef, [undef, $attr_name],
4378            ->append_child ($title_el);               $token->{attributes}->{$attr_name}->{value});
         $self->{content_model_flag} = 'RCDATA';  
         delete $self->{escape}; # MUST  
           
         my $text = '';  
         !!!next-token;  
         while ($token->{type} eq 'character') {  
           $text .= $token->{data};  
           !!!next-token;  
         }  
         if (length $text) {  
           $title_el->manakai_append_text ($text);  
         }  
           
         $self->{content_model_flag} = 'PCDATA';  
           
         if ($token->{type} eq 'end tag' and  
             $token->{tag_name} eq 'title') {  
           ## Ignore the token  
         } else {  
           !!!parse-error (type => 'in RCDATA:#'.$token->{type});  
           ## ISSUE: And ignore?  
         }  
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'body') {  
         !!!parse-error (type => 'in body:body');  
                 
         if (@{$self->{open_elements}} == 1 or  
             $self->{open_elements}->[1]->[1] ne 'body') {  
           ## Ignore the token  
         } else {  
           my $body_el = $self->{open_elements}->[1]->[0];  
           for my $attr_name (keys %{$token->{attributes}}) {  
             unless ($body_el->has_attribute_ns (undef, $attr_name)) {  
               $body_el->set_attribute_ns  
                 (undef, [undef, $attr_name],  
                  $token->{attributes}->{$attr_name}->{value});  
             }  
           }  
         }  
         !!!next-token;  
         return;  
       } elsif ({  
                 address => 1, blockquote => 1, center => 1, dir => 1,  
                 div => 1, dl => 1, fieldset => 1, listing => 1,  
                 menu => 1, ol => 1, p => 1, ul => 1,  
                 pre => 1,  
                }->{$token->{tag_name}}) {  
         ## has a p element in scope  
         INSCOPE: for (reverse @{$self->{open_elements}}) {  
           if ($_->[1] eq 'p') {  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'p'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$_->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         if ($token->{tag_name} eq 'pre') {  
           !!!next-token;  
           if ($token->{type} eq 'character') {  
             $token->{data} =~ s/^\x0A//;  
             unless (length $token->{data}) {  
               !!!next-token;  
             }  
           }  
         } else {  
           !!!next-token;  
         }  
         return;  
       } elsif ($token->{tag_name} eq 'form') {  
         if (defined $self->{form_element}) {  
           !!!parse-error (type => 'in form:form');  
           ## Ignore the token  
           !!!next-token;  
           return;  
         } else {  
           ## has a p element in scope  
           INSCOPE: for (reverse @{$self->{open_elements}}) {  
             if ($_->[1] eq 'p') {  
               !!!back-token;  
               $token = {type => 'end tag', tag_name => 'p'};  
               return;  
             } elsif ({  
                       table => 1, caption => 1, td => 1, th => 1,  
                       button => 1, marquee => 1, object => 1, html => 1,  
                      }->{$_->[1]}) {  
               last INSCOPE;  
             }  
           } # INSCOPE  
               
           !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
           $self->{form_element} = $self->{open_elements}->[-1]->[0];  
           !!!next-token;  
           return;  
4379          }          }
4380        } elsif ($token->{tag_name} eq 'li') {        }
4381          ## has a p element in scope        !!!nack ('t84.1');
4382          INSCOPE: for (reverse @{$self->{open_elements}}) {        !!!next-token;
4383            if ($_->[1] eq 'p') {        next B;
4384              !!!back-token;      } elsif ($token->{type} == COMMENT_TOKEN) {
4385              $token = {type => 'end tag', tag_name => 'p'};        my $comment = $self->{document}->create_comment ($token->{data});
4386              return;        if ($self->{insertion_mode} & AFTER_HTML_IMS) {
4387            } elsif ({          !!!cp ('t85');
4388                      table => 1, caption => 1, td => 1, th => 1,          $self->{document}->append_child ($comment);
4389                      button => 1, marquee => 1, object => 1, html => 1,        } elsif ($self->{insertion_mode} == AFTER_BODY_IM) {
4390                     }->{$_->[1]}) {          !!!cp ('t86');
4391              last INSCOPE;          $self->{open_elements}->[0]->[0]->append_child ($comment);
4392            }        } else {
4393          } # INSCOPE          !!!cp ('t87');
4394                      $self->{open_elements}->[-1]->[0]->append_child ($comment);
4395          ## Step 1        }
4396          my $i = -1;        !!!next-token;
4397          my $node = $self->{open_elements}->[$i];        next B;
4398          LI: {      } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {
4399            ## Step 2        if ($token->{type} == CHARACTER_TOKEN) {
4400            if ($node->[1] eq 'li') {          !!!cp ('t87.1');
4401              if ($i != -1) {          $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
               !!!parse-error (type => 'end tag missing:'.  
                               $self->{open_elements}->[-1]->[1]);  
               ## TODO: test  
             }  
             splice @{$self->{open_elements}}, $i;  
             last LI;  
           }  
             
           ## Step 3  
           if (not $formatting_category->{$node->[1]} and  
               #not $phrasing_category->{$node->[1]} and  
               ($special_category->{$node->[1]} or  
                $scoping_category->{$node->[1]}) and  
               $node->[1] ne 'address' and $node->[1] ne 'div') {  
             last LI;  
           }  
             
           ## Step 4  
           $i--;  
           $node = $self->{open_elements}->[$i];  
           redo LI;  
         } # LI  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'dd' or $token->{tag_name} eq 'dt') {  
         ## has a p element in scope  
         INSCOPE: for (reverse @{$self->{open_elements}}) {  
           if ($_->[1] eq 'p') {  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'p'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$_->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         ## Step 1  
         my $i = -1;  
         my $node = $self->{open_elements}->[$i];  
         LI: {  
           ## Step 2  
           if ($node->[1] eq 'dt' or $node->[1] eq 'dd') {  
             if ($i != -1) {  
               !!!parse-error (type => 'end tag missing:'.  
                               $self->{open_elements}->[-1]->[1]);  
               ## TODO: test  
             }  
             splice @{$self->{open_elements}}, $i;  
             last LI;  
           }  
             
           ## Step 3  
           if (not $formatting_category->{$node->[1]} and  
               #not $phrasing_category->{$node->[1]} and  
               ($special_category->{$node->[1]} or  
                $scoping_category->{$node->[1]}) and  
               $node->[1] ne 'address' and $node->[1] ne 'div') {  
             last LI;  
           }  
             
           ## Step 4  
           $i--;  
           $node = $self->{open_elements}->[$i];  
           redo LI;  
         } # LI  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'plaintext') {  
         ## has a p element in scope  
         INSCOPE: for (reverse @{$self->{open_elements}}) {  
           if ($_->[1] eq 'p') {  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'p'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$_->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
             
         $self->{content_model_flag} = 'PLAINTEXT';  
             
         !!!next-token;  
         return;  
       } elsif ({  
                 h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,  
                }->{$token->{tag_name}}) {  
         ## has a p element in scope  
         INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
           my $node = $self->{open_elements}->[$_];  
           if ($node->[1] eq 'p') {  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'p'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         ## NOTE: See <http://html5.org/tools/web-apps-tracker?from=925&to=926>  
         ## has an element in scope  
         #my $i;  
         #INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
         #  my $node = $self->{open_elements}->[$_];  
         #  if ({  
         #       h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,  
         #      }->{$node->[1]}) {  
         #    $i = $_;  
         #    last INSCOPE;  
         #  } elsif ({  
         #            table => 1, caption => 1, td => 1, th => 1,  
         #            button => 1, marquee => 1, object => 1, html => 1,  
         #           }->{$node->[1]}) {  
         #    last INSCOPE;  
         #  }  
         #} # INSCOPE  
         #    
         #if (defined $i) {  
         #  !!! parse-error (type => 'in hn:hn');  
         #  splice @{$self->{open_elements}}, $i;  
         #}  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
             
4402          !!!next-token;          !!!next-token;
4403          return;          next B;
4404        } elsif ($token->{tag_name} eq 'a') {        } elsif ($token->{type} == START_TAG_TOKEN) {
4405          AFE: for my $i (reverse 0..$#$active_formatting_elements) {          if ((not {mglyph => 1, malignmark => 1}->{$token->{tag_name}} and
4406            my $node = $active_formatting_elements->[$i];               $self->{open_elements}->[-1]->[1] & FOREIGN_FLOW_CONTENT_EL) or
4407            if ($node->[1] eq 'a') {              not ($self->{open_elements}->[-1]->[1] & FOREIGN_EL) or
4408              !!!parse-error (type => 'in a:a');              ($token->{tag_name} eq 'svg' and
4409                             $self->{open_elements}->[-1]->[1] & MML_AXML_EL)) {
4410              !!!back-token;            ## NOTE: "using the rules for secondary insertion mode"then"continue"
4411              $token = {type => 'end tag', tag_name => 'a'};            !!!cp ('t87.2');
4412              $formatting_end_tag->($token->{tag_name});            #
4413                        } elsif ({
4414              AFE2: for (reverse 0..$#$active_formatting_elements) {                    b => 1, big => 1, blockquote => 1, body => 1, br => 1,
4415                if ($active_formatting_elements->[$_]->[0] eq $node->[0]) {                    center => 1, code => 1, dd => 1, div => 1, dl => 1, dt => 1,
4416                  splice @$active_formatting_elements, $_, 1;                    em => 1, embed => 1, font => 1, h1 => 1, h2 => 1, h3 => 1,
4417                  last AFE2;                    h4 => 1, h5 => 1, h6 => 1, head => 1, hr => 1, i => 1,
4418                }                    img => 1, li => 1, listing => 1, menu => 1, meta => 1,
4419              } # AFE2                    nobr => 1, ol => 1, p => 1, pre => 1, ruby => 1, s => 1,
4420              OE: for (reverse 0..$#{$self->{open_elements}}) {                    small => 1, span => 1, strong => 1, strike => 1, sub => 1,
4421                if ($self->{open_elements}->[$_]->[0] eq $node->[0]) {                    sup => 1, table => 1, tt => 1, u => 1, ul => 1, var => 1,
4422                  splice @{$self->{open_elements}}, $_, 1;                   }->{$token->{tag_name}}) {
4423                  last OE;            !!!cp ('t87.2');
4424                }            !!!parse-error (type => 'not closed',
4425              } # OE                            text => $self->{open_elements}->[-1]->[0]
4426              last AFE;                                ->manakai_local_name,
4427            } elsif ($node->[0] eq '#marker') {                            token => $token);
4428              last AFE;  
4429              pop @{$self->{open_elements}}
4430                  while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;
4431    
4432              $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
4433              ## Reprocess.
4434              next B;
4435            } else {
4436              my $nsuri = $self->{open_elements}->[-1]->[0]->namespace_uri;
4437              my $tag_name = $token->{tag_name};
4438              if ($nsuri eq $SVG_NS) {
4439                $tag_name = {
4440                   altglyph => 'altGlyph',
4441                   altglyphdef => 'altGlyphDef',
4442                   altglyphitem => 'altGlyphItem',
4443                   animatecolor => 'animateColor',
4444                   animatemotion => 'animateMotion',
4445                   animatetransform => 'animateTransform',
4446                   clippath => 'clipPath',
4447                   feblend => 'feBlend',
4448                   fecolormatrix => 'feColorMatrix',
4449                   fecomponenttransfer => 'feComponentTransfer',
4450                   fecomposite => 'feComposite',
4451                   feconvolvematrix => 'feConvolveMatrix',
4452                   fediffuselighting => 'feDiffuseLighting',
4453                   fedisplacementmap => 'feDisplacementMap',
4454                   fedistantlight => 'feDistantLight',
4455                   feflood => 'feFlood',
4456                   fefunca => 'feFuncA',
4457                   fefuncb => 'feFuncB',
4458                   fefuncg => 'feFuncG',
4459                   fefuncr => 'feFuncR',
4460                   fegaussianblur => 'feGaussianBlur',
4461                   feimage => 'feImage',
4462                   femerge => 'feMerge',
4463                   femergenode => 'feMergeNode',
4464                   femorphology => 'feMorphology',
4465                   feoffset => 'feOffset',
4466                   fepointlight => 'fePointLight',
4467                   fespecularlighting => 'feSpecularLighting',
4468                   fespotlight => 'feSpotLight',
4469                   fetile => 'feTile',
4470                   feturbulence => 'feTurbulence',
4471                   foreignobject => 'foreignObject',
4472                   glyphref => 'glyphRef',
4473                   lineargradient => 'linearGradient',
4474                   radialgradient => 'radialGradient',
4475                   #solidcolor => 'solidColor', ## NOTE: Commented in spec (SVG1.2)
4476                   textpath => 'textPath',  
4477                }->{$tag_name} || $tag_name;
4478            }            }
         } # AFE  
             
         $reconstruct_active_formatting_elements->($insert_to_current);  
4479    
4480          !!!insert-element-t ($token->{tag_name}, $token->{attributes});            ## "adjust SVG attributes" (SVG only) - done in insert-element-f
         push @$active_formatting_elements, $self->{open_elements}->[-1];  
4481    
4482          !!!next-token;            ## "adjust foreign attributes" - done in insert-element-f
         return;  
       } elsif ({  
                 b => 1, big => 1, em => 1, font => 1, i => 1,  
                 s => 1, small => 1, strile => 1,  
                 strong => 1, tt => 1, u => 1,  
                }->{$token->{tag_name}}) {  
         $reconstruct_active_formatting_elements->($insert_to_current);  
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         push @$active_formatting_elements, $self->{open_elements}->[-1];  
           
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'nobr') {  
         $reconstruct_active_formatting_elements->($insert_to_current);  
4483    
4484          ## has a |nobr| element in scope            !!!insert-element-f ($nsuri, $tag_name, $token->{attributes}, $token);
         INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
           my $node = $self->{open_elements}->[$_];  
           if ($node->[1] eq 'nobr') {  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'nobr'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         push @$active_formatting_elements, $self->{open_elements}->[-1];  
           
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'button') {  
         ## has a button element in scope  
         INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
           my $node = $self->{open_elements}->[$_];  
           if ($node->[1] eq 'button') {  
             !!!parse-error (type => 'in button:button');  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'button'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         $reconstruct_active_formatting_elements->($insert_to_current);  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         push @$active_formatting_elements, ['#marker', ''];  
4485    
4486          !!!next-token;            if ($self->{self_closing}) {
4487          return;              pop @{$self->{open_elements}};
4488        } elsif ($token->{tag_name} eq 'marquee' or              !!!ack ('t87.3');
                $token->{tag_name} eq 'object') {  
         $reconstruct_active_formatting_elements->($insert_to_current);  
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         push @$active_formatting_elements, ['#marker', ''];  
           
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'xmp') {  
         $reconstruct_active_formatting_elements->($insert_to_current);  
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
           
         $self->{content_model_flag} = 'CDATA';  
         delete $self->{escape}; # MUST  
           
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'table') {  
         ## has a p element in scope  
         INSCOPE: for (reverse @{$self->{open_elements}}) {  
           if ($_->[1] eq 'p') {  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'p'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$_->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
             
         $self->{insertion_mode} = 'in table';  
             
         !!!next-token;  
         return;  
       } elsif ({  
                 area => 1, basefont => 1, bgsound => 1, br => 1,  
                 embed => 1, img => 1, param => 1, spacer => 1, wbr => 1,  
                 image => 1,  
                }->{$token->{tag_name}}) {  
         if ($token->{tag_name} eq 'image') {  
           !!!parse-error (type => 'image');  
           $token->{tag_name} = 'img';  
         }  
           
         $reconstruct_active_formatting_elements->($insert_to_current);  
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         pop @{$self->{open_elements}};  
           
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'hr') {  
         ## has a p element in scope  
         INSCOPE: for (reverse @{$self->{open_elements}}) {  
           if ($_->[1] eq 'p') {  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'p'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$_->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         pop @{$self->{open_elements}};  
             
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'input') {  
         $reconstruct_active_formatting_elements->($insert_to_current);  
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         ## TODO: associate with $self->{form_element} if defined  
         pop @{$self->{open_elements}};  
           
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'isindex') {  
         !!!parse-error (type => 'isindex');  
           
         if (defined $self->{form_element}) {  
           ## Ignore the token  
           !!!next-token;  
           return;  
         } else {  
           my $at = $token->{attributes};  
           my $form_attrs;  
           $form_attrs->{action} = $at->{action} if $at->{action};  
           my $prompt_attr = $at->{prompt};  
           $at->{name} = {name => 'name', value => 'isindex'};  
           delete $at->{action};  
           delete $at->{prompt};  
           my @tokens = (  
                         {type => 'start tag', tag_name => 'form',  
                          attributes => $form_attrs},  
                         {type => 'start tag', tag_name => 'hr'},  
                         {type => 'start tag', tag_name => 'p'},  
                         {type => 'start tag', tag_name => 'label'},  
                        );  
           if ($prompt_attr) {  
             push @tokens, {type => 'character', data => $prompt_attr->{value}};  
4489            } else {            } else {
4490              push @tokens, {type => 'character',              !!!cp ('t87.4');
                            data => 'This is a searchable index. Insert your search keywords here: '}; # SHOULD  
             ## TODO: make this configurable  
4491            }            }
4492            push @tokens,  
                         {type => 'start tag', tag_name => 'input', attributes => $at},  
                         #{type => 'character', data => ''}, # SHOULD  
                         {type => 'end tag', tag_name => 'label'},  
                         {type => 'end tag', tag_name => 'p'},  
                         {type => 'start tag', tag_name => 'hr'},  
                         {type => 'end tag', tag_name => 'form'};  
           $token = shift @tokens;  
           !!!back-token (@tokens);  
           return;  
         }  
       } elsif ({  
                 textarea => 1,  
                 iframe => 1,  
                 noembed => 1,  
                 noframes => 1,  
                 noscript => 0, ## TODO: 1 if scripting is enabled  
                }->{$token->{tag_name}}) {  
         my $tag_name = $token->{tag_name};  
         my $el;  
         !!!create-element ($el, $token->{tag_name}, $token->{attributes});  
           
         if ($token->{tag_name} eq 'textarea') {  
           ## TODO: $self->{form_element} if defined  
           $self->{content_model_flag} = 'RCDATA';  
         } else {  
           $self->{content_model_flag} = 'CDATA';  
         }  
         delete $self->{escape}; # MUST  
           
         $insert->($el);  
           
         my $text = '';  
         if ($token->{tag_name} eq 'textarea') {  
           !!!next-token;  
           if ($token->{type} eq 'character') {  
             $token->{data} =~ s/^\x0A//;  
             unless (length $token->{data}) {  
               !!!next-token;  
             }  
           }  
         } else {  
           !!!next-token;  
         }  
         while ($token->{type} eq 'character') {  
           $text .= $token->{data};  
4493            !!!next-token;            !!!next-token;
4494              next B;
4495          }          }
4496          if (length $text) {        } elsif ($token->{type} == END_TAG_TOKEN) {
4497            $el->manakai_append_text ($text);          ## NOTE: "using the rules for secondary insertion mode" then "continue"
4498          }          !!!cp ('t87.5');
4499                    #
4500          $self->{content_model_flag} = 'PCDATA';        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
4501                    !!!cp ('t87.6');
4502          if ($token->{type} eq 'end tag' and          !!!parse-error (type => 'not closed',
4503              $token->{tag_name} eq $tag_name) {                          text => $self->{open_elements}->[-1]->[0]
4504            ## Ignore the token                              ->manakai_local_name,
4505          } else {                          token => $token);
4506            if ($token->{tag_name} eq 'textarea') {  
4507              !!!parse-error (type => 'in RCDATA:#'.$token->{type});          pop @{$self->{open_elements}}
4508            } else {              while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;
4509              !!!parse-error (type => 'in CDATA:#'.$token->{type});  
4510            }          $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
4511            ## ISSUE: And ignore?          ## Reprocess.
4512          }          next B;
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'select') {  
         $reconstruct_active_formatting_elements->($insert_to_current);  
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
           
         $self->{insertion_mode} = 'in select';  
         !!!next-token;  
         return;  
       } elsif ({  
                 caption => 1, col => 1, colgroup => 1, frame => 1,  
                 frameset => 1, head => 1, option => 1, optgroup => 1,  
                 tbody => 1, td => 1, tfoot => 1, th => 1,  
                 thead => 1, tr => 1,  
                }->{$token->{tag_name}}) {  
         !!!parse-error (type => 'in body:'.$token->{tag_name});  
         ## Ignore the token  
         !!!next-token;  
         return;  
           
         ## ISSUE: An issue on HTML5 new elements in the spec.  
4513        } else {        } else {
4514          $reconstruct_active_formatting_elements->($insert_to_current);          die "$0: $token->{type}: Unknown token type";        
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
           
         !!!next-token;  
         return;  
4515        }        }
4516      } elsif ($token->{type} eq 'end tag') {      }
       if ($token->{tag_name} eq 'body') {  
         if (@{$self->{open_elements}} > 1 and  
             $self->{open_elements}->[1]->[1] eq 'body') {  
           for (@{$self->{open_elements}}) {  
             unless ({  
                        dd => 1, dt => 1, li => 1, p => 1, td => 1,  
                        th => 1, tr => 1, body => 1, html => 1,  
                     }->{$_->[1]}) {  
               !!!parse-error (type => 'not closed:'.$_->[1]);  
             }  
           }  
4517    
4518            $self->{insertion_mode} = 'after body';      if ($self->{insertion_mode} & HEAD_IMS) {
4519            !!!next-token;        if ($token->{type} == CHARACTER_TOKEN) {
4520            return;          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
4521          } else {            unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4522            !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});              !!!cp ('t88.2');
4523            ## Ignore the token              $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
4524            !!!next-token;              #
4525            return;            } else {
4526          }              !!!cp ('t88.1');
4527        } elsif ($token->{tag_name} eq 'html') {              ## Ignore the token.
4528          if (@{$self->{open_elements}} > 1 and $self->{open_elements}->[1]->[1] eq 'body') {              #
           ## ISSUE: There is an issue in the spec.  
           if ($self->{open_elements}->[-1]->[1] ne 'body') {  
             !!!parse-error (type => 'not closed:'.$self->{open_elements}->[1]->[1]);  
           }  
           $self->{insertion_mode} = 'after body';  
           ## reprocess  
           return;  
         } else {  
           !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
           ## Ignore the token  
           !!!next-token;  
           return;  
         }  
       } elsif ({  
                 address => 1, blockquote => 1, center => 1, dir => 1,  
                 div => 1, dl => 1, fieldset => 1, listing => 1,  
                 menu => 1, ol => 1, pre => 1, ul => 1,  
                 p => 1,  
                 dd => 1, dt => 1, li => 1,  
                 button => 1, marquee => 1, object => 1,  
                }->{$token->{tag_name}}) {  
         ## has an element in scope  
         my $i;  
         INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
           my $node = $self->{open_elements}->[$_];  
           if ($node->[1] eq $token->{tag_name}) {  
             ## generate implied end tags  
             if ({  
                  dd => ($token->{tag_name} ne 'dd'),  
                  dt => ($token->{tag_name} ne 'dt'),  
                  li => ($token->{tag_name} ne 'li'),  
                  p => ($token->{tag_name} ne 'p'),  
                  td => 1, th => 1, tr => 1,  
                 }->{$self->{open_elements}->[-1]->[1]}) {  
               !!!back-token;  
               $token = {type => 'end tag',  
                         tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
               return;  
             }  
             $i = $_;  
             last INSCOPE unless $token->{tag_name} eq 'p';  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
             last INSCOPE;  
4529            }            }
4530          } # INSCOPE            unless (length $token->{data}) {
4531                        !!!cp ('t88');
4532          if ($self->{open_elements}->[-1]->[1] ne $token->{tag_name}) {              !!!next-token;
4533            !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);              next B;
         }  
           
         splice @{$self->{open_elements}}, $i if defined $i;  
         $clear_up_to_marker->()  
           if {  
             button => 1, marquee => 1, object => 1,  
           }->{$token->{tag_name}};  
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'form') {  
         ## has an element in scope  
         INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
           my $node = $self->{open_elements}->[$_];  
           if ($node->[1] eq $token->{tag_name}) {  
             ## generate implied end tags  
             if ({  
                  dd => 1, dt => 1, li => 1, p => 1,  
                  td => 1, th => 1, tr => 1,  
                 }->{$self->{open_elements}->[-1]->[1]}) {  
               !!!back-token;  
               $token = {type => 'end tag',  
                         tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
               return;  
             }  
             last INSCOPE;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
             last INSCOPE;  
4534            }            }
4535          } # INSCOPE  ## TODO: set $token->{column} appropriately
           
         if ($self->{open_elements}->[-1]->[1] eq $token->{tag_name}) {  
           pop @{$self->{open_elements}};  
         } else {  
           !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
4536          }          }
4537    
4538          undef $self->{form_element};          if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4539          !!!next-token;            !!!cp ('t89');
4540          return;            ## As if <head>
4541        } elsif ({            !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4542                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,            $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4543                 }->{$token->{tag_name}}) {            push @{$self->{open_elements}},
4544          ## has an element in scope                [$self->{head_element}, $el_category->{head}];
         my $i;  
         INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
           my $node = $self->{open_elements}->[$_];  
           if ({  
                h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,  
               }->{$node->[1]}) {  
             ## generate implied end tags  
             if ({  
                  dd => 1, dt => 1, li => 1, p => 1,  
                  td => 1, th => 1, tr => 1,  
                 }->{$self->{open_elements}->[-1]->[1]}) {  
               !!!back-token;  
               $token = {type => 'end tag',  
                         tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
               return;  
             }  
             $i = $_;  
             last INSCOPE;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
           
         if ($self->{open_elements}->[-1]->[1] ne $token->{tag_name}) {  
           !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
         }  
           
         splice @{$self->{open_elements}}, $i if defined $i;  
         !!!next-token;  
         return;  
       } elsif ({  
                 a => 1,  
                 b => 1, big => 1, em => 1, font => 1, i => 1,  
                 nobr => 1, s => 1, small => 1, strile => 1,  
                 strong => 1, tt => 1, u => 1,  
                }->{$token->{tag_name}}) {  
         $formatting_end_tag->($token->{tag_name});  
 ## TODO: <http://html5.org/tools/web-apps-tracker?from=883&to=884>  
         return;  
       } elsif ({  
                 caption => 1, col => 1, colgroup => 1, frame => 1,  
                 frameset => 1, head => 1, option => 1, optgroup => 1,  
                 tbody => 1, td => 1, tfoot => 1, th => 1,  
                 thead => 1, tr => 1,  
                 area => 1, basefont => 1, bgsound => 1, br => 1,  
                 embed => 1, hr => 1, iframe => 1, image => 1,  
                 img => 1, input => 1, isindex => 1, noembed => 1,  
                 noframes => 1, param => 1, select => 1, spacer => 1,  
                 table => 1, textarea => 1, wbr => 1,  
                 noscript => 0, ## TODO: if scripting is enabled  
                }->{$token->{tag_name}}) {  
         !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
         ## Ignore the token  
         !!!next-token;  
         return;  
           
         ## ISSUE: Issue on HTML5 new elements in spec  
           
       } else {  
         ## Step 1  
         my $node_i = -1;  
         my $node = $self->{open_elements}->[$node_i];  
4545    
4546          ## Step 2            ## Reprocess in the "in head" insertion mode...
4547          S2: {            pop @{$self->{open_elements}};
           if ($node->[1] eq $token->{tag_name}) {  
             ## Step 1  
             ## generate implied end tags  
             if ({  
                  dd => 1, dt => 1, li => 1, p => 1,  
                  td => 1, th => 1, tr => 1,  
                 }->{$self->{open_elements}->[-1]->[1]}) {  
               !!!back-token;  
               $token = {type => 'end tag',  
                         tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
               return;  
             }  
           
             ## Step 2  
             if ($token->{tag_name} ne $self->{open_elements}->[-1]->[1]) {  
               !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
             }  
               
             ## Step 3  
             splice @{$self->{open_elements}}, $node_i;  
4548    
4549              !!!next-token;            ## Reprocess in the "after head" insertion mode...
4550              last S2;          } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4551            } else {            !!!cp ('t90');
4552              ## Step 3            ## As if </noscript>
4553              if (not $formatting_category->{$node->[1]} and            pop @{$self->{open_elements}};
4554                  #not $phrasing_category->{$node->[1]} and            !!!parse-error (type => 'in noscript:#text', token => $token);
                 ($special_category->{$node->[1]} or  
                  $scoping_category->{$node->[1]})) {  
               !!!parse-error (type => 'not closed:'.$node->[1]);  
               ## Ignore the token  
               !!!next-token;  
               last S2;  
             }  
           }  
             
           ## Step 4  
           $node_i--;  
           $node = $self->{open_elements}->[$node_i];  
4555                        
4556            ## Step 5;            ## Reprocess in the "in head" insertion mode...
4557            redo S2;            ## As if </head>
4558          } # S2            pop @{$self->{open_elements}};
         return;  
       }  
     }  
   }; # $in_body  
4559    
4560    B: {            ## Reprocess in the "after head" insertion mode...
4561      if ($phase eq 'main') {          } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4562        if ($token->{type} eq 'DOCTYPE') {            !!!cp ('t91');
4563          !!!parse-error (type => 'in html:#DOCTYPE');            pop @{$self->{open_elements}};
         ## Ignore the token  
         ## Stay in the phase  
         !!!next-token;  
         redo B;  
       } elsif ($token->{type} eq 'start tag' and  
                $token->{tag_name} eq 'html') {  
         ## TODO: unless it is the first start tag token, parse-error  
         my $top_el = $self->{open_elements}->[0]->[0];  
         for my $attr_name (keys %{$token->{attributes}}) {  
           unless ($top_el->has_attribute_ns (undef, $attr_name)) {  
             $top_el->set_attribute_ns  
               (undef, [undef, $attr_name],  
                $token->{attributes}->{$attr_name}->{value});  
           }  
         }  
         !!!next-token;  
         redo B;  
       } elsif ($token->{type} eq 'end-of-file') {  
         ## Generate implied end tags  
         if ({  
              dd => 1, dt => 1, li => 1, p => 1, td => 1, th => 1, tr => 1,  
             }->{$self->{open_elements}->[-1]->[1]}) {  
           !!!back-token;  
           $token = {type => 'end tag', tag_name => $self->{open_elements}->[-1]->[1]};  
           redo B;  
         }  
           
         if (@{$self->{open_elements}} > 2 or  
             (@{$self->{open_elements}} == 2 and $self->{open_elements}->[1]->[1] ne 'body')) {  
           !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
         } elsif (defined $self->{inner_html_node} and  
                  @{$self->{open_elements}} > 1 and  
                  $self->{open_elements}->[1]->[1] ne 'body') {  
           !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
         }  
4564    
4565          ## Stop parsing            ## Reprocess in the "after head" insertion mode...
4566          last B;          } else {
4567              !!!cp ('t92');
4568            }
4569    
4570          ## ISSUE: There is an issue in the spec.          ## "after head" insertion mode
4571        } else {          ## As if <body>
4572          if ($self->{insertion_mode} eq 'before head') {          !!!insert-element ('body',, $token);
4573            if ($token->{type} eq 'character') {          $self->{insertion_mode} = IN_BODY_IM;
4574              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          ## reprocess
4575                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);          next B;
4576                unless (length $token->{data}) {        } elsif ($token->{type} == START_TAG_TOKEN) {
4577                  !!!next-token;          if ($token->{tag_name} eq 'head') {
4578                  redo B;            if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4579                }              !!!cp ('t93');
4580              }              !!!create-element ($self->{head_element}, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);
4581              ## As if <head>              $self->{open_elements}->[-1]->[0]->append_child
4582              !!!create-element ($self->{head_element}, 'head');                  ($self->{head_element});
4583              $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});              push @{$self->{open_elements}},
4584              push @{$self->{open_elements}}, [$self->{head_element}, 'head'];                  [$self->{head_element}, $el_category->{head}];
4585              $self->{insertion_mode} = 'in head';              $self->{insertion_mode} = IN_HEAD_IM;
4586              ## reprocess              !!!nack ('t93.1');
             redo B;  
           } elsif ($token->{type} eq 'comment') {  
             my $comment = $self->{document}->create_comment ($token->{data});  
             $self->{open_elements}->[-1]->[0]->append_child ($comment);  
4587              !!!next-token;              !!!next-token;
4588              redo B;              next B;
4589            } elsif ($token->{type} eq 'start tag') {            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4590              my $attr = $token->{tag_name} eq 'head' ? $token->{attributes} : {};              !!!cp ('t93.2');
4591              !!!create-element ($self->{head_element}, 'head', $attr);              !!!parse-error (type => 'after head', text => 'head',
4592              $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});                              token => $token);
4593              push @{$self->{open_elements}}, [$self->{head_element}, 'head'];              ## Ignore the token
4594              $self->{insertion_mode} = 'in head';              !!!nack ('t93.3');
4595              if ($token->{tag_name} eq 'head') {              !!!next-token;
4596                !!!next-token;              next B;
             #} elsif ({  
             #          base => 1, link => 1, meta => 1,  
             #          script => 1, style => 1, title => 1,  
             #         }->{$token->{tag_name}}) {  
             #  ## reprocess  
             } else {  
               ## reprocess  
             }  
             redo B;  
           } elsif ($token->{type} eq 'end tag') {  
             if ({head => 1, body => 1, html => 1}->{$token->{tag_name}}) {  
               ## As if <head>  
               !!!create-element ($self->{head_element}, 'head');  
               $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});  
               push @{$self->{open_elements}}, [$self->{head_element}, 'head'];  
               $self->{insertion_mode} = 'in head';  
               ## reprocess  
               redo B;  
             } else {  
               !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
               ## Ignore the token ## ISSUE: An issue in the spec.  
               !!!next-token;  
               redo B;  
             }  
4597            } else {            } else {
4598              die "$0: $token->{type}: Unknown type";              !!!cp ('t95');
4599            }              !!!parse-error (type => 'in head:head',
4600          } elsif ($self->{insertion_mode} eq 'in head') {                              token => $token); # or in head noscript
4601            if ($token->{type} eq 'character') {              ## Ignore the token
4602              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {              !!!nack ('t95.1');
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);  
               unless (length $token->{data}) {  
                 !!!next-token;  
                 redo B;  
               }  
             }  
               
             #  
           } elsif ($token->{type} eq 'comment') {  
             my $comment = $self->{document}->create_comment ($token->{data});  
             $self->{open_elements}->[-1]->[0]->append_child ($comment);  
4603              !!!next-token;              !!!next-token;
4604              redo B;              next B;
4605            } elsif ($token->{type} eq 'start tag') {            }
4606              if ($token->{tag_name} eq 'title') {          } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4607                ## NOTE: There is an "as if in head" code clone            !!!cp ('t96');
4608                my $title_el;            ## As if <head>
4609                !!!create-element ($title_el, 'title', $token->{attributes});            !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4610                (defined $self->{head_element} ? $self->{head_element} : $self->{open_elements}->[-1]->[0])            $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4611                  ->append_child ($title_el);            push @{$self->{open_elements}},
4612                $self->{content_model_flag} = 'RCDATA';                [$self->{head_element}, $el_category->{head}];
4613                delete $self->{escape}; # MUST  
4614              $self->{insertion_mode} = IN_HEAD_IM;
4615                my $text = '';            ## Reprocess in the "in head" insertion mode...
4616                !!!next-token;          } else {
4617                while ($token->{type} eq 'character') {            !!!cp ('t97');
4618                  $text .= $token->{data};          }
4619                  !!!next-token;  
4620                }              if ($token->{tag_name} eq 'base') {
4621                if (length $text) {                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4622                  $title_el->manakai_append_text ($text);                  !!!cp ('t98');
4623                }                  ## As if </noscript>
4624                                  pop @{$self->{open_elements}};
4625                $self->{content_model_flag} = 'PCDATA';                  !!!parse-error (type => 'in noscript', text => 'base',
4626                                    token => $token);
4627                                
4628                if ($token->{type} eq 'end tag' and                  $self->{insertion_mode} = IN_HEAD_IM;
4629                    $token->{tag_name} eq 'title') {                  ## Reprocess in the "in head" insertion mode...
                 ## Ignore the token  
               } else {  
                 !!!parse-error (type => 'in RCDATA:#'.$token->{type});  
                 ## ISSUE: And ignore?  
               }  
               !!!next-token;  
               redo B;  
             } elsif ($token->{tag_name} eq 'style') {  
               $style_start_tag->();  
               redo B;  
             } elsif ($token->{tag_name} eq 'script') {  
               $script_start_tag->();  
               redo B;  
             } elsif ({base => 1, link => 1, meta => 1}->{$token->{tag_name}}) {  
               ## NOTE: There are "as if in head" code clones  
               my $el;  
               !!!create-element ($el, $token->{tag_name}, $token->{attributes});  
               if ($self->{insertion_mode} eq 'in head' and  
                   defined $self->{head_element}) {  
                 $self->{head_element}->append_child ($el);  
4630                } else {                } else {
4631                  $self->{open_elements}->[-1]->[0]->append_child ($el);                  !!!cp ('t99');
4632                }                }
4633    
4634                !!!next-token;                ## NOTE: There is a "as if in head" code clone.
4635                redo B;                if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4636              } elsif ($token->{tag_name} eq 'head') {                  !!!cp ('t100');
4637                !!!parse-error (type => 'in head:head');                  !!!parse-error (type => 'after head',
4638                ## Ignore the token                                  text => $token->{tag_name}, token => $token);
4639                !!!next-token;                  push @{$self->{open_elements}},
4640                redo B;                      [$self->{head_element}, $el_category->{head}];
             } else {  
               #  
             }  
           } elsif ($token->{type} eq 'end tag') {  
             if ($token->{tag_name} eq 'head') {  
               if ($self->{open_elements}->[-1]->[1] eq 'head') {  
                 pop @{$self->{open_elements}};  
4641                } else {                } else {
4642                  !!!parse-error (type => 'unmatched end tag:head');                  !!!cp ('t101');
4643                }                }
4644                $self->{insertion_mode} = 'after head';                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4645                  pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
4646                  pop @{$self->{open_elements}} # <head>
4647                      if $self->{insertion_mode} == AFTER_HEAD_IM;
4648                  !!!nack ('t101.1');
4649                !!!next-token;                !!!next-token;
4650                redo B;                next B;
4651              } elsif ($token->{tag_name} eq 'body' or              } elsif ($token->{tag_name} eq 'link') {
4652                       $token->{tag_name} eq 'html') {                ## NOTE: There is a "as if in head" code clone.
4653                #                if ($self->{insertion_mode} == AFTER_HEAD_IM) {
4654              } else {                  !!!cp ('t102');
4655                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                  !!!parse-error (type => 'after head',
4656                ## Ignore the token                                  text => $token->{tag_name}, token => $token);
4657                !!!next-token;                  push @{$self->{open_elements}},
4658                redo B;                      [$self->{head_element}, $el_category->{head}];
4659              }                } else {
4660            } else {                  !!!cp ('t103');
             #  
           }  
   
           if ($self->{open_elements}->[-1]->[1] eq 'head') {  
             ## As if </head>  
             pop @{$self->{open_elements}};  
           }  
           $self->{insertion_mode} = 'after head';  
           ## reprocess  
           redo B;  
   
           ## ISSUE: An issue in the spec.  
         } elsif ($self->{insertion_mode} eq 'after head') {  
           if ($token->{type} eq 'character') {  
             if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {  
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);  
               unless (length $token->{data}) {  
                 !!!next-token;  
                 redo B;  
4661                }                }
4662              }                !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
4663                              pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
4664              #                pop @{$self->{open_elements}} # <head>
4665            } elsif ($token->{type} eq 'comment') {                    if $self->{insertion_mode} == AFTER_HEAD_IM;