/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.22 by wakaba, Sat Jun 23 14:55:45 2007 UTC revision 1.237 by wakaba, Sun Sep 6 08:29:32 2009 UTC
# Line 1  Line 1 
1  package Whatpm::HTML;  package Whatpm::HTML;
2  use strict;  use strict;
3  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};
4    use Error qw(:try);
5    
6    use Whatpm::HTML::Tokenizer;
7    
8    ## NOTE: This module don't check all HTML5 parse errors; character
9    ## encoding related parse errors are expected to be handled by relevant
10    ## modules.
11    ## Parse errors for control characters that are not allowed in HTML5
12    ## documents, for surrogate code points, and for noncharacter code
13    ## points, as well as U+FFFD substitions for characters whose code points
14    ## is higher than U+10FFFF may be detected by combining the parser with
15    ## the checker implemented by Whatpm::Charset::UnicodeChecker (for its
16    ## usage example, see |t/HTML-tree.t| in the Whatpm package or the
17    ## WebHACC::Language::HTML module in the WebHACC package).
18    
19  ## ISSUE:  ## ISSUE:
20  ## var doc = implementation.createDocument (null, null, null);  ## var doc = implementation.createDocument (null, null, null);
21  ## doc.write ('');  ## doc.write ('');
22  ## alert (doc.compatMode);  ## alert (doc.compatMode);
23    
24  my $permitted_slash_tag_name = {  require IO::Handle;
   base => 1,  
   link => 1,  
   meta => 1,  
   hr => 1,  
   br => 1,  
   img=> 1,  
   embed => 1,  
   param => 1,  
   area => 1,  
   col => 1,  
   input => 1,  
 };  
25    
26  my $c1_entity_char = {  ## Namespace URLs
27    0x80 => 0x20AC,  
28    0x81 => 0xFFFD,  my $HTML_NS = q<http://www.w3.org/1999/xhtml>;
29    0x82 => 0x201A,  my $MML_NS = q<http://www.w3.org/1998/Math/MathML>;
30    0x83 => 0x0192,  my $SVG_NS = q<http://www.w3.org/2000/svg>;
31    0x84 => 0x201E,  my $XLINK_NS = q<http://www.w3.org/1999/xlink>;
32    0x85 => 0x2026,  my $XML_NS = q<http://www.w3.org/XML/1998/namespace>;
33    0x86 => 0x2020,  my $XMLNS_NS = q<http://www.w3.org/2000/xmlns/>;
34    0x87 => 0x2021,  
35    0x88 => 0x02C6,  ## Element categories
36    0x89 => 0x2030,  
37    0x8A => 0x0160,  ## Bits 12-15
38    0x8B => 0x2039,  sub SPECIAL_EL () { 0b1_000000000000000 }
39    0x8C => 0x0152,  sub SCOPING_EL () { 0b1_00000000000000 }
40    0x8D => 0xFFFD,  sub FORMATTING_EL () { 0b1_0000000000000 }
41    0x8E => 0x017D,  sub PHRASING_EL () { 0b1_000000000000 }
42    0x8F => 0xFFFD,  
43    0x90 => 0xFFFD,  ## Bits 10-11
44    0x91 => 0x2018,  #sub FOREIGN_EL () { 0b1_00000000000 } # see Whatpm::HTML::Tokenizer
45    0x92 => 0x2019,  sub FOREIGN_FLOW_CONTENT_EL () { 0b1_0000000000 }
46    0x93 => 0x201C,  
47    0x94 => 0x201D,  ## Bits 6-9
48    0x95 => 0x2022,  sub TABLE_SCOPING_EL () { 0b1_000000000 }
49    0x96 => 0x2013,  sub TABLE_ROWS_SCOPING_EL () { 0b1_00000000 }
50    0x97 => 0x2014,  sub TABLE_ROW_SCOPING_EL () { 0b1_0000000 }
51    0x98 => 0x02DC,  sub TABLE_ROWS_EL () { 0b1_000000 }
52    0x99 => 0x2122,  
53    0x9A => 0x0161,  ## Bit 5
54    0x9B => 0x203A,  sub ADDRESS_DIV_P_EL () { 0b1_00000 }
55    0x9C => 0x0153,  
56    0x9D => 0xFFFD,  ## NOTE: Used in </body> and EOF algorithms.
57    0x9E => 0x017E,  ## Bit 4
58    0x9F => 0x0178,  sub ALL_END_TAG_OPTIONAL_EL () { 0b1_0000 }
59  }; # $c1_entity_char  
60    ## NOTE: Used in "generate implied end tags" algorithm.
61  my $special_category = {  ## NOTE: There is a code where a modified version of
62    address => 1, area => 1, base => 1, basefont => 1, bgsound => 1,  ## END_TAG_OPTIONAL_EL is used in "generate implied end tags"
63    blockquote => 1, body => 1, br => 1, center => 1, col => 1, colgroup => 1,  ## implementation (search for the algorithm name).
64    dd => 1, dir => 1, div => 1, dl => 1, dt => 1, embed => 1, fieldset => 1,  ## Bit 3
65    form => 1, frame => 1, frameset => 1, h1 => 1, h2 => 1, h3 => 1,  sub END_TAG_OPTIONAL_EL () { 0b1_000 }
66    h4 => 1, h5 => 1, h6 => 1, head => 1, hr => 1, iframe => 1, image => 1,  
67    img => 1, input => 1, isindex => 1, li => 1, link => 1, listing => 1,  ## Bits 0-2
68    menu => 1, meta => 1, noembed => 1, noframes => 1, noscript => 1,  
69    ol => 1, optgroup => 1, option => 1, p => 1, param => 1, plaintext => 1,  sub MISC_SPECIAL_EL () { SPECIAL_EL | 0b000 }
70    pre => 1, script => 1, select => 1, spacer => 1, style => 1, tbody => 1,  sub FORM_EL () { SPECIAL_EL | 0b001 }
71    textarea => 1, tfoot => 1, thead => 1, title => 1, tr => 1, ul => 1, wbr => 1,  sub FRAMESET_EL () { SPECIAL_EL | 0b010 }
72  };  sub HEADING_EL () { SPECIAL_EL | 0b011 }
73  my $scoping_category = {  sub SELECT_EL () { SPECIAL_EL | 0b100 }
74    button => 1, caption => 1, html => 1, marquee => 1, object => 1,  sub SCRIPT_EL () { SPECIAL_EL | 0b101 }
75    table => 1, td => 1, th => 1,  
76    sub ADDRESS_DIV_EL () { SPECIAL_EL | ADDRESS_DIV_P_EL | 0b001 }
77    sub BODY_EL () { SPECIAL_EL | ALL_END_TAG_OPTIONAL_EL | 0b001 }
78    
79    sub DTDD_EL () {
80      SPECIAL_EL |
81      END_TAG_OPTIONAL_EL |
82      ALL_END_TAG_OPTIONAL_EL |
83      0b010
84    }
85    sub LI_EL () {
86      SPECIAL_EL |
87      END_TAG_OPTIONAL_EL |
88      ALL_END_TAG_OPTIONAL_EL |
89      0b100
90    }
91    sub P_EL () {
92      SPECIAL_EL |
93      ADDRESS_DIV_P_EL |
94      END_TAG_OPTIONAL_EL |
95      ALL_END_TAG_OPTIONAL_EL |
96      0b001
97    }
98    
99    sub TABLE_ROW_EL () {
100      SPECIAL_EL |
101      TABLE_ROWS_EL |
102      TABLE_ROW_SCOPING_EL |
103      ALL_END_TAG_OPTIONAL_EL |
104      0b001
105    }
106    sub TABLE_ROW_GROUP_EL () {
107      SPECIAL_EL |
108      TABLE_ROWS_EL |
109      TABLE_ROWS_SCOPING_EL |
110      ALL_END_TAG_OPTIONAL_EL |
111      0b001
112    }
113    
114    sub MISC_SCOPING_EL () { SCOPING_EL | 0b000 }
115    sub BUTTON_EL () { SCOPING_EL | 0b001 }
116    sub CAPTION_EL () { SCOPING_EL | 0b010 }
117    sub HTML_EL () {
118      SCOPING_EL |
119      TABLE_SCOPING_EL |
120      TABLE_ROWS_SCOPING_EL |
121      TABLE_ROW_SCOPING_EL |
122      ALL_END_TAG_OPTIONAL_EL |
123      0b001
124    }
125    sub TABLE_EL () {
126      SCOPING_EL |
127      TABLE_ROWS_EL |
128      TABLE_SCOPING_EL |
129      0b001
130    }
131    sub TABLE_CELL_EL () {
132      SCOPING_EL |
133      TABLE_ROW_SCOPING_EL |
134      ALL_END_TAG_OPTIONAL_EL |
135      0b001
136    }
137    
138    sub MISC_FORMATTING_EL () { FORMATTING_EL | 0b000 }
139    sub A_EL () { FORMATTING_EL | 0b001 }
140    sub NOBR_EL () { FORMATTING_EL | 0b010 }
141    
142    sub RUBY_EL () { PHRASING_EL | 0b001 }
143    
144    ## ISSUE: ALL_END_TAG_OPTIONAL_EL?
145    sub OPTGROUP_EL () { PHRASING_EL | END_TAG_OPTIONAL_EL | 0b001 }
146    sub OPTION_EL () { PHRASING_EL | END_TAG_OPTIONAL_EL | 0b010 }
147    sub RUBY_COMPONENT_EL () { PHRASING_EL | END_TAG_OPTIONAL_EL | 0b100 }
148    
149    sub MML_AXML_EL () { PHRASING_EL | FOREIGN_EL | 0b001 }
150    
151    my $el_category = {
152      a => A_EL,
153      address => ADDRESS_DIV_EL,
154      applet => MISC_SCOPING_EL,
155      area => MISC_SPECIAL_EL,
156      article => MISC_SPECIAL_EL,
157      aside => MISC_SPECIAL_EL,
158      b => FORMATTING_EL,
159      base => MISC_SPECIAL_EL,
160      basefont => MISC_SPECIAL_EL,
161      bgsound => MISC_SPECIAL_EL,
162      big => FORMATTING_EL,
163      blockquote => MISC_SPECIAL_EL,
164      body => BODY_EL,
165      br => MISC_SPECIAL_EL,
166      button => BUTTON_EL,
167      caption => CAPTION_EL,
168      center => MISC_SPECIAL_EL,
169      col => MISC_SPECIAL_EL,
170      colgroup => MISC_SPECIAL_EL,
171      command => MISC_SPECIAL_EL,
172      datagrid => MISC_SPECIAL_EL,
173      dd => DTDD_EL,
174      details => MISC_SPECIAL_EL,
175      dialog => MISC_SPECIAL_EL,
176      dir => MISC_SPECIAL_EL,
177      div => ADDRESS_DIV_EL,
178      dl => MISC_SPECIAL_EL,
179      dt => DTDD_EL,
180      em => FORMATTING_EL,
181      embed => MISC_SPECIAL_EL,
182      fieldset => MISC_SPECIAL_EL,
183      figure => MISC_SPECIAL_EL,
184      font => FORMATTING_EL,
185      footer => MISC_SPECIAL_EL,
186      form => FORM_EL,
187      frame => MISC_SPECIAL_EL,
188      frameset => FRAMESET_EL,
189      h1 => HEADING_EL,
190      h2 => HEADING_EL,
191      h3 => HEADING_EL,
192      h4 => HEADING_EL,
193      h5 => HEADING_EL,
194      h6 => HEADING_EL,
195      head => MISC_SPECIAL_EL,
196      header => MISC_SPECIAL_EL,
197      hgroup => MISC_SPECIAL_EL,
198      hr => MISC_SPECIAL_EL,
199      html => HTML_EL,
200      i => FORMATTING_EL,
201      iframe => MISC_SPECIAL_EL,
202      img => MISC_SPECIAL_EL,
203      #image => MISC_SPECIAL_EL, ## NOTE: Commented out in the spec.
204      input => MISC_SPECIAL_EL,
205      isindex => MISC_SPECIAL_EL,
206      ## XXX keygen? (Whether a void element is in Special or not does not
207      ## affect to the processing, however.)
208      li => LI_EL,
209      link => MISC_SPECIAL_EL,
210      listing => MISC_SPECIAL_EL,
211      marquee => MISC_SCOPING_EL,
212      menu => MISC_SPECIAL_EL,
213      meta => MISC_SPECIAL_EL,
214      nav => MISC_SPECIAL_EL,
215      nobr => NOBR_EL,
216      noembed => MISC_SPECIAL_EL,
217      noframes => MISC_SPECIAL_EL,
218      noscript => MISC_SPECIAL_EL,
219      object => MISC_SCOPING_EL,
220      ol => MISC_SPECIAL_EL,
221      optgroup => OPTGROUP_EL,
222      option => OPTION_EL,
223      p => P_EL,
224      param => MISC_SPECIAL_EL,
225      plaintext => MISC_SPECIAL_EL,
226      pre => MISC_SPECIAL_EL,
227      rp => RUBY_COMPONENT_EL,
228      rt => RUBY_COMPONENT_EL,
229      ruby => RUBY_EL,
230      s => FORMATTING_EL,
231      script => MISC_SPECIAL_EL,
232      select => SELECT_EL,
233      section => MISC_SPECIAL_EL,
234      small => FORMATTING_EL,
235      spacer => MISC_SPECIAL_EL,
236      strike => FORMATTING_EL,
237      strong => FORMATTING_EL,
238      style => MISC_SPECIAL_EL,
239      table => TABLE_EL,
240      tbody => TABLE_ROW_GROUP_EL,
241      td => TABLE_CELL_EL,
242      textarea => MISC_SPECIAL_EL,
243      tfoot => TABLE_ROW_GROUP_EL,
244      th => TABLE_CELL_EL,
245      thead => TABLE_ROW_GROUP_EL,
246      title => MISC_SPECIAL_EL,
247      tr => TABLE_ROW_EL,
248      tt => FORMATTING_EL,
249      u => FORMATTING_EL,
250      ul => MISC_SPECIAL_EL,
251      wbr => MISC_SPECIAL_EL,
252      xmp => MISC_SPECIAL_EL,
253  };  };
254  my $formatting_category = {  
255    a => 1, b => 1, big => 1, em => 1, font => 1, i => 1, nobr => 1,  my $el_category_f = {
256    s => 1, small => 1, strile => 1, strong => 1, tt => 1, u => 1,    $MML_NS => {
257        'annotation-xml' => MML_AXML_EL,
258        mi => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
259        mo => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
260        mn => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
261        ms => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
262        mtext => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
263      },
264      $SVG_NS => {
265        foreignObject => SCOPING_EL | FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
266        desc => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
267        title => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL,
268      },
269      ## NOTE: In addition, FOREIGN_EL is set to non-HTML elements.
270  };  };
 # $phrasing_category: all other elements  
271    
272  sub parse_string ($$$;$) {  my $svg_attr_name = {
273    my $self = shift->new;    attributename => 'attributeName',
274    my $s = \$_[0];    attributetype => 'attributeType',
275    $self->{document} = $_[1];    basefrequency => 'baseFrequency',
276      baseprofile => 'baseProfile',
277      calcmode => 'calcMode',
278      clippathunits => 'clipPathUnits',
279      contentscripttype => 'contentScriptType',
280      contentstyletype => 'contentStyleType',
281      diffuseconstant => 'diffuseConstant',
282      edgemode => 'edgeMode',
283      externalresourcesrequired => 'externalResourcesRequired',
284      filterres => 'filterRes',
285      filterunits => 'filterUnits',
286      glyphref => 'glyphRef',
287      gradienttransform => 'gradientTransform',
288      gradientunits => 'gradientUnits',
289      kernelmatrix => 'kernelMatrix',
290      kernelunitlength => 'kernelUnitLength',
291      keypoints => 'keyPoints',
292      keysplines => 'keySplines',
293      keytimes => 'keyTimes',
294      lengthadjust => 'lengthAdjust',
295      limitingconeangle => 'limitingConeAngle',
296      markerheight => 'markerHeight',
297      markerunits => 'markerUnits',
298      markerwidth => 'markerWidth',
299      maskcontentunits => 'maskContentUnits',
300      maskunits => 'maskUnits',
301      numoctaves => 'numOctaves',
302      pathlength => 'pathLength',
303      patterncontentunits => 'patternContentUnits',
304      patterntransform => 'patternTransform',
305      patternunits => 'patternUnits',
306      pointsatx => 'pointsAtX',
307      pointsaty => 'pointsAtY',
308      pointsatz => 'pointsAtZ',
309      preservealpha => 'preserveAlpha',
310      preserveaspectratio => 'preserveAspectRatio',
311      primitiveunits => 'primitiveUnits',
312      refx => 'refX',
313      refy => 'refY',
314      repeatcount => 'repeatCount',
315      repeatdur => 'repeatDur',
316      requiredextensions => 'requiredExtensions',
317      requiredfeatures => 'requiredFeatures',
318      specularconstant => 'specularConstant',
319      specularexponent => 'specularExponent',
320      spreadmethod => 'spreadMethod',
321      startoffset => 'startOffset',
322      stddeviation => 'stdDeviation',
323      stitchtiles => 'stitchTiles',
324      surfacescale => 'surfaceScale',
325      systemlanguage => 'systemLanguage',
326      tablevalues => 'tableValues',
327      targetx => 'targetX',
328      targety => 'targetY',
329      textlength => 'textLength',
330      viewbox => 'viewBox',
331      viewtarget => 'viewTarget',
332      xchannelselector => 'xChannelSelector',
333      ychannelselector => 'yChannelSelector',
334      zoomandpan => 'zoomAndPan',
335    };
336    
337    ## NOTE: |set_inner_html| copies most of this method's code  my $foreign_attr_xname = {
338      'xlink:actuate' => [$XLINK_NS, ['xlink', 'actuate']],
339      'xlink:arcrole' => [$XLINK_NS, ['xlink', 'arcrole']],
340      'xlink:href' => [$XLINK_NS, ['xlink', 'href']],
341      'xlink:role' => [$XLINK_NS, ['xlink', 'role']],
342      'xlink:show' => [$XLINK_NS, ['xlink', 'show']],
343      'xlink:title' => [$XLINK_NS, ['xlink', 'title']],
344      'xlink:type' => [$XLINK_NS, ['xlink', 'type']],
345      'xml:base' => [$XML_NS, ['xml', 'base']],
346      'xml:lang' => [$XML_NS, ['xml', 'lang']],
347      'xml:space' => [$XML_NS, ['xml', 'space']],
348      'xmlns' => [$XMLNS_NS, [undef, 'xmlns']],
349      'xmlns:xlink' => [$XMLNS_NS, ['xmlns', 'xlink']],
350    };
351    
352    my $i = 0;  ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error.
   my $line = 1;  
   my $column = 0;  
   $self->{set_next_input_character} = sub {  
     my $self = shift;  
353    
354      pop @{$self->{prev_input_character}};  ## TODO: Invoke the reset algorithm when a resettable element is
355      unshift @{$self->{prev_input_character}}, $self->{next_input_character};  ## created (cf. HTML5 revision 2259).
356    
357      $self->{next_input_character} = -1 and return if $i >= length $$s;  sub parse_byte_string ($$$$;$) {
358      $self->{next_input_character} = ord substr $$s, $i++, 1;    my $self = shift;
359      $column++;    my $charset_name = shift;
360          open my $input, '<', ref $_[0] ? $_[0] : \($_[0]);
361      if ($self->{next_input_character} == 0x000A) { # LF    return $self->parse_byte_stream ($charset_name, $input, @_[1..$#_]);
362        $line++;  } # parse_byte_string
363        $column = 0;  
364      } elsif ($self->{next_input_character} == 0x000D) { # CR  sub parse_byte_stream ($$$$;$$) {
365        $i++ if substr ($$s, $i, 1) eq "\x0A";    # my ($self, $charset_name, $byte_stream, $doc, $onerror, $get_wrapper) = @_;
366        $self->{next_input_character} = 0x000A; # LF # MUST    my $self = ref $_[0] ? shift : shift->new;
367        $line++;    my $charset_name = shift;
368        $column = 0;    my $byte_stream = $_[0];
     } elsif ($self->{next_input_character} > 0x10FFFF) {  
       $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST  
     } elsif ($self->{next_input_character} == 0x0000) { # NULL  
       !!!parse-error (type => 'NULL');  
       $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST  
     }  
   };  
   $self->{prev_input_character} = [-1, -1, -1];  
   $self->{next_input_character} = -1;  
369    
370    my $onerror = $_[2] || sub {    my $onerror = $_[2] || sub {
371      my (%opt) = @_;      my (%opt) = @_;
372      warn "Parse error ($opt{type}) at line $opt{line} column $opt{column}\n";      warn "Parse error ($opt{type})\n";
   };  
   $self->{parse_error} = sub {  
     $onerror->(@_, line => $line, column => $column);  
373    };    };
374      $self->{parse_error} = $onerror; # updated later by parse_char_string
375    
376    $self->_initialize_tokenizer;    my $get_wrapper = $_[3] || sub ($) {
377    $self->_initialize_tree_constructor;      return $_[0]; # $_[0] = byte stream handle, returned = arg to char handle
   $self->_construct_tree;  
   $self->_terminate_tree_constructor;  
   
   return $self->{document};  
 } # parse_string  
   
 sub new ($) {  
   my $class = shift;  
   my $self = bless {}, $class;  
   $self->{set_next_input_character} = sub {  
     $self->{next_input_character} = -1;  
   };  
   $self->{parse_error} = sub {  
     #  
378    };    };
   return $self;  
 } # new  
   
 ## Implementations MUST act as if state machine in the spec  
   
 sub _initialize_tokenizer ($) {  
   my $self = shift;  
   $self->{state} = 'data'; # MUST  
   $self->{content_model_flag} = 'PCDATA'; # be  
   undef $self->{current_token}; # start tag, end tag, comment, or DOCTYPE  
   undef $self->{current_attribute};  
   undef $self->{last_emitted_start_tag_name};  
   undef $self->{last_attribute_value_state};  
   $self->{char} = [];  
   # $self->{next_input_character}  
   !!!next-input-character;  
   $self->{token} = [];  
   # $self->{escape}  
 } # _initialize_tokenizer  
   
 ## A token has:  
 ##   ->{type} eq 'DOCTYPE', 'start tag', 'end tag', 'comment',  
 ##       'character', or 'end-of-file'  
 ##   ->{name} (DOCTYPE, start tag (tag name), end tag (tag name))  
 ##   ->{public_identifier} (DOCTYPE)  
 ##   ->{system_identifier} (DOCTYPE)  
 ##   ->{correct} == 1 or 0 (DOCTYPE)  
 ##   ->{attributes} isa HASH (start tag, end tag)  
 ##   ->{data} (comment, character)  
   
 ## Emitted token MUST immediately be handled by the tree construction state.  
   
 ## Before each step, UA MAY check to see if either one of the scripts in  
 ## "list of scripts that will execute as soon as possible" or the first  
 ## script in the "list of scripts that will execute asynchronously",  
 ## has completed loading.  If one has, then it MUST be executed  
 ## and removed from the list.  
   
 sub _get_next_token ($) {  
   my $self = shift;  
   if (@{$self->{token}}) {  
     return shift @{$self->{token}};  
   }  
   
   A: {  
     if ($self->{state} eq 'data') {  
       if ($self->{next_input_character} == 0x0026) { # &  
         if ($self->{content_model_flag} eq 'PCDATA' or  
             $self->{content_model_flag} eq 'RCDATA') {  
           $self->{state} = 'entity data';  
           !!!next-input-character;  
           redo A;  
         } else {  
           #  
         }  
       } elsif ($self->{next_input_character} == 0x002D) { # -  
         if ($self->{content_model_flag} eq 'RCDATA' or  
             $self->{content_model_flag} eq 'CDATA') {  
           unless ($self->{escape}) {  
             if ($self->{prev_input_character}->[0] == 0x002D and # -  
                 $self->{prev_input_character}->[1] == 0x0021 and # !  
                 $self->{prev_input_character}->[2] == 0x003C) { # <  
               $self->{escape} = 1;  
             }  
           }  
         }  
           
         #  
       } elsif ($self->{next_input_character} == 0x003C) { # <  
         if ($self->{content_model_flag} eq 'PCDATA' or  
             (($self->{content_model_flag} eq 'CDATA' or  
               $self->{content_model_flag} eq 'RCDATA') and  
              not $self->{escape})) {  
           $self->{state} = 'tag open';  
           !!!next-input-character;  
           redo A;  
         } else {  
           #  
         }  
       } elsif ($self->{next_input_character} == 0x003E) { # >  
         if ($self->{escape} and  
             ($self->{content_model_flag} eq 'RCDATA' or  
              $self->{content_model_flag} eq 'CDATA')) {  
           if ($self->{prev_input_character}->[0] == 0x002D and # -  
               $self->{prev_input_character}->[1] == 0x002D) { # -  
             delete $self->{escape};  
           }  
         }  
           
         #  
       } elsif ($self->{next_input_character} == -1) {  
         !!!emit ({type => 'end-of-file'});  
         last A; ## TODO: ok?  
       }  
       # Anything else  
       my $token = {type => 'character',  
                    data => chr $self->{next_input_character}};  
       ## Stay in the data state  
       !!!next-input-character;  
   
       !!!emit ($token);  
   
       redo A;  
     } elsif ($self->{state} eq 'entity data') {  
       ## (cannot happen in CDATA state)  
         
       my $token = $self->_tokenize_attempt_to_consume_an_entity;  
379    
380        $self->{state} = 'data';    ## HTML5 encoding sniffing algorithm
381        # next-input-character is already done    require Message::Charset::Info;
382      my $charset;
383      my $buffer;
384      my ($char_stream, $e_status);
385    
386      SNIFFING: {
387        ## NOTE: By setting |allow_fallback| option true when the
388        ## |get_decode_handle| method is invoked, we ignore what the HTML5
389        ## spec requires, i.e. unsupported encoding should be ignored.
390          ## TODO: We should not do this unless the parser is invoked
391          ## in the conformance checking mode, in which this behavior
392          ## would be useful.
393    
394        unless (defined $token) {      ## Step 1
395          !!!emit ({type => 'character', data => '&'});      if (defined $charset_name) {
396          $charset = Message::Charset::Info->get_by_html_name ($charset_name);
397              ## TODO: Is this ok?  Transfer protocol's parameter should be
398              ## interpreted in its semantics?
399    
400          ($char_stream, $e_status) = $charset->get_decode_handle
401              ($byte_stream, allow_error_reporting => 1,
402               allow_fallback => 1);
403          if ($char_stream) {
404            $self->{confident} = 1;
405            last SNIFFING;
406        } else {        } else {
407          !!!emit ($token);          !!!parse-error (type => 'charset:not supported',
408                            layer => 'encode',
409                            line => 1, column => 1,
410                            value => $charset_name,
411                            level => $self->{level}->{uncertain});
412        }        }
413        }
414    
415        redo A;      ## Step 2
416      } elsif ($self->{state} eq 'tag open') {      my $byte_buffer = '';
417        if ($self->{content_model_flag} eq 'RCDATA' or      for (1..1024) {
418            $self->{content_model_flag} eq 'CDATA') {        my $char = $byte_stream->getc;
419          if ($self->{next_input_character} == 0x002F) { # /        last unless defined $char;
420            !!!next-input-character;        $byte_buffer .= $char;
421            $self->{state} = 'close tag open';      } ## TODO: timeout
           redo A;  
         } else {  
           ## reconsume  
           $self->{state} = 'data';  
   
           !!!emit ({type => 'character', data => '<'});  
   
           redo A;  
         }  
       } elsif ($self->{content_model_flag} eq 'PCDATA') {  
         if ($self->{next_input_character} == 0x0021) { # !  
           $self->{state} = 'markup declaration open';  
           !!!next-input-character;  
           redo A;  
         } elsif ($self->{next_input_character} == 0x002F) { # /  
           $self->{state} = 'close tag open';  
           !!!next-input-character;  
           redo A;  
         } elsif (0x0041 <= $self->{next_input_character} and  
                  $self->{next_input_character} <= 0x005A) { # A..Z  
           $self->{current_token}  
             = {type => 'start tag',  
                tag_name => chr ($self->{next_input_character} + 0x0020)};  
           $self->{state} = 'tag name';  
           !!!next-input-character;  
           redo A;  
         } elsif (0x0061 <= $self->{next_input_character} and  
                  $self->{next_input_character} <= 0x007A) { # a..z  
           $self->{current_token} = {type => 'start tag',  
                             tag_name => chr ($self->{next_input_character})};  
           $self->{state} = 'tag name';  
           !!!next-input-character;  
           redo A;  
         } elsif ($self->{next_input_character} == 0x003E) { # >  
           !!!parse-error (type => 'empty start tag');  
           $self->{state} = 'data';  
           !!!next-input-character;  
   
           !!!emit ({type => 'character', data => '<>'});  
   
           redo A;  
         } elsif ($self->{next_input_character} == 0x003F) { # ?  
           !!!parse-error (type => 'pio');  
           $self->{state} = 'bogus comment';  
           ## $self->{next_input_character} is intentionally left as is  
           redo A;  
         } else {  
           !!!parse-error (type => 'bare stago');  
           $self->{state} = 'data';  
           ## reconsume  
422    
423            !!!emit ({type => 'character', data => '<'});      ## Step 3
424        if ($byte_buffer =~ /^\xFE\xFF/) {
425          $charset = Message::Charset::Info->get_by_html_name ('utf-16be');
426          ($char_stream, $e_status) = $charset->get_decode_handle
427              ($byte_stream, allow_error_reporting => 1,
428               allow_fallback => 1, byte_buffer => \$byte_buffer);
429          $self->{confident} = 1;
430          last SNIFFING;
431        } elsif ($byte_buffer =~ /^\xFF\xFE/) {
432          $charset = Message::Charset::Info->get_by_html_name ('utf-16le');
433          ($char_stream, $e_status) = $charset->get_decode_handle
434              ($byte_stream, allow_error_reporting => 1,
435               allow_fallback => 1, byte_buffer => \$byte_buffer);
436          $self->{confident} = 1;
437          last SNIFFING;
438        } elsif ($byte_buffer =~ /^\xEF\xBB\xBF/) {
439          $charset = Message::Charset::Info->get_by_html_name ('utf-8');
440          ($char_stream, $e_status) = $charset->get_decode_handle
441              ($byte_stream, allow_error_reporting => 1,
442               allow_fallback => 1, byte_buffer => \$byte_buffer);
443          $self->{confident} = 1;
444          last SNIFFING;
445        }
446    
447            redo A;      ## Step 4
448          }      ## TODO: <meta charset>
       } else {  
         die "$0: $self->{content_model_flag}: Unknown content model flag";  
       }  
     } elsif ($self->{state} eq 'close tag open') {  
       if ($self->{content_model_flag} eq 'RCDATA' or  
           $self->{content_model_flag} eq 'CDATA') {  
         my @next_char;  
         TAGNAME: for (my $i = 0; $i < length $self->{last_emitted_start_tag_name}; $i++) {  
           push @next_char, $self->{next_input_character};  
           my $c = ord substr ($self->{last_emitted_start_tag_name}, $i, 1);  
           my $C = 0x0061 <= $c && $c <= 0x007A ? $c - 0x0020 : $c;  
           if ($self->{next_input_character} == $c or $self->{next_input_character} == $C) {  
             !!!next-input-character;  
             next TAGNAME;  
           } else {  
             $self->{next_input_character} = shift @next_char; # reconsume  
             !!!back-next-input-character (@next_char);  
             $self->{state} = 'data';  
449    
450              !!!emit ({type => 'character', data => '</'});      ## Step 5
451        ## TODO: from history
452    
453              redo A;      ## Step 6
454            }      require Whatpm::Charset::UniversalCharDet;
455          }      $charset_name = Whatpm::Charset::UniversalCharDet->detect_byte_string
456          push @next_char, $self->{next_input_character};          ($byte_buffer);
457            if (defined $charset_name) {
458          unless ($self->{next_input_character} == 0x0009 or # HT        $charset = Message::Charset::Info->get_by_html_name ($charset_name);
459                  $self->{next_input_character} == 0x000A or # LF  
460                  $self->{next_input_character} == 0x000B or # VT        require Whatpm::Charset::DecodeHandle;
461                  $self->{next_input_character} == 0x000C or # FF        $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new
462                  $self->{next_input_character} == 0x0020 or # SP            ($byte_stream);
463                  $self->{next_input_character} == 0x003E or # >        ($char_stream, $e_status) = $charset->get_decode_handle
464                  $self->{next_input_character} == 0x002F or # /            ($buffer, allow_error_reporting => 1,
465                  $self->{next_input_character} == -1) {             allow_fallback => 1, byte_buffer => \$byte_buffer);
466            $self->{next_input_character} = shift @next_char; # reconsume        if ($char_stream) {
467            !!!back-next-input-character (@next_char);          $buffer->{buffer} = $byte_buffer;
468            $self->{state} = 'data';          !!!parse-error (type => 'sniffing:chardet',
469                            text => $charset_name,
470            !!!emit ({type => 'character', data => '</'});                          level => $self->{level}->{info},
471                            layer => 'encode',
472            redo A;                          line => 1, column => 1);
473          } else {          $self->{confident} = 0;
474            $self->{next_input_character} = shift @next_char;          last SNIFFING;
           !!!back-next-input-character (@next_char);  
           # and consume...  
         }  
475        }        }
476              }
       if (0x0041 <= $self->{next_input_character} and  
           $self->{next_input_character} <= 0x005A) { # A..Z  
         $self->{current_token} = {type => 'end tag',  
                           tag_name => chr ($self->{next_input_character} + 0x0020)};  
         $self->{state} = 'tag name';  
         !!!next-input-character;  
         redo A;  
       } elsif (0x0061 <= $self->{next_input_character} and  
                $self->{next_input_character} <= 0x007A) { # a..z  
         $self->{current_token} = {type => 'end tag',  
                           tag_name => chr ($self->{next_input_character})};  
         $self->{state} = 'tag name';  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == 0x003E) { # >  
         !!!parse-error (type => 'empty end tag');  
         $self->{state} = 'data';  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == -1) {  
         !!!parse-error (type => 'bare etago');  
         $self->{state} = 'data';  
         # reconsume  
   
         !!!emit ({type => 'character', data => '</'});  
   
         redo A;  
       } else {  
         !!!parse-error (type => 'bogus end tag');  
         $self->{state} = 'bogus comment';  
         ## $self->{next_input_character} is intentionally left as is  
         redo A;  
       }  
     } elsif ($self->{state} eq 'tag name') {  
       if ($self->{next_input_character} == 0x0009 or # HT  
           $self->{next_input_character} == 0x000A or # LF  
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
         $self->{state} = 'before attribute name';  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == 0x003E) { # >  
         if ($self->{current_token}->{type} eq 'start tag') {  
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} eq 'end tag') {  
           $self->{content_model_flag} = 'PCDATA'; # MUST  
           if ($self->{current_token}->{attributes}) {  
             !!!parse-error (type => 'end tag attribute');  
           }  
         } else {  
           die "$0: $self->{current_token}->{type}: Unknown token type";  
         }  
         $self->{state} = 'data';  
         !!!next-input-character;  
   
         !!!emit ($self->{current_token}); # start tag or end tag  
         undef $self->{current_token};  
   
         redo A;  
       } elsif (0x0041 <= $self->{next_input_character} and  
                $self->{next_input_character} <= 0x005A) { # A..Z  
         $self->{current_token}->{tag_name} .= chr ($self->{next_input_character} + 0x0020);  
           # start tag or end tag  
         ## Stay in this state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         if ($self->{current_token}->{type} eq 'start tag') {  
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} eq 'end tag') {  
           $self->{content_model_flag} = 'PCDATA'; # MUST  
           if ($self->{current_token}->{attributes}) {  
             !!!parse-error (type => 'end tag attribute');  
           }  
         } else {  
           die "$0: $self->{current_token}->{type}: Unknown token type";  
         }  
         $self->{state} = 'data';  
         # reconsume  
   
         !!!emit ($self->{current_token}); # start tag or end tag  
         undef $self->{current_token};  
   
         redo A;  
       } elsif ($self->{next_input_character} == 0x002F) { # /  
         !!!next-input-character;  
         if ($self->{next_input_character} == 0x003E and # >  
             $self->{current_token}->{type} eq 'start tag' and  
             $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {  
           # permitted slash  
           #  
         } else {  
           !!!parse-error (type => 'nestc');  
         }  
         $self->{state} = 'before attribute name';  
         # next-input-character is already done  
         redo A;  
       } else {  
         $self->{current_token}->{tag_name} .= chr $self->{next_input_character};  
           # start tag or end tag  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} eq 'before attribute name') {  
       if ($self->{next_input_character} == 0x0009 or # HT  
           $self->{next_input_character} == 0x000A or # LF  
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == 0x003E) { # >  
         if ($self->{current_token}->{type} eq 'start tag') {  
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} eq 'end tag') {  
           $self->{content_model_flag} = 'PCDATA'; # MUST  
           if ($self->{current_token}->{attributes}) {  
             !!!parse-error (type => 'end tag attribute');  
           }  
         } else {  
           die "$0: $self->{current_token}->{type}: Unknown token type";  
         }  
         $self->{state} = 'data';  
         !!!next-input-character;  
   
         !!!emit ($self->{current_token}); # start tag or end tag  
         undef $self->{current_token};  
   
         redo A;  
       } elsif (0x0041 <= $self->{next_input_character} and  
                $self->{next_input_character} <= 0x005A) { # A..Z  
         $self->{current_attribute} = {name => chr ($self->{next_input_character} + 0x0020),  
                               value => ''};  
         $self->{state} = 'attribute name';  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == 0x002F) { # /  
         !!!next-input-character;  
         if ($self->{next_input_character} == 0x003E and # >  
             $self->{current_token}->{type} eq 'start tag' and  
             $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {  
           # permitted slash  
           #  
         } else {  
           !!!parse-error (type => 'nestc');  
         }  
         ## Stay in the state  
         # next-input-character is already done  
         redo A;  
       } elsif ($self->{next_input_character} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         if ($self->{current_token}->{type} eq 'start tag') {  
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} eq 'end tag') {  
           $self->{content_model_flag} = 'PCDATA'; # MUST  
           if ($self->{current_token}->{attributes}) {  
             !!!parse-error (type => 'end tag attribute');  
           }  
         } else {  
           die "$0: $self->{current_token}->{type}: Unknown token type";  
         }  
         $self->{state} = 'data';  
         # reconsume  
   
         !!!emit ($self->{current_token}); # start tag or end tag  
         undef $self->{current_token};  
   
         redo A;  
       } else {  
         $self->{current_attribute} = {name => chr ($self->{next_input_character}),  
                               value => ''};  
         $self->{state} = 'attribute name';  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} eq 'attribute name') {  
       my $before_leave = sub {  
         if (exists $self->{current_token}->{attributes} # start tag or end tag  
             ->{$self->{current_attribute}->{name}}) { # MUST  
           !!!parse-error (type => 'dupulicate attribute');  
           ## Discard $self->{current_attribute} # MUST  
         } else {  
           $self->{current_token}->{attributes}->{$self->{current_attribute}->{name}}  
             = $self->{current_attribute};  
         }  
       }; # $before_leave  
   
       if ($self->{next_input_character} == 0x0009 or # HT  
           $self->{next_input_character} == 0x000A or # LF  
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
         $before_leave->();  
         $self->{state} = 'after attribute name';  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == 0x003D) { # =  
         $before_leave->();  
         $self->{state} = 'before attribute value';  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == 0x003E) { # >  
         $before_leave->();  
         if ($self->{current_token}->{type} eq 'start tag') {  
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} eq 'end tag') {  
           $self->{content_model_flag} = 'PCDATA'; # MUST  
           if ($self->{current_token}->{attributes}) {  
             !!!parse-error (type => 'end tag attribute');  
           }  
         } else {  
           die "$0: $self->{current_token}->{type}: Unknown token type";  
         }  
         $self->{state} = 'data';  
         !!!next-input-character;  
   
         !!!emit ($self->{current_token}); # start tag or end tag  
         undef $self->{current_token};  
   
         redo A;  
       } elsif (0x0041 <= $self->{next_input_character} and  
                $self->{next_input_character} <= 0x005A) { # A..Z  
         $self->{current_attribute}->{name} .= chr ($self->{next_input_character} + 0x0020);  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == 0x002F) { # /  
         $before_leave->();  
         !!!next-input-character;  
         if ($self->{next_input_character} == 0x003E and # >  
             $self->{current_token}->{type} eq 'start tag' and  
             $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {  
           # permitted slash  
           #  
         } else {  
           !!!parse-error (type => 'nestc');  
         }  
         $self->{state} = 'before attribute name';  
         # next-input-character is already done  
         redo A;  
       } elsif ($self->{next_input_character} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         $before_leave->();  
         if ($self->{current_token}->{type} eq 'start tag') {  
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} eq 'end tag') {  
           $self->{content_model_flag} = 'PCDATA'; # MUST  
           if ($self->{current_token}->{attributes}) {  
             !!!parse-error (type => 'end tag attribute');  
           }  
         } else {  
           die "$0: $self->{current_token}->{type}: Unknown token type";  
         }  
         $self->{state} = 'data';  
         # reconsume  
   
         !!!emit ($self->{current_token}); # start tag or end tag  
         undef $self->{current_token};  
   
         redo A;  
       } else {  
         $self->{current_attribute}->{name} .= chr ($self->{next_input_character});  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} eq 'after attribute name') {  
       if ($self->{next_input_character} == 0x0009 or # HT  
           $self->{next_input_character} == 0x000A or # LF  
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == 0x003D) { # =  
         $self->{state} = 'before attribute value';  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == 0x003E) { # >  
         if ($self->{current_token}->{type} eq 'start tag') {  
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} eq 'end tag') {  
           $self->{content_model_flag} = 'PCDATA'; # MUST  
           if ($self->{current_token}->{attributes}) {  
             !!!parse-error (type => 'end tag attribute');  
           }  
         } else {  
           die "$0: $self->{current_token}->{type}: Unknown token type";  
         }  
         $self->{state} = 'data';  
         !!!next-input-character;  
   
         !!!emit ($self->{current_token}); # start tag or end tag  
         undef $self->{current_token};  
   
         redo A;  
       } elsif (0x0041 <= $self->{next_input_character} and  
                $self->{next_input_character} <= 0x005A) { # A..Z  
         $self->{current_attribute} = {name => chr ($self->{next_input_character} + 0x0020),  
                               value => ''};  
         $self->{state} = 'attribute name';  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == 0x002F) { # /  
         !!!next-input-character;  
         if ($self->{next_input_character} == 0x003E and # >  
             $self->{current_token}->{type} eq 'start tag' and  
             $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {  
           # permitted slash  
           #  
         } else {  
           !!!parse-error (type => 'nestc');  
         }  
         $self->{state} = 'before attribute name';  
         # next-input-character is already done  
         redo A;  
       } elsif ($self->{next_input_character} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         if ($self->{current_token}->{type} eq 'start tag') {  
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} eq 'end tag') {  
           $self->{content_model_flag} = 'PCDATA'; # MUST  
           if ($self->{current_token}->{attributes}) {  
             !!!parse-error (type => 'end tag attribute');  
           }  
         } else {  
           die "$0: $self->{current_token}->{type}: Unknown token type";  
         }  
         $self->{state} = 'data';  
         # reconsume  
   
         !!!emit ($self->{current_token}); # start tag or end tag  
         undef $self->{current_token};  
   
         redo A;  
       } else {  
         $self->{current_attribute} = {name => chr ($self->{next_input_character}),  
                               value => ''};  
         $self->{state} = 'attribute name';  
         !!!next-input-character;  
         redo A;          
       }  
     } elsif ($self->{state} eq 'before attribute value') {  
       if ($self->{next_input_character} == 0x0009 or # HT  
           $self->{next_input_character} == 0x000A or # LF  
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP        
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == 0x0022) { # "  
         $self->{state} = 'attribute value (double-quoted)';  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == 0x0026) { # &  
         $self->{state} = 'attribute value (unquoted)';  
         ## reconsume  
         redo A;  
       } elsif ($self->{next_input_character} == 0x0027) { # '  
         $self->{state} = 'attribute value (single-quoted)';  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == 0x003E) { # >  
         if ($self->{current_token}->{type} eq 'start tag') {  
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} eq 'end tag') {  
           $self->{content_model_flag} = 'PCDATA'; # MUST  
           if ($self->{current_token}->{attributes}) {  
             !!!parse-error (type => 'end tag attribute');  
           }  
         } else {  
           die "$0: $self->{current_token}->{type}: Unknown token type";  
         }  
         $self->{state} = 'data';  
         !!!next-input-character;  
   
         !!!emit ($self->{current_token}); # start tag or end tag  
         undef $self->{current_token};  
   
         redo A;  
       } elsif ($self->{next_input_character} == -1) {  
         !!!parse-error (type => 'unclosed tag');  
         if ($self->{current_token}->{type} eq 'start tag') {  
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} eq 'end tag') {  
           $self->{content_model_flag} = 'PCDATA'; # MUST  
           if ($self->{current_token}->{attributes}) {  
             !!!parse-error (type => 'end tag attribute');  
           }  
         } else {  
           die "$0: $self->{current_token}->{type}: Unknown token type";  
         }  
         $self->{state} = 'data';  
         ## reconsume  
   
         !!!emit ($self->{current_token}); # start tag or end tag  
         undef $self->{current_token};  
   
         redo A;  
       } else {  
         $self->{current_attribute}->{value} .= chr ($self->{next_input_character});  
         $self->{state} = 'attribute value (unquoted)';  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} eq 'attribute value (double-quoted)') {  
       if ($self->{next_input_character} == 0x0022) { # "  
         $self->{state} = 'before attribute name';  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == 0x0026) { # &  
         $self->{last_attribute_value_state} = 'attribute value (double-quoted)';  
         $self->{state} = 'entity in attribute value';  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == -1) {  
         !!!parse-error (type => 'unclosed attribute value');  
         if ($self->{current_token}->{type} eq 'start tag') {  
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} eq 'end tag') {  
           $self->{content_model_flag} = 'PCDATA'; # MUST  
           if ($self->{current_token}->{attributes}) {  
             !!!parse-error (type => 'end tag attribute');  
           }  
         } else {  
           die "$0: $self->{current_token}->{type}: Unknown token type";  
         }  
         $self->{state} = 'data';  
         ## reconsume  
   
         !!!emit ($self->{current_token}); # start tag or end tag  
         undef $self->{current_token};  
   
         redo A;  
       } else {  
         $self->{current_attribute}->{value} .= chr ($self->{next_input_character});  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} eq 'attribute value (single-quoted)') {  
       if ($self->{next_input_character} == 0x0027) { # '  
         $self->{state} = 'before attribute name';  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == 0x0026) { # &  
         $self->{last_attribute_value_state} = 'attribute value (single-quoted)';  
         $self->{state} = 'entity in attribute value';  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == -1) {  
         !!!parse-error (type => 'unclosed attribute value');  
         if ($self->{current_token}->{type} eq 'start tag') {  
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} eq 'end tag') {  
           $self->{content_model_flag} = 'PCDATA'; # MUST  
           if ($self->{current_token}->{attributes}) {  
             !!!parse-error (type => 'end tag attribute');  
           }  
         } else {  
           die "$0: $self->{current_token}->{type}: Unknown token type";  
         }  
         $self->{state} = 'data';  
         ## reconsume  
   
         !!!emit ($self->{current_token}); # start tag or end tag  
         undef $self->{current_token};  
477    
478          redo A;      ## Step 7: default
479        } else {      ## TODO: Make this configurable.
480          $self->{current_attribute}->{value} .= chr ($self->{next_input_character});      $charset = Message::Charset::Info->get_by_html_name ('windows-1252');
481          ## Stay in the state          ## NOTE: We choose |windows-1252| here, since |utf-8| should be
482          !!!next-input-character;          ## detectable in the step 6.
483          redo A;      require Whatpm::Charset::DecodeHandle;
484        }      $buffer = Whatpm::Charset::DecodeHandle::ByteBuffer->new
485      } elsif ($self->{state} eq 'attribute value (unquoted)') {          ($byte_stream);
486        if ($self->{next_input_character} == 0x0009 or # HT      ($char_stream, $e_status)
487            $self->{next_input_character} == 0x000A or # LF          = $charset->get_decode_handle ($buffer,
488            $self->{next_input_character} == 0x000B or # HT                                         allow_error_reporting => 1,
489            $self->{next_input_character} == 0x000C or # FF                                         allow_fallback => 1,
490            $self->{next_input_character} == 0x0020) { # SP                                         byte_buffer => \$byte_buffer);
491          $self->{state} = 'before attribute name';      $buffer->{buffer} = $byte_buffer;
492          !!!next-input-character;      !!!parse-error (type => 'sniffing:default',
493          redo A;                      text => 'windows-1252',
494        } elsif ($self->{next_input_character} == 0x0026) { # &                      level => $self->{level}->{info},
495          $self->{last_attribute_value_state} = 'attribute value (unquoted)';                      line => 1, column => 1,
496          $self->{state} = 'entity in attribute value';                      layer => 'encode');
497          !!!next-input-character;      $self->{confident} = 0;
498          redo A;    } # SNIFFING
499        } elsif ($self->{next_input_character} == 0x003E) { # >  
500          if ($self->{current_token}->{type} eq 'start tag') {    if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
501            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};      $self->{input_encoding} = $charset->get_iana_name; ## TODO: Should we set actual charset decoder's encoding name?
502          } elsif ($self->{current_token}->{type} eq 'end tag') {      !!!parse-error (type => 'chardecode:fallback',
503            $self->{content_model_flag} = 'PCDATA'; # MUST                      #text => $self->{input_encoding},
504            if ($self->{current_token}->{attributes}) {                      level => $self->{level}->{uncertain},
505              !!!parse-error (type => 'end tag attribute');                      line => 1, column => 1,
506            }                      layer => 'encode');
507          } else {    } elsif (not ($e_status &
508            die "$0: $self->{current_token}->{type}: Unknown token type";                  Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) {
509          }      $self->{input_encoding} = $charset->get_iana_name;
510          $self->{state} = 'data';      !!!parse-error (type => 'chardecode:no error',
511          !!!next-input-character;                      text => $self->{input_encoding},
512                        level => $self->{level}->{uncertain},
513          !!!emit ($self->{current_token}); # start tag or end tag                      line => 1, column => 1,
514          undef $self->{current_token};                      layer => 'encode');
515      } else {
516          redo A;      $self->{input_encoding} = $charset->get_iana_name;
517        } elsif ($self->{next_input_character} == -1) {    }
         !!!parse-error (type => 'unclosed tag');  
         if ($self->{current_token}->{type} eq 'start tag') {  
           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};  
         } elsif ($self->{current_token}->{type} eq 'end tag') {  
           $self->{content_model_flag} = 'PCDATA'; # MUST  
           if ($self->{current_token}->{attributes}) {  
             !!!parse-error (type => 'end tag attribute');  
           }  
         } else {  
           die "$0: $self->{current_token}->{type}: Unknown token type";  
         }  
         $self->{state} = 'data';  
         ## reconsume  
518    
519          !!!emit ($self->{current_token}); # start tag or end tag    $self->{change_encoding} = sub {
520          undef $self->{current_token};      my $self = shift;
521        $charset_name = shift;
522        my $token = shift;
523    
524          redo A;      $charset = Message::Charset::Info->get_by_html_name ($charset_name);
525        } else {      ($char_stream, $e_status) = $charset->get_decode_handle
526          $self->{current_attribute}->{value} .= chr ($self->{next_input_character});          ($byte_stream, allow_error_reporting => 1, allow_fallback => 1,
527          ## Stay in the state           byte_buffer => \ $buffer->{buffer});
528          !!!next-input-character;      
529          redo A;      if ($char_stream) { # if supported
530          ## "Change the encoding" algorithm:
531          
532          ## Step 1
533          if (defined $self->{input_encoding} and
534              $self->{input_encoding} eq $charset_name) {
535            !!!parse-error (type => 'charset label:matching',
536                            text => $charset_name,
537                            level => $self->{level}->{info});
538            $self->{confident} = 1;
539            return;
540        }        }
     } elsif ($self->{state} eq 'entity in attribute value') {  
       my $token = $self->_tokenize_attempt_to_consume_an_entity;  
541    
542        unless (defined $token) {        ## Step 2 (HTML5 revision 3205)
543          $self->{current_attribute}->{value} .= '&';        if (defined $self->{input_encoding} and
544        } else {            Message::Charset::Info->get_by_html_name ($self->{input_encoding})
545          $self->{current_attribute}->{value} .= $token->{data};            ->{category} & Message::Charset::Info::CHARSET_CATEGORY_UTF16 ()) {
546          ## ISSUE: spec says "append the returned character token to the current attribute's value"          $self->{confident} = 1;
547            return;
548        }        }
549    
550        $self->{state} = $self->{last_attribute_value_state};        ## Step 3
551        # next-input-character is already done        if ($charset->{category} &
552        redo A;            Message::Charset::Info::CHARSET_CATEGORY_UTF16 ()) {
553      } elsif ($self->{state} eq 'bogus comment') {          $charset = Message::Charset::Info->get_by_html_name ('utf-8');
554        ## (only happen if PCDATA state)          ($char_stream, $e_status) = $charset->get_decode_handle
555                      ($byte_stream,
556        my $token = {type => 'comment', data => ''};               byte_buffer => \ $buffer->{buffer});
557          }
558        BC: {        $charset_name = $charset->get_iana_name;
559          if ($self->{next_input_character} == 0x003E) { # >  
560            $self->{state} = 'data';        !!!parse-error (type => 'charset label detected',
561            !!!next-input-character;                        text => $self->{input_encoding},
562                          value => $charset_name,
563            !!!emit ($token);                        level => $self->{level}->{warn},
564                          token => $token);
           redo A;  
         } elsif ($self->{next_input_character} == -1) {  
           $self->{state} = 'data';  
           ## reconsume  
   
           !!!emit ($token);  
   
           redo A;  
         } else {  
           $token->{data} .= chr ($self->{next_input_character});  
           !!!next-input-character;  
           redo BC;  
         }  
       } # BC  
     } elsif ($self->{state} eq 'markup declaration open') {  
       ## (only happen if PCDATA state)  
   
       my @next_char;  
       push @next_char, $self->{next_input_character};  
565                
566        if ($self->{next_input_character} == 0x002D) { # -        ## Step 4
567          !!!next-input-character;        # if (can) {
568          push @next_char, $self->{next_input_character};          ## change the encoding on the fly.
569          if ($self->{next_input_character} == 0x002D) { # -          #$self->{confident} = 1;
570            $self->{current_token} = {type => 'comment', data => ''};          #return;
571            $self->{state} = 'comment';        # }
           !!!next-input-character;  
           redo A;  
         }  
       } elsif ($self->{next_input_character} == 0x0044 or # D  
                $self->{next_input_character} == 0x0064) { # d  
         !!!next-input-character;  
         push @next_char, $self->{next_input_character};  
         if ($self->{next_input_character} == 0x004F or # O  
             $self->{next_input_character} == 0x006F) { # o  
           !!!next-input-character;  
           push @next_char, $self->{next_input_character};  
           if ($self->{next_input_character} == 0x0043 or # C  
               $self->{next_input_character} == 0x0063) { # c  
             !!!next-input-character;  
             push @next_char, $self->{next_input_character};  
             if ($self->{next_input_character} == 0x0054 or # T  
                 $self->{next_input_character} == 0x0074) { # t  
               !!!next-input-character;  
               push @next_char, $self->{next_input_character};  
               if ($self->{next_input_character} == 0x0059 or # Y  
                   $self->{next_input_character} == 0x0079) { # y  
                 !!!next-input-character;  
                 push @next_char, $self->{next_input_character};  
                 if ($self->{next_input_character} == 0x0050 or # P  
                     $self->{next_input_character} == 0x0070) { # p  
                   !!!next-input-character;  
                   push @next_char, $self->{next_input_character};  
                   if ($self->{next_input_character} == 0x0045 or # E  
                       $self->{next_input_character} == 0x0065) { # e  
                     ## ISSUE: What a stupid code this is!  
                     $self->{state} = 'DOCTYPE';  
                     !!!next-input-character;  
                     redo A;  
                   }  
                 }  
               }  
             }  
           }  
         }  
       }  
   
       !!!parse-error (type => 'bogus comment open');  
       $self->{next_input_character} = shift @next_char;  
       !!!back-next-input-character (@next_char);  
       $self->{state} = 'bogus comment';  
       redo A;  
572                
573        ## ISSUE: typos in spec: chacacters, is is a parse error        ## Step 5
574        ## ISSUE: spec is somewhat unclear on "is the first character that will be in the comment"; what is "that will be in the comment" is what the algorithm defines, isn't it?        throw Whatpm::HTML::RestartParser ();
575      } elsif ($self->{state} eq 'comment') {      }
576        if ($self->{next_input_character} == 0x002D) { # -    }; # $self->{change_encoding}
         $self->{state} = 'comment dash';  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == -1) {  
         !!!parse-error (type => 'unclosed comment');  
         $self->{state} = 'data';  
         ## reconsume  
   
         !!!emit ($self->{current_token}); # comment  
         undef $self->{current_token};  
   
         redo A;  
       } else {  
         $self->{current_token}->{data} .= chr ($self->{next_input_character}); # comment  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} eq 'comment dash') {  
       if ($self->{next_input_character} == 0x002D) { # -  
         $self->{state} = 'comment end';  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == -1) {  
         !!!parse-error (type => 'unclosed comment');  
         $self->{state} = 'data';  
         ## reconsume  
   
         !!!emit ($self->{current_token}); # comment  
         undef $self->{current_token};  
   
         redo A;  
       } else {  
         $self->{current_token}->{data} .= '-' . chr ($self->{next_input_character}); # comment  
         $self->{state} = 'comment';  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} eq 'comment end') {  
       if ($self->{next_input_character} == 0x003E) { # >  
         $self->{state} = 'data';  
         !!!next-input-character;  
   
         !!!emit ($self->{current_token}); # comment  
         undef $self->{current_token};  
   
         redo A;  
       } elsif ($self->{next_input_character} == 0x002D) { # -  
         !!!parse-error (type => 'dash in comment');  
         $self->{current_token}->{data} .= '-'; # comment  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == -1) {  
         !!!parse-error (type => 'unclosed comment');  
         $self->{state} = 'data';  
         ## reconsume  
   
         !!!emit ($self->{current_token}); # comment  
         undef $self->{current_token};  
   
         redo A;  
       } else {  
         !!!parse-error (type => 'dash in comment');  
         $self->{current_token}->{data} .= '--' . chr ($self->{next_input_character}); # comment  
         $self->{state} = 'comment';  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} eq 'DOCTYPE') {  
       if ($self->{next_input_character} == 0x0009 or # HT  
           $self->{next_input_character} == 0x000A or # LF  
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
         $self->{state} = 'before DOCTYPE name';  
         !!!next-input-character;  
         redo A;  
       } else {  
         !!!parse-error (type => 'no space before DOCTYPE name');  
         $self->{state} = 'before DOCTYPE name';  
         ## reconsume  
         redo A;  
       }  
     } elsif ($self->{state} eq 'before DOCTYPE name') {  
       if ($self->{next_input_character} == 0x0009 or # HT  
           $self->{next_input_character} == 0x000A or # LF  
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == 0x003E) { # >  
         !!!parse-error (type => 'no DOCTYPE name');  
         $self->{state} = 'data';  
         !!!next-input-character;  
   
         !!!emit ({type => 'DOCTYPE'}); # incorrect  
   
         redo A;  
       } elsif ($self->{next_input_character} == -1) {  
         !!!parse-error (type => 'no DOCTYPE name');  
         $self->{state} = 'data';  
         ## reconsume  
   
         !!!emit ({type => 'DOCTYPE'}); # incorrect  
   
         redo A;  
       } else {  
         $self->{current_token}  
             = {type => 'DOCTYPE',  
                name => chr ($self->{next_input_character}),  
                correct => 1};  
 ## ISSUE: "Set the token's name name to the" in the spec  
         $self->{state} = 'DOCTYPE name';  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} eq 'DOCTYPE name') {  
 ## ISSUE: Redundant "First," in the spec.  
       if ($self->{next_input_character} == 0x0009 or # HT  
           $self->{next_input_character} == 0x000A or # LF  
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
         $self->{state} = 'after DOCTYPE name';  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == 0x003E) { # >  
         $self->{state} = 'data';  
         !!!next-input-character;  
   
         !!!emit ($self->{current_token}); # DOCTYPE  
         undef $self->{current_token};  
   
         redo A;  
       } elsif ($self->{next_input_character} == -1) {  
         !!!parse-error (type => 'unclosed DOCTYPE');  
         $self->{state} = 'data';  
         ## reconsume  
   
         delete $self->{current_token}->{correct};  
         !!!emit ($self->{current_token}); # DOCTYPE  
         undef $self->{current_token};  
   
         redo A;  
       } else {  
         $self->{current_token}->{name}  
           .= chr ($self->{next_input_character}); # DOCTYPE  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} eq 'after DOCTYPE name') {  
       if ($self->{next_input_character} == 0x0009 or # HT  
           $self->{next_input_character} == 0x000A or # LF  
           $self->{next_input_character} == 0x000B or # VT  
           $self->{next_input_character} == 0x000C or # FF  
           $self->{next_input_character} == 0x0020) { # SP  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == 0x003E) { # >  
         $self->{state} = 'data';  
         !!!next-input-character;  
   
         !!!emit ($self->{current_token}); # DOCTYPE  
         undef $self->{current_token};  
   
         redo A;  
       } elsif ($self->{next_input_character} == -1) {  
         !!!parse-error (type => 'unclosed DOCTYPE');  
         $self->{state} = 'data';  
         ## reconsume  
   
         delete $self->{current_token}->{correct};  
         !!!emit ($self->{current_token}); # DOCTYPE  
         undef $self->{current_token};  
   
         redo A;  
       } elsif ($self->{next_input_character} == 0x0050 or # P  
                $self->{next_input_character} == 0x0070) { # p  
         !!!next-input-character;  
         if ($self->{next_input_character} == 0x0055 or # U  
             $self->{next_input_character} == 0x0075) { # u  
           !!!next-input-character;  
           if ($self->{next_input_character} == 0x0042 or # B  
               $self->{next_input_character} == 0x0062) { # b  
             !!!next-input-character;  
             if ($self->{next_input_character} == 0x004C or # L  
                 $self->{next_input_character} == 0x006C) { # l  
               !!!next-input-character;  
               if ($self->{next_input_character} == 0x0049 or # I  
                   $self->{next_input_character} == 0x0069) { # i  
                 !!!next-input-character;  
                 if ($self->{next_input_character} == 0x0043 or # C  
                     $self->{next_input_character} == 0x0063) { # c  
                   $self->{state} = 'before DOCTYPE public identifier';  
                   !!!next-input-character;  
                   redo A;  
                 }  
               }  
             }  
           }  
         }  
   
         #  
       } elsif ($self->{next_input_character} == 0x0053 or # S  
                $self->{next_input_character} == 0x0073) { # s  
         !!!next-input-character;  
         if ($self->{next_input_character} == 0x0059 or # Y  
             $self->{next_input_character} == 0x0079) { # y  
           !!!next-input-character;  
           if ($self->{next_input_character} == 0x0053 or # S  
               $self->{next_input_character} == 0x0073) { # s  
             !!!next-input-character;  
             if ($self->{next_input_character} == 0x0054 or # T  
                 $self->{next_input_character} == 0x0074) { # t  
               !!!next-input-character;  
               if ($self->{next_input_character} == 0x0045 or # E  
                   $self->{next_input_character} == 0x0065) { # e  
                 !!!next-input-character;  
                 if ($self->{next_input_character} == 0x004D or # M  
                     $self->{next_input_character} == 0x006D) { # m  
                   $self->{state} = 'before DOCTYPE system identifier';  
                   !!!next-input-character;  
                   redo A;  
                 }  
               }  
             }  
           }  
         }  
577    
578          #    my $char_onerror = sub {
579        } else {      my (undef, $type, %opt) = @_;
580          !!!next-input-character;      !!!parse-error (layer => 'encode',
581          #                      line => $self->{line}, column => $self->{column} + 1,
582        }                      %opt, type => $type);
583        if ($opt{octets}) {
584          ${$opt{octets}} = "\x{FFFD}"; # relacement character
585        }
586      };
587    
588        !!!parse-error (type => 'string after DOCTYPE name');    my $wrapped_char_stream = $get_wrapper->($char_stream);
589        $self->{state} = 'bogus DOCTYPE';    $wrapped_char_stream->onerror ($char_onerror);
       # next-input-character is already done  
       redo A;  
     } elsif ($self->{state} eq 'before DOCTYPE public identifier') {  
       if ({  
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_input_character}}) {  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} eq 0x0022) { # "  
         $self->{current_token}->{public_identifier} = ''; # DOCTYPE  
         $self->{state} = 'DOCTYPE public identifier (double-quoted)';  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} eq 0x0027) { # '  
         $self->{current_token}->{public_identifier} = ''; # DOCTYPE  
         $self->{state} = 'DOCTYPE public identifier (single-quoted)';  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} eq 0x003E) { # >  
         !!!parse-error (type => 'no PUBLIC literal');  
   
         $self->{state} = 'data';  
         !!!next-input-character;  
   
         delete $self->{current_token}->{correct};  
         !!!emit ($self->{current_token}); # DOCTYPE  
         undef $self->{current_token};  
   
         redo A;  
       } elsif ($self->{next_input_character} == -1) {  
         !!!parse-error (type => 'unclosed DOCTYPE');  
   
         $self->{state} = 'data';  
         ## reconsume  
   
         delete $self->{current_token}->{correct};  
         !!!emit ($self->{current_token}); # DOCTYPE  
         undef $self->{current_token};  
590    
591          redo A;    my @args = ($_[1], $_[2]); # $doc, $onerror - $get_wrapper = undef;
592        } else {    my $return;
593          !!!parse-error (type => 'string after PUBLIC');    try {
594          $self->{state} = 'bogus DOCTYPE';      $return = $self->parse_char_stream ($wrapped_char_stream, @args);  
595          !!!next-input-character;    } catch Whatpm::HTML::RestartParser with {
596          redo A;      ## NOTE: Invoked after {change_encoding}.
597        }  
598      } elsif ($self->{state} eq 'DOCTYPE public identifier (double-quoted)') {      if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
599        if ($self->{next_input_character} == 0x0022) { # "        $self->{input_encoding} = $charset->get_iana_name; ## TODO: Should we set actual charset decoder's encoding name?
600          $self->{state} = 'after DOCTYPE public identifier';        !!!parse-error (type => 'chardecode:fallback',
601          !!!next-input-character;                        level => $self->{level}->{uncertain},
602          redo A;                        #text => $self->{input_encoding},
603        } elsif ($self->{next_input_character} == -1) {                        line => 1, column => 1,
604          !!!parse-error (type => 'unclosed PUBLIC literal');                        layer => 'encode');
605        } elsif (not ($e_status &
606          $self->{state} = 'data';                    Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL ())) {
607          ## reconsume        $self->{input_encoding} = $charset->get_iana_name;
608          !!!parse-error (type => 'chardecode:no error',
609          delete $self->{current_token}->{correct};                        text => $self->{input_encoding},
610          !!!emit ($self->{current_token}); # DOCTYPE                        level => $self->{level}->{uncertain},
611          undef $self->{current_token};                        line => 1, column => 1,
612                          layer => 'encode');
613        } else {
614          $self->{input_encoding} = $charset->get_iana_name;
615        }
616        $self->{confident} = 1;
617    
618          redo A;      $wrapped_char_stream = $get_wrapper->($char_stream);
619        } else {      $wrapped_char_stream->onerror ($char_onerror);
         $self->{current_token}->{public_identifier} # DOCTYPE  
             .= chr $self->{next_input_character};  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} eq 'DOCTYPE public identifier (single-quoted)') {  
       if ($self->{next_input_character} == 0x0027) { # '  
         $self->{state} = 'after DOCTYPE public identifier';  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == -1) {  
         !!!parse-error (type => 'unclosed PUBLIC literal');  
   
         $self->{state} = 'data';  
         ## reconsume  
   
         delete $self->{current_token}->{correct};  
         !!!emit ($self->{current_token}); # DOCTYPE  
         undef $self->{current_token};  
620    
621          redo A;      $return = $self->parse_char_stream ($wrapped_char_stream, @args);
622        } else {    };
623          $self->{current_token}->{public_identifier} # DOCTYPE    return $return;
624              .= chr $self->{next_input_character};  } # parse_byte_stream
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} eq 'after DOCTYPE public identifier') {  
       if ({  
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_input_character}}) {  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == 0x0022) { # "  
         $self->{current_token}->{system_identifier} = ''; # DOCTYPE  
         $self->{state} = 'DOCTYPE system identifier (double-quoted)';  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == 0x0027) { # '  
         $self->{current_token}->{system_identifier} = ''; # DOCTYPE  
         $self->{state} = 'DOCTYPE system identifier (single-quoted)';  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == 0x003E) { # >  
         $self->{state} = 'data';  
         !!!next-input-character;  
   
         !!!emit ($self->{current_token}); # DOCTYPE  
         undef $self->{current_token};  
   
         redo A;  
       } elsif ($self->{next_input_character} == -1) {  
         !!!parse-error (type => 'unclosed DOCTYPE');  
   
         $self->{state} = 'data';  
         ## recomsume  
   
         delete $self->{current_token}->{correct};  
         !!!emit ($self->{current_token}); # DOCTYPE  
         undef $self->{current_token};  
625    
626          redo A;  ## NOTE: HTML5 spec says that the encoding layer MUST NOT strip BOM
627        } else {  ## and the HTML layer MUST ignore it.  However, we does strip BOM in
628          !!!parse-error (type => 'string after PUBLIC literal');  ## the encoding layer and the HTML layer does not ignore any U+FEFF,
629          $self->{state} = 'bogus DOCTYPE';  ## because the core part of our HTML parser expects a string of character,
630          !!!next-input-character;  ## not a string of bytes or code units or anything which might contain a BOM.
631          redo A;  ## Therefore, any parser interface that accepts a string of bytes,
632        }  ## such as |parse_byte_string| in this module, must ensure that it does
633      } elsif ($self->{state} eq 'before DOCTYPE system identifier') {  ## strip the BOM and never strip any ZWNBSP.
       if ({  
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_input_character}}) {  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == 0x0022) { # "  
         $self->{current_token}->{system_identifier} = ''; # DOCTYPE  
         $self->{state} = 'DOCTYPE system identifier (double-quoted)';  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == 0x0027) { # '  
         $self->{current_token}->{system_identifier} = ''; # DOCTYPE  
         $self->{state} = 'DOCTYPE system identifier (single-quoted)';  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == 0x003E) { # >  
         !!!parse-error (type => 'no SYSTEM literal');  
         $self->{state} = 'data';  
         !!!next-input-character;  
   
         delete $self->{current_token}->{correct};  
         !!!emit ($self->{current_token}); # DOCTYPE  
         undef $self->{current_token};  
   
         redo A;  
       } elsif ($self->{next_input_character} == -1) {  
         !!!parse-error (type => 'unclosed DOCTYPE');  
   
         $self->{state} = 'data';  
         ## recomsume  
   
         delete $self->{current_token}->{correct};  
         !!!emit ($self->{current_token}); # DOCTYPE  
         undef $self->{current_token};  
634    
635          redo A;  sub parse_char_string ($$$;$$) {
636        } else {    #my ($self, $s, $doc, $onerror, $get_wrapper) = @_;
637          !!!parse-error (type => 'string after PUBLIC literal');    my $self = shift;
638          $self->{state} = 'bogus DOCTYPE';    my $s = ref $_[0] ? $_[0] : \($_[0]);
639          !!!next-input-character;    require Whatpm::Charset::DecodeHandle;
640          redo A;    my $input = Whatpm::Charset::DecodeHandle::CharString->new ($s);
641        }    return $self->parse_char_stream ($input, @_[1..$#_]);
642      } elsif ($self->{state} eq 'DOCTYPE system identifier (double-quoted)') {  } # parse_char_string
643        if ($self->{next_input_character} == 0x0022) { # "  *parse_string = \&parse_char_string; ## NOTE: Alias for backward compatibility.
644          $self->{state} = 'after DOCTYPE system identifier';  
645          !!!next-input-character;  sub parse_char_stream ($$$;$$) {
646          redo A;    my $self = ref $_[0] ? shift : shift->new;
647        } elsif ($self->{next_input_character} == -1) {    my $input = $_[0];
648          !!!parse-error (type => 'unclosed SYSTEM literal');    $self->{document} = $_[1];
649      @{$self->{document}->child_nodes} = ();
         $self->{state} = 'data';  
         ## reconsume  
   
         delete $self->{current_token}->{correct};  
         !!!emit ($self->{current_token}); # DOCTYPE  
         undef $self->{current_token};  
650    
651          redo A;    ## NOTE: |set_inner_html| copies most of this method's code
       } else {  
         $self->{current_token}->{system_identifier} # DOCTYPE  
             .= chr $self->{next_input_character};  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} eq 'DOCTYPE system identifier (single-quoted)') {  
       if ($self->{next_input_character} == 0x0027) { # '  
         $self->{state} = 'after DOCTYPE system identifier';  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == -1) {  
         !!!parse-error (type => 'unclosed SYSTEM literal');  
   
         $self->{state} = 'data';  
         ## reconsume  
   
         delete $self->{current_token}->{correct};  
         !!!emit ($self->{current_token}); # DOCTYPE  
         undef $self->{current_token};  
652    
653          redo A;    ## Confidence: irrelevant.
654        } else {    $self->{confident} = 1 unless exists $self->{confident};
         $self->{current_token}->{system_identifier} # DOCTYPE  
             .= chr $self->{next_input_character};  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       }  
     } elsif ($self->{state} eq 'after DOCTYPE system identifier') {  
       if ({  
             0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,  
             #0x000D => 1, # HT, LF, VT, FF, SP, CR  
           }->{$self->{next_input_character}}) {  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
       } elsif ($self->{next_input_character} == 0x003E) { # >  
         $self->{state} = 'data';  
         !!!next-input-character;  
   
         !!!emit ($self->{current_token}); # DOCTYPE  
         undef $self->{current_token};  
   
         redo A;  
       } elsif ($self->{next_input_character} == -1) {  
         !!!parse-error (type => 'unclosed DOCTYPE');  
   
         $self->{state} = 'data';  
         ## recomsume  
   
         delete $self->{current_token}->{correct};  
         !!!emit ($self->{current_token}); # DOCTYPE  
         undef $self->{current_token};  
655    
656          redo A;    $self->{document}->input_encoding ($self->{input_encoding})
657        } else {        if defined $self->{input_encoding};
658          !!!parse-error (type => 'string after SYSTEM literal');  ## TODO: |{input_encoding}| is needless?
659          $self->{state} = 'bogus DOCTYPE';  
660          !!!next-input-character;    $self->{line_prev} = $self->{line} = 1;
661          redo A;    $self->{column_prev} = -1;
662        }    $self->{column} = 0;
663      } elsif ($self->{state} eq 'bogus DOCTYPE') {    $self->{set_nc} = sub {
664        if ($self->{next_input_character} == 0x003E) { # >      my $self = shift;
         $self->{state} = 'data';  
         !!!next-input-character;  
   
         delete $self->{current_token}->{correct};  
         !!!emit ($self->{current_token}); # DOCTYPE  
         undef $self->{current_token};  
   
         redo A;  
       } elsif ($self->{next_input_character} == -1) {  
         !!!parse-error (type => 'unclosed DOCTYPE');  
         $self->{state} = 'data';  
         ## reconsume  
   
         delete $self->{current_token}->{correct};  
         !!!emit ($self->{current_token}); # DOCTYPE  
         undef $self->{current_token};  
665    
666          redo A;      my $char = '';
667        } else {      if (defined $self->{next_nc}) {
668          ## Stay in the state        $char = $self->{next_nc};
669          !!!next-input-character;        delete $self->{next_nc};
670          redo A;        $self->{nc} = ord $char;
       }  
671      } else {      } else {
672        die "$0: $self->{state}: Unknown state";        $self->{char_buffer} = '';
673      }        $self->{char_buffer_pos} = 0;
   } # A    
674    
675    die "$0: _get_next_token: unexpected case";        my $count = $input->manakai_read_until
676  } # _get_next_token           ($self->{char_buffer}, qr/[^\x00\x0A\x0D]/, $self->{char_buffer_pos});
677          if ($count) {
678  sub _tokenize_attempt_to_consume_an_entity ($) {          $self->{line_prev} = $self->{line};
679    my $self = shift;          $self->{column_prev} = $self->{column};
680            $self->{column}++;
681    if ({          $self->{nc}
682         0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, # HT, LF, VT, FF,              = ord substr ($self->{char_buffer}, $self->{char_buffer_pos}++, 1);
683         0x0020 => 1, 0x003C => 1, 0x0026 => 1, -1 => 1, # SP, <, & # 0x000D # CR          return;
       }->{$self->{next_input_character}}) {  
     ## Don't consume  
     ## No error  
     return undef;  
   } elsif ($self->{next_input_character} == 0x0023) { # #  
     !!!next-input-character;  
     if ($self->{next_input_character} == 0x0078 or # x  
         $self->{next_input_character} == 0x0058) { # X  
       my $num;  
       X: {  
         my $x_char = $self->{next_input_character};  
         !!!next-input-character;  
         if (0x0030 <= $self->{next_input_character} and  
             $self->{next_input_character} <= 0x0039) { # 0..9  
           $num ||= 0;  
           $num *= 0x10;  
           $num += $self->{next_input_character} - 0x0030;  
           redo X;  
         } elsif (0x0061 <= $self->{next_input_character} and  
                  $self->{next_input_character} <= 0x0066) { # a..f  
           ## ISSUE: the spec says U+0078, which is apparently incorrect  
           $num ||= 0;  
           $num *= 0x10;  
           $num += $self->{next_input_character} - 0x0060 + 9;  
           redo X;  
         } elsif (0x0041 <= $self->{next_input_character} and  
                  $self->{next_input_character} <= 0x0046) { # A..F  
           ## ISSUE: the spec says U+0058, which is apparently incorrect  
           $num ||= 0;  
           $num *= 0x10;  
           $num += $self->{next_input_character} - 0x0040 + 9;  
           redo X;  
         } elsif (not defined $num) { # no hexadecimal digit  
           !!!parse-error (type => 'bare hcro');  
           $self->{next_input_character} = 0x0023; # #  
           !!!back-next-input-character ($x_char);  
           return undef;  
         } elsif ($self->{next_input_character} == 0x003B) { # ;  
           !!!next-input-character;  
         } else {  
           !!!parse-error (type => 'no refc');  
         }  
   
         ## TODO: check the definition for |a valid Unicode character|.  
         ## <http://lists.whatwg.org/pipermail/whatwg-whatwg.org/2006-December/thread.html#8189>  
         if ($num > 1114111 or $num == 0) {  
           $num = 0xFFFD; # REPLACEMENT CHARACTER  
           ## ISSUE: Why this is not an error?  
         } elsif (0x80 <= $num and $num <= 0x9F) {  
           !!!parse-error (type => sprintf 'c1 entity:U+%04X', $num);  
           $num = $c1_entity_char->{$num};  
         }  
   
         return {type => 'character', data => chr $num};  
       } # X  
     } elsif (0x0030 <= $self->{next_input_character} and  
              $self->{next_input_character} <= 0x0039) { # 0..9  
       my $code = $self->{next_input_character} - 0x0030;  
       !!!next-input-character;  
         
       while (0x0030 <= $self->{next_input_character} and  
                 $self->{next_input_character} <= 0x0039) { # 0..9  
         $code *= 10;  
         $code += $self->{next_input_character} - 0x0030;  
           
         !!!next-input-character;  
684        }        }
685    
686        if ($self->{next_input_character} == 0x003B) { # ;        if ($input->read ($char, 1)) {
687          !!!next-input-character;          $self->{nc} = ord $char;
688        } else {        } else {
689          !!!parse-error (type => 'no refc');          $self->{nc} = -1;
690            return;
691        }        }
692        }
693    
694        ## TODO: check the definition for |a valid Unicode character|.      ($self->{line_prev}, $self->{column_prev})
695        if ($code > 1114111 or $code == 0) {          = ($self->{line}, $self->{column});
696          $code = 0xFFFD; # REPLACEMENT CHARACTER      $self->{column}++;
697          ## ISSUE: Why this is not an error?      
698        } elsif (0x80 <= $code and $code <= 0x9F) {      if ($self->{nc} == 0x000A) { # LF
699          !!!parse-error (type => sprintf 'c1 entity:U+%04X', $code);        !!!cp ('j1');
700          $code = $c1_entity_char->{$code};        $self->{line}++;
701        }        $self->{column} = 0;
702              } elsif ($self->{nc} == 0x000D) { # CR
703        return {type => 'character', data => chr $code};        !!!cp ('j2');
704      } else {  ## TODO: support for abort/streaming
705        !!!parse-error (type => 'bare nero');        my $next = '';
706        !!!back-next-input-character ($self->{next_input_character});        if ($input->read ($next, 1) and $next ne "\x0A") {
707        $self->{next_input_character} = 0x0023; # #          $self->{next_nc} = $next;
708        return undef;        }
709          $self->{nc} = 0x000A; # LF # MUST
710          $self->{line}++;
711          $self->{column} = 0;
712        } elsif ($self->{nc} == 0x0000) { # NULL
713          !!!cp ('j4');
714          !!!parse-error (type => 'NULL');
715          $self->{nc} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
716      }      }
717    } elsif ((0x0041 <= $self->{next_input_character} and    };
718              $self->{next_input_character} <= 0x005A) or  
719             (0x0061 <= $self->{next_input_character} and    $self->{read_until} = sub {
720              $self->{next_input_character} <= 0x007A)) {      #my ($scalar, $specials_range, $offset) = @_;
721      my $entity_name = chr $self->{next_input_character};      return 0 if defined $self->{next_nc};
722      !!!next-input-character;  
723        my $pattern = qr/[^$_[1]\x00\x0A\x0D]/;
724      my $value = $entity_name;      my $offset = $_[2] || 0;
725      my $match;  
726      require Whatpm::_NamedEntityList;      if ($self->{char_buffer_pos} < length $self->{char_buffer}) {
727      our $EntityChar;        pos ($self->{char_buffer}) = $self->{char_buffer_pos};
728          if ($self->{char_buffer} =~ /\G(?>$pattern)+/) {
729      while (length $entity_name < 10 and          substr ($_[0], $offset)
730             ## NOTE: Some number greater than the maximum length of entity name              = substr ($self->{char_buffer}, $-[0], $+[0] - $-[0]);
731             ((0x0041 <= $self->{next_input_character} and # a          my $count = $+[0] - $-[0];
732               $self->{next_input_character} <= 0x005A) or # x          if ($count) {
733              (0x0061 <= $self->{next_input_character} and # a            $self->{column} += $count;
734               $self->{next_input_character} <= 0x007A) or # z            $self->{char_buffer_pos} += $count;
735              (0x0030 <= $self->{next_input_character} and # 0            $self->{line_prev} = $self->{line};
736               $self->{next_input_character} <= 0x0039) or # 9            $self->{column_prev} = $self->{column} - 1;
737              $self->{next_input_character} == 0x003B)) { # ;            $self->{nc} = -1;
       $entity_name .= chr $self->{next_input_character};  
       if (defined $EntityChar->{$entity_name}) {  
         $value = $EntityChar->{$entity_name};  
         if ($self->{next_input_character} == 0x003B) { # ;  
           $match = 1;  
           !!!next-input-character;  
           last;  
         } else {  
           $match = -1;  
738          }          }
739            return $count;
740        } else {        } else {
741          $value .= chr $self->{next_input_character};          return 0;
742        }        }
       !!!next-input-character;  
     }  
       
     if ($match > 0) {  
       return {type => 'character', data => $value};  
     } elsif ($match < 0) {  
       !!!parse-error (type => 'refc');  
       return {type => 'character', data => $value};  
743      } else {      } else {
744        !!!parse-error (type => 'bare ero');        my $count = $input->manakai_read_until ($_[0], $pattern, $_[2]);
745        ## NOTE: No characters are consumed in the spec.        if ($count) {
746        !!!back-token ({type => 'character', data => $value});          $self->{column} += $count;
747        return undef;          $self->{line_prev} = $self->{line};
748            $self->{column_prev} = $self->{column} - 1;
749            $self->{nc} = -1;
750          }
751          return $count;
752      }      }
753      }; # $self->{read_until}
754    
755      my $onerror = $_[2] || sub {
756        my (%opt) = @_;
757        my $line = $opt{token} ? $opt{token}->{line} : $opt{line};
758        my $column = $opt{token} ? $opt{token}->{column} : $opt{column};
759        warn "Parse error ($opt{type}) at line $line column $column\n";
760      };
761      $self->{parse_error} = sub {
762        $onerror->(line => $self->{line}, column => $self->{column}, @_);
763      };
764    
765      my $char_onerror = sub {
766        my (undef, $type, %opt) = @_;
767        !!!parse-error (layer => 'encode',
768                        line => $self->{line}, column => $self->{column} + 1,
769                        %opt, type => $type);
770      }; # $char_onerror
771    
772      if ($_[3]) {
773        $input = $_[3]->($input);
774        $input->onerror ($char_onerror);
775    } else {    } else {
776      ## no characters are consumed      $input->onerror ($char_onerror) unless defined $input->onerror;
     !!!parse-error (type => 'bare ero');  
     return undef;  
777    }    }
778  } # _tokenize_attempt_to_consume_an_entity  
779      $self->_initialize_tokenizer;
780      $self->_initialize_tree_constructor;
781      $self->_construct_tree;
782      $self->_terminate_tree_constructor;
783    
784      delete $self->{parse_error}; # remove loop
785    
786      return $self->{document};
787    } # parse_char_stream
788    
789    sub new ($) {
790      my $class = shift;
791      my $self = bless {
792        level => {must => 'm',
793                  should => 's',
794                  warn => 'w',
795                  info => 'i',
796                  uncertain => 'u'},
797      }, $class;
798      $self->{set_nc} = sub {
799        $self->{nc} = -1;
800      };
801      $self->{parse_error} = sub {
802        #
803      };
804      $self->{change_encoding} = sub {
805        # if ($_[0] is a supported encoding) {
806        #   run "change the encoding" algorithm;
807        #   throw Whatpm::HTML::RestartParser (charset => $new_encoding);
808        # }
809      };
810      $self->{application_cache_selection} = sub {
811        #
812      };
813      return $self;
814    } # new
815    
816    ## Insertion modes
817    
818    sub AFTER_HTML_IMS () { 0b100 }
819    sub HEAD_IMS ()       { 0b1000 }
820    sub BODY_IMS ()       { 0b10000 }
821    sub BODY_TABLE_IMS () { 0b100000 }
822    sub TABLE_IMS ()      { 0b1000000 }
823    sub ROW_IMS ()        { 0b10000000 }
824    sub BODY_AFTER_IMS () { 0b100000000 }
825    sub FRAME_IMS ()      { 0b1000000000 }
826    sub SELECT_IMS ()     { 0b10000000000 }
827    #sub IN_FOREIGN_CONTENT_IM () { 0b100000000000 } # see Whatpm::HTML::Tokenizer
828        ## NOTE: "in foreign content" insertion mode is special; it is combined
829        ## with the secondary insertion mode.  In this parser, they are stored
830        ## together in the bit-or'ed form.
831    sub IN_CDATA_RCDATA_IM () { 0b1000000000000 }
832        ## NOTE: "in CDATA/RCDATA" insertion mode is also special; it is
833        ## combined with the original insertion mode.  In thie parser,
834        ## they are stored together in the bit-or'ed form.
835    
836    sub IM_MASK () { 0b11111111111 }
837    
838    ## NOTE: "initial" and "before html" insertion modes have no constants.
839    
840    ## NOTE: "after after body" insertion mode.
841    sub AFTER_HTML_BODY_IM () { AFTER_HTML_IMS | BODY_AFTER_IMS }
842    
843    ## NOTE: "after after frameset" insertion mode.
844    sub AFTER_HTML_FRAMESET_IM () { AFTER_HTML_IMS | FRAME_IMS }
845    
846    sub IN_HEAD_IM () { HEAD_IMS | 0b00 }
847    sub IN_HEAD_NOSCRIPT_IM () { HEAD_IMS | 0b01 }
848    sub AFTER_HEAD_IM () { HEAD_IMS | 0b10 }
849    sub BEFORE_HEAD_IM () { HEAD_IMS | 0b11 }
850    sub IN_BODY_IM () { BODY_IMS }
851    sub IN_CELL_IM () { BODY_IMS | BODY_TABLE_IMS | 0b01 }
852    sub IN_CAPTION_IM () { BODY_IMS | BODY_TABLE_IMS | 0b10 }
853    sub IN_ROW_IM () { TABLE_IMS | ROW_IMS | 0b01 }
854    sub IN_TABLE_BODY_IM () { TABLE_IMS | ROW_IMS | 0b10 }
855    sub IN_TABLE_IM () { TABLE_IMS }
856    sub AFTER_BODY_IM () { BODY_AFTER_IMS }
857    sub IN_FRAMESET_IM () { FRAME_IMS | 0b01 }
858    sub AFTER_FRAMESET_IM () { FRAME_IMS | 0b10 }
859    sub IN_SELECT_IM () { SELECT_IMS | 0b01 }
860    sub IN_SELECT_IN_TABLE_IM () { SELECT_IMS | 0b10 }
861    sub IN_COLUMN_GROUP_IM () { 0b10 }
862    
863  sub _initialize_tree_constructor ($) {  sub _initialize_tree_constructor ($) {
864    my $self = shift;    my $self = shift;
# Line 1702  sub _initialize_tree_constructor ($) { Line 867  sub _initialize_tree_constructor ($) {
867    ## TODO: Turn mutation events off # MUST    ## TODO: Turn mutation events off # MUST
868    ## TODO: Turn loose Document option (manakai extension) on    ## TODO: Turn loose Document option (manakai extension) on
869    $self->{document}->manakai_is_html (1); # MUST    $self->{document}->manakai_is_html (1); # MUST
870      $self->{document}->set_user_data (manakai_source_line => 1);
871      $self->{document}->set_user_data (manakai_source_column => 1);
872  } # _initialize_tree_constructor  } # _initialize_tree_constructor
873    
874  sub _terminate_tree_constructor ($) {  sub _terminate_tree_constructor ($) {
# Line 1721  sub _construct_tree ($) { Line 888  sub _construct_tree ($) {
888    ## When an interactive UA render the $self->{document} available    ## When an interactive UA render the $self->{document} available
889    ## to the user, or when it begin accepting user input, are    ## to the user, or when it begin accepting user input, are
890    ## not defined.    ## not defined.
   
   ## Append a character: collect it and all subsequent consecutive  
   ## characters and insert one Text node whose data is concatenation  
   ## of all those characters. # MUST  
891        
892    !!!next-token;    !!!next-token;
893    
   $self->{insertion_mode} = 'before head';  
894    undef $self->{form_element};    undef $self->{form_element};
895    undef $self->{head_element};    undef $self->{head_element};
896      undef $self->{head_element_inserted};
897    $self->{open_elements} = [];    $self->{open_elements} = [];
898    undef $self->{inner_html_node};    undef $self->{inner_html_node};
899      undef $self->{ignore_newline};
900    
901      ## NOTE: The "initial" insertion mode.
902    $self->_tree_construction_initial; # MUST    $self->_tree_construction_initial; # MUST
903    
904      ## NOTE: The "before html" insertion mode.
905    $self->_tree_construction_root_element;    $self->_tree_construction_root_element;
906      $self->{insertion_mode} = BEFORE_HEAD_IM;
907    
908      ## NOTE: The "before head" insertion mode and so on.
909    $self->_tree_construction_main;    $self->_tree_construction_main;
910  } # _construct_tree  } # _construct_tree
911    
912  sub _tree_construction_initial ($) {  sub _tree_construction_initial ($) {
913    my $self = shift;    my $self = shift;
914    
915      ## NOTE: "initial" insertion mode
916    
917    INITIAL: {    INITIAL: {
918      if ($token->{type} eq 'DOCTYPE') {      if ($token->{type} == DOCTYPE_TOKEN) {
919        ## NOTE: Conformance checkers MAY, instead of reporting "not HTML5"        ## NOTE: Conformance checkers MAY, instead of reporting "not
920        ## error, switch to a conformance checking mode for another        ## HTML5" error, switch to a conformance checking mode for
921        ## language.        ## another language.  (We don't support such mode switchings; it
922          ## is nonsense to do anything different from what browsers do.)
923        my $doctype_name = $token->{name};        my $doctype_name = $token->{name};
924        $doctype_name = '' unless defined $doctype_name;        $doctype_name = '' unless defined $doctype_name;
925        $doctype_name =~ tr/a-z/A-Z/;        my $doctype = $self->{document}->create_document_type_definition
926        if (not defined $token->{name} or # <!DOCTYPE>            ($doctype_name);
927            defined $token->{public_identifier} or  
928            defined $token->{system_identifier}) {        $doctype_name =~ tr/A-Z/a-z/; # ASCII case-insensitive
929          !!!parse-error (type => 'not HTML5');        if ($doctype_name ne 'html') {
930        } elsif ($doctype_name ne 'HTML') {          !!!cp ('t1');
931          ## ISSUE: ASCII case-insensitive? (in fact it does not matter)          !!!parse-error (type => 'not HTML5', token => $token);
932          !!!parse-error (type => 'not HTML5');        } elsif (defined $token->{pubid}) {
933            !!!cp ('t2');
934            ## XXX Obsolete permitted DOCTYPEs
935            !!!parse-error (type => 'not HTML5', token => $token);
936          } elsif (defined $token->{sysid}) {
937            if ($token->{sysid} eq 'about:legacy-compat') {
938              !!!cp ('t1.2'); ## <!DOCTYPE HTML SYSTEM "about:legacy-compat">
939              !!!parse-error (type => 'XSLT-compat', token => $token,
940                              level => $self->{level}->{should});
941            } else {
942              !!!parse-error (type => 'not HTML5', token => $token);
943            }
944          } else { ## <!DOCTYPE HTML>
945            !!!cp ('t3');
946            #
947        }        }
948                
949        my $doctype = $self->{document}->create_document_type_definition        ## NOTE: Default value for both |public_id| and |system_id| attributes
950          ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?        ## are empty strings, so that we don't set any value in missing cases.
951        $doctype->public_id ($token->{public_identifier})        $doctype->public_id ($token->{pubid}) if defined $token->{pubid};
952            if defined $token->{public_identifier};        $doctype->system_id ($token->{sysid}) if defined $token->{sysid};
953        $doctype->system_id ($token->{system_identifier})  
           if defined $token->{system_identifier};  
954        ## NOTE: Other DocumentType attributes are null or empty lists.        ## NOTE: Other DocumentType attributes are null or empty lists.
955        ## ISSUE: internalSubset = null??        ## In Firefox3, |internalSubset| attribute is set to the empty
956          ## string, while |null| is an allowed value for the attribute
957          ## according to DOM3 Core.
958        $self->{document}->append_child ($doctype);        $self->{document}->append_child ($doctype);
959                
960        if (not $token->{correct} or $doctype_name ne 'HTML') {        if ($token->{quirks} or $doctype_name ne 'html') {
961            !!!cp ('t4');
962          $self->{document}->manakai_compat_mode ('quirks');          $self->{document}->manakai_compat_mode ('quirks');
963        } elsif (defined $token->{public_identifier}) {        } elsif (defined $token->{pubid}) {
964          my $pubid = $token->{public_identifier};          my $pubid = $token->{pubid};
965          $pubid =~ tr/a-z/A-z/;          $pubid =~ tr/a-z/A-z/;
966          if ({          my $prefix = [
967            "+//SILMARIL//DTD HTML PRO V0R11 19970101//EN" => 1,            "+//SILMARIL//DTD HTML PRO V0R11 19970101//",
968            "-//ADVASOFT LTD//DTD HTML 3.0 ASWEDIT + EXTENSIONS//EN" => 1,            "-//ADVASOFT LTD//DTD HTML 3.0 ASWEDIT + EXTENSIONS//",
969            "-//AS//DTD HTML 3.0 ASWEDIT + EXTENSIONS//EN" => 1,            "-//AS//DTD HTML 3.0 ASWEDIT + EXTENSIONS//",
970            "-//IETF//DTD HTML 2.0 LEVEL 1//EN" => 1,            "-//IETF//DTD HTML 2.0 LEVEL 1//",
971            "-//IETF//DTD HTML 2.0 LEVEL 2//EN" => 1,            "-//IETF//DTD HTML 2.0 LEVEL 2//",
972            "-//IETF//DTD HTML 2.0 STRICT LEVEL 1//EN" => 1,            "-//IETF//DTD HTML 2.0 STRICT LEVEL 1//",
973            "-//IETF//DTD HTML 2.0 STRICT LEVEL 2//EN" => 1,            "-//IETF//DTD HTML 2.0 STRICT LEVEL 2//",
974            "-//IETF//DTD HTML 2.0 STRICT//EN" => 1,            "-//IETF//DTD HTML 2.0 STRICT//",
975            "-//IETF//DTD HTML 2.0//EN" => 1,            "-//IETF//DTD HTML 2.0//",
976            "-//IETF//DTD HTML 2.1E//EN" => 1,            "-//IETF//DTD HTML 2.1E//",
977            "-//IETF//DTD HTML 3.0//EN" => 1,            "-//IETF//DTD HTML 3.0//",
978            "-//IETF//DTD HTML 3.0//EN//" => 1,            "-//IETF//DTD HTML 3.2 FINAL//",
979            "-//IETF//DTD HTML 3.2 FINAL//EN" => 1,            "-//IETF//DTD HTML 3.2//",
980            "-//IETF//DTD HTML 3.2//EN" => 1,            "-//IETF//DTD HTML 3//",
981            "-//IETF//DTD HTML 3//EN" => 1,            "-//IETF//DTD HTML LEVEL 0//",
982            "-//IETF//DTD HTML LEVEL 0//EN" => 1,            "-//IETF//DTD HTML LEVEL 1//",
983            "-//IETF//DTD HTML LEVEL 0//EN//2.0" => 1,            "-//IETF//DTD HTML LEVEL 2//",
984            "-//IETF//DTD HTML LEVEL 1//EN" => 1,            "-//IETF//DTD HTML LEVEL 3//",
985            "-//IETF//DTD HTML LEVEL 1//EN//2.0" => 1,            "-//IETF//DTD HTML STRICT LEVEL 0//",
986            "-//IETF//DTD HTML LEVEL 2//EN" => 1,            "-//IETF//DTD HTML STRICT LEVEL 1//",
987            "-//IETF//DTD HTML LEVEL 2//EN//2.0" => 1,            "-//IETF//DTD HTML STRICT LEVEL 2//",
988            "-//IETF//DTD HTML LEVEL 3//EN" => 1,            "-//IETF//DTD HTML STRICT LEVEL 3//",
989            "-//IETF//DTD HTML LEVEL 3//EN//3.0" => 1,            "-//IETF//DTD HTML STRICT//",
990            "-//IETF//DTD HTML STRICT LEVEL 0//EN" => 1,            "-//IETF//DTD HTML//",
991            "-//IETF//DTD HTML STRICT LEVEL 0//EN//2.0" => 1,            "-//METRIUS//DTD METRIUS PRESENTATIONAL//",
992            "-//IETF//DTD HTML STRICT LEVEL 1//EN" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML STRICT//",
993            "-//IETF//DTD HTML STRICT LEVEL 1//EN//2.0" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML//",
994            "-//IETF//DTD HTML STRICT LEVEL 2//EN" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 TABLES//",
995            "-//IETF//DTD HTML STRICT LEVEL 2//EN//2.0" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML STRICT//",
996            "-//IETF//DTD HTML STRICT LEVEL 3//EN" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML//",
997            "-//IETF//DTD HTML STRICT LEVEL 3//EN//3.0" => 1,            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 TABLES//",
998            "-//IETF//DTD HTML STRICT//EN" => 1,            "-//NETSCAPE COMM. CORP.//DTD HTML//",
999            "-//IETF//DTD HTML STRICT//EN//2.0" => 1,            "-//NETSCAPE COMM. CORP.//DTD STRICT HTML//",
1000            "-//IETF//DTD HTML STRICT//EN//3.0" => 1,            "-//O'REILLY AND ASSOCIATES//DTD HTML 2.0//",
1001            "-//IETF//DTD HTML//EN" => 1,            "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED 1.0//",
1002            "-//IETF//DTD HTML//EN//2.0" => 1,            "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED RELAXED 1.0//",
1003            "-//IETF//DTD HTML//EN//3.0" => 1,            "-//SOFTQUAD SOFTWARE//DTD HOTMETAL PRO 6.0::19990601::EXTENSIONS TO HTML 4.0//",
1004            "-//METRIUS//DTD METRIUS PRESENTATIONAL//EN" => 1,            "-//SOFTQUAD//DTD HOTMETAL PRO 4.0::19971010::EXTENSIONS TO HTML 4.0//",
1005            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML STRICT//EN" => 1,            "-//SPYGLASS//DTD HTML 2.0 EXTENDED//",
1006            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML//EN" => 1,            "-//SQ//DTD HTML 2.0 HOTMETAL + EXTENSIONS//",
1007            "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 TABLES//EN" => 1,            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA HTML//",
1008            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML STRICT//EN" => 1,            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA STRICT HTML//",
1009            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML//EN" => 1,            "-//W3C//DTD HTML 3 1995-03-24//",
1010            "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 TABLES//EN" => 1,            "-//W3C//DTD HTML 3.2 DRAFT//",
1011            "-//NETSCAPE COMM. CORP.//DTD HTML//EN" => 1,            "-//W3C//DTD HTML 3.2 FINAL//",
1012            "-//NETSCAPE COMM. CORP.//DTD STRICT HTML//EN" => 1,            "-//W3C//DTD HTML 3.2//",
1013            "-//O'REILLY AND ASSOCIATES//DTD HTML 2.0//EN" => 1,            "-//W3C//DTD HTML 3.2S DRAFT//",
1014            "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED 1.0//EN" => 1,            "-//W3C//DTD HTML 4.0 FRAMESET//",
1015            "-//SPYGLASS//DTD HTML 2.0 EXTENDED//EN" => 1,            "-//W3C//DTD HTML 4.0 TRANSITIONAL//",
1016            "-//SQ//DTD HTML 2.0 HOTMETAL + EXTENSIONS//EN" => 1,            "-//W3C//DTD HTML EXPERIMETNAL 19960712//",
1017            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA HTML//EN" => 1,            "-//W3C//DTD HTML EXPERIMENTAL 970421//",
1018            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA STRICT HTML//EN" => 1,            "-//W3C//DTD W3 HTML//",
1019            "-//W3C//DTD HTML 3 1995-03-24//EN" => 1,            "-//W3O//DTD W3 HTML 3.0//",
1020            "-//W3C//DTD HTML 3.2 DRAFT//EN" => 1,            "-//WEBTECHS//DTD MOZILLA HTML 2.0//",
1021            "-//W3C//DTD HTML 3.2 FINAL//EN" => 1,            "-//WEBTECHS//DTD MOZILLA HTML//",
1022            "-//W3C//DTD HTML 3.2//EN" => 1,          ]; # $prefix
1023            "-//W3C//DTD HTML 3.2S DRAFT//EN" => 1,          my $match;
1024            "-//W3C//DTD HTML 4.0 FRAMESET//EN" => 1,          for (@$prefix) {
1025            "-//W3C//DTD HTML 4.0 TRANSITIONAL//EN" => 1,            if (substr ($prefix, 0, length $_) eq $_) {
1026            "-//W3C//DTD HTML EXPERIMETNAL 19960712//EN" => 1,              $match = 1;
1027            "-//W3C//DTD HTML EXPERIMENTAL 970421//EN" => 1,              last;
1028            "-//W3C//DTD W3 HTML//EN" => 1,            }
1029            "-//W3O//DTD W3 HTML 3.0//EN" => 1,          }
1030            "-//W3O//DTD W3 HTML 3.0//EN//" => 1,          if ($match or
1031            "-//W3O//DTD W3 HTML STRICT 3.0//EN//" => 1,              $pubid eq "-//W3O//DTD W3 HTML STRICT 3.0//EN//" or
1032            "-//WEBTECHS//DTD MOZILLA HTML 2.0//EN" => 1,              $pubid eq "-/W3C/DTD HTML 4.0 TRANSITIONAL/EN" or
1033            "-//WEBTECHS//DTD MOZILLA HTML//EN" => 1,              $pubid eq "HTML") {
1034            "-/W3C/DTD HTML 4.0 TRANSITIONAL/EN" => 1,            !!!cp ('t5');
           "HTML" => 1,  
         }->{$pubid}) {  
1035            $self->{document}->manakai_compat_mode ('quirks');            $self->{document}->manakai_compat_mode ('quirks');
1036          } elsif ($pubid eq "-//W3C//DTD HTML 4.01 FRAMESET//EN" or          } elsif ($pubid =~ m[^-//W3C//DTD HTML 4.01 FRAMESET//] or
1037                   $pubid eq "-//W3C//DTD HTML 4.01 TRANSITIONAL//EN") {                   $pubid =~ m[^-//W3C//DTD HTML 4.01 TRANSITIONAL//]) {
1038            if (defined $token->{system_identifier}) {            if (defined $token->{sysid}) {
1039                !!!cp ('t6');
1040              $self->{document}->manakai_compat_mode ('quirks');              $self->{document}->manakai_compat_mode ('quirks');
1041            } else {            } else {
1042                !!!cp ('t7');
1043              $self->{document}->manakai_compat_mode ('limited quirks');              $self->{document}->manakai_compat_mode ('limited quirks');
1044            }            }
1045          } elsif ($pubid eq "-//W3C//DTD XHTML 1.0 Frameset//EN" or          } elsif ($pubid =~ m[^-//W3C//DTD XHTML 1.0 FRAMESET//] or
1046                   $pubid eq "-//W3C//DTD XHTML 1.0 Transitional//EN") {                   $pubid =~ m[^-//W3C//DTD XHTML 1.0 TRANSITIONAL//]) {
1047              !!!cp ('t8');
1048            $self->{document}->manakai_compat_mode ('limited quirks');            $self->{document}->manakai_compat_mode ('limited quirks');
1049            } else {
1050              !!!cp ('t9');
1051          }          }
1052          } else {
1053            !!!cp ('t10');
1054        }        }
1055        if (defined $token->{system_identifier}) {        if (defined $token->{sysid}) {
1056          my $sysid = $token->{system_identifier};          my $sysid = $token->{sysid};
1057          $sysid =~ tr/A-Z/a-z/;          $sysid =~ tr/A-Z/a-z/;
1058          if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {          if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {
1059              ## NOTE: Ensure that |PUBLIC "(limited quirks)" "(quirks)"| is
1060              ## marked as quirks.
1061            $self->{document}->manakai_compat_mode ('quirks');            $self->{document}->manakai_compat_mode ('quirks');
1062              !!!cp ('t11');
1063            } else {
1064              !!!cp ('t12');
1065          }          }
1066          } else {
1067            !!!cp ('t13');
1068        }        }
1069                
1070        ## Go to the root element phase.        ## Go to the "before html" insertion mode.
1071        !!!next-token;        !!!next-token;
1072        return;        return;
1073      } elsif ({      } elsif ({
1074                'start tag' => 1,                START_TAG_TOKEN, 1,
1075                'end tag' => 1,                END_TAG_TOKEN, 1,
1076                'end-of-file' => 1,                END_OF_FILE_TOKEN, 1,
1077               }->{$token->{type}}) {               }->{$token->{type}}) {
1078        !!!parse-error (type => 'no DOCTYPE');        !!!cp ('t14');
1079          !!!parse-error (type => 'no DOCTYPE', token => $token);
1080        $self->{document}->manakai_compat_mode ('quirks');        $self->{document}->manakai_compat_mode ('quirks');
1081        ## Go to the root element phase        ## Go to the "before html" insertion mode.
1082        ## reprocess        ## reprocess
1083          !!!ack-later;
1084        return;        return;
1085      } elsif ($token->{type} eq 'character') {      } elsif ($token->{type} == CHARACTER_TOKEN) {
1086        if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D        if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
1087          ## Ignore the token          ## Ignore the token
1088    
1089          unless (length $token->{data}) {          unless (length $token->{data}) {
1090            ## Stay in the phase            !!!cp ('t15');
1091              ## Stay in the insertion mode.
1092            !!!next-token;            !!!next-token;
1093            redo INITIAL;            redo INITIAL;
1094            } else {
1095              !!!cp ('t16');
1096          }          }
1097          } else {
1098            !!!cp ('t17');
1099        }        }
1100    
1101        !!!parse-error (type => 'no DOCTYPE');        !!!parse-error (type => 'no DOCTYPE', token => $token);
1102        $self->{document}->manakai_compat_mode ('quirks');        $self->{document}->manakai_compat_mode ('quirks');
1103        ## Go to the root element phase        ## Go to the "before html" insertion mode.
1104        ## reprocess        ## reprocess
1105        return;        return;
1106      } elsif ($token->{type} eq 'comment') {      } elsif ($token->{type} == COMMENT_TOKEN) {
1107          !!!cp ('t18');
1108        my $comment = $self->{document}->create_comment ($token->{data});        my $comment = $self->{document}->create_comment ($token->{data});
1109        $self->{document}->append_child ($comment);        $self->{document}->append_child ($comment);
1110                
1111        ## Stay in the phase.        ## Stay in the insertion mode.
1112        !!!next-token;        !!!next-token;
1113        redo INITIAL;        redo INITIAL;
1114      } else {      } else {
1115        die "$0: $token->{type}: Unknown token";        die "$0: $token->{type}: Unknown token type";
1116      }      }
1117    } # INITIAL    } # INITIAL
1118    
1119      die "$0: _tree_construction_initial: This should be never reached";
1120  } # _tree_construction_initial  } # _tree_construction_initial
1121    
1122  sub _tree_construction_root_element ($) {  sub _tree_construction_root_element ($) {
1123    my $self = shift;    my $self = shift;
1124    
1125      ## NOTE: "before html" insertion mode.
1126        
1127    B: {    B: {
1128        if ($token->{type} eq 'DOCTYPE') {        if ($token->{type} == DOCTYPE_TOKEN) {
1129          !!!parse-error (type => 'in html:#DOCTYPE');          !!!cp ('t19');
1130            !!!parse-error (type => 'in html:#DOCTYPE', token => $token);
1131          ## Ignore the token          ## Ignore the token
1132          ## Stay in the phase          ## Stay in the insertion mode.
1133          !!!next-token;          !!!next-token;
1134          redo B;          redo B;
1135        } elsif ($token->{type} eq 'comment') {        } elsif ($token->{type} == COMMENT_TOKEN) {
1136            !!!cp ('t20');
1137          my $comment = $self->{document}->create_comment ($token->{data});          my $comment = $self->{document}->create_comment ($token->{data});
1138          $self->{document}->append_child ($comment);          $self->{document}->append_child ($comment);
1139          ## Stay in the phase          ## Stay in the insertion mode.
1140          !!!next-token;          !!!next-token;
1141          redo B;          redo B;
1142        } elsif ($token->{type} eq 'character') {        } elsif ($token->{type} == CHARACTER_TOKEN) {
1143          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
1144            $self->{document}->manakai_append_text ($1);            ## Ignore the token.
1145            ## ISSUE: DOM3 Core does not allow Document > Text  
1146            unless (length $token->{data}) {            unless (length $token->{data}) {
1147              ## Stay in the phase              !!!cp ('t21');
1148                ## Stay in the insertion mode.
1149              !!!next-token;              !!!next-token;
1150              redo B;              redo B;
1151              } else {
1152                !!!cp ('t22');
1153            }            }
1154            } else {
1155              !!!cp ('t23');
1156          }          }
1157    
1158            $self->{application_cache_selection}->(undef);
1159    
1160          #          #
1161          } elsif ($token->{type} == START_TAG_TOKEN) {
1162            if ($token->{tag_name} eq 'html') {
1163              my $root_element;
1164              !!!create-element ($root_element, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);
1165              $self->{document}->append_child ($root_element);
1166              push @{$self->{open_elements}},
1167                  [$root_element, $el_category->{html}];
1168    
1169              if ($token->{attributes}->{manifest}) {
1170                !!!cp ('t24');
1171                $self->{application_cache_selection}
1172                    ->($token->{attributes}->{manifest}->{value});
1173                ## ISSUE: Spec is unclear on relative references.
1174                ## According to Hixie (#whatwg 2008-03-19), it should be
1175                ## resolved against the base URI of the document in HTML
1176                ## or xml:base of the element in XHTML.
1177              } else {
1178                !!!cp ('t25');
1179                $self->{application_cache_selection}->(undef);
1180              }
1181    
1182              !!!nack ('t25c');
1183    
1184              !!!next-token;
1185              return; ## Go to the "before head" insertion mode.
1186            } else {
1187              !!!cp ('t25.1');
1188              #
1189            }
1190        } elsif ({        } elsif ({
1191                  'start tag' => 1,                  END_TAG_TOKEN, 1,
1192                  'end tag' => 1,                  END_OF_FILE_TOKEN, 1,
                 'end-of-file' => 1,  
1193                 }->{$token->{type}}) {                 }->{$token->{type}}) {
1194          ## ISSUE: There is an issue in the spec          !!!cp ('t26');
1195          #          #
1196        } else {        } else {
1197          die "$0: $token->{type}: Unknown token";          die "$0: $token->{type}: Unknown token type";
1198        }        }
1199        my $root_element; !!!create-element ($root_element, 'html');  
1200        $self->{document}->append_child ($root_element);      my $root_element;
1201        push @{$self->{open_elements}}, [$root_element, 'html'];      !!!create-element ($root_element, $HTML_NS, 'html',, $token);
1202        #$phase = 'main';      $self->{document}->append_child ($root_element);
1203        ## reprocess      push @{$self->{open_elements}}, [$root_element, $el_category->{html}];
1204        #redo B;  
1205        return;      $self->{application_cache_selection}->(undef);
1206    
1207        ## NOTE: Reprocess the token.
1208        !!!ack-later;
1209        return; ## Go to the "before head" insertion mode.
1210    } # B    } # B
1211    
1212      die "$0: _tree_construction_root_element: This should never be reached";
1213  } # _tree_construction_root_element  } # _tree_construction_root_element
1214    
1215  sub _reset_insertion_mode ($) {  sub _reset_insertion_mode ($) {
# Line 1965  sub _reset_insertion_mode ($) { Line 1224  sub _reset_insertion_mode ($) {
1224            
1225      ## Step 3      ## Step 3
1226      S3: {      S3: {
1227        $last = 1 if $self->{open_elements}->[0]->[0] eq $node->[0];        if ($self->{open_elements}->[0]->[0] eq $node->[0]) {
1228        if (defined $self->{inner_html_node}) {          $last = 1;
1229          if ($self->{inner_html_node}->[1] eq 'td' or          if (defined $self->{inner_html_node}) {
1230              $self->{inner_html_node}->[1] eq 'th') {            !!!cp ('t28');
1231              $node = $self->{inner_html_node};
1232            } else {
1233              die "_reset_insertion_mode: t27";
1234            }
1235          }
1236          
1237          ## Step 4..14
1238          my $new_mode;
1239          if ($node->[1] & FOREIGN_EL) {
1240            !!!cp ('t28.1');
1241            ## NOTE: Strictly spaking, the line below only applies to MathML and
1242            ## SVG elements.  Currently the HTML syntax supports only MathML and
1243            ## SVG elements as foreigners.
1244            $new_mode = IN_BODY_IM | IN_FOREIGN_CONTENT_IM;
1245          } elsif ($node->[1] == TABLE_CELL_EL) {
1246            if ($last) {
1247              !!!cp ('t28.2');
1248            #            #
1249          } else {          } else {
1250            $node = $self->{inner_html_node};            !!!cp ('t28.3');
1251              $new_mode = IN_CELL_IM;
1252          }          }
1253          } else {
1254            !!!cp ('t28.4');
1255            $new_mode = {
1256                          select => IN_SELECT_IM,
1257                          ## NOTE: |option| and |optgroup| do not set
1258                          ## insertion mode to "in select" by themselves.
1259                          tr => IN_ROW_IM,
1260                          tbody => IN_TABLE_BODY_IM,
1261                          thead => IN_TABLE_BODY_IM,
1262                          tfoot => IN_TABLE_BODY_IM,
1263                          caption => IN_CAPTION_IM,
1264                          colgroup => IN_COLUMN_GROUP_IM,
1265                          table => IN_TABLE_IM,
1266                          head => IN_BODY_IM, # not in head!
1267                          body => IN_BODY_IM,
1268                          frameset => IN_FRAMESET_IM,
1269                         }->{$node->[0]->manakai_local_name};
1270        }        }
       
       ## Step 4..13  
       my $new_mode = {  
                       select => 'in select',  
                       td => 'in cell',  
                       th => 'in cell',  
                       tr => 'in row',  
                       tbody => 'in table body',  
                       thead => 'in table head',  
                       tfoot => 'in table foot',  
                       caption => 'in caption',  
                       colgroup => 'in column group',  
                       table => 'in table',  
                       head => 'in body', # not in head!  
                       body => 'in body',  
                       frameset => 'in frameset',  
                      }->{$node->[1]};  
1271        $self->{insertion_mode} = $new_mode and return if defined $new_mode;        $self->{insertion_mode} = $new_mode and return if defined $new_mode;
1272                
1273        ## Step 14        ## Step 15
1274        if ($node->[1] eq 'html') {        if ($node->[1] == HTML_EL) {
1275          unless (defined $self->{head_element}) {          unless (defined $self->{head_element}) {
1276            $self->{insertion_mode} = 'before head';            !!!cp ('t29');
1277              $self->{insertion_mode} = BEFORE_HEAD_IM;
1278          } else {          } else {
1279            $self->{insertion_mode} = 'after head';            ## ISSUE: Can this state be reached?
1280              !!!cp ('t30');
1281              $self->{insertion_mode} = AFTER_HEAD_IM;
1282          }          }
1283          return;          return;
1284          } else {
1285            !!!cp ('t31');
1286        }        }
1287                
       ## Step 15  
       $self->{insertion_mode} = 'in body' and return if $last;  
         
1288        ## Step 16        ## Step 16
1289          $self->{insertion_mode} = IN_BODY_IM and return if $last;
1290          
1291          ## Step 17
1292        $i--;        $i--;
1293        $node = $self->{open_elements}->[$i];        $node = $self->{open_elements}->[$i];
1294                
1295        ## Step 17        ## Step 18
1296        redo S3;        redo S3;
1297      } # S3      } # S3
1298    
1299      die "$0: _reset_insertion_mode: This line should never be reached";
1300  } # _reset_insertion_mode  } # _reset_insertion_mode
1301    
1302  sub _tree_construction_main ($) {  sub _tree_construction_main ($) {
1303    my $self = shift;    my $self = shift;
1304    
   my $phase = 'main';  
   
1305    my $active_formatting_elements = [];    my $active_formatting_elements = [];
1306    
1307    my $reconstruct_active_formatting_elements = sub { # MUST    my $reconstruct_active_formatting_elements = sub { # MUST
# Line 2036  sub _tree_construction_main ($) { Line 1318  sub _tree_construction_main ($) {
1318      return if $entry->[0] eq '#marker';      return if $entry->[0] eq '#marker';
1319      for (@{$self->{open_elements}}) {      for (@{$self->{open_elements}}) {
1320        if ($entry->[0] eq $_->[0]) {        if ($entry->[0] eq $_->[0]) {
1321            !!!cp ('t32');
1322          return;          return;
1323        }        }
1324      }      }
# Line 2050  sub _tree_construction_main ($) { Line 1333  sub _tree_construction_main ($) {
1333    
1334        ## Step 6        ## Step 6
1335        if ($entry->[0] eq '#marker') {        if ($entry->[0] eq '#marker') {
1336            !!!cp ('t33_1');
1337          #          #
1338        } else {        } else {
1339          my $in_open_elements;          my $in_open_elements;
1340          OE: for (@{$self->{open_elements}}) {          OE: for (@{$self->{open_elements}}) {
1341            if ($entry->[0] eq $_->[0]) {            if ($entry->[0] eq $_->[0]) {
1342                !!!cp ('t33');
1343              $in_open_elements = 1;              $in_open_elements = 1;
1344              last OE;              last OE;
1345            }            }
1346          }          }
1347          if ($in_open_elements) {          if ($in_open_elements) {
1348              !!!cp ('t34');
1349            #            #
1350          } else {          } else {
1351              ## NOTE: <!DOCTYPE HTML><p><b><i><u></p> <p>X
1352              !!!cp ('t35');
1353            redo S4;            redo S4;
1354          }          }
1355        }        }
# Line 2084  sub _tree_construction_main ($) { Line 1372  sub _tree_construction_main ($) {
1372    
1373        ## Step 11        ## Step 11
1374        unless ($clone->[0] eq $active_formatting_elements->[-1]->[0]) {        unless ($clone->[0] eq $active_formatting_elements->[-1]->[0]) {
1375            !!!cp ('t36');
1376          ## Step 7'          ## Step 7'
1377          $i++;          $i++;
1378          $entry = $active_formatting_elements->[$i];          $entry = $active_formatting_elements->[$i];
1379                    
1380          redo S7;          redo S7;
1381        }        }
1382    
1383          !!!cp ('t37');
1384      } # S7      } # S7
1385    }; # $reconstruct_active_formatting_elements    }; # $reconstruct_active_formatting_elements
1386    
1387    my $clear_up_to_marker = sub {    my $clear_up_to_marker = sub {
1388      for (reverse 0..$#$active_formatting_elements) {      for (reverse 0..$#$active_formatting_elements) {
1389        if ($active_formatting_elements->[$_]->[0] eq '#marker') {        if ($active_formatting_elements->[$_]->[0] eq '#marker') {
1390            !!!cp ('t38');
1391          splice @$active_formatting_elements, $_;          splice @$active_formatting_elements, $_;
1392          return;          return;
1393        }        }
1394      }      }
1395    
1396        !!!cp ('t39');
1397    }; # $clear_up_to_marker    }; # $clear_up_to_marker
1398    
1399    my $style_start_tag = sub {    my $insert;
1400      my $style_el; !!!create-element ($style_el, 'style', $token->{attributes});  
1401      ## $self->{insertion_mode} eq 'in head' and ... (always true)    my $parse_rcdata = sub ($) {
1402      (($self->{insertion_mode} eq 'in head' and defined $self->{head_element})      my ($content_model_flag) = @_;
1403       ? $self->{head_element} : $self->{open_elements}->[-1]->[0])  
1404        ->append_child ($style_el);      ## Step 1
1405      $self->{content_model_flag} = 'CDATA';      my $start_tag_name = $token->{tag_name};
1406        !!!insert-element-t ($token->{tag_name}, $token->{attributes}, $token);
1407    
1408        ## Step 2
1409        $self->{content_model} = $content_model_flag; # CDATA or RCDATA
1410      delete $self->{escape}; # MUST      delete $self->{escape}; # MUST
1411                  
1412      my $text = '';      ## Step 3, 4
1413      !!!next-token;      $self->{insertion_mode} |= IN_CDATA_RCDATA_IM;
1414      while ($token->{type} eq 'character') {  
1415        $text .= $token->{data};      !!!nack ('t40.1');
       !!!next-token;  
     } # stop if non-character token or tokenizer stops tokenising  
     if (length $text) {  
       $style_el->manakai_append_text ($text);  
     }  
       
     $self->{content_model_flag} = 'PCDATA';  
                 
     if ($token->{type} eq 'end tag' and $token->{tag_name} eq 'style') {  
       ## Ignore the token  
     } else {  
       !!!parse-error (type => 'in CDATA:#'.$token->{type});  
       ## ISSUE: And ignore?  
     }  
1416      !!!next-token;      !!!next-token;
1417    }; # $style_start_tag    }; # $parse_rcdata
1418    
1419    my $script_start_tag = sub {    my $script_start_tag = sub () {
1420        ## Step 1
1421      my $script_el;      my $script_el;
1422      !!!create-element ($script_el, 'script', $token->{attributes});      !!!create-element ($script_el, $HTML_NS, 'script', $token->{attributes}, $token);
1423    
1424        ## Step 2
1425      ## TODO: mark as "parser-inserted"      ## TODO: mark as "parser-inserted"
1426    
1427      $self->{content_model_flag} = 'CDATA';      ## Step 3
1428        ## TODO: Mark as "already executed", if ...
1429    
1430        ## Step 4 (HTML5 revision 2702)
1431        $insert->($script_el);
1432        push @{$self->{open_elements}}, [$script_el, $el_category->{script}];
1433    
1434        ## Step 5
1435        $self->{content_model} = CDATA_CONTENT_MODEL;
1436      delete $self->{escape}; # MUST      delete $self->{escape}; # MUST
       
     my $text = '';  
     !!!next-token;  
     while ($token->{type} eq 'character') {  
       $text .= $token->{data};  
       !!!next-token;  
     } # stop if non-character token or tokenizer stops tokenising  
     if (length $text) {  
       $script_el->manakai_append_text ($text);  
     }  
                 
     $self->{content_model_flag} = 'PCDATA';  
1437    
1438      if ($token->{type} eq 'end tag' and      ## Step 6-7
1439          $token->{tag_name} eq 'script') {      $self->{insertion_mode} |= IN_CDATA_RCDATA_IM;
1440        ## Ignore the token  
1441      } else {      !!!nack ('t40.2');
       !!!parse-error (type => 'in CDATA:#'.$token->{type});  
       ## ISSUE: And ignore?  
       ## TODO: mark as "already executed"  
     }  
       
     if (defined $self->{inner_html_node}) {  
       ## TODO: mark as "already executed"  
     } else {  
       ## TODO: $old_insertion_point = current insertion point  
       ## TODO: insertion point = just before the next input character  
         
       (($self->{insertion_mode} eq 'in head' and defined $self->{head_element})  
        ? $self->{head_element} : $self->{open_elements}->[-1]->[0])->append_child ($script_el);  
         
       ## TODO: insertion point = $old_insertion_point (might be "undefined")  
         
       ## TODO: if there is a script that will execute as soon as the parser resume, then...  
     }  
       
1442      !!!next-token;      !!!next-token;
1443    }; # $script_start_tag    }; # $script_start_tag
1444    
1445      ## NOTE: $open_tables->[-1]->[0] is the "current table" element node.
1446      ## NOTE: $open_tables->[-1]->[1] is the "tainted" flag (OBSOLETE; unused).
1447      ## NOTE: $open_tables->[-1]->[2] is set false when non-Text node inserted.
1448      my $open_tables = [[$self->{open_elements}->[0]->[0]]];
1449    
1450    my $formatting_end_tag = sub {    my $formatting_end_tag = sub {
1451      my $tag_name = shift;      my $end_tag_token = shift;
1452        my $tag_name = $end_tag_token->{tag_name};
1453    
1454        ## NOTE: The adoption agency algorithm (AAA).
1455    
1456      FET: {      FET: {
1457        ## Step 1        ## Step 1
1458        my $formatting_element;        my $formatting_element;
1459        my $formatting_element_i_in_active;        my $formatting_element_i_in_active;
1460        AFE: for (reverse 0..$#$active_formatting_elements) {        AFE: for (reverse 0..$#$active_formatting_elements) {
1461          if ($active_formatting_elements->[$_]->[1] eq $tag_name) {          if ($active_formatting_elements->[$_]->[0] eq '#marker') {
1462              !!!cp ('t52');
1463              last AFE;
1464            } elsif ($active_formatting_elements->[$_]->[0]->manakai_local_name
1465                         eq $tag_name) {
1466              !!!cp ('t51');
1467            $formatting_element = $active_formatting_elements->[$_];            $formatting_element = $active_formatting_elements->[$_];
1468            $formatting_element_i_in_active = $_;            $formatting_element_i_in_active = $_;
1469            last AFE;            last AFE;
         } elsif ($active_formatting_elements->[$_]->[0] eq '#marker') {  
           last AFE;  
1470          }          }
1471        } # AFE        } # AFE
1472        unless (defined $formatting_element) {        unless (defined $formatting_element) {
1473          !!!parse-error (type => 'unmatched end tag:'.$tag_name);          !!!cp ('t53');
1474            !!!parse-error (type => 'unmatched end tag', text => $tag_name, token => $end_tag_token);
1475          ## Ignore the token          ## Ignore the token
1476          !!!next-token;          !!!next-token;
1477          return;          return;
# Line 2207  sub _tree_construction_main ($) { Line 1483  sub _tree_construction_main ($) {
1483          my $node = $self->{open_elements}->[$_];          my $node = $self->{open_elements}->[$_];
1484          if ($node->[0] eq $formatting_element->[0]) {          if ($node->[0] eq $formatting_element->[0]) {
1485            if ($in_scope) {            if ($in_scope) {
1486                !!!cp ('t54');
1487              $formatting_element_i_in_open = $_;              $formatting_element_i_in_open = $_;
1488              last INSCOPE;              last INSCOPE;
1489            } else { # in open elements but not in scope            } else { # in open elements but not in scope
1490              !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});              !!!cp ('t55');
1491                !!!parse-error (type => 'unmatched end tag',
1492                                text => $token->{tag_name},
1493                                token => $end_tag_token);
1494              ## Ignore the token              ## Ignore the token
1495              !!!next-token;              !!!next-token;
1496              return;              return;
1497            }            }
1498          } elsif ({          } elsif ($node->[1] & SCOPING_EL) {
1499                    table => 1, caption => 1, td => 1, th => 1,            !!!cp ('t56');
                   button => 1, marquee => 1, object => 1, html => 1,  
                  }->{$node->[1]}) {  
1500            $in_scope = 0;            $in_scope = 0;
1501          }          }
1502        } # INSCOPE        } # INSCOPE
1503        unless (defined $formatting_element_i_in_open) {        unless (defined $formatting_element_i_in_open) {
1504          !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});          !!!cp ('t57');
1505            !!!parse-error (type => 'unmatched end tag',
1506                            text => $token->{tag_name},
1507                            token => $end_tag_token);
1508          pop @$active_formatting_elements; # $formatting_element          pop @$active_formatting_elements; # $formatting_element
1509          !!!next-token; ## TODO: ok?          !!!next-token; ## TODO: ok?
1510          return;          return;
1511        }        }
1512        if (not $self->{open_elements}->[-1]->[0] eq $formatting_element->[0]) {        if (not $self->{open_elements}->[-1]->[0] eq $formatting_element->[0]) {
1513          !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);          !!!cp ('t58');
1514            !!!parse-error (type => 'not closed',
1515                            text => $self->{open_elements}->[-1]->[0]
1516                                ->manakai_local_name,
1517                            token => $end_tag_token);
1518        }        }
1519                
1520        ## Step 2        ## Step 2
# Line 2237  sub _tree_construction_main ($) { Line 1522  sub _tree_construction_main ($) {
1522        my $furthest_block_i_in_open;        my $furthest_block_i_in_open;
1523        OE: for (reverse 0..$#{$self->{open_elements}}) {        OE: for (reverse 0..$#{$self->{open_elements}}) {
1524          my $node = $self->{open_elements}->[$_];          my $node = $self->{open_elements}->[$_];
1525          if (not $formatting_category->{$node->[1]} and          if (not ($node->[1] & FORMATTING_EL) and
1526              #not $phrasing_category->{$node->[1]} and              #not $phrasing_category->{$node->[1]} and
1527              ($special_category->{$node->[1]} or              ($node->[1] & SPECIAL_EL or
1528               $scoping_category->{$node->[1]})) {               $node->[1] & SCOPING_EL)) { ## Scoping is redundant, maybe
1529              !!!cp ('t59');
1530            $furthest_block = $node;            $furthest_block = $node;
1531            $furthest_block_i_in_open = $_;            $furthest_block_i_in_open = $_;
1532              ## NOTE: The topmost (eldest) node.
1533          } elsif ($node->[0] eq $formatting_element->[0]) {          } elsif ($node->[0] eq $formatting_element->[0]) {
1534              !!!cp ('t60');
1535            last OE;            last OE;
1536          }          }
1537        } # OE        } # OE
1538                
1539        ## Step 3        ## Step 3
1540        unless (defined $furthest_block) { # MUST        unless (defined $furthest_block) { # MUST
1541            !!!cp ('t61');
1542          splice @{$self->{open_elements}}, $formatting_element_i_in_open;          splice @{$self->{open_elements}}, $formatting_element_i_in_open;
1543          splice @$active_formatting_elements, $formatting_element_i_in_active, 1;          splice @$active_formatting_elements, $formatting_element_i_in_active, 1;
1544          !!!next-token;          !!!next-token;
# Line 2262  sub _tree_construction_main ($) { Line 1551  sub _tree_construction_main ($) {
1551        ## Step 5        ## Step 5
1552        my $furthest_block_parent = $furthest_block->[0]->parent_node;        my $furthest_block_parent = $furthest_block->[0]->parent_node;
1553        if (defined $furthest_block_parent) {        if (defined $furthest_block_parent) {
1554            !!!cp ('t62');
1555          $furthest_block_parent->remove_child ($furthest_block->[0]);          $furthest_block_parent->remove_child ($furthest_block->[0]);
1556        }        }
1557                
# Line 2284  sub _tree_construction_main ($) { Line 1574  sub _tree_construction_main ($) {
1574          S7S2: {          S7S2: {
1575            for (reverse 0..$#$active_formatting_elements) {            for (reverse 0..$#$active_formatting_elements) {
1576              if ($active_formatting_elements->[$_]->[0] eq $node->[0]) {              if ($active_formatting_elements->[$_]->[0] eq $node->[0]) {
1577                  !!!cp ('t63');
1578                $node_i_in_active = $_;                $node_i_in_active = $_;
1579                last S7S2;                last S7S2;
1580              }              }
# Line 2297  sub _tree_construction_main ($) { Line 1588  sub _tree_construction_main ($) {
1588                    
1589          ## Step 4          ## Step 4
1590          if ($last_node->[0] eq $furthest_block->[0]) {          if ($last_node->[0] eq $furthest_block->[0]) {
1591              !!!cp ('t64');
1592            $bookmark_prev_el = $node->[0];            $bookmark_prev_el = $node->[0];
1593          }          }
1594                    
1595          ## Step 5          ## Step 5
1596          if ($node->[0]->has_child_nodes ()) {          if ($node->[0]->has_child_nodes ()) {
1597              !!!cp ('t65');
1598            my $clone = [$node->[0]->clone_node (0), $node->[1]];            my $clone = [$node->[0]->clone_node (0), $node->[1]];
1599            $active_formatting_elements->[$node_i_in_active] = $clone;            $active_formatting_elements->[$node_i_in_active] = $clone;
1600            $self->{open_elements}->[$node_i_in_open] = $clone;            $self->{open_elements}->[$node_i_in_open] = $clone;
# Line 2319  sub _tree_construction_main ($) { Line 1612  sub _tree_construction_main ($) {
1612        } # S7          } # S7  
1613                
1614        ## Step 8        ## Step 8
1615        $common_ancestor_node->[0]->append_child ($last_node->[0]);        if ($common_ancestor_node->[1] & TABLE_ROWS_EL) {
1616            ## Foster parenting.
1617            my $foster_parent_element;
1618            my $next_sibling;
1619            OE: for (reverse 0..$#{$self->{open_elements}}) {
1620              if ($self->{open_elements}->[$_]->[1] == TABLE_EL) {
1621                !!!cp ('t65.2');
1622                $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];
1623                $next_sibling = $self->{open_elements}->[$_]->[0];
1624                undef $next_sibling
1625                    unless $next_sibling->parent_node eq $foster_parent_element;
1626                last OE;
1627              }
1628            } # OE
1629            $foster_parent_element ||= $self->{open_elements}->[0]->[0];
1630    
1631            $foster_parent_element->insert_before ($last_node->[0], $next_sibling);
1632            $open_tables->[-1]->[1] = 1; # tainted
1633          } else {
1634            !!!cp ('t65.3');
1635            $common_ancestor_node->[0]->append_child ($last_node->[0]);
1636          }
1637                
1638        ## Step 9        ## Step 9
1639        my $clone = [$formatting_element->[0]->clone_node (0),        my $clone = [$formatting_element->[0]->clone_node (0),
# Line 2336  sub _tree_construction_main ($) { Line 1650  sub _tree_construction_main ($) {
1650        my $i;        my $i;
1651        AFE: for (reverse 0..$#$active_formatting_elements) {        AFE: for (reverse 0..$#$active_formatting_elements) {
1652          if ($active_formatting_elements->[$_]->[0] eq $formatting_element->[0]) {          if ($active_formatting_elements->[$_]->[0] eq $formatting_element->[0]) {
1653              !!!cp ('t66');
1654            splice @$active_formatting_elements, $_, 1;            splice @$active_formatting_elements, $_, 1;
1655            $i-- and last AFE if defined $i;            $i-- and last AFE if defined $i;
1656          } elsif ($active_formatting_elements->[$_]->[0] eq $bookmark_prev_el) {          } elsif ($active_formatting_elements->[$_]->[0] eq $bookmark_prev_el) {
1657              !!!cp ('t67');
1658            $i = $_;            $i = $_;
1659          }          }
1660        } # AFE        } # AFE
# Line 2348  sub _tree_construction_main ($) { Line 1664  sub _tree_construction_main ($) {
1664        undef $i;        undef $i;
1665        OE: for (reverse 0..$#{$self->{open_elements}}) {        OE: for (reverse 0..$#{$self->{open_elements}}) {
1666          if ($self->{open_elements}->[$_]->[0] eq $formatting_element->[0]) {          if ($self->{open_elements}->[$_]->[0] eq $formatting_element->[0]) {
1667              !!!cp ('t68');
1668            splice @{$self->{open_elements}}, $_, 1;            splice @{$self->{open_elements}}, $_, 1;
1669            $i-- and last OE if defined $i;            $i-- and last OE if defined $i;
1670          } elsif ($self->{open_elements}->[$_]->[0] eq $furthest_block->[0]) {          } elsif ($self->{open_elements}->[$_]->[0] eq $furthest_block->[0]) {
1671              !!!cp ('t69');
1672            $i = $_;            $i = $_;
1673          }          }
1674        } # OE        } # OE
1675        splice @{$self->{open_elements}}, $i + 1, 1, $clone;        splice @{$self->{open_elements}}, $i + 1, 0, $clone;
1676                
1677        ## Step 14        ## Step 14
1678        redo FET;        redo FET;
1679      } # FET      } # FET
1680    }; # $formatting_end_tag    }; # $formatting_end_tag
1681    
1682    my $insert_to_current = sub {    $insert = my $insert_to_current = sub {
1683      $self->{open_elements}->[-1]->[0]->append_child (shift);      $self->{open_elements}->[-1]->[0]->append_child ($_[0]);
1684    }; # $insert_to_current    }; # $insert_to_current
1685    
1686      ## Foster parenting.  Note that there are three "foster parenting"
1687      ## code in the parser: for elements (this one), for texts, and for
1688      ## elements in the AAA code.
1689    my $insert_to_foster = sub {    my $insert_to_foster = sub {
1690                         my $child = shift;      my $child = shift;
1691                         if ({      if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {
1692                              table => 1, tbody => 1, tfoot => 1,        # MUST
1693                              thead => 1, tr => 1,        my $foster_parent_element;
1694                             }->{$self->{open_elements}->[-1]->[1]}) {        my $next_sibling;
1695                           # MUST        OE: for (reverse 0..$#{$self->{open_elements}}) {
1696                           my $foster_parent_element;          if ($self->{open_elements}->[$_]->[1] == TABLE_EL) {
1697                           my $next_sibling;            !!!cp ('t71');
1698                           OE: for (reverse 0..$#{$self->{open_elements}}) {            $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];
1699                             if ($self->{open_elements}->[$_]->[1] eq 'table') {            $next_sibling = $self->{open_elements}->[$_]->[0];
1700                               my $parent = $self->{open_elements}->[$_]->[0]->parent_node;            undef $next_sibling
1701                               if (defined $parent and $parent->node_type == 1) {                unless $next_sibling->parent_node eq $foster_parent_element;
1702                                 $foster_parent_element = $parent;            last OE;
1703                                 $next_sibling = $self->{open_elements}->[$_]->[0];          }
1704                               } else {        } # OE
1705                                 $foster_parent_element        $foster_parent_element ||= $self->{open_elements}->[0]->[0];
1706                                   = $self->{open_elements}->[$_ - 1]->[0];  
1707                               }        $foster_parent_element->insert_before ($child, $next_sibling);
1708                               last OE;        $open_tables->[-1]->[1] = 1; # tainted
1709                             }      } else {
1710                           } # OE        !!!cp ('t72');
1711                           $foster_parent_element = $self->{open_elements}->[0]->[0]        $self->{open_elements}->[-1]->[0]->append_child ($child);
1712                             unless defined $foster_parent_element;      }
                          $foster_parent_element->insert_before  
                            ($child, $next_sibling);  
                        } else {  
                          $self->{open_elements}->[-1]->[0]->append_child ($child);  
                        }  
1713    }; # $insert_to_foster    }; # $insert_to_foster
1714    
1715    my $in_body = sub {    ## NOTE: Insert a character (MUST): When a character is inserted, if
1716      my $insert = shift;    ## the last node that was inserted by the parser is a Text node and
1717      if ($token->{type} eq 'start tag') {    ## the character has to be inserted after that node, then the
1718        if ($token->{tag_name} eq 'script') {    ## character is appended to the Text node.  However, if any other
1719          $script_start_tag->();    ## node is inserted by the parser, then a new Text node is created
1720          return;    ## and the character is appended as that Text node.  If I'm not
1721        } elsif ($token->{tag_name} eq 'style') {    ## wrong, for a parser with scripting disabled, there are only two
1722          $style_start_tag->();    ## cases where this occurs.  One is the case where an element node
1723          return;    ## is inserted to the |head| element.  This is covered by using the
1724        } elsif ({    ## |$self->{head_element_inserted}| flag.  Another is the case where
1725                  base => 1, link => 1, meta => 1,    ## an element or comment is inserted into the |table| subtree while
1726                 }->{$token->{tag_name}}) {    ## foster parenting happens.  This is covered by using the [2] flag
1727          !!!parse-error (type => 'in body:'.$token->{tag_name});    ## of the |$open_tables| structure.  All other cases are handled
1728          ## NOTE: This is an "as if in head" code clone    ## simply by calling |manakai_append_text| method.
1729          my $el;  
1730          !!!create-element ($el, $token->{tag_name}, $token->{attributes});    ## TODO: |<body><script>document.write("a<br>");
1731          if (defined $self->{head_element}) {    ## document.body.removeChild (document.body.lastChild);
1732            $self->{head_element}->append_child ($el);    ## document.write ("b")</script>|
1733          } else {  
1734            $insert->($el);    B: while (1) {
1735          }  
1736                ## The "in table text" insertion mode.
1737          !!!next-token;      if ($self->{insertion_mode} & TABLE_IMS and
1738          return;          not $self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and
1739        } elsif ($token->{tag_name} eq 'title') {          not $self->{insertion_mode} & IN_CDATA_RCDATA_IM) {
1740          !!!parse-error (type => 'in body:title');        C: {
1741          ## NOTE: There is an "as if in head" code clone          my $s;
1742          my $title_el;          if ($token->{type} == CHARACTER_TOKEN) {
1743          !!!create-element ($title_el, 'title', $token->{attributes});            !!!cp ('t194');
1744          (defined $self->{head_element} ? $self->{head_element} : $self->{open_elements}->[-1]->[0])            $self->{pending_chars} ||= [];
1745            ->append_child ($title_el);            push @{$self->{pending_chars}}, $token;
         $self->{content_model_flag} = 'RCDATA';  
         delete $self->{escape}; # MUST  
           
         my $text = '';  
         !!!next-token;  
         while ($token->{type} eq 'character') {  
           $text .= $token->{data};  
1746            !!!next-token;            !!!next-token;
1747          }            next B;
         if (length $text) {  
           $title_el->manakai_append_text ($text);  
         }  
           
         $self->{content_model_flag} = 'PCDATA';  
           
         if ($token->{type} eq 'end tag' and  
             $token->{tag_name} eq 'title') {  
           ## Ignore the token  
         } else {  
           !!!parse-error (type => 'in RCDATA:#'.$token->{type});  
           ## ISSUE: And ignore?  
         }  
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'body') {  
         !!!parse-error (type => 'in body:body');  
                 
         if (@{$self->{open_elements}} == 1 or  
             $self->{open_elements}->[1]->[1] ne 'body') {  
           ## Ignore the token  
1748          } else {          } else {
1749            my $body_el = $self->{open_elements}->[1]->[0];            if ($self->{pending_chars}) {
1750            for my $attr_name (keys %{$token->{attributes}}) {              $s = join '', map { $_->{data} } @{$self->{pending_chars}};
1751              unless ($body_el->has_attribute_ns (undef, $attr_name)) {              delete $self->{pending_chars};
1752                $body_el->set_attribute_ns              if ($s =~ /[^\x09\x0A\x0C\x0D\x20]/) {
1753                  (undef, [undef, $attr_name],                !!!cp ('t195');
1754                   $token->{attributes}->{$attr_name}->{value});                #
1755                } else {
1756                  !!!cp ('t195.1');
1757                  #$self->{open_elements}->[-1]->[0]->manakai_append_text ($s);
1758                  $self->{open_elements}->[-1]->[0]->append_child
1759                      ($self->{document}->create_text_node ($s));
1760                  last C;
1761              }              }
1762              } else {
1763                !!!cp ('t195.2');
1764                last C;
1765            }            }
1766          }          }
1767          !!!next-token;  
1768          return;          ## Foster parenting.
1769        } elsif ({          !!!parse-error (type => 'in table:#text', token => $token);
1770                  address => 1, blockquote => 1, center => 1, dir => 1,  
1771                  div => 1, dl => 1, fieldset => 1, listing => 1,          ## NOTE: As if in body, but insert into the foster parent element.
1772                  menu => 1, ol => 1, p => 1, ul => 1,          $reconstruct_active_formatting_elements->($insert_to_foster);
1773                  pre => 1,              
1774                 }->{$token->{tag_name}}) {          if ($self->{open_elements}->[-1]->[1] & TABLE_ROWS_EL) {
1775          ## has a p element in scope            # MUST
1776          INSCOPE: for (reverse @{$self->{open_elements}}) {            my $foster_parent_element;
1777            if ($_->[1] eq 'p') {            my $next_sibling;
1778              !!!back-token;            OE: for (reverse 0..$#{$self->{open_elements}}) {
1779              $token = {type => 'end tag', tag_name => 'p'};              if ($self->{open_elements}->[$_]->[1] == TABLE_EL) {
1780              return;                !!!cp ('t197');
1781            } elsif ({                $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];
1782                      table => 1, caption => 1, td => 1, th => 1,                $next_sibling = $self->{open_elements}->[$_]->[0];
1783                      button => 1, marquee => 1, object => 1, html => 1,                undef $next_sibling
1784                     }->{$_->[1]}) {                  unless $next_sibling->parent_node eq $foster_parent_element;
1785              last INSCOPE;                last OE;
1786            }              }
1787          } # INSCOPE            } # OE
1788                        $foster_parent_element ||= $self->{open_elements}->[0]->[0];
1789          !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
1790          if ($token->{tag_name} eq 'pre') {            !!!cp ('t199');
1791            !!!next-token;            $foster_parent_element->insert_before
1792            if ($token->{type} eq 'character') {                ($self->{document}->create_text_node ($s), $next_sibling);
1793              $token->{data} =~ s/^\x0A//;  
1794              unless (length $token->{data}) {            $open_tables->[-1]->[1] = 1; # tainted
1795                !!!next-token;            $open_tables->[-1]->[2] = 1; # ~node inserted
1796              }          } else {
1797            }            ## NOTE: Fragment case or in a foster parent'ed element
1798          } else {            ## (e.g. |<table><span>a|).  In fragment case, whether the
1799            !!!next-token;            ## character is appended to existing node or a new node is
1800              ## created is irrelevant, since the foster parent'ed nodes
1801              ## are discarded and fragment parsing does not invoke any
1802              ## script.
1803              !!!cp ('t200');
1804              $self->{open_elements}->[-1]->[0]->manakai_append_text ($s);
1805            }
1806          } # C
1807        } # TABLE_IMS
1808    
1809        if ($token->{type} == DOCTYPE_TOKEN) {
1810          !!!cp ('t73');
1811          !!!parse-error (type => 'in html:#DOCTYPE', token => $token);
1812          ## Ignore the token
1813          ## Stay in the phase
1814          !!!next-token;
1815          next B;
1816        } elsif ($token->{type} == START_TAG_TOKEN and
1817                 $token->{tag_name} eq 'html') {
1818          if ($self->{insertion_mode} == AFTER_HTML_BODY_IM) {
1819            !!!cp ('t79');
1820            !!!parse-error (type => 'after html', text => 'html', token => $token);
1821            $self->{insertion_mode} = AFTER_BODY_IM;
1822          } elsif ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {
1823            !!!cp ('t80');
1824            !!!parse-error (type => 'after html', text => 'html', token => $token);
1825            $self->{insertion_mode} = AFTER_FRAMESET_IM;
1826          } else {
1827            !!!cp ('t81');
1828          }
1829    
1830          !!!cp ('t82');
1831          !!!parse-error (type => 'not first start tag', token => $token);
1832          my $top_el = $self->{open_elements}->[0]->[0];
1833          for my $attr_name (keys %{$token->{attributes}}) {
1834            unless ($top_el->has_attribute_ns (undef, $attr_name)) {
1835              !!!cp ('t84');
1836              $top_el->set_attribute_ns
1837                (undef, [undef, $attr_name],
1838                 $token->{attributes}->{$attr_name}->{value});
1839          }          }
1840          return;        }
1841        } elsif ($token->{tag_name} eq 'form') {        !!!nack ('t84.1');
1842          if (defined $self->{form_element}) {        !!!next-token;
1843            !!!parse-error (type => 'in form:form');        next B;
1844            ## Ignore the token      } elsif ($token->{type} == COMMENT_TOKEN) {
1845            !!!next-token;        my $comment = $self->{document}->create_comment ($token->{data});
1846            return;        if ($self->{insertion_mode} & AFTER_HTML_IMS) {
1847            !!!cp ('t85');
1848            $self->{document}->append_child ($comment);
1849          } elsif ($self->{insertion_mode} == AFTER_BODY_IM) {
1850            !!!cp ('t86');
1851            $self->{open_elements}->[0]->[0]->append_child ($comment);
1852          } else {
1853            !!!cp ('t87');
1854            $self->{open_elements}->[-1]->[0]->append_child ($comment);
1855            $open_tables->[-1]->[2] = 0 if @$open_tables; # ~node inserted
1856          }
1857          !!!next-token;
1858          next B;
1859        } elsif ($self->{insertion_mode} & IN_CDATA_RCDATA_IM) {
1860          if ($token->{type} == CHARACTER_TOKEN) {
1861            $token->{data} =~ s/^\x0A// if $self->{ignore_newline};
1862            delete $self->{ignore_newline};
1863    
1864            if (length $token->{data}) {
1865              !!!cp ('t43');
1866              $self->{open_elements}->[-1]->[0]->manakai_append_text
1867                  ($token->{data});
1868          } else {          } else {
1869            ## has a p element in scope            !!!cp ('t43.1');
           INSCOPE: for (reverse @{$self->{open_elements}}) {  
             if ($_->[1] eq 'p') {  
               !!!back-token;  
               $token = {type => 'end tag', tag_name => 'p'};  
               return;  
             } elsif ({  
                       table => 1, caption => 1, td => 1, th => 1,  
                       button => 1, marquee => 1, object => 1, html => 1,  
                      }->{$_->[1]}) {  
               last INSCOPE;  
             }  
           } # INSCOPE  
               
           !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
           $self->{form_element} = $self->{open_elements}->[-1]->[0];  
           !!!next-token;  
           return;  
1870          }          }
       } elsif ($token->{tag_name} eq 'li') {  
         ## has a p element in scope  
         INSCOPE: for (reverse @{$self->{open_elements}}) {  
           if ($_->[1] eq 'p') {  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'p'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$_->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         ## Step 1  
         my $i = -1;  
         my $node = $self->{open_elements}->[$i];  
         LI: {  
           ## Step 2  
           if ($node->[1] eq 'li') {  
             if ($i != -1) {  
               !!!parse-error (type => 'end tag missing:'.  
                               $self->{open_elements}->[-1]->[1]);  
               ## TODO: test  
             }  
             splice @{$self->{open_elements}}, $i;  
             last LI;  
           }  
             
           ## Step 3  
           if (not $formatting_category->{$node->[1]} and  
               #not $phrasing_category->{$node->[1]} and  
               ($special_category->{$node->[1]} or  
                $scoping_category->{$node->[1]}) and  
               $node->[1] ne 'address' and $node->[1] ne 'div') {  
             last LI;  
           }  
             
           ## Step 4  
           $i--;  
           $node = $self->{open_elements}->[$i];  
           redo LI;  
         } # LI  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
1871          !!!next-token;          !!!next-token;
1872          return;          next B;
1873        } elsif ($token->{tag_name} eq 'dd' or $token->{tag_name} eq 'dt') {        } elsif ($token->{type} == END_TAG_TOKEN) {
1874          ## has a p element in scope          delete $self->{ignore_newline};
1875          INSCOPE: for (reverse @{$self->{open_elements}}) {  
1876            if ($_->[1] eq 'p') {          if ($token->{tag_name} eq 'script') {
1877              !!!back-token;            !!!cp ('t50');
             $token = {type => 'end tag', tag_name => 'p'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$_->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         ## Step 1  
         my $i = -1;  
         my $node = $self->{open_elements}->[$i];  
         LI: {  
           ## Step 2  
           if ($node->[1] eq 'dt' or $node->[1] eq 'dd') {  
             if ($i != -1) {  
               !!!parse-error (type => 'end tag missing:'.  
                               $self->{open_elements}->[-1]->[1]);  
               ## TODO: test  
             }  
             splice @{$self->{open_elements}}, $i;  
             last LI;  
           }  
             
           ## Step 3  
           if (not $formatting_category->{$node->[1]} and  
               #not $phrasing_category->{$node->[1]} and  
               ($special_category->{$node->[1]} or  
                $scoping_category->{$node->[1]}) and  
               $node->[1] ne 'address' and $node->[1] ne 'div') {  
             last LI;  
           }  
             
           ## Step 4  
           $i--;  
           $node = $self->{open_elements}->[$i];  
           redo LI;  
         } # LI  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'plaintext') {  
         ## has a p element in scope  
         INSCOPE: for (reverse @{$self->{open_elements}}) {  
           if ($_->[1] eq 'p') {  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'p'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$_->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
             
         $self->{content_model_flag} = 'PLAINTEXT';  
             
         !!!next-token;  
         return;  
       } elsif ({  
                 h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,  
                }->{$token->{tag_name}}) {  
         ## has a p element in scope  
         INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
           my $node = $self->{open_elements}->[$_];  
           if ($node->[1] eq 'p') {  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'p'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         ## has an element in scope  
         my $i;  
         INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
           my $node = $self->{open_elements}->[$_];  
           if ({  
                h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,  
               }->{$node->[1]}) {  
             $i = $_;  
             last INSCOPE;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         if (defined $i) {  
           !!!parse-error (type => 'in hn:hn');  
           splice @{$self->{open_elements}}, $i;  
         }  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
1878                        
1879          !!!next-token;            ## Para 1-2
1880          return;            my $script = pop @{$self->{open_elements}};
       } elsif ($token->{tag_name} eq 'a') {  
         AFE: for my $i (reverse 0..$#$active_formatting_elements) {  
           my $node = $active_formatting_elements->[$i];  
           if ($node->[1] eq 'a') {  
             !!!parse-error (type => 'in a:a');  
               
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'a'};  
             $formatting_end_tag->($token->{tag_name});  
               
             AFE2: for (reverse 0..$#$active_formatting_elements) {  
               if ($active_formatting_elements->[$_]->[0] eq $node->[0]) {  
                 splice @$active_formatting_elements, $_, 1;  
                 last AFE2;  
               }  
             } # AFE2  
             OE: for (reverse 0..$#{$self->{open_elements}}) {  
               if ($self->{open_elements}->[$_]->[0] eq $node->[0]) {  
                 splice @{$self->{open_elements}}, $_, 1;  
                 last OE;  
               }  
             } # OE  
             last AFE;  
           } elsif ($node->[0] eq '#marker') {  
             last AFE;  
           }  
         } # AFE  
1881                        
1882          $reconstruct_active_formatting_elements->($insert_to_current);            ## Para 3
1883              $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
1884    
1885          !!!insert-element-t ($token->{tag_name}, $token->{attributes});            ## Para 4
1886          push @$active_formatting_elements, $self->{open_elements}->[-1];            ## TODO: $old_insertion_point = $current_insertion_point;
1887              ## TODO: $current_insertion_point = just before $self->{nc};
1888    
1889              ## Para 5
1890              ## TODO: Run the $script->[0].
1891    
1892              ## Para 6
1893              ## TODO: $current_insertion_point = $old_insertion_point;
1894    
1895              ## Para 7
1896              ## TODO: if ($pending_external_script) {
1897                ## TODO: ...
1898              ## TODO: }
1899    
1900          !!!next-token;            !!!next-token;
1901          return;            next B;
1902        } elsif ({          } else {
1903                  b => 1, big => 1, em => 1, font => 1, i => 1,            !!!cp ('t42');
1904                  s => 1, small => 1, strile => 1,  
1905                  strong => 1, tt => 1, u => 1,            pop @{$self->{open_elements}};
                }->{$token->{tag_name}}) {  
         $reconstruct_active_formatting_elements->($insert_to_current);  
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         push @$active_formatting_elements, $self->{open_elements}->[-1];  
           
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'nobr') {  
         $reconstruct_active_formatting_elements->($insert_to_current);  
   
         ## has a |nobr| element in scope  
         INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
           my $node = $self->{open_elements}->[$_];  
           if ($node->[1] eq 'nobr') {  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'nobr'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         push @$active_formatting_elements, $self->{open_elements}->[-1];  
           
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'button') {  
         ## has a button element in scope  
         INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
           my $node = $self->{open_elements}->[$_];  
           if ($node->[1] eq 'button') {  
             !!!parse-error (type => 'in button:button');  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'button'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         $reconstruct_active_formatting_elements->($insert_to_current);  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         push @$active_formatting_elements, ['#marker', ''];  
1906    
1907          !!!next-token;            $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
1908          return;            !!!next-token;
1909        } elsif ($token->{tag_name} eq 'marquee' or            next B;
                $token->{tag_name} eq 'object') {  
         $reconstruct_active_formatting_elements->($insert_to_current);  
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         push @$active_formatting_elements, ['#marker', ''];  
           
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'xmp') {  
         $reconstruct_active_formatting_elements->($insert_to_current);  
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
           
         $self->{content_model_flag} = 'CDATA';  
         delete $self->{escape}; # MUST  
           
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'table') {  
         ## has a p element in scope  
         INSCOPE: for (reverse @{$self->{open_elements}}) {  
           if ($_->[1] eq 'p') {  
             !!!back-token;  
             $token = {type => 'end tag', tag_name => 'p'};  
             return;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$_->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
             
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
             
         $self->{insertion_mode} = 'in table';  
             
         !!!next-token;  
         return;  
       } elsif ({  
                 area => 1, basefont => 1, bgsound => 1, br => 1,  
                 embed => 1, img => 1, param => 1, spacer => 1, wbr => 1,  
                 image => 1,  
                }->{$token->{tag_name}}) {  
         if ($token->{tag_name} eq 'image') {  
           !!!parse-error (type => 'image');  
           $token->{tag_name} = 'img';  
1910          }          }
1911                  } elsif ($token->{type} == END_OF_FILE_TOKEN) {
1912          $reconstruct_active_formatting_elements->($insert_to_current);          delete $self->{ignore_newline};
1913            
1914          !!!insert-element-t ($token->{tag_name}, $token->{attributes});          !!!cp ('t44');
1915            !!!parse-error (type => 'not closed',
1916                            text => $self->{open_elements}->[-1]->[0]
1917                                ->manakai_local_name,
1918                            token => $token);
1919    
1920            #if ($self->{open_elements}->[-1]->[1] == SCRIPT_EL) {
1921            #  ## TODO: Mark as "already executed"
1922            #}
1923    
1924          pop @{$self->{open_elements}};          pop @{$self->{open_elements}};
1925            
1926            $self->{insertion_mode} &= ~ IN_CDATA_RCDATA_IM;
1927            ## Reprocess.
1928            next B;
1929          } else {
1930            die "$0: $token->{type}: In CDATA/RCDATA: Unknown token type";        
1931          }
1932        } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM) {
1933          if ($token->{type} == CHARACTER_TOKEN) {
1934            !!!cp ('t87.1');
1935            $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});
1936          !!!next-token;          !!!next-token;
1937          return;          next B;
1938        } elsif ($token->{tag_name} eq 'hr') {        } elsif ($token->{type} == START_TAG_TOKEN) {
1939          ## has a p element in scope          if ((not {mglyph => 1, malignmark => 1}->{$token->{tag_name}} and
1940          INSCOPE: for (reverse @{$self->{open_elements}}) {               $self->{open_elements}->[-1]->[1] & FOREIGN_FLOW_CONTENT_EL) or
1941            if ($_->[1] eq 'p') {              not ($self->{open_elements}->[-1]->[1] & FOREIGN_EL) or
1942              !!!back-token;              ($token->{tag_name} eq 'svg' and
1943              $token = {type => 'end tag', tag_name => 'p'};               $self->{open_elements}->[-1]->[1] == MML_AXML_EL)) {
1944              return;            ## NOTE: "using the rules for secondary insertion mode"then"continue"
1945            } elsif ({            !!!cp ('t87.2');
1946                      table => 1, caption => 1, td => 1, th => 1,            #
1947                      button => 1, marquee => 1, object => 1, html => 1,          } elsif ({
1948                     }->{$_->[1]}) {                    b => 1, big => 1, blockquote => 1, body => 1, br => 1,
1949              last INSCOPE;                    center => 1, code => 1, dd => 1, div => 1, dl => 1, dt => 1,
1950                      em => 1, embed => 1, h1 => 1, h2 => 1, h3 => 1,
1951                      h4 => 1, h5 => 1, h6 => 1, head => 1, hr => 1, i => 1,
1952                      img => 1, li => 1, listing => 1, menu => 1, meta => 1,
1953                      nobr => 1, ol => 1, p => 1, pre => 1, ruby => 1, s => 1,
1954                      small => 1, span => 1, strong => 1, strike => 1, sub => 1,
1955                      sup => 1, table => 1, tt => 1, u => 1, ul => 1, var => 1,
1956                     }->{$token->{tag_name}} or
1957                     ($token->{tag_name} eq 'font' and
1958                      ($token->{attributes}->{color} or
1959                       $token->{attributes}->{face} or
1960                       $token->{attributes}->{size}))) {
1961              !!!cp ('t87.2');
1962              !!!parse-error (type => 'not closed',
1963                              text => $self->{open_elements}->[-1]->[0]
1964                                  ->manakai_local_name,
1965                              token => $token);
1966    
1967              pop @{$self->{open_elements}}
1968                  while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;
1969    
1970              $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
1971              ## Reprocess.
1972              next B;
1973            } else {
1974              my $nsuri = $self->{open_elements}->[-1]->[0]->namespace_uri;
1975              my $tag_name = $token->{tag_name};
1976              if ($nsuri eq $SVG_NS) {
1977                $tag_name = {
1978                   altglyph => 'altGlyph',
1979                   altglyphdef => 'altGlyphDef',
1980                   altglyphitem => 'altGlyphItem',
1981                   animatecolor => 'animateColor',
1982                   animatemotion => 'animateMotion',
1983                   animatetransform => 'animateTransform',
1984                   clippath => 'clipPath',
1985                   feblend => 'feBlend',
1986                   fecolormatrix => 'feColorMatrix',
1987                   fecomponenttransfer => 'feComponentTransfer',
1988                   fecomposite => 'feComposite',
1989                   feconvolvematrix => 'feConvolveMatrix',
1990                   fediffuselighting => 'feDiffuseLighting',
1991                   fedisplacementmap => 'feDisplacementMap',
1992                   fedistantlight => 'feDistantLight',
1993                   feflood => 'feFlood',
1994                   fefunca => 'feFuncA',
1995                   fefuncb => 'feFuncB',
1996                   fefuncg => 'feFuncG',
1997                   fefuncr => 'feFuncR',
1998                   fegaussianblur => 'feGaussianBlur',
1999                   feimage => 'feImage',
2000                   femerge => 'feMerge',
2001                   femergenode => 'feMergeNode',
2002                   femorphology => 'feMorphology',
2003                   feoffset => 'feOffset',
2004                   fepointlight => 'fePointLight',
2005                   fespecularlighting => 'feSpecularLighting',
2006                   fespotlight => 'feSpotLight',
2007                   fetile => 'feTile',
2008                   feturbulence => 'feTurbulence',
2009                   foreignobject => 'foreignObject',
2010                   glyphref => 'glyphRef',
2011                   lineargradient => 'linearGradient',
2012                   radialgradient => 'radialGradient',
2013                   #solidcolor => 'solidColor', ## NOTE: Commented in spec (SVG1.2)
2014                   textpath => 'textPath',  
2015                }->{$tag_name} || $tag_name;
2016            }            }
2017          } # INSCOPE  
2018                        ## "adjust SVG attributes" (SVG only) - done in insert-element-f
2019          !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
2020          pop @{$self->{open_elements}};            ## "adjust foreign attributes" - done in insert-element-f
2021              
2022          !!!next-token;            !!!insert-element-f ($nsuri, $tag_name, $token->{attributes}, $token);
2023          return;  
2024        } elsif ($token->{tag_name} eq 'input') {            if ($self->{self_closing}) {
2025          $reconstruct_active_formatting_elements->($insert_to_current);              pop @{$self->{open_elements}};
2026                        !!!ack ('t87.3');
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
         ## TODO: associate with $self->{form_element} if defined  
         pop @{$self->{open_elements}};  
           
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'isindex') {  
         !!!parse-error (type => 'isindex');  
           
         if (defined $self->{form_element}) {  
           ## Ignore the token  
           !!!next-token;  
           return;  
         } else {  
           my $at = $token->{attributes};  
           my $form_attrs;  
           $form_attrs->{action} = $at->{action} if $at->{action};  
           my $prompt_attr = $at->{prompt};  
           $at->{name} = {name => 'name', value => 'isindex'};  
           delete $at->{action};  
           delete $at->{prompt};  
           my @tokens = (  
                         {type => 'start tag', tag_name => 'form',  
                          attributes => $form_attrs},  
                         {type => 'start tag', tag_name => 'hr'},  
                         {type => 'start tag', tag_name => 'p'},  
                         {type => 'start tag', tag_name => 'label'},  
                        );  
           if ($prompt_attr) {  
             push @tokens, {type => 'character', data => $prompt_attr->{value}};  
2027            } else {            } else {
2028              push @tokens, {type => 'character',              !!!cp ('t87.4');
                            data => 'This is a searchable index. Insert your search keywords here: '}; # SHOULD  
             ## TODO: make this configurable  
           }  
           push @tokens,  
                         {type => 'start tag', tag_name => 'input', attributes => $at},  
                         #{type => 'character', data => ''}, # SHOULD  
                         {type => 'end tag', tag_name => 'label'},  
                         {type => 'end tag', tag_name => 'p'},  
                         {type => 'start tag', tag_name => 'hr'},  
                         {type => 'end tag', tag_name => 'form'};  
           $token = shift @tokens;  
           !!!back-token (@tokens);  
           return;  
         }  
       } elsif ({  
                 textarea => 1,  
                 iframe => 1,  
                 noembed => 1,  
                 noframes => 1,  
                 noscript => 0, ## TODO: 1 if scripting is enabled  
                }->{$token->{tag_name}}) {  
         my $tag_name = $token->{tag_name};  
         my $el;  
         !!!create-element ($el, $token->{tag_name}, $token->{attributes});  
           
         if ($token->{tag_name} eq 'textarea') {  
           ## TODO: $self->{form_element} if defined  
           $self->{content_model_flag} = 'RCDATA';  
         } else {  
           $self->{content_model_flag} = 'CDATA';  
         }  
         delete $self->{escape}; # MUST  
           
         $insert->($el);  
           
         my $text = '';  
         if ($token->{tag_name} eq 'textarea') {  
           !!!next-token;  
           if ($token->{type} eq 'character') {  
             $token->{data} =~ s/^\x0A//;  
             unless (length $token->{data}) {  
               !!!next-token;  
             }  
2029            }            }
2030          } else {  
           !!!next-token;  
         }  
         while ($token->{type} eq 'character') {  
           $text .= $token->{data};  
2031            !!!next-token;            !!!next-token;
2032              next B;
2033          }          }
2034          if (length $text) {        } elsif ($token->{type} == END_TAG_TOKEN) {
2035            $el->manakai_append_text ($text);          ## NOTE: "using the rules for secondary insertion mode" then "continue"
2036          }          if ($token->{tag_name} eq 'script') {
2037                      !!!cp ('t87.41');
2038          $self->{content_model_flag} = 'PCDATA';            #
2039                      ## XXXscript: Execute script here.
         if ($token->{type} eq 'end tag' and  
             $token->{tag_name} eq $tag_name) {  
           ## Ignore the token  
2040          } else {          } else {
2041            if ($token->{tag_name} eq 'textarea') {            !!!cp ('t87.5');
2042              !!!parse-error (type => 'in RCDATA:#'.$token->{type});            #
           } else {  
             !!!parse-error (type => 'in CDATA:#'.$token->{type});  
           }  
           ## ISSUE: And ignore?  
2043          }          }
2044          !!!next-token;        } elsif ($token->{type} == END_OF_FILE_TOKEN) {
2045          return;          !!!cp ('t87.6');
2046        } elsif ($token->{tag_name} eq 'select') {          !!!parse-error (type => 'not closed',
2047          $reconstruct_active_formatting_elements->($insert_to_current);                          text => $self->{open_elements}->[-1]->[0]
2048                                        ->manakai_local_name,
2049          !!!insert-element-t ($token->{tag_name}, $token->{attributes});                          token => $token);
2050            
2051          $self->{insertion_mode} = 'in select';          pop @{$self->{open_elements}}
2052          !!!next-token;              while $self->{open_elements}->[-1]->[1] & FOREIGN_EL;
2053          return;  
2054        } elsif ({          ## NOTE: |<span><svg>| ... two parse errors, |<svg>| ... a parse error.
2055                  caption => 1, col => 1, colgroup => 1, frame => 1,  
2056                  frameset => 1, head => 1, option => 1, optgroup => 1,          $self->{insertion_mode} &= ~ IN_FOREIGN_CONTENT_IM;
2057                  tbody => 1, td => 1, tfoot => 1, th => 1,          ## Reprocess.
2058                  thead => 1, tr => 1,          next B;
                }->{$token->{tag_name}}) {  
         !!!parse-error (type => 'in body:'.$token->{tag_name});  
         ## Ignore the token  
         !!!next-token;  
         return;  
           
         ## ISSUE: An issue on HTML5 new elements in the spec.  
2059        } else {        } else {
2060          $reconstruct_active_formatting_elements->($insert_to_current);          die "$0: $token->{type}: Unknown token type";        
           
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
           
         !!!next-token;  
         return;  
2061        }        }
2062      } elsif ($token->{type} eq 'end tag') {      }
       if ($token->{tag_name} eq 'body') {  
         if (@{$self->{open_elements}} > 1 and  
             $self->{open_elements}->[1]->[1] eq 'body') {  
           for (@{$self->{open_elements}}) {  
             unless ({  
                        dd => 1, dt => 1, li => 1, p => 1, td => 1,  
                        th => 1, tr => 1, body => 1, html => 1,  
                     }->{$_->[1]}) {  
               !!!parse-error (type => 'not closed:'.$_->[1]);  
             }  
           }  
2063    
2064            $self->{insertion_mode} = 'after body';      if ($self->{insertion_mode} & HEAD_IMS) {
2065            !!!next-token;        if ($token->{type} == CHARACTER_TOKEN) {
2066            return;          if ($token->{data} =~ s/^([\x09\x0A\x0C\x20]+)//) {
2067          } else {            unless ($self->{insertion_mode} == BEFORE_HEAD_IM) {
2068            !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});              if ($self->{head_element_inserted}) {
2069            ## Ignore the token                !!!cp ('t88.3');
2070            !!!next-token;                $self->{open_elements}->[-1]->[0]->append_child
2071            return;                  ($self->{document}->create_text_node ($1));
2072          }                delete $self->{head_element_inserted};
2073        } elsif ($token->{tag_name} eq 'html') {                ## NOTE: |</head> <link> |
2074          if (@{$self->{open_elements}} > 1 and $self->{open_elements}->[1]->[1] eq 'body') {                #
2075            ## ISSUE: There is an issue in the spec.              } else {
2076            if ($self->{open_elements}->[-1]->[1] ne 'body') {                !!!cp ('t88.2');
2077              !!!parse-error (type => 'not closed:'.$self->{open_elements}->[1]->[1]);                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
2078            }                ## NOTE: |</head> &#x20;|
2079            $self->{insertion_mode} = 'after body';                #
           ## reprocess  
           return;  
         } else {  
           !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
           ## Ignore the token  
           !!!next-token;  
           return;  
         }  
       } elsif ({  
                 address => 1, blockquote => 1, center => 1, dir => 1,  
                 div => 1, dl => 1, fieldset => 1, listing => 1,  
                 menu => 1, ol => 1, pre => 1, ul => 1,  
                 p => 1,  
                 dd => 1, dt => 1, li => 1,  
                 button => 1, marquee => 1, object => 1,  
                }->{$token->{tag_name}}) {  
         ## has an element in scope  
         my $i;  
         INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
           my $node = $self->{open_elements}->[$_];  
           if ($node->[1] eq $token->{tag_name}) {  
             ## generate implied end tags  
             if ({  
                  dd => ($token->{tag_name} ne 'dd'),  
                  dt => ($token->{tag_name} ne 'dt'),  
                  li => ($token->{tag_name} ne 'li'),  
                  p => ($token->{tag_name} ne 'p'),  
                  td => 1, th => 1, tr => 1,  
                 }->{$self->{open_elements}->[-1]->[1]}) {  
               !!!back-token;  
               $token = {type => 'end tag',  
                         tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
               return;  
2080              }              }
2081              $i = $_;            } else {
2082              last INSCOPE unless $token->{tag_name} eq 'p';              !!!cp ('t88.1');
2083            } elsif ({              ## Ignore the token.
2084                      table => 1, caption => 1, td => 1, th => 1,              #
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
             last INSCOPE;  
2085            }            }
2086          } # INSCOPE            unless (length $token->{data}) {
2087                        !!!cp ('t88');
2088          if ($self->{open_elements}->[-1]->[1] ne $token->{tag_name}) {              !!!next-token;
2089            !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);              next B;
         }  
           
         splice @{$self->{open_elements}}, $i if defined $i;  
         $clear_up_to_marker->()  
           if {  
             button => 1, marquee => 1, object => 1,  
           }->{$token->{tag_name}};  
         !!!next-token;  
         return;  
       } elsif ($token->{tag_name} eq 'form') {  
         ## has an element in scope  
         INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
           my $node = $self->{open_elements}->[$_];  
           if ($node->[1] eq $token->{tag_name}) {  
             ## generate implied end tags  
             if ({  
                  dd => 1, dt => 1, li => 1, p => 1,  
                  td => 1, th => 1, tr => 1,  
                 }->{$self->{open_elements}->[-1]->[1]}) {  
               !!!back-token;  
               $token = {type => 'end tag',  
                         tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
               return;  
             }  
             last INSCOPE;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
             last INSCOPE;  
2090            }            }
2091          } # INSCOPE  ## TODO: set $token->{column} appropriately
           
         if ($self->{open_elements}->[-1]->[1] eq $token->{tag_name}) {  
           pop @{$self->{open_elements}};  
         } else {  
           !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
2092          }          }
2093    
2094          undef $self->{form_element};          if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
2095          !!!next-token;            !!!cp ('t89');
2096          return;            ## As if <head>
2097        } elsif ({            !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
2098                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,            $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
2099                 }->{$token->{tag_name}}) {            push @{$self->{open_elements}},
2100          ## has an element in scope                [$self->{head_element}, $el_category->{head}];
         my $i;  
         INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {  
           my $node = $self->{open_elements}->[$_];  
           if ({  
                h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,  
               }->{$node->[1]}) {  
             ## generate implied end tags  
             if ({  
                  dd => 1, dt => 1, li => 1, p => 1,  
                  td => 1, th => 1, tr => 1,  
                 }->{$self->{open_elements}->[-1]->[1]}) {  
               !!!back-token;  
               $token = {type => 'end tag',  
                         tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
               return;  
             }  
             $i = $_;  
             last INSCOPE;  
           } elsif ({  
                     table => 1, caption => 1, td => 1, th => 1,  
                     button => 1, marquee => 1, object => 1, html => 1,  
                    }->{$node->[1]}) {  
             last INSCOPE;  
           }  
         } # INSCOPE  
           
         if ($self->{open_elements}->[-1]->[1] ne $token->{tag_name}) {  
           !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
         }  
           
         splice @{$self->{open_elements}}, $i if defined $i;  
         !!!next-token;  
         return;  
       } elsif ({  
                 a => 1,  
                 b => 1, big => 1, em => 1, font => 1, i => 1,  
                 nobr => 1, s => 1, small => 1, strile => 1,  
                 strong => 1, tt => 1, u => 1,  
                }->{$token->{tag_name}}) {  
         $formatting_end_tag->($token->{tag_name});  
 ## TODO: <http://html5.org/tools/web-apps-tracker?from=883&to=884>  
         return;  
       } elsif ({  
                 caption => 1, col => 1, colgroup => 1, frame => 1,  
                 frameset => 1, head => 1, option => 1, optgroup => 1,  
                 tbody => 1, td => 1, tfoot => 1, th => 1,  
                 thead => 1, tr => 1,  
                 area => 1, basefont => 1, bgsound => 1, br => 1,  
                 embed => 1, hr => 1, iframe => 1, image => 1,  
                 img => 1, input => 1, isindex => 1, noembed => 1,  
                 noframes => 1, param => 1, select => 1, spacer => 1,  
                 table => 1, textarea => 1, wbr => 1,  
                 noscript => 0, ## TODO: if scripting is enabled  
                }->{$token->{tag_name}}) {  
         !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
         ## Ignore the token  
         !!!next-token;  
         return;  
           
         ## ISSUE: Issue on HTML5 new elements in spec  
           
       } else {  
         ## Step 1  
         my $node_i = -1;  
         my $node = $self->{open_elements}->[$node_i];  
2101    
2102          ## Step 2            ## Reprocess in the "in head" insertion mode...
2103          S2: {            pop @{$self->{open_elements}};
           if ($node->[1] eq $token->{tag_name}) {  
             ## Step 1  
             ## generate implied end tags  
             if ({  
                  dd => 1, dt => 1, li => 1, p => 1,  
                  td => 1, th => 1, tr => 1,  
                 }->{$self->{open_elements}->[-1]->[1]}) {  
               !!!back-token;  
               $token = {type => 'end tag',  
                         tag_name => $self->{open_elements}->[-1]->[1]}; # MUST  
               return;  
             }  
           
             ## Step 2  
             if ($token->{tag_name} ne $self->{open_elements}->[-1]->[1]) {  
               !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
             }  
               
             ## Step 3  
             splice @{$self->{open_elements}}, $node_i;  
2104    
2105              !!!next-token;            ## Reprocess in the "after head" insertion mode...
2106              last S2;          } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2107            } else {            !!!cp ('t90');
2108              ## Step 3            ## As if </noscript>
2109              if (not $formatting_category->{$node->[1]} and            pop @{$self->{open_elements}};
2110                  #not $phrasing_category->{$node->[1]} and            !!!parse-error (type => 'in noscript:#text', token => $token);
                 ($special_category->{$node->[1]} or  
                  $scoping_category->{$node->[1]})) {  
               !!!parse-error (type => 'not closed:'.$node->[1]);  
               ## Ignore the token  
               !!!next-token;  
               last S2;  
             }  
           }  
             
           ## Step 4  
           $node_i--;  
           $node = $self->{open_elements}->[$node_i];  
2111                        
2112            ## Step 5;            ## Reprocess in the "in head" insertion mode...
2113            redo S2;            ## As if </head>
2114          } # S2            pop @{$self->{open_elements}};
         return;  
       }  
     }  
   }; # $in_body  
2115    
2116    B: {            ## Reprocess in the "after head" insertion mode...
2117      if ($phase eq 'main') {          } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
2118        if ($token->{type} eq 'DOCTYPE') {            !!!cp ('t91');
2119          !!!parse-error (type => 'in html:#DOCTYPE');            pop @{$self->{open_elements}};
         ## Ignore the token  
         ## Stay in the phase  
         !!!next-token;  
         redo B;  
       } elsif ($token->{type} eq 'start tag' and  
                $token->{tag_name} eq 'html') {  
         ## TODO: unless it is the first start tag token, parse-error  
         my $top_el = $self->{open_elements}->[0]->[0];  
         for my $attr_name (keys %{$token->{attributes}}) {  
           unless ($top_el->has_attribute_ns (undef, $attr_name)) {  
             $top_el->set_attribute_ns  
               (undef, [undef, $attr_name],  
                $token->{attributes}->{$attr_name}->{value});  
           }  
         }  
         !!!next-token;  
         redo B;  
       } elsif ($token->{type} eq 'end-of-file') {  
         ## Generate implied end tags  
         if ({  
              dd => 1, dt => 1, li => 1, p => 1, td => 1, th => 1, tr => 1,  
             }->{$self->{open_elements}->[-1]->[1]}) {  
           !!!back-token;  
           $token = {type => 'end tag', tag_name => $self->{open_elements}->[-1]->[1]};  
           redo B;  
         }  
           
         if (@{$self->{open_elements}} > 2 or  
             (@{$self->{open_elements}} == 2 and $self->{open_elements}->[1]->[1] ne 'body')) {  
           !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
         } elsif (defined $self->{inner_html_node} and  
                  @{$self->{open_elements}} > 1 and  
                  $self->{open_elements}->[1]->[1] ne 'body') {  
           !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
         }  
2120    
2121          ## Stop parsing            ## Reprocess in the "after head" insertion mode...
2122          last B;          } else {
2123              !!!cp ('t92');
2124            }
2125    
2126          ## ISSUE: There is an issue in the spec.          ## "after head" insertion mode
2127        } else {          ## As if <body>
2128          if ($self->{insertion_mode} eq 'before head') {          !!!insert-element ('body',, $token);
2129            if ($token->{type} eq 'character') {          $self->{insertion_mode} = IN_BODY_IM;
2130              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          ## reprocess
2131                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);          next B;
2132                unless (length $token->{data}) {        } elsif ($token->{type} == START_TAG_TOKEN) {
2133                  !!!next-token;          if ($token->{tag_name} eq 'head') {
2134                  redo B;            if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
2135                }              !!!cp ('t93');
2136              }              !!!create-element ($self->{head_element}, $HTML_NS, $token->{tag_name}, $token->{attributes}, $token);
2137              ## As if <head>              $self->{open_elements}->[-1]->[0]->append_child
2138              !!!create-element ($self->{head_element}, 'head');                  ($self->{head_element});
2139              $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});              push @{$self->{open_elements}},
2140              push @{$self->{open_elements}}, [$self->{head_element}, 'head'];                  [$self->{head_element}, $el_category->{head}];
2141              $self->{insertion_mode} = 'in head';              $self->{insertion_mode} = IN_HEAD_IM;
2142              ## reprocess              !!!nack ('t93.1');
             redo B;  
           } elsif ($token->{type} eq 'comment') {  
             my $comment = $self->{document}->create_comment ($token->{data});  
             $self->{open_elements}->[-1]->[0]->append_child ($comment);  
2143              !!!next-token;              !!!next-token;
2144              redo B;              next B;
2145            } elsif ($token->{type} eq 'start tag') {            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
2146              my $attr = $token->{tag_name} eq 'head' ? $token->{attributes} : {};              !!!cp ('t93.2');
2147              !!!create-element ($self->{head_element}, 'head', $attr);              !!!parse-error (type => 'after head', text => 'head',
2148              $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});                              token => $token);
2149              push @{$self->{open_elements}}, [$self->{head_element}, 'head'];              ## Ignore the token
2150              $self->{insertion_mode} = 'in head';              !!!nack ('t93.3');
2151              if ($token->{tag_name} eq 'head') {              !!!next-token;
2152                !!!next-token;              next B;
             #} elsif ({  
             #          base => 1, link => 1, meta => 1,  
             #          script => 1, style => 1, title => 1,  
             #         }->{$token->{tag_name}}) {  
             #  ## reprocess  
             } else {  
               ## reprocess  
             }  
             redo B;  
           } elsif ($token->{type} eq 'end tag') {  
             if ({head => 1, body => 1, html => 1}->{$token->{tag_name}}) {  
               ## As if <head>  
               !!!create-element ($self->{head_element}, 'head');  
               $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});  
               push @{$self->{open_elements}}, [$self->{head_element}, 'head'];  
               $self->{insertion_mode} = 'in head';  
               ## reprocess  
               redo B;  
             } else {  
               !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
               ## Ignore the token ## ISSUE: An issue in the spec.  
               !!!next-token;  
               redo B;  
             }  
2153            } else {            } else {
2154              die "$0: $token->{type}: Unknown type";              !!!cp ('t95');
2155            }              !!!parse-error (type => 'in head:head',
2156          } elsif ($self->{insertion_mode} eq 'in head') {                              token => $token); # or in head noscript
2157            if ($token->{type} eq 'character') {              ## Ignore the token
2158              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {              !!!nack ('t95.1');
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);  
               unless (length $token->{data}) {  
                 !!!next-token;  
                 redo B;  
               }  
             }  
               
             #  
           } elsif ($token->{type} eq 'comment') {  
             my $comment = $self->{document}->create_comment ($token->{data});  
             $self->{open_elements}->[-1]->[0]->append_child ($comment);  
2159              !!!next-token;              !!!next-token;
2160              redo B;              next B;
2161            } elsif ($token->{type} eq 'start tag') {            }
2162              if ($token->{tag_name} eq 'title') {          } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {
2163                ## NOTE: There is an "as if in head" code clone            !!!cp ('t96');
2164                my $title_el;            ## As if <head>
2165                !!!create-element ($title_el, 'title', $token->{attributes});            !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
2166                (defined $self->{head_element} ? $self->{head_element} : $self->{open_elements}->[-1]->[0])            $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
2167                  ->append_child ($title_el);            push @{$self->{open_elements}},
2168                $self->{content_model_flag} = 'RCDATA';                [$self->{head_element}, $el_category->{head}];
2169                delete $self->{escape}; # MUST  
2170              $self->{insertion_mode} = IN_HEAD_IM;
2171                my $text = '';            ## Reprocess in the "in head" insertion mode...
2172                !!!next-token;          } else {
2173                while ($token->{type} eq 'character') {            !!!cp ('t97');
2174                  $text .= $token->{data};          }
2175                  !!!next-token;  
2176                }          if ($token->{tag_name} eq 'base') {
2177                if (length $text) {            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2178                  $title_el->manakai_append_text ($text);              !!!cp ('t98');
2179                }              ## As if </noscript>
2180                              pop @{$self->{open_elements}};
2181                $self->{content_model_flag} = 'PCDATA';              !!!parse-error (type => 'in noscript', text => 'base',
2182                                              token => $token);
2183                if ($token->{type} eq 'end tag' and            
2184                    $token->{tag_name} eq 'title') {              $self->{insertion_mode} = IN_HEAD_IM;
2185                  ## Ignore the token              ## Reprocess in the "in head" insertion mode...
               } else {  
                 !!!parse-error (type => 'in RCDATA:#'.$token->{type});  
                 ## ISSUE: And ignore?  
               }  
               !!!next-token;  
               redo B;  
             } elsif ($token->{tag_name} eq 'style') {  
               $style_start_tag->();  
               redo B;  
             } elsif ($token->{tag_name} eq 'script') {  
               $script_start_tag->();  
               redo B;  
             } elsif ({base => 1, link => 1, meta => 1}->{$token->{tag_name}}) {  
               ## NOTE: There are "as if in head" code clones  
               my $el;  
               !!!create-element ($el, $token->{tag_name}, $token->{attributes});  
               (defined $self->{head_element} ? $self->{head_element} : $self->{open_elements}->[-1]->[0])  
                 ->append_child ($el);  
   
               !!!next-token;  
               redo B;  
             } elsif ($token->{tag_name} eq 'head') {  
               !!!parse-error (type => 'in head:head');  
               ## Ignore the token  
               !!!next-token;  
               redo B;  
             } else {  
               #  
             }  
           } elsif ($token->{type} eq 'end tag') {  
             if ($token->{tag_name} eq 'head') {  
               if ($self->{open_elements}->[-1]->[1] eq 'head') {  
                 pop @{$self->{open_elements}};  
               } else {  
                 !!!parse-error (type => 'unmatched end tag:head');  
               }  
               $self->{insertion_mode} = 'after head';  
               !!!next-token;  
               redo B;  
             } elsif ($token->{tag_name} eq 'body' or  
                      $token->{tag_name} eq 'html') {  
               #  
             } else {  
               !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});  
               ## Ignore the token  
               !!!next-token;  
               redo B;  
             }  
2186            } else {            } else {
2187              #              !!!cp ('t99');
2188            }            }
2189    
2190            if ($self->{open_elements}->[-1]->[1] eq 'head') {            ## NOTE: There is a "as if in head" code clone.
2191              ## As if </head>            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
2192              pop @{$self->{open_elements}};              !!!cp ('t100');
2193                !!!parse-error (type => 'after head',
2194                                text => $token->{tag_name}, token => $token);
2195                push @{$self->{open_elements}},
2196                    [$self->{head_element}, $el_category->{head}];
2197                $self->{head_element_inserted} = 1;
2198              } else {
2199                !!!cp ('t101');
2200            }            }
2201            $self->{insertion_mode} = 'after head';            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
2202            ## reprocess            pop @{$self->{open_elements}};
2203            redo B;            pop @{$self->{open_elements}} # <head>
2204                  if $self->{insertion_mode} == AFTER_HEAD_IM;
2205              !!!nack ('t101.1');
2206              !!!next-token;
2207              next B;
2208            } elsif ($token->{tag_name} eq 'link') {
2209              ## NOTE: There is a "as if in head" code clone.
2210              if ($self->{insertion_mode} == AFTER_HEAD_IM) {
2211                !!!cp ('t102');
2212                !!!parse-error (type => 'after head',
2213                                text => $token->{tag_name}, token => $token);
2214                push @{$self->{open_elements}},
2215                    [$self->{head_element}, $el_category->{head}];
2216                $self->{head_element_inserted} = 1;
2217              } else {
2218                !!!cp ('t103');
2219              }
2220              !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
2221              pop @{$self->{open_elements}};
2222              pop @{$self->{open_elements}} # <head>
2223                  if $self->{insertion_mode} == AFTER_HEAD_IM;
2224              !!!ack ('t103.1');
2225              !!!next-token;
2226              next B;
2227            } elsif ($token->{tag_name} eq 'command') {
2228              if ($self->{insertion_mode} == IN_HEAD_IM) {
2229                ## NOTE: If the insertion mode at the time of the emission
2230                ## of the token was "before head", $self->{insertion_mode}
2231                ## is already changed to |IN_HEAD_IM|.
2232    
2233            ## ISSUE: An issue in the spec.              ## NOTE: There is a "as if in head" code clone.
2234          } elsif ($self->{insertion_mode} eq 'after head') {              !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
2235            if ($token->{type} eq 'character') {              pop @{$self->{open_elements}};
2236              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {              pop @{$self->{open_elements}} # <head>
2237                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);                  if $self->{insertion_mode} == AFTER_HEAD_IM;
2238                unless (length $token->{data}) {              !!!ack ('t103.2');
                 !!!next-token;  
                 redo B;  
               }  
             }  
               
             #  
           } elsif ($token->{type} eq 'comment') {  
             my $comment = $self->{document}->create_comment ($token->{data});  
             $self->{open_elements}->[-1]->[0]->append_child ($comment);  
2239              !!!next-token;              !!!next-token;
2240              redo B;              next B;
           } elsif ($token->{type} eq 'start tag') {  
             if ($token->{tag_name} eq 'body') {  
               !!!insert-element ('body', $token->{attributes});  
               $self->{insertion_mode} = 'in body';  
               !!!next-token;  
               redo B;  
             } elsif ($token->{tag_name} eq 'frameset') {  
               !!!insert-element ('frameset', $token->{attributes});  
               $self->{insertion_mode} = 'in frameset';  
               !!!next-token;  
               redo B;  
             } elsif ({  
                       base => 1, link => 1, meta => 1,  
                       script => 1, style => 1, title => 1,  
                      }->{$token->{tag_name}}) {  
               !!!parse-error (type => 'after head:'.$token->{tag_name});  
               $self->{insertion_mode} = 'in head';  
               ## reprocess  
               redo B;  
             } else {  
               #  
             }  
2241            } else {            } else {
2242                ## NOTE: "in head noscript" or "after head" insertion mode
2243                ## - in these cases, these tags are treated as same as
2244                ## normal in-body tags.
2245                !!!cp ('t103.3');
2246              #              #
2247            }            }
2248                      } elsif ($token->{tag_name} eq 'meta') {
2249            ## As if <body>            ## NOTE: There is a "as if in head" code clone.
2250            !!!insert-element ('body');            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
2251            $self->{insertion_mode} = 'in body';              !!!cp ('t104');
2252            ## reprocess              !!!parse-error (type => 'after head',
2253            redo B;                              text => $token->{tag_name}, token => $token);
2254          } elsif ($self->{insertion_mode} eq 'in body') {              push @{$self->{open_elements}},
2255            if ($token->{type} eq 'character') {                  [$self->{head_element}, $el_category->{head}];
2256              ## NOTE: There is a code clone of "character in body".              $self->{head_element_inserted} = 1;
             $reconstruct_active_formatting_elements->($insert_to_current);  
               
             $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});  
   
             !!!next-token;  
             redo B;  
           } elsif ($token->{type} eq 'comment') {  
             ## NOTE: There is a code clone of "comment in body".  
             my $comment = $self->{document}->create_comment ($token->{data});  
             $self->{open_elements}->[-1]->[0]->append_child ($comment);  
             !!!next-token;  
             redo B;  
2257            } else {            } else {
2258              $in_body->($insert_to_current);              !!!cp ('t105');
             redo B;  
2259            }            }
2260          } elsif ($self->{insertion_mode} eq 'in table') {            !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
2261            if ($token->{type} eq 'character') {            my $meta_el = pop @{$self->{open_elements}};
             ## NOTE: There are "character in table" code clones.  
             if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {  
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);  
                 
               unless (length $token->{data}) {  
                 !!!next-token;  
                 redo B;  
               }  
             }  
   
             !!!parse-error (type => 'in table:#character');  
2262    
2263              ## As if in body, but insert into foster parent element                unless ($self->{confident}) {
2264              ## ISSUE: Spec says that "whenever a node would be inserted                  if ($token->{attributes}->{charset}) {
2265              ## into the current node" while characters might not be                    !!!cp ('t106');
2266              ## result in a new Text node.                    ## NOTE: Whether the encoding is supported or not is handled
2267              $reconstruct_active_formatting_elements->($insert_to_foster);                    ## in the {change_encoding} callback.
2268                                  $self->{change_encoding}
2269              if ({                        ->($self, $token->{attributes}->{charset}->{value},
2270                   table => 1, tbody => 1, tfoot => 1,                           $token);
2271                   thead => 1, tr => 1,                    
2272                  }->{$self->{open_elements}->[-1]->[1]}) {                    $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
2273                # MUST                        ->set_user_data (manakai_has_reference =>
2274                my $foster_parent_element;                                             $token->{attributes}->{charset}
2275                my $next_sibling;                                                 ->{has_reference});
2276                my $prev_sibling;                  } elsif ($token->{attributes}->{content}) {
2277                OE: for (reverse 0..$#{$self->{open_elements}}) {                    if ($token->{attributes}->{content}->{value}
2278                  if ($self->{open_elements}->[$_]->[1] eq 'table') {                        =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
2279                    my $parent = $self->{open_elements}->[$_]->[0]->parent_node;                            [\x09\x0A\x0C\x0D\x20]*=
2280                    if (defined $parent and $parent->node_type == 1) {                            [\x09\x0A\x0C\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
2281                      $foster_parent_element = $parent;                            ([^"'\x09\x0A\x0C\x0D\x20]
2282                      $next_sibling = $self->{open_elements}->[$_]->[0];                             [^\x09\x0A\x0C\x0D\x20\x3B]*))/x) {
2283                      $prev_sibling = $next_sibling->previous_sibling;                      !!!cp ('t107');
2284                        ## NOTE: Whether the encoding is supported or not is handled
2285                        ## in the {change_encoding} callback.
2286                        $self->{change_encoding}
2287                            ->($self, defined $1 ? $1 : defined $2 ? $2 : $3,
2288                               $token);
2289                        $meta_el->[0]->get_attribute_node_ns (undef, 'content')
2290                            ->set_user_data (manakai_has_reference =>
2291                                                 $token->{attributes}->{content}
2292                                                       ->{has_reference});
2293                    } else {                    } else {
2294                      $foster_parent_element = $self->{open_elements}->[$_ - 1]->[0];                      !!!cp ('t108');
                     $prev_sibling = $foster_parent_element->last_child;  
2295                    }                    }
                   last OE;  
2296                  }                  }
               } # OE  
               $foster_parent_element = $self->{open_elements}->[0]->[0] and  
               $prev_sibling = $foster_parent_element->last_child  
                 unless defined $foster_parent_element;  
               if (defined $prev_sibling and  
                   $prev_sibling->node_type == 3) {  
                 $prev_sibling->manakai_append_text ($token->{data});  
2297                } else {                } else {
2298                  $foster_parent_element->insert_before                  if ($token->{attributes}->{charset}) {
2299                    ($self->{document}->create_text_node ($token->{data}),                    !!!cp ('t109');
2300                     $next_sibling);                    $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
2301                }                        ->set_user_data (manakai_has_reference =>
2302              } else {                                             $token->{attributes}->{charset}
2303                $self->{open_elements}->[-1]->[0]->manakai_append_text ($token->{data});                                                 ->{has_reference});
2304              }                  }
2305                                if ($token->{attributes}->{content}) {
2306              !!!next-token;                    !!!cp ('t110');
2307              redo B;                    $meta_el->[0]->get_attribute_node_ns (undef, 'content')
2308            } elsif ($token->{type} eq 'comment') {                        ->set_user_data (manakai_has_reference =>
2309              my $comment = $self->{document}->create_comment ($token->{data});                                             $token->{attributes}->{content}
2310              $self->{open_elements}->[-1]->[0]->append_child ($comment);                                                 ->{has_reference});
2311              !!!next-token;                  }
             redo B;  
           } elsif ($token->{type} eq 'start tag') {  
             if ({  
                  caption => 1,  
                  colgroup => 1,  
                  tbody => 1, tfoot => 1, thead => 1,  
                 }->{$token->{tag_name}}) {  
               ## Clear back to table context  
               while ($self->{open_elements}->[-1]->[1] ne 'table' and  
                      $self->{open_elements}->[-1]->[1] ne 'html') {  
                 !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
                 pop @{$self->{open_elements}};  
2312                }                }
2313    
2314                push @$active_formatting_elements, ['#marker', '']                pop @{$self->{open_elements}} # <head>
2315                  if $token->{tag_name} eq 'caption';                    if $self->{insertion_mode} == AFTER_HEAD_IM;
2316                  !!!ack ('t110.1');
               !!!insert-element ($token->{tag_name}, $token->{attributes});  
               $self->{insertion_mode} = {  
                                  caption => 'in caption',  
                                  colgroup => 'in column group',  
                                  tbody => 'in table body',  
                                  tfoot => 'in table body',  
                                  thead => 'in table body',  
                                 }->{$token->{tag_name}};  
2317                !!!next-token;                !!!next-token;
2318                redo B;                next B;
2319              } elsif ({          } elsif ($token->{tag_name} eq 'title') {
2320                        col => 1,            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2321                        td => 1, th => 1, tr => 1,              !!!cp ('t111');
2322                       }->{$token->{tag_name}}) {              ## As if </noscript>
2323                ## Clear back to table context              pop @{$self->{open_elements}};
2324                while ($self->{open_elements}->[-1]->[1] ne 'table' and              !!!parse-error (type => 'in noscript', text => 'title',
2325                       $self->{open_elements}->[-1]->[1] ne 'html') {                              token => $token);
2326                  !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);            
2327                  pop @{$self->{open_elements}};              $self->{insertion_mode} = IN_HEAD_IM;
2328                }              ## Reprocess in the "in head" insertion mode...
2329              } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
2330                !!!cp ('t112');
2331                !!!parse-error (type => 'after head',
2332                                text => $token->{tag_name}, token => $token);
2333                push @{$self->{open_elements}},
2334                    [$self->{head_element}, $el_category->{head}];
2335                $self->{head_element_inserted} = 1;
2336              } else {
2337                !!!cp ('t113');
2338              }
2339    
2340                !!!insert-element ($token->{tag_name} eq 'col' ? 'colgroup' : 'tbody');            ## NOTE: There is a "as if in head" code clone.
2341                $self->{insertion_mode} = $token->{tag_name} eq 'col'            $parse_rcdata->(RCDATA_CONTENT_MODEL);
                 ? 'in column group' : 'in table body';  
               ## reprocess  
               redo B;  
             } elsif ($token->{tag_name} eq 'table') {  
               ## NOTE: There are code clones for this "table in table"  
               !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);  
2342    
2343                ## As if </table>            ## NOTE: At this point the stack of open elements contain
2344                ## have a table element in table scope            ## the |head| element (index == -2) and the |script| element
2345                my $i;            ## (index == -1).  In the "after head" insertion mode the
2346                INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {            ## |head| element is inserted only for the purpose of
2347                  my $node = $self->{open_elements}->[$_];            ## providing the context for the |script| element, and
2348                  if ($node->[1] eq 'table') {            ## therefore we can now and have to remove the element from
2349                    $i = $_;            ## the stack.
2350                    last INSCOPE;            splice @{$self->{open_elements}}, -2, 1, () # <head>
2351                  } elsif ({                if ($self->{insertion_mode} & IM_MASK) == AFTER_HEAD_IM;
2352                            table => 1, html => 1,            next B;
2353                           }->{$node->[1]}) {          } elsif ($token->{tag_name} eq 'style' or
2354                    last INSCOPE;                   $token->{tag_name} eq 'noframes') {
2355                  }            ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and
2356                } # INSCOPE            ## insertion mode IN_HEAD_IM)
2357                unless (defined $i) {            ## NOTE: There is a "as if in head" code clone.
2358                  !!!parse-error (type => 'unmatched end tag:table');            if ($self->{insertion_mode} == AFTER_HEAD_IM) {
2359                  ## Ignore tokens </table><table>              !!!cp ('t114');
2360                !!!parse-error (type => 'after head',
2361                                text => $token->{tag_name}, token => $token);
2362                push @{$self->{open_elements}},
2363                    [$self->{head_element}, $el_category->{head}];
2364                $self->{head_element_inserted} = 1;
2365              } else {
2366                !!!cp ('t115');
2367              }
2368              $parse_rcdata->(CDATA_CONTENT_MODEL);
2369              ## ISSUE: A spec bug [Bug 6038]
2370              splice @{$self->{open_elements}}, -2, 1, () # <head>
2371                  if ($self->{insertion_mode} & IM_MASK) == AFTER_HEAD_IM;
2372              next B;
2373            } elsif ($token->{tag_name} eq 'noscript') {
2374                  if ($self->{insertion_mode} == IN_HEAD_IM) {
2375                    !!!cp ('t116');
2376                    ## NOTE: and scripting is disalbed
2377                    !!!insert-element ($token->{tag_name}, $token->{attributes}, $token);
2378                    $self->{insertion_mode} = IN_HEAD_NOSCRIPT_IM;
2379                    !!!nack ('t116.1');
2380                  !!!next-token;                  !!!next-token;
2381                  redo B;                  next B;
2382                  } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2383                    !!!cp ('t117');
2384                    !!!parse-error (type => 'in noscript', text => 'noscript',
2385                                    token => $token);
2386                    ## Ignore the token
2387                    !!!nack ('t117.1');
2388                    !!!next-token;
2389                    next B;
2390                  } else {
2391                    !!!cp ('t118');
2392                    #
2393                }                }
2394                          } elsif ($token->{tag_name} eq 'script') {
2395                ## generate implied end tags            if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
2396                if ({              !!!cp ('t119');
2397                     dd => 1, dt => 1, li => 1, p => 1,              ## As if </noscript>
2398                     td => 1, th => 1, tr => 1,              pop @{$self->{open_elements}};
2399                    }->{$self->{open_elements}->[-1]->[1]}) {              !!!parse-error (type => 'in noscript', text => 'script',
2400                  !!!back-token; # <table>                              token => $token);
2401                  $token = {type => 'end tag', tag_name => 'table'};            
2402                  !!!back-token;              $self->{insertion_mode} = IN_HEAD_IM;
2403                  $token = {type => 'end tag',              ## Reprocess in the "in head" insertion mode...
2404                            tag_name => $self->{open_elements}->[-1]->[1]}; # MUST            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
2405                  redo B;              !!!cp ('t120');
2406                !!!parse-error (type => 'after head',
2407                                text => $token->{tag_name}, token => $token);
2408                push @{$self->{open_elements}},
2409                    [$self->{head_element}, $el_category->{head}];
2410                $self->{head_element_inserted} = 1;
2411              } else {
2412                !!!cp ('t121');
2413              }
2414    
2415              ## NOTE: There is a "as if in head" code clone.
2416              $script_start_tag->();
2417              ## ISSUE: A spec bug  [Bug 6038]
2418              splice @{$self->{open_elements}}, -2, 1 # <head>
2419