/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.11 by wakaba, Sat Jun 23 03:53:35 2007 UTC revision 1.30 by wakaba, Sat Jun 30 13:12:32 2007 UTC
# Line 2  package Whatpm::HTML; Line 2  package Whatpm::HTML;
2  use strict;  use strict;
3  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};
4    
5  ## This is an early version of an HTML parser.  ## ISSUE:
6    ## var doc = implementation.createDocument (null, null, null);
7    ## doc.write ('');
8    ## alert (doc.compatMode);
9    
10  my $permitted_slash_tag_name = {  my $permitted_slash_tag_name = {
11    base => 1,    base => 1,
# Line 18  my $permitted_slash_tag_name = { Line 21  my $permitted_slash_tag_name = {
21    input => 1,    input => 1,
22  };  };
23    
 my $entity_char = {  
   AElig => "\x{00C6}",  
   Aacute => "\x{00C1}",  
   Acirc => "\x{00C2}",  
   Agrave => "\x{00C0}",  
   Alpha => "\x{0391}",  
   Aring => "\x{00C5}",  
   Atilde => "\x{00C3}",  
   Auml => "\x{00C4}",  
   Beta => "\x{0392}",  
   Ccedil => "\x{00C7}",  
   Chi => "\x{03A7}",  
   Dagger => "\x{2021}",  
   Delta => "\x{0394}",  
   ETH => "\x{00D0}",  
   Eacute => "\x{00C9}",  
   Ecirc => "\x{00CA}",  
   Egrave => "\x{00C8}",  
   Epsilon => "\x{0395}",  
   Eta => "\x{0397}",  
   Euml => "\x{00CB}",  
   Gamma => "\x{0393}",  
   Iacute => "\x{00CD}",  
   Icirc => "\x{00CE}",  
   Igrave => "\x{00CC}",  
   Iota => "\x{0399}",  
   Iuml => "\x{00CF}",  
   Kappa => "\x{039A}",  
   Lambda => "\x{039B}",  
   Mu => "\x{039C}",  
   Ntilde => "\x{00D1}",  
   Nu => "\x{039D}",  
   OElig => "\x{0152}",  
   Oacute => "\x{00D3}",  
   Ocirc => "\x{00D4}",  
   Ograve => "\x{00D2}",  
   Omega => "\x{03A9}",  
   Omicron => "\x{039F}",  
   Oslash => "\x{00D8}",  
   Otilde => "\x{00D5}",  
   Ouml => "\x{00D6}",  
   Phi => "\x{03A6}",  
   Pi => "\x{03A0}",  
   Prime => "\x{2033}",  
   Psi => "\x{03A8}",  
   Rho => "\x{03A1}",  
   Scaron => "\x{0160}",  
   Sigma => "\x{03A3}",  
   THORN => "\x{00DE}",  
   Tau => "\x{03A4}",  
   Theta => "\x{0398}",  
   Uacute => "\x{00DA}",  
   Ucirc => "\x{00DB}",  
   Ugrave => "\x{00D9}",  
   Upsilon => "\x{03A5}",  
   Uuml => "\x{00DC}",  
   Xi => "\x{039E}",  
   Yacute => "\x{00DD}",  
   Yuml => "\x{0178}",  
   Zeta => "\x{0396}",  
   aacute => "\x{00E1}",  
   acirc => "\x{00E2}",  
   acute => "\x{00B4}",  
   aelig => "\x{00E6}",  
   agrave => "\x{00E0}",  
   alefsym => "\x{2135}",  
   alpha => "\x{03B1}",  
   amp => "\x{0026}",  
   AMP => "\x{0026}",  
   and => "\x{2227}",  
   ang => "\x{2220}",  
   apos => "\x{0027}",  
   aring => "\x{00E5}",  
   asymp => "\x{2248}",  
   atilde => "\x{00E3}",  
   auml => "\x{00E4}",  
   bdquo => "\x{201E}",  
   beta => "\x{03B2}",  
   brvbar => "\x{00A6}",  
   bull => "\x{2022}",  
   cap => "\x{2229}",  
   ccedil => "\x{00E7}",  
   cedil => "\x{00B8}",  
   cent => "\x{00A2}",  
   chi => "\x{03C7}",  
   circ => "\x{02C6}",  
   clubs => "\x{2663}",  
   cong => "\x{2245}",  
   copy => "\x{00A9}",  
   COPY => "\x{00A9}",  
   crarr => "\x{21B5}",  
   cup => "\x{222A}",  
   curren => "\x{00A4}",  
   dArr => "\x{21D3}",  
   dagger => "\x{2020}",  
   darr => "\x{2193}",  
   deg => "\x{00B0}",  
   delta => "\x{03B4}",  
   diams => "\x{2666}",  
   divide => "\x{00F7}",  
   eacute => "\x{00E9}",  
   ecirc => "\x{00EA}",  
   egrave => "\x{00E8}",  
   empty => "\x{2205}",  
   emsp => "\x{2003}",  
   ensp => "\x{2002}",  
   epsilon => "\x{03B5}",  
   equiv => "\x{2261}",  
   eta => "\x{03B7}",  
   eth => "\x{00F0}",  
   euml => "\x{00EB}",  
   euro => "\x{20AC}",  
   exist => "\x{2203}",  
   fnof => "\x{0192}",  
   forall => "\x{2200}",  
   frac12 => "\x{00BD}",  
   frac14 => "\x{00BC}",  
   frac34 => "\x{00BE}",  
   frasl => "\x{2044}",  
   gamma => "\x{03B3}",  
   ge => "\x{2265}",  
   gt => "\x{003E}",  
   GT => "\x{003E}",  
   hArr => "\x{21D4}",  
   harr => "\x{2194}",  
   hearts => "\x{2665}",  
   hellip => "\x{2026}",  
   iacute => "\x{00ED}",  
   icirc => "\x{00EE}",  
   iexcl => "\x{00A1}",  
   igrave => "\x{00EC}",  
   image => "\x{2111}",  
   infin => "\x{221E}",  
   int => "\x{222B}",  
   iota => "\x{03B9}",  
   iquest => "\x{00BF}",  
   isin => "\x{2208}",  
   iuml => "\x{00EF}",  
   kappa => "\x{03BA}",  
   lArr => "\x{21D0}",  
   lambda => "\x{03BB}",  
   lang => "\x{2329}",  
   laquo => "\x{00AB}",  
   larr => "\x{2190}",  
   lceil => "\x{2308}",  
   ldquo => "\x{201C}",  
   le => "\x{2264}",  
   lfloor => "\x{230A}",  
   lowast => "\x{2217}",  
   loz => "\x{25CA}",  
   lrm => "\x{200E}",  
   lsaquo => "\x{2039}",  
   lsquo => "\x{2018}",  
   lt => "\x{003C}",  
   LT => "\x{003C}",  
   macr => "\x{00AF}",  
   mdash => "\x{2014}",  
   micro => "\x{00B5}",  
   middot => "\x{00B7}",  
   minus => "\x{2212}",  
   mu => "\x{03BC}",  
   nabla => "\x{2207}",  
   nbsp => "\x{00A0}",  
   ndash => "\x{2013}",  
   ne => "\x{2260}",  
   ni => "\x{220B}",  
   not => "\x{00AC}",  
   notin => "\x{2209}",  
   nsub => "\x{2284}",  
   ntilde => "\x{00F1}",  
   nu => "\x{03BD}",  
   oacute => "\x{00F3}",  
   ocirc => "\x{00F4}",  
   oelig => "\x{0153}",  
   ograve => "\x{00F2}",  
   oline => "\x{203E}",  
   omega => "\x{03C9}",  
   omicron => "\x{03BF}",  
   oplus => "\x{2295}",  
   or => "\x{2228}",  
   ordf => "\x{00AA}",  
   ordm => "\x{00BA}",  
   oslash => "\x{00F8}",  
   otilde => "\x{00F5}",  
   otimes => "\x{2297}",  
   ouml => "\x{00F6}",  
   para => "\x{00B6}",  
   part => "\x{2202}",  
   permil => "\x{2030}",  
   perp => "\x{22A5}",  
   phi => "\x{03C6}",  
   pi => "\x{03C0}",  
   piv => "\x{03D6}",  
   plusmn => "\x{00B1}",  
   pound => "\x{00A3}",  
   prime => "\x{2032}",  
   prod => "\x{220F}",  
   prop => "\x{221D}",  
   psi => "\x{03C8}",  
   quot => "\x{0022}",  
   QUOT => "\x{0022}",  
   rArr => "\x{21D2}",  
   radic => "\x{221A}",  
   rang => "\x{232A}",  
   raquo => "\x{00BB}",  
   rarr => "\x{2192}",  
   rceil => "\x{2309}",  
   rdquo => "\x{201D}",  
   real => "\x{211C}",  
   reg => "\x{00AE}",  
   REG => "\x{00AE}",  
   rfloor => "\x{230B}",  
   rho => "\x{03C1}",  
   rlm => "\x{200F}",  
   rsaquo => "\x{203A}",  
   rsquo => "\x{2019}",  
   sbquo => "\x{201A}",  
   scaron => "\x{0161}",  
   sdot => "\x{22C5}",  
   sect => "\x{00A7}",  
   shy => "\x{00AD}",  
   sigma => "\x{03C3}",  
   sigmaf => "\x{03C2}",  
   sim => "\x{223C}",  
   spades => "\x{2660}",  
   sub => "\x{2282}",  
   sube => "\x{2286}",  
   sum => "\x{2211}",  
   sup => "\x{2283}",  
   sup1 => "\x{00B9}",  
   sup2 => "\x{00B2}",  
   sup3 => "\x{00B3}",  
   supe => "\x{2287}",  
   szlig => "\x{00DF}",  
   tau => "\x{03C4}",  
   there4 => "\x{2234}",  
   theta => "\x{03B8}",  
   thetasym => "\x{03D1}",  
   thinsp => "\x{2009}",  
   thorn => "\x{00FE}",  
   tilde => "\x{02DC}",  
   times => "\x{00D7}",  
   trade => "\x{2122}",  
   uArr => "\x{21D1}",  
   uacute => "\x{00FA}",  
   uarr => "\x{2191}",  
   ucirc => "\x{00FB}",  
   ugrave => "\x{00F9}",  
   uml => "\x{00A8}",  
   upsih => "\x{03D2}",  
   upsilon => "\x{03C5}",  
   uuml => "\x{00FC}",  
   weierp => "\x{2118}",  
   xi => "\x{03BE}",  
   yacute => "\x{00FD}",  
   yen => "\x{00A5}",  
   yuml => "\x{00FF}",  
   zeta => "\x{03B6}",  
   zwj => "\x{200D}",  
   zwnj => "\x{200C}",  
 }; # $entity_char  
   
24  my $c1_entity_char = {  my $c1_entity_char = {
25    0x80 => 0x20AC,    0x80 => 0x20AC,
26    0x81 => 0xFFFD,    0x81 => 0xFFFD,
# Line 349  sub parse_string ($$$;$) { Line 90  sub parse_string ($$$;$) {
90    my $column = 0;    my $column = 0;
91    $self->{set_next_input_character} = sub {    $self->{set_next_input_character} = sub {
92      my $self = shift;      my $self = shift;
93    
94        pop @{$self->{prev_input_character}};
95        unshift @{$self->{prev_input_character}}, $self->{next_input_character};
96    
97      $self->{next_input_character} = -1 and return if $i >= length $$s;      $self->{next_input_character} = -1 and return if $i >= length $$s;
98      $self->{next_input_character} = ord substr $$s, $i++, 1;      $self->{next_input_character} = ord substr $$s, $i++, 1;
99      $column++;      $column++;
# Line 357  sub parse_string ($$$;$) { Line 102  sub parse_string ($$$;$) {
102        $line++;        $line++;
103        $column = 0;        $column = 0;
104      } elsif ($self->{next_input_character} == 0x000D) { # CR      } elsif ($self->{next_input_character} == 0x000D) { # CR
105        if ($i >= length $$s) {        $i++ if substr ($$s, $i, 1) eq "\x0A";
         #  
       } else {  
         my $next_char = ord substr $$s, $i++, 1;  
         if ($next_char == 0x000A) { # LF  
           #  
         } else {  
           push @{$self->{char}}, $next_char;  
         }  
       }  
106        $self->{next_input_character} = 0x000A; # LF # MUST        $self->{next_input_character} = 0x000A; # LF # MUST
107        $line++;        $line++;
108        $column = 0;        $column = 0;
# Line 377  sub parse_string ($$$;$) { Line 113  sub parse_string ($$$;$) {
113        $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST        $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
114      }      }
115    };    };
116      $self->{prev_input_character} = [-1, -1, -1];
117      $self->{next_input_character} = -1;
118    
119    my $onerror = $_[2] || sub {    my $onerror = $_[2] || sub {
120      my (%opt) = @_;      my (%opt) = @_;
# Line 420  sub _initialize_tokenizer ($) { Line 158  sub _initialize_tokenizer ($) {
158    # $self->{next_input_character}    # $self->{next_input_character}
159    !!!next-input-character;    !!!next-input-character;
160    $self->{token} = [];    $self->{token} = [];
161      # $self->{escape}
162  } # _initialize_tokenizer  } # _initialize_tokenizer
163    
164  ## A token has:  ## A token has:
165  ##   ->{type} eq 'DOCTYPE', 'start tag', 'end tag', 'comment',  ##   ->{type} eq 'DOCTYPE', 'start tag', 'end tag', 'comment',
166  ##       'character', or 'end-of-file'  ##       'character', or 'end-of-file'
167  ##   ->{name} (DOCTYPE, start tag (tagname), end tag (tagname))  ##   ->{name} (DOCTYPE, start tag (tag name), end tag (tag name))
168      ## ISSUE: the spec need s/tagname/tag name/  ##   ->{public_identifier} (DOCTYPE)
169  ##   ->{error} == 1 or 0 (DOCTYPE)  ##   ->{system_identifier} (DOCTYPE)
170    ##   ->{correct} == 1 or 0 (DOCTYPE)
171  ##   ->{attributes} isa HASH (start tag, end tag)  ##   ->{attributes} isa HASH (start tag, end tag)
172  ##   ->{data} (comment, character)  ##   ->{data} (comment, character)
173    
 ## Macros  
 ##   Macros MUST be preceded by three EXCLAMATION MARKs.  
 ##   emit ($token)  
 ##     Emits the specified token.  
   
174  ## Emitted token MUST immediately be handled by the tree construction state.  ## Emitted token MUST immediately be handled by the tree construction state.
175    
176  ## Before each step, UA MAY check to see if either one of the scripts in  ## Before each step, UA MAY check to see if either one of the scripts in
# Line 461  sub _get_next_token ($) { Line 196  sub _get_next_token ($) {
196          } else {          } else {
197            #            #
198          }          }
199          } elsif ($self->{next_input_character} == 0x002D) { # -
200            if ($self->{content_model_flag} eq 'RCDATA' or
201                $self->{content_model_flag} eq 'CDATA') {
202              unless ($self->{escape}) {
203                if ($self->{prev_input_character}->[0] == 0x002D and # -
204                    $self->{prev_input_character}->[1] == 0x0021 and # !
205                    $self->{prev_input_character}->[2] == 0x003C) { # <
206                  $self->{escape} = 1;
207                }
208              }
209            }
210            
211            #
212        } elsif ($self->{next_input_character} == 0x003C) { # <        } elsif ($self->{next_input_character} == 0x003C) { # <
213          if ($self->{content_model_flag} ne 'PLAINTEXT') {          if ($self->{content_model_flag} eq 'PCDATA' or
214                (($self->{content_model_flag} eq 'CDATA' or
215                  $self->{content_model_flag} eq 'RCDATA') and
216                 not $self->{escape})) {
217            $self->{state} = 'tag open';            $self->{state} = 'tag open';
218            !!!next-input-character;            !!!next-input-character;
219            redo A;            redo A;
220          } else {          } else {
221            #            #
222          }          }
223          } elsif ($self->{next_input_character} == 0x003E) { # >
224            if ($self->{escape} and
225                ($self->{content_model_flag} eq 'RCDATA' or
226                 $self->{content_model_flag} eq 'CDATA')) {
227              if ($self->{prev_input_character}->[0] == 0x002D and # -
228                  $self->{prev_input_character}->[1] == 0x002D) { # -
229                delete $self->{escape};
230              }
231            }
232            
233            #
234        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
235          !!!emit ({type => 'end-of-file'});          !!!emit ({type => 'end-of-file'});
236          last A; ## TODO: ok?          last A; ## TODO: ok?
# Line 485  sub _get_next_token ($) { Line 247  sub _get_next_token ($) {
247      } elsif ($self->{state} eq 'entity data') {      } elsif ($self->{state} eq 'entity data') {
248        ## (cannot happen in CDATA state)        ## (cannot happen in CDATA state)
249                
250        my $token = $self->_tokenize_attempt_to_consume_an_entity;        my $token = $self->_tokenize_attempt_to_consume_an_entity (0);
251    
252        $self->{state} = 'data';        $self->{state} = 'data';
253        # next-input-character is already done        # next-input-character is already done
# Line 564  sub _get_next_token ($) { Line 326  sub _get_next_token ($) {
326      } elsif ($self->{state} eq 'close tag open') {      } elsif ($self->{state} eq 'close tag open') {
327        if ($self->{content_model_flag} eq 'RCDATA' or        if ($self->{content_model_flag} eq 'RCDATA' or
328            $self->{content_model_flag} eq 'CDATA') {            $self->{content_model_flag} eq 'CDATA') {
329          my @next_char;          if (defined $self->{last_emitted_start_tag_name}) {
330          TAGNAME: for (my $i = 0; $i < length $self->{last_emitted_start_tag_name}; $i++) {            ## NOTE: <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>
331              my @next_char;
332              TAGNAME: for (my $i = 0; $i < length $self->{last_emitted_start_tag_name}; $i++) {
333                push @next_char, $self->{next_input_character};
334                my $c = ord substr ($self->{last_emitted_start_tag_name}, $i, 1);
335                my $C = 0x0061 <= $c && $c <= 0x007A ? $c - 0x0020 : $c;
336                if ($self->{next_input_character} == $c or $self->{next_input_character} == $C) {
337                  !!!next-input-character;
338                  next TAGNAME;
339                } else {
340                  $self->{next_input_character} = shift @next_char; # reconsume
341                  !!!back-next-input-character (@next_char);
342                  $self->{state} = 'data';
343    
344                  !!!emit ({type => 'character', data => '</'});
345      
346                  redo A;
347                }
348              }
349            push @next_char, $self->{next_input_character};            push @next_char, $self->{next_input_character};
350            my $c = ord substr ($self->{last_emitted_start_tag_name}, $i, 1);        
351            my $C = 0x0061 <= $c && $c <= 0x007A ? $c - 0x0020 : $c;            unless ($self->{next_input_character} == 0x0009 or # HT
352            if ($self->{next_input_character} == $c or $self->{next_input_character} == $C) {                    $self->{next_input_character} == 0x000A or # LF
353              !!!next-input-character;                    $self->{next_input_character} == 0x000B or # VT
354              next TAGNAME;                    $self->{next_input_character} == 0x000C or # FF
355            } else {                    $self->{next_input_character} == 0x0020 or # SP
356              !!!parse-error (type => 'unmatched end tag');                    $self->{next_input_character} == 0x003E or # >
357                      $self->{next_input_character} == 0x002F or # /
358                      $self->{next_input_character} == -1) {
359              $self->{next_input_character} = shift @next_char; # reconsume              $self->{next_input_character} = shift @next_char; # reconsume
360              !!!back-next-input-character (@next_char);              !!!back-next-input-character (@next_char);
361              $self->{state} = 'data';              $self->{state} = 'data';
   
362              !!!emit ({type => 'character', data => '</'});              !!!emit ({type => 'character', data => '</'});
   
363              redo A;              redo A;
364              } else {
365                $self->{next_input_character} = shift @next_char;
366                !!!back-next-input-character (@next_char);
367                # and consume...
368            }            }
369          }          } else {
370          push @next_char, $self->{next_input_character};            ## No start tag token has ever been emitted
371                  # next-input-character is already done
         unless ($self->{next_input_character} == 0x0009 or # HT  
                 $self->{next_input_character} == 0x000A or # LF  
                 $self->{next_input_character} == 0x000B or # VT  
                 $self->{next_input_character} == 0x000C or # FF  
                 $self->{next_input_character} == 0x0020 or # SP  
                 $self->{next_input_character} == 0x003E or # >  
                 $self->{next_input_character} == 0x002F or # /  
                 $self->{next_input_character} == 0x003C or # <  
                 $self->{next_input_character} == -1) {  
           !!!parse-error (type => 'unmatched end tag');  
           $self->{next_input_character} = shift @next_char; # reconsume  
           !!!back-next-input-character (@next_char);  
372            $self->{state} = 'data';            $self->{state} = 'data';
   
373            !!!emit ({type => 'character', data => '</'});            !!!emit ({type => 'character', data => '</'});
   
374            redo A;            redo A;
         } else {  
           $self->{next_input_character} = shift @next_char;  
           !!!back-next-input-character (@next_char);  
           # and consume...  
375          }          }
376        }        }
377                
# Line 653  sub _get_next_token ($) { Line 419  sub _get_next_token ($) {
419          redo A;          redo A;
420        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
421          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
422              $self->{current_token}->{first_start_tag}
423                  = not defined $self->{last_emitted_start_tag_name};
424            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
425          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
426            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 666  sub _get_next_token ($) { Line 434  sub _get_next_token ($) {
434          !!!next-input-character;          !!!next-input-character;
435    
436          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
437    
438          redo A;          redo A;
439        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{next_input_character} and
# Line 676  sub _get_next_token ($) { Line 443  sub _get_next_token ($) {
443          ## Stay in this state          ## Stay in this state
444          !!!next-input-character;          !!!next-input-character;
445          redo A;          redo A;
446        } elsif ($self->{next_input_character} == 0x003C or # <        } elsif ($self->{next_input_character} == -1) {
                $self->{next_input_character} == -1) {  
447          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
448          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
449              $self->{current_token}->{first_start_tag}
450                  = not defined $self->{last_emitted_start_tag_name};
451            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
452          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
453            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 693  sub _get_next_token ($) { Line 461  sub _get_next_token ($) {
461          # reconsume          # reconsume
462    
463          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
464    
465          redo A;          redo A;
466        } elsif ($self->{next_input_character} == 0x002F) { # /        } elsif ($self->{next_input_character} == 0x002F) { # /
# Line 727  sub _get_next_token ($) { Line 494  sub _get_next_token ($) {
494          redo A;          redo A;
495        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
496          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
497              $self->{current_token}->{first_start_tag}
498                  = not defined $self->{last_emitted_start_tag_name};
499            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
500          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
501            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 740  sub _get_next_token ($) { Line 509  sub _get_next_token ($) {
509          !!!next-input-character;          !!!next-input-character;
510    
511          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
512    
513          redo A;          redo A;
514        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{next_input_character} and
# Line 763  sub _get_next_token ($) { Line 531  sub _get_next_token ($) {
531          ## Stay in the state          ## Stay in the state
532          # next-input-character is already done          # next-input-character is already done
533          redo A;          redo A;
534        } elsif ($self->{next_input_character} == 0x003C or # <        } elsif ($self->{next_input_character} == -1) {
                $self->{next_input_character} == -1) {  
535          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
536          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
537              $self->{current_token}->{first_start_tag}
538                  = not defined $self->{last_emitted_start_tag_name};
539            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
540          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
541            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 780  sub _get_next_token ($) { Line 549  sub _get_next_token ($) {
549          # reconsume          # reconsume
550    
551          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
552    
553          redo A;          redo A;
554        } else {        } else {
# Line 819  sub _get_next_token ($) { Line 587  sub _get_next_token ($) {
587        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
588          $before_leave->();          $before_leave->();
589          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
590              $self->{current_token}->{first_start_tag}
591                  = not defined $self->{last_emitted_start_tag_name};
592            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
593          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
594            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 832  sub _get_next_token ($) { Line 602  sub _get_next_token ($) {
602          !!!next-input-character;          !!!next-input-character;
603    
604          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
605    
606          redo A;          redo A;
607        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{next_input_character} and
# Line 855  sub _get_next_token ($) { Line 624  sub _get_next_token ($) {
624          $self->{state} = 'before attribute name';          $self->{state} = 'before attribute name';
625          # next-input-character is already done          # next-input-character is already done
626          redo A;          redo A;
627        } elsif ($self->{next_input_character} == 0x003C or # <        } elsif ($self->{next_input_character} == -1) {
                $self->{next_input_character} == -1) {  
628          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
629          $before_leave->();          $before_leave->();
630          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
631              $self->{current_token}->{first_start_tag}
632                  = not defined $self->{last_emitted_start_tag_name};
633            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
634          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
635            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 873  sub _get_next_token ($) { Line 643  sub _get_next_token ($) {
643          # reconsume          # reconsume
644    
645          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
646    
647          redo A;          redo A;
648        } else {        } else {
# Line 897  sub _get_next_token ($) { Line 666  sub _get_next_token ($) {
666          redo A;          redo A;
667        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
668          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
669              $self->{current_token}->{first_start_tag}
670                  = not defined $self->{last_emitted_start_tag_name};
671            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
672          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
673            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 910  sub _get_next_token ($) { Line 681  sub _get_next_token ($) {
681          !!!next-input-character;          !!!next-input-character;
682    
683          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
684    
685          redo A;          redo A;
686        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{next_input_character} and
# Line 933  sub _get_next_token ($) { Line 703  sub _get_next_token ($) {
703          $self->{state} = 'before attribute name';          $self->{state} = 'before attribute name';
704          # next-input-character is already done          # next-input-character is already done
705          redo A;          redo A;
706        } elsif ($self->{next_input_character} == 0x003C or # <        } elsif ($self->{next_input_character} == -1) {
                $self->{next_input_character} == -1) {  
707          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
708          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
709              $self->{current_token}->{first_start_tag}
710                  = not defined $self->{last_emitted_start_tag_name};
711            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
712          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
713            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 950  sub _get_next_token ($) { Line 721  sub _get_next_token ($) {
721          # reconsume          # reconsume
722    
723          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
724    
725          redo A;          redo A;
726        } else {        } else {
# Line 983  sub _get_next_token ($) { Line 753  sub _get_next_token ($) {
753          redo A;          redo A;
754        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
755          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
756              $self->{current_token}->{first_start_tag}
757                  = not defined $self->{last_emitted_start_tag_name};
758            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
759          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
760            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 996  sub _get_next_token ($) { Line 768  sub _get_next_token ($) {
768          !!!next-input-character;          !!!next-input-character;
769    
770          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
771    
772          redo A;          redo A;
773        } elsif ($self->{next_input_character} == 0x003C or # <        } elsif ($self->{next_input_character} == -1) {
                $self->{next_input_character} == -1) {  
774          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
775          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
776              $self->{current_token}->{first_start_tag}
777                  = not defined $self->{last_emitted_start_tag_name};
778            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
779          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
780            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 1016  sub _get_next_token ($) { Line 788  sub _get_next_token ($) {
788          ## reconsume          ## reconsume
789    
790          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
791    
792          redo A;          redo A;
793        } else {        } else {
# Line 1038  sub _get_next_token ($) { Line 809  sub _get_next_token ($) {
809        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
810          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
811          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
812              $self->{current_token}->{first_start_tag}
813                  = not defined $self->{last_emitted_start_tag_name};
814            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
815          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
816            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 1051  sub _get_next_token ($) { Line 824  sub _get_next_token ($) {
824          ## reconsume          ## reconsume
825    
826          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
827    
828          redo A;          redo A;
829        } else {        } else {
# Line 1073  sub _get_next_token ($) { Line 845  sub _get_next_token ($) {
845        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
846          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
847          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
848              $self->{current_token}->{first_start_tag}
849                  = not defined $self->{last_emitted_start_tag_name};
850            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
851          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
852            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 1086  sub _get_next_token ($) { Line 860  sub _get_next_token ($) {
860          ## reconsume          ## reconsume
861    
862          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
863    
864          redo A;          redo A;
865        } else {        } else {
# Line 1111  sub _get_next_token ($) { Line 884  sub _get_next_token ($) {
884          redo A;          redo A;
885        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
886          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
887              $self->{current_token}->{first_start_tag}
888                  = not defined $self->{last_emitted_start_tag_name};
889            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
890          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
891            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 1124  sub _get_next_token ($) { Line 899  sub _get_next_token ($) {
899          !!!next-input-character;          !!!next-input-character;
900    
901          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
902    
903          redo A;          redo A;
904        } elsif ($self->{next_input_character} == 0x003C or # <        } elsif ($self->{next_input_character} == -1) {
                $self->{next_input_character} == -1) {  
905          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
906          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
907              $self->{current_token}->{first_start_tag}
908                  = not defined $self->{last_emitted_start_tag_name};
909            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
910          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
911            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 1144  sub _get_next_token ($) { Line 919  sub _get_next_token ($) {
919          ## reconsume          ## reconsume
920    
921          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
922    
923          redo A;          redo A;
924        } else {        } else {
# Line 1154  sub _get_next_token ($) { Line 928  sub _get_next_token ($) {
928          redo A;          redo A;
929        }        }
930      } elsif ($self->{state} eq 'entity in attribute value') {      } elsif ($self->{state} eq 'entity in attribute value') {
931        my $token = $self->_tokenize_attempt_to_consume_an_entity;        my $token = $self->_tokenize_attempt_to_consume_an_entity (1);
932    
933        unless (defined $token) {        unless (defined $token) {
934          $self->{current_attribute}->{value} .= '&';          $self->{current_attribute}->{value} .= '&';
# Line 1203  sub _get_next_token ($) { Line 977  sub _get_next_token ($) {
977          push @next_char, $self->{next_input_character};          push @next_char, $self->{next_input_character};
978          if ($self->{next_input_character} == 0x002D) { # -          if ($self->{next_input_character} == 0x002D) { # -
979            $self->{current_token} = {type => 'comment', data => ''};            $self->{current_token} = {type => 'comment', data => ''};
980            $self->{state} = 'comment';            $self->{state} = 'comment start';
981            !!!next-input-character;            !!!next-input-character;
982            redo A;            redo A;
983          }          }
# Line 1245  sub _get_next_token ($) { Line 1019  sub _get_next_token ($) {
1019          }          }
1020        }        }
1021    
1022        !!!parse-error (type => 'bogus comment open');        !!!parse-error (type => 'bogus comment');
1023        $self->{next_input_character} = shift @next_char;        $self->{next_input_character} = shift @next_char;
1024        !!!back-next-input-character (@next_char);        !!!back-next-input-character (@next_char);
1025        $self->{state} = 'bogus comment';        $self->{state} = 'bogus comment';
# Line 1253  sub _get_next_token ($) { Line 1027  sub _get_next_token ($) {
1027                
1028        ## ISSUE: typos in spec: chacacters, is is a parse error        ## ISSUE: typos in spec: chacacters, is is a parse error
1029        ## ISSUE: spec is somewhat unclear on "is the first character that will be in the comment"; what is "that will be in the comment" is what the algorithm defines, isn't it?        ## ISSUE: spec is somewhat unclear on "is the first character that will be in the comment"; what is "that will be in the comment" is what the algorithm defines, isn't it?
1030        } elsif ($self->{state} eq 'comment start') {
1031          if ($self->{next_input_character} == 0x002D) { # -
1032            $self->{state} = 'comment start dash';
1033            !!!next-input-character;
1034            redo A;
1035          } elsif ($self->{next_input_character} == 0x003E) { # >
1036            !!!parse-error (type => 'bogus comment');
1037            $self->{state} = 'data';
1038            !!!next-input-character;
1039    
1040            !!!emit ($self->{current_token}); # comment
1041    
1042            redo A;
1043          } elsif ($self->{next_input_character} == -1) {
1044            !!!parse-error (type => 'unclosed comment');
1045            $self->{state} = 'data';
1046            ## reconsume
1047    
1048            !!!emit ($self->{current_token}); # comment
1049    
1050            redo A;
1051          } else {
1052            $self->{current_token}->{data} # comment
1053                .= chr ($self->{next_input_character});
1054            $self->{state} = 'comment';
1055            !!!next-input-character;
1056            redo A;
1057          }
1058        } elsif ($self->{state} eq 'comment start dash') {
1059          if ($self->{next_input_character} == 0x002D) { # -
1060            $self->{state} = 'comment end';
1061            !!!next-input-character;
1062            redo A;
1063          } elsif ($self->{next_input_character} == 0x003E) { # >
1064            !!!parse-error (type => 'bogus comment');
1065            $self->{state} = 'data';
1066            !!!next-input-character;
1067    
1068            !!!emit ($self->{current_token}); # comment
1069    
1070            redo A;
1071          } elsif ($self->{next_input_character} == -1) {
1072            !!!parse-error (type => 'unclosed comment');
1073            $self->{state} = 'data';
1074            ## reconsume
1075    
1076            !!!emit ($self->{current_token}); # comment
1077    
1078            redo A;
1079          } else {
1080            $self->{current_token}->{data} # comment
1081                .= chr ($self->{next_input_character});
1082            $self->{state} = 'comment';
1083            !!!next-input-character;
1084            redo A;
1085          }
1086      } elsif ($self->{state} eq 'comment') {      } elsif ($self->{state} eq 'comment') {
1087        if ($self->{next_input_character} == 0x002D) { # -        if ($self->{next_input_character} == 0x002D) { # -
1088          $self->{state} = 'comment dash';          $self->{state} = 'comment end dash';
1089          !!!next-input-character;          !!!next-input-character;
1090          redo A;          redo A;
1091        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
# Line 1264  sub _get_next_token ($) { Line 1094  sub _get_next_token ($) {
1094          ## reconsume          ## reconsume
1095    
1096          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{current_token}); # comment
         undef $self->{current_token};  
1097    
1098          redo A;          redo A;
1099        } else {        } else {
# Line 1273  sub _get_next_token ($) { Line 1102  sub _get_next_token ($) {
1102          !!!next-input-character;          !!!next-input-character;
1103          redo A;          redo A;
1104        }        }
1105      } elsif ($self->{state} eq 'comment dash') {      } elsif ($self->{state} eq 'comment end dash') {
1106        if ($self->{next_input_character} == 0x002D) { # -        if ($self->{next_input_character} == 0x002D) { # -
1107          $self->{state} = 'comment end';          $self->{state} = 'comment end';
1108          !!!next-input-character;          !!!next-input-character;
# Line 1284  sub _get_next_token ($) { Line 1113  sub _get_next_token ($) {
1113          ## reconsume          ## reconsume
1114    
1115          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{current_token}); # comment
         undef $self->{current_token};  
1116    
1117          redo A;          redo A;
1118        } else {        } else {
# Line 1299  sub _get_next_token ($) { Line 1127  sub _get_next_token ($) {
1127          !!!next-input-character;          !!!next-input-character;
1128    
1129          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{current_token}); # comment
         undef $self->{current_token};  
1130    
1131          redo A;          redo A;
1132        } elsif ($self->{next_input_character} == 0x002D) { # -        } elsif ($self->{next_input_character} == 0x002D) { # -
# Line 1314  sub _get_next_token ($) { Line 1141  sub _get_next_token ($) {
1141          ## reconsume          ## reconsume
1142    
1143          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{current_token}); # comment
         undef $self->{current_token};  
1144    
1145          redo A;          redo A;
1146        } else {        } else {
# Line 1348  sub _get_next_token ($) { Line 1174  sub _get_next_token ($) {
1174          ## Stay in the state          ## Stay in the state
1175          !!!next-input-character;          !!!next-input-character;
1176          redo A;          redo A;
       } elsif (0x0061 <= $self->{next_input_character} and  
                $self->{next_input_character} <= 0x007A) { # a..z  
 ## ISSUE: "Set the token's name name to the" in the spec  
         $self->{current_token} = {type => 'DOCTYPE',  
                           name => chr ($self->{next_input_character} - 0x0020),  
                           error => 1};  
         $self->{state} = 'DOCTYPE name';  
         !!!next-input-character;  
         redo A;  
1177        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
1178          !!!parse-error (type => 'no DOCTYPE name');          !!!parse-error (type => 'no DOCTYPE name');
1179          $self->{state} = 'data';          $self->{state} = 'data';
1180          !!!next-input-character;          !!!next-input-character;
1181    
1182          !!!emit ({type => 'DOCTYPE', name => '', error => 1});          !!!emit ({type => 'DOCTYPE'}); # incorrect
1183    
1184          redo A;          redo A;
1185        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
# Line 1370  sub _get_next_token ($) { Line 1187  sub _get_next_token ($) {
1187          $self->{state} = 'data';          $self->{state} = 'data';
1188          ## reconsume          ## reconsume
1189    
1190          !!!emit ({type => 'DOCTYPE', name => '', error => 1});          !!!emit ({type => 'DOCTYPE'}); # incorrect
1191    
1192          redo A;          redo A;
1193        } else {        } else {
1194          $self->{current_token} = {type => 'DOCTYPE',          $self->{current_token}
1195                            name => chr ($self->{next_input_character}),              = {type => 'DOCTYPE',
1196                            error => 1};                 name => chr ($self->{next_input_character}),
1197                   correct => 1};
1198  ## ISSUE: "Set the token's name name to the" in the spec  ## ISSUE: "Set the token's name name to the" in the spec
1199          $self->{state} = 'DOCTYPE name';          $self->{state} = 'DOCTYPE name';
1200          !!!next-input-character;          !!!next-input-character;
1201          redo A;          redo A;
1202        }        }
1203      } elsif ($self->{state} eq 'DOCTYPE name') {      } elsif ($self->{state} eq 'DOCTYPE name') {
1204    ## ISSUE: Redundant "First," in the spec.
1205        if ($self->{next_input_character} == 0x0009 or # HT        if ($self->{next_input_character} == 0x0009 or # HT
1206            $self->{next_input_character} == 0x000A or # LF            $self->{next_input_character} == 0x000A or # LF
1207            $self->{next_input_character} == 0x000B or # VT            $self->{next_input_character} == 0x000B or # VT
1208            $self->{next_input_character} == 0x000C or # FF            $self->{next_input_character} == 0x000C or # FF
1209            $self->{next_input_character} == 0x0020) { # SP            $self->{next_input_character} == 0x0020) { # SP
         $self->{current_token}->{error} = ($self->{current_token}->{name} ne 'HTML'); # DOCTYPE  
1210          $self->{state} = 'after DOCTYPE name';          $self->{state} = 'after DOCTYPE name';
1211          !!!next-input-character;          !!!next-input-character;
1212          redo A;          redo A;
1213        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
         $self->{current_token}->{error} = ($self->{current_token}->{name} ne 'HTML'); # DOCTYPE  
1214          $self->{state} = 'data';          $self->{state} = 'data';
1215          !!!next-input-character;          !!!next-input-character;
1216    
1217          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1218    
1219          redo A;          redo A;
       } elsif (0x0061 <= $self->{next_input_character} and  
                $self->{next_input_character} <= 0x007A) { # a..z  
         $self->{current_token}->{name} .= chr ($self->{next_input_character} - 0x0020); # DOCTYPE  
         #$self->{current_token}->{error} = ($self->{current_token}->{name} ne 'HTML');  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
1220        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
1221          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
         $self->{current_token}->{error} = ($self->{current_token}->{name} ne 'HTML'); # DOCTYPE  
1222          $self->{state} = 'data';          $self->{state} = 'data';
1223          ## reconsume          ## reconsume
1224    
1225          !!!emit ($self->{current_token});          delete $self->{current_token}->{correct};
1226          undef $self->{current_token};          !!!emit ($self->{current_token}); # DOCTYPE
1227    
1228          redo A;          redo A;
1229        } else {        } else {
1230          $self->{current_token}->{name}          $self->{current_token}->{name}
1231            .= chr ($self->{next_input_character}); # DOCTYPE            .= chr ($self->{next_input_character}); # DOCTYPE
         #$self->{current_token}->{error} = ($self->{current_token}->{name} ne 'HTML');  
1232          ## Stay in the state          ## Stay in the state
1233          !!!next-input-character;          !!!next-input-character;
1234          redo A;          redo A;
# Line 1440  sub _get_next_token ($) { Line 1247  sub _get_next_token ($) {
1247          !!!next-input-character;          !!!next-input-character;
1248    
1249          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1250    
1251          redo A;          redo A;
1252        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
# Line 1448  sub _get_next_token ($) { Line 1254  sub _get_next_token ($) {
1254          $self->{state} = 'data';          $self->{state} = 'data';
1255          ## reconsume          ## reconsume
1256    
1257            delete $self->{current_token}->{correct};
1258            !!!emit ($self->{current_token}); # DOCTYPE
1259    
1260            redo A;
1261          } elsif ($self->{next_input_character} == 0x0050 or # P
1262                   $self->{next_input_character} == 0x0070) { # p
1263            !!!next-input-character;
1264            if ($self->{next_input_character} == 0x0055 or # U
1265                $self->{next_input_character} == 0x0075) { # u
1266              !!!next-input-character;
1267              if ($self->{next_input_character} == 0x0042 or # B
1268                  $self->{next_input_character} == 0x0062) { # b
1269                !!!next-input-character;
1270                if ($self->{next_input_character} == 0x004C or # L
1271                    $self->{next_input_character} == 0x006C) { # l
1272                  !!!next-input-character;
1273                  if ($self->{next_input_character} == 0x0049 or # I
1274                      $self->{next_input_character} == 0x0069) { # i
1275                    !!!next-input-character;
1276                    if ($self->{next_input_character} == 0x0043 or # C
1277                        $self->{next_input_character} == 0x0063) { # c
1278                      $self->{state} = 'before DOCTYPE public identifier';
1279                      !!!next-input-character;
1280                      redo A;
1281                    }
1282                  }
1283                }
1284              }
1285            }
1286    
1287            #
1288          } elsif ($self->{next_input_character} == 0x0053 or # S
1289                   $self->{next_input_character} == 0x0073) { # s
1290            !!!next-input-character;
1291            if ($self->{next_input_character} == 0x0059 or # Y
1292                $self->{next_input_character} == 0x0079) { # y
1293              !!!next-input-character;
1294              if ($self->{next_input_character} == 0x0053 or # S
1295                  $self->{next_input_character} == 0x0073) { # s
1296                !!!next-input-character;
1297                if ($self->{next_input_character} == 0x0054 or # T
1298                    $self->{next_input_character} == 0x0074) { # t
1299                  !!!next-input-character;
1300                  if ($self->{next_input_character} == 0x0045 or # E
1301                      $self->{next_input_character} == 0x0065) { # e
1302                    !!!next-input-character;
1303                    if ($self->{next_input_character} == 0x004D or # M
1304                        $self->{next_input_character} == 0x006D) { # m
1305                      $self->{state} = 'before DOCTYPE system identifier';
1306                      !!!next-input-character;
1307                      redo A;
1308                    }
1309                  }
1310                }
1311              }
1312            }
1313    
1314            #
1315          } else {
1316            !!!next-input-character;
1317            #
1318          }
1319    
1320          !!!parse-error (type => 'string after DOCTYPE name');
1321          $self->{state} = 'bogus DOCTYPE';
1322          # next-input-character is already done
1323          redo A;
1324        } elsif ($self->{state} eq 'before DOCTYPE public identifier') {
1325          if ({
1326                0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
1327                #0x000D => 1, # HT, LF, VT, FF, SP, CR
1328              }->{$self->{next_input_character}}) {
1329            ## Stay in the state
1330            !!!next-input-character;
1331            redo A;
1332          } elsif ($self->{next_input_character} eq 0x0022) { # "
1333            $self->{current_token}->{public_identifier} = ''; # DOCTYPE
1334            $self->{state} = 'DOCTYPE public identifier (double-quoted)';
1335            !!!next-input-character;
1336            redo A;
1337          } elsif ($self->{next_input_character} eq 0x0027) { # '
1338            $self->{current_token}->{public_identifier} = ''; # DOCTYPE
1339            $self->{state} = 'DOCTYPE public identifier (single-quoted)';
1340            !!!next-input-character;
1341            redo A;
1342          } elsif ($self->{next_input_character} eq 0x003E) { # >
1343            !!!parse-error (type => 'no PUBLIC literal');
1344    
1345            $self->{state} = 'data';
1346            !!!next-input-character;
1347    
1348            delete $self->{current_token}->{correct};
1349            !!!emit ($self->{current_token}); # DOCTYPE
1350    
1351            redo A;
1352          } elsif ($self->{next_input_character} == -1) {
1353            !!!parse-error (type => 'unclosed DOCTYPE');
1354    
1355            $self->{state} = 'data';
1356            ## reconsume
1357    
1358            delete $self->{current_token}->{correct};
1359            !!!emit ($self->{current_token}); # DOCTYPE
1360    
1361            redo A;
1362          } else {
1363            !!!parse-error (type => 'string after PUBLIC');
1364            $self->{state} = 'bogus DOCTYPE';
1365            !!!next-input-character;
1366            redo A;
1367          }
1368        } elsif ($self->{state} eq 'DOCTYPE public identifier (double-quoted)') {
1369          if ($self->{next_input_character} == 0x0022) { # "
1370            $self->{state} = 'after DOCTYPE public identifier';
1371            !!!next-input-character;
1372            redo A;
1373          } elsif ($self->{next_input_character} == -1) {
1374            !!!parse-error (type => 'unclosed PUBLIC literal');
1375    
1376            $self->{state} = 'data';
1377            ## reconsume
1378    
1379            delete $self->{current_token}->{correct};
1380            !!!emit ($self->{current_token}); # DOCTYPE
1381    
1382            redo A;
1383          } else {
1384            $self->{current_token}->{public_identifier} # DOCTYPE
1385                .= chr $self->{next_input_character};
1386            ## Stay in the state
1387            !!!next-input-character;
1388            redo A;
1389          }
1390        } elsif ($self->{state} eq 'DOCTYPE public identifier (single-quoted)') {
1391          if ($self->{next_input_character} == 0x0027) { # '
1392            $self->{state} = 'after DOCTYPE public identifier';
1393            !!!next-input-character;
1394            redo A;
1395          } elsif ($self->{next_input_character} == -1) {
1396            !!!parse-error (type => 'unclosed PUBLIC literal');
1397    
1398            $self->{state} = 'data';
1399            ## reconsume
1400    
1401            delete $self->{current_token}->{correct};
1402            !!!emit ($self->{current_token}); # DOCTYPE
1403    
1404            redo A;
1405          } else {
1406            $self->{current_token}->{public_identifier} # DOCTYPE
1407                .= chr $self->{next_input_character};
1408            ## Stay in the state
1409            !!!next-input-character;
1410            redo A;
1411          }
1412        } elsif ($self->{state} eq 'after DOCTYPE public identifier') {
1413          if ({
1414                0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
1415                #0x000D => 1, # HT, LF, VT, FF, SP, CR
1416              }->{$self->{next_input_character}}) {
1417            ## Stay in the state
1418            !!!next-input-character;
1419            redo A;
1420          } elsif ($self->{next_input_character} == 0x0022) { # "
1421            $self->{current_token}->{system_identifier} = ''; # DOCTYPE
1422            $self->{state} = 'DOCTYPE system identifier (double-quoted)';
1423            !!!next-input-character;
1424            redo A;
1425          } elsif ($self->{next_input_character} == 0x0027) { # '
1426            $self->{current_token}->{system_identifier} = ''; # DOCTYPE
1427            $self->{state} = 'DOCTYPE system identifier (single-quoted)';
1428            !!!next-input-character;
1429            redo A;
1430          } elsif ($self->{next_input_character} == 0x003E) { # >
1431            $self->{state} = 'data';
1432            !!!next-input-character;
1433    
1434            !!!emit ($self->{current_token}); # DOCTYPE
1435    
1436            redo A;
1437          } elsif ($self->{next_input_character} == -1) {
1438            !!!parse-error (type => 'unclosed DOCTYPE');
1439    
1440            $self->{state} = 'data';
1441            ## reconsume
1442    
1443            delete $self->{current_token}->{correct};
1444            !!!emit ($self->{current_token}); # DOCTYPE
1445    
1446            redo A;
1447          } else {
1448            !!!parse-error (type => 'string after PUBLIC literal');
1449            $self->{state} = 'bogus DOCTYPE';
1450            !!!next-input-character;
1451            redo A;
1452          }
1453        } elsif ($self->{state} eq 'before DOCTYPE system identifier') {
1454          if ({
1455                0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
1456                #0x000D => 1, # HT, LF, VT, FF, SP, CR
1457              }->{$self->{next_input_character}}) {
1458            ## Stay in the state
1459            !!!next-input-character;
1460            redo A;
1461          } elsif ($self->{next_input_character} == 0x0022) { # "
1462            $self->{current_token}->{system_identifier} = ''; # DOCTYPE
1463            $self->{state} = 'DOCTYPE system identifier (double-quoted)';
1464            !!!next-input-character;
1465            redo A;
1466          } elsif ($self->{next_input_character} == 0x0027) { # '
1467            $self->{current_token}->{system_identifier} = ''; # DOCTYPE
1468            $self->{state} = 'DOCTYPE system identifier (single-quoted)';
1469            !!!next-input-character;
1470            redo A;
1471          } elsif ($self->{next_input_character} == 0x003E) { # >
1472            !!!parse-error (type => 'no SYSTEM literal');
1473            $self->{state} = 'data';
1474            !!!next-input-character;
1475    
1476            delete $self->{current_token}->{correct};
1477            !!!emit ($self->{current_token}); # DOCTYPE
1478    
1479            redo A;
1480          } elsif ($self->{next_input_character} == -1) {
1481            !!!parse-error (type => 'unclosed DOCTYPE');
1482    
1483            $self->{state} = 'data';
1484            ## reconsume
1485    
1486            delete $self->{current_token}->{correct};
1487          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1488    
1489          redo A;          redo A;
1490        } else {        } else {
1491          !!!parse-error (type => 'string after DOCTYPE name');          !!!parse-error (type => 'string after SYSTEM');
1492          $self->{current_token}->{error} = 1; # DOCTYPE          $self->{state} = 'bogus DOCTYPE';
1493            !!!next-input-character;
1494            redo A;
1495          }
1496        } elsif ($self->{state} eq 'DOCTYPE system identifier (double-quoted)') {
1497          if ($self->{next_input_character} == 0x0022) { # "
1498            $self->{state} = 'after DOCTYPE system identifier';
1499            !!!next-input-character;
1500            redo A;
1501          } elsif ($self->{next_input_character} == -1) {
1502            !!!parse-error (type => 'unclosed SYSTEM literal');
1503    
1504            $self->{state} = 'data';
1505            ## reconsume
1506    
1507            delete $self->{current_token}->{correct};
1508            !!!emit ($self->{current_token}); # DOCTYPE
1509    
1510            redo A;
1511          } else {
1512            $self->{current_token}->{system_identifier} # DOCTYPE
1513                .= chr $self->{next_input_character};
1514            ## Stay in the state
1515            !!!next-input-character;
1516            redo A;
1517          }
1518        } elsif ($self->{state} eq 'DOCTYPE system identifier (single-quoted)') {
1519          if ($self->{next_input_character} == 0x0027) { # '
1520            $self->{state} = 'after DOCTYPE system identifier';
1521            !!!next-input-character;
1522            redo A;
1523          } elsif ($self->{next_input_character} == -1) {
1524            !!!parse-error (type => 'unclosed SYSTEM literal');
1525    
1526            $self->{state} = 'data';
1527            ## reconsume
1528    
1529            delete $self->{current_token}->{correct};
1530            !!!emit ($self->{current_token}); # DOCTYPE
1531    
1532            redo A;
1533          } else {
1534            $self->{current_token}->{system_identifier} # DOCTYPE
1535                .= chr $self->{next_input_character};
1536            ## Stay in the state
1537            !!!next-input-character;
1538            redo A;
1539          }
1540        } elsif ($self->{state} eq 'after DOCTYPE system identifier') {
1541          if ({
1542                0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
1543                #0x000D => 1, # HT, LF, VT, FF, SP, CR
1544              }->{$self->{next_input_character}}) {
1545            ## Stay in the state
1546            !!!next-input-character;
1547            redo A;
1548          } elsif ($self->{next_input_character} == 0x003E) { # >
1549            $self->{state} = 'data';
1550            !!!next-input-character;
1551    
1552            !!!emit ($self->{current_token}); # DOCTYPE
1553    
1554            redo A;
1555          } elsif ($self->{next_input_character} == -1) {
1556            !!!parse-error (type => 'unclosed DOCTYPE');
1557    
1558            $self->{state} = 'data';
1559            ## reconsume
1560    
1561            delete $self->{current_token}->{correct};
1562            !!!emit ($self->{current_token}); # DOCTYPE
1563    
1564            redo A;
1565          } else {
1566            !!!parse-error (type => 'string after SYSTEM literal');
1567          $self->{state} = 'bogus DOCTYPE';          $self->{state} = 'bogus DOCTYPE';
1568          !!!next-input-character;          !!!next-input-character;
1569          redo A;          redo A;
# Line 1464  sub _get_next_token ($) { Line 1573  sub _get_next_token ($) {
1573          $self->{state} = 'data';          $self->{state} = 'data';
1574          !!!next-input-character;          !!!next-input-character;
1575    
1576            delete $self->{current_token}->{correct};
1577          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1578    
1579          redo A;          redo A;
1580        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
# Line 1473  sub _get_next_token ($) { Line 1582  sub _get_next_token ($) {
1582          $self->{state} = 'data';          $self->{state} = 'data';
1583          ## reconsume          ## reconsume
1584    
1585            delete $self->{current_token}->{correct};
1586          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1587    
1588          redo A;          redo A;
1589        } else {        } else {
# Line 1490  sub _get_next_token ($) { Line 1599  sub _get_next_token ($) {
1599    die "$0: _get_next_token: unexpected case";    die "$0: _get_next_token: unexpected case";
1600  } # _get_next_token  } # _get_next_token
1601    
1602  sub _tokenize_attempt_to_consume_an_entity ($) {  sub _tokenize_attempt_to_consume_an_entity ($$) {
1603    my $self = shift;    my ($self, $in_attr) = @_;
1604      
1605    if ($self->{next_input_character} == 0x0023) { # #    if ({
1606           0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, # HT, LF, VT, FF,
1607           0x0020 => 1, 0x003C => 1, 0x0026 => 1, -1 => 1, # SP, <, & # 0x000D # CR
1608          }->{$self->{next_input_character}}) {
1609        ## Don't consume
1610        ## No error
1611        return undef;
1612      } elsif ($self->{next_input_character} == 0x0023) { # #
1613      !!!next-input-character;      !!!next-input-character;
1614      if ($self->{next_input_character} == 0x0078 or # x      if ($self->{next_input_character} == 0x0078 or # x
1615          $self->{next_input_character} == 0x0058) { # X          $self->{next_input_character} == 0x0058) { # X
1616        my $num;        my $code;
1617        X: {        X: {
1618          my $x_char = $self->{next_input_character};          my $x_char = $self->{next_input_character};
1619          !!!next-input-character;          !!!next-input-character;
1620          if (0x0030 <= $self->{next_input_character} and          if (0x0030 <= $self->{next_input_character} and
1621              $self->{next_input_character} <= 0x0039) { # 0..9              $self->{next_input_character} <= 0x0039) { # 0..9
1622            $num ||= 0;            $code ||= 0;
1623            $num *= 0x10;            $code *= 0x10;
1624            $num += $self->{next_input_character} - 0x0030;            $code += $self->{next_input_character} - 0x0030;
1625            redo X;            redo X;
1626          } elsif (0x0061 <= $self->{next_input_character} and          } elsif (0x0061 <= $self->{next_input_character} and
1627                   $self->{next_input_character} <= 0x0066) { # a..f                   $self->{next_input_character} <= 0x0066) { # a..f
1628            ## ISSUE: the spec says U+0078, which is apparently incorrect            $code ||= 0;
1629            $num ||= 0;            $code *= 0x10;
1630            $num *= 0x10;            $code += $self->{next_input_character} - 0x0060 + 9;
           $num += $self->{next_input_character} - 0x0060 + 9;  
1631            redo X;            redo X;
1632          } elsif (0x0041 <= $self->{next_input_character} and          } elsif (0x0041 <= $self->{next_input_character} and
1633                   $self->{next_input_character} <= 0x0046) { # A..F                   $self->{next_input_character} <= 0x0046) { # A..F
1634            ## ISSUE: the spec says U+0058, which is apparently incorrect            $code ||= 0;
1635            $num ||= 0;            $code *= 0x10;
1636            $num *= 0x10;            $code += $self->{next_input_character} - 0x0040 + 9;
           $num += $self->{next_input_character} - 0x0040 + 9;  
1637            redo X;            redo X;
1638          } elsif (not defined $num) { # no hexadecimal digit          } elsif (not defined $code) { # no hexadecimal digit
1639            !!!parse-error (type => 'bare hcro');            !!!parse-error (type => 'bare hcro');
1640            $self->{next_input_character} = 0x0023; # #            $self->{next_input_character} = 0x0023; # #
1641            !!!back-next-input-character ($x_char);            !!!back-next-input-character ($x_char);
# Line 1532  sub _tokenize_attempt_to_consume_an_enti Line 1646  sub _tokenize_attempt_to_consume_an_enti
1646            !!!parse-error (type => 'no refc');            !!!parse-error (type => 'no refc');
1647          }          }
1648    
1649          ## TODO: check the definition for |a valid Unicode character|.          if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {
1650          ## <http://lists.whatwg.org/pipermail/whatwg-whatwg.org/2006-December/thread.html#8189>            !!!parse-error (type => sprintf 'invalid character reference:U+%04X', $code);
1651          if ($num > 1114111 or $num == 0) {            $code = 0xFFFD;
1652            $num = 0xFFFD; # REPLACEMENT CHARACTER          } elsif ($code > 0x10FFFF) {
1653            ## ISSUE: Why this is not an error?            !!!parse-error (type => sprintf 'invalid character reference:U-%08X', $code);
1654          } elsif (0x80 <= $num and $num <= 0x9F) {            $code = 0xFFFD;
1655            !!!parse-error (type => sprintf 'c1 entity:U+%04X', $num);          } elsif ($code == 0x000D) {
1656            $num = $c1_entity_char->{$num};            !!!parse-error (type => 'CR character reference');
1657              $code = 0x000A;
1658            } elsif (0x80 <= $code and $code <= 0x9F) {
1659              !!!parse-error (type => sprintf 'C1 character reference:U+%04X', $code);
1660              $code = $c1_entity_char->{$code};
1661          }          }
1662    
1663          return {type => 'character', data => chr $num};          return {type => 'character', data => chr $code};
1664        } # X        } # X
1665      } elsif (0x0030 <= $self->{next_input_character} and      } elsif (0x0030 <= $self->{next_input_character} and
1666               $self->{next_input_character} <= 0x0039) { # 0..9               $self->{next_input_character} <= 0x0039) { # 0..9
# Line 1563  sub _tokenize_attempt_to_consume_an_enti Line 1681  sub _tokenize_attempt_to_consume_an_enti
1681          !!!parse-error (type => 'no refc');          !!!parse-error (type => 'no refc');
1682        }        }
1683    
1684        ## TODO: check the definition for |a valid Unicode character|.        if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {
1685        if ($code > 1114111 or $code == 0) {          !!!parse-error (type => sprintf 'invalid character reference:U+%04X', $code);
1686          $code = 0xFFFD; # REPLACEMENT CHARACTER          $code = 0xFFFD;
1687          ## ISSUE: Why this is not an error?        } elsif ($code > 0x10FFFF) {
1688            !!!parse-error (type => sprintf 'invalid character reference:U-%08X', $code);
1689            $code = 0xFFFD;
1690          } elsif ($code == 0x000D) {
1691            !!!parse-error (type => 'CR character reference');
1692            $code = 0x000A;
1693        } elsif (0x80 <= $code and $code <= 0x9F) {        } elsif (0x80 <= $code and $code <= 0x9F) {
1694          !!!parse-error (type => sprintf 'c1 entity:U+%04X', $code);          !!!parse-error (type => sprintf 'C1 character reference:U+%04X', $code);
1695          $code = $c1_entity_char->{$code};          $code = $c1_entity_char->{$code};
1696        }        }
1697                
# Line 1588  sub _tokenize_attempt_to_consume_an_enti Line 1711  sub _tokenize_attempt_to_consume_an_enti
1711    
1712      my $value = $entity_name;      my $value = $entity_name;
1713      my $match;      my $match;
1714        require Whatpm::_NamedEntityList;
1715        our $EntityChar;
1716    
1717      while (length $entity_name < 10 and      while (length $entity_name < 10 and
1718             ## NOTE: Some number greater than the maximum length of entity name             ## NOTE: Some number greater than the maximum length of entity name
1719             ((0x0041 <= $self->{next_input_character} and             ((0x0041 <= $self->{next_input_character} and # a
1720               $self->{next_input_character} <= 0x005A) or               $self->{next_input_character} <= 0x005A) or # x
1721              (0x0061 <= $self->{next_input_character} and              (0x0061 <= $self->{next_input_character} and # a
1722               $self->{next_input_character} <= 0x007A) or               $self->{next_input_character} <= 0x007A) or # z
1723              (0x0030 <= $self->{next_input_character} and              (0x0030 <= $self->{next_input_character} and # 0
1724               $self->{next_input_character} <= 0x0039))) {               $self->{next_input_character} <= 0x0039) or # 9
1725                $self->{next_input_character} == 0x003B)) { # ;
1726        $entity_name .= chr $self->{next_input_character};        $entity_name .= chr $self->{next_input_character};
1727        if (defined $entity_char->{$entity_name}) {        if (defined $EntityChar->{$entity_name}) {
1728          $value = $entity_char->{$entity_name};          if ($self->{next_input_character} == 0x003B) { # ;
1729          $match = 1;            $value = $EntityChar->{$entity_name};
1730              $match = 1;
1731              !!!next-input-character;
1732              last;
1733            } elsif (not $in_attr) {
1734              $value = $EntityChar->{$entity_name};
1735              $match = -1;
1736            } else {
1737              $value .= chr $self->{next_input_character};
1738            }
1739        } else {        } else {
1740          $value .= chr $self->{next_input_character};          $value .= chr $self->{next_input_character};
1741        }        }
1742        !!!next-input-character;        !!!next-input-character;
1743      }      }
1744            
1745      if ($match) {      if ($match > 0) {
1746        if ($self->{next_input_character} == 0x003B) { # ;        return {type => 'character', data => $value};
1747          !!!next-input-character;      } elsif ($match < 0) {
1748        } else {        !!!parse-error (type => 'no refc');
         !!!parse-error (type => 'refc');  
       }  
   
1749        return {type => 'character', data => $value};        return {type => 'character', data => $value};
1750      } else {      } else {
1751        !!!parse-error (type => 'bare ero');        !!!parse-error (type => 'bare ero');
1752        ## NOTE: No characters are consumed in the spec.        ## NOTE: No characters are consumed in the spec.
1753        !!!back-token ({type => 'character', data => $value});        return {type => 'character', data => '&'.$value};
       return undef;  
1754      }      }
1755    } else {    } else {
1756      ## no characters are consumed      ## no characters are consumed
# Line 1634  sub _initialize_tree_constructor ($) { Line 1765  sub _initialize_tree_constructor ($) {
1765    $self->{document}->strict_error_checking (0);    $self->{document}->strict_error_checking (0);
1766    ## TODO: Turn mutation events off # MUST    ## TODO: Turn mutation events off # MUST
1767    ## TODO: Turn loose Document option (manakai extension) on    ## TODO: Turn loose Document option (manakai extension) on
1768    ## TODO: Mark the Document as an HTML document # MUST    $self->{document}->manakai_is_html (1); # MUST
1769  } # _initialize_tree_constructor  } # _initialize_tree_constructor
1770    
1771  sub _terminate_tree_constructor ($) {  sub _terminate_tree_constructor ($) {
# Line 1674  sub _construct_tree ($) { Line 1805  sub _construct_tree ($) {
1805    
1806  sub _tree_construction_initial ($) {  sub _tree_construction_initial ($) {
1807    my $self = shift;    my $self = shift;
1808    B: {    INITIAL: {
1809        if ($token->{type} eq 'DOCTYPE') {      if ($token->{type} eq 'DOCTYPE') {
1810          if ($token->{error}) {        ## NOTE: Conformance checkers MAY, instead of reporting "not HTML5"
1811            ## ISSUE: Spec currently left this case undefined.        ## error, switch to a conformance checking mode for another
1812            !!!parse-error (type => 'bogus DOCTYPE');        ## language.
1813          }        my $doctype_name = $token->{name};
1814          my $doctype = $self->{document}->create_document_type_definition        $doctype_name = '' unless defined $doctype_name;
1815            ($token->{name});        $doctype_name =~ tr/a-z/A-Z/;
1816          $self->{document}->append_child ($doctype);        if (not defined $token->{name} or # <!DOCTYPE>
1817          #$phase = 'root element';            defined $token->{public_identifier} or
1818          !!!next-token;            defined $token->{system_identifier}) {
1819          #redo B;          !!!parse-error (type => 'not HTML5');
1820          return;        } elsif ($doctype_name ne 'HTML') {
1821        } elsif ({          ## ISSUE: ASCII case-insensitive? (in fact it does not matter)
1822                  comment => 1,          !!!parse-error (type => 'not HTML5');
1823                  'start tag' => 1,        }
1824                  'end tag' => 1,        
1825                  'end-of-file' => 1,        my $doctype = $self->{document}->create_document_type_definition
1826                 }->{$token->{type}}) {          ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?
1827          ## ISSUE: Spec currently left this case undefined.        $doctype->public_id ($token->{public_identifier})
1828          !!!parse-error (type => 'missing DOCTYPE');            if defined $token->{public_identifier};
1829          #$phase = 'root element';        $doctype->system_id ($token->{system_identifier})
1830          ## reprocess            if defined $token->{system_identifier};
1831          #redo B;        ## NOTE: Other DocumentType attributes are null or empty lists.
1832          return;        ## ISSUE: internalSubset = null??
1833        } elsif ($token->{type} eq 'character') {        $self->{document}->append_child ($doctype);
1834          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {        
1835            $self->{document}->manakai_append_text ($1);        if (not $token->{correct} or $doctype_name ne 'HTML') {
1836            ## ISSUE: DOM3 Core does not allow Document > Text          $self->{document}->manakai_compat_mode ('quirks');
1837            unless (length $token->{data}) {        } elsif (defined $token->{public_identifier}) {
1838              ## Stay in the phase          my $pubid = $token->{public_identifier};
1839              !!!next-token;          $pubid =~ tr/a-z/A-z/;
1840              redo B;          if ({
1841              "+//SILMARIL//DTD HTML PRO V0R11 19970101//EN" => 1,
1842              "-//ADVASOFT LTD//DTD HTML 3.0 ASWEDIT + EXTENSIONS//EN" => 1,
1843              "-//AS//DTD HTML 3.0 ASWEDIT + EXTENSIONS//EN" => 1,
1844              "-//IETF//DTD HTML 2.0 LEVEL 1//EN" => 1,
1845              "-//IETF//DTD HTML 2.0 LEVEL 2//EN" => 1,
1846              "-//IETF//DTD HTML 2.0 STRICT LEVEL 1//EN" => 1,
1847              "-//IETF//DTD HTML 2.0 STRICT LEVEL 2//EN" => 1,
1848              "-//IETF//DTD HTML 2.0 STRICT//EN" => 1,
1849              "-//IETF//DTD HTML 2.0//EN" => 1,
1850              "-//IETF//DTD HTML 2.1E//EN" => 1,
1851              "-//IETF//DTD HTML 3.0//EN" => 1,
1852              "-//IETF//DTD HTML 3.0//EN//" => 1,
1853              "-//IETF//DTD HTML 3.2 FINAL//EN" => 1,
1854              "-//IETF//DTD HTML 3.2//EN" => 1,
1855              "-//IETF//DTD HTML 3//EN" => 1,
1856              "-//IETF//DTD HTML LEVEL 0//EN" => 1,
1857              "-//IETF//DTD HTML LEVEL 0//EN//2.0" => 1,
1858              "-//IETF//DTD HTML LEVEL 1//EN" => 1,
1859              "-//IETF//DTD HTML LEVEL 1//EN//2.0" => 1,
1860              "-//IETF//DTD HTML LEVEL 2//EN" => 1,
1861              "-//IETF//DTD HTML LEVEL 2//EN//2.0" => 1,
1862              "-//IETF//DTD HTML LEVEL 3//EN" => 1,
1863              "-//IETF//DTD HTML LEVEL 3//EN//3.0" => 1,
1864              "-//IETF//DTD HTML STRICT LEVEL 0//EN" => 1,
1865              "-//IETF//DTD HTML STRICT LEVEL 0//EN//2.0" => 1,
1866              "-//IETF//DTD HTML STRICT LEVEL 1//EN" => 1,
1867              "-//IETF//DTD HTML STRICT LEVEL 1//EN//2.0" => 1,
1868              "-//IETF//DTD HTML STRICT LEVEL 2//EN" => 1,
1869              "-//IETF//DTD HTML STRICT LEVEL 2//EN//2.0" => 1,
1870              "-//IETF//DTD HTML STRICT LEVEL 3//EN" => 1,
1871              "-//IETF//DTD HTML STRICT LEVEL 3//EN//3.0" => 1,
1872              "-//IETF//DTD HTML STRICT//EN" => 1,
1873              "-//IETF//DTD HTML STRICT//EN//2.0" => 1,
1874              "-//IETF//DTD HTML STRICT//EN//3.0" => 1,
1875              "-//IETF//DTD HTML//EN" => 1,
1876              "-//IETF//DTD HTML//EN//2.0" => 1,
1877              "-//IETF//DTD HTML//EN//3.0" => 1,
1878              "-//METRIUS//DTD METRIUS PRESENTATIONAL//EN" => 1,
1879              "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML STRICT//EN" => 1,
1880              "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML//EN" => 1,
1881              "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 TABLES//EN" => 1,
1882              "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML STRICT//EN" => 1,
1883              "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML//EN" => 1,
1884              "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 TABLES//EN" => 1,
1885              "-//NETSCAPE COMM. CORP.//DTD HTML//EN" => 1,
1886              "-//NETSCAPE COMM. CORP.//DTD STRICT HTML//EN" => 1,
1887              "-//O'REILLY AND ASSOCIATES//DTD HTML 2.0//EN" => 1,
1888              "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED 1.0//EN" => 1,
1889              "-//SPYGLASS//DTD HTML 2.0 EXTENDED//EN" => 1,
1890              "-//SQ//DTD HTML 2.0 HOTMETAL + EXTENSIONS//EN" => 1,
1891              "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA HTML//EN" => 1,
1892              "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA STRICT HTML//EN" => 1,
1893              "-//W3C//DTD HTML 3 1995-03-24//EN" => 1,
1894              "-//W3C//DTD HTML 3.2 DRAFT//EN" => 1,
1895              "-//W3C//DTD HTML 3.2 FINAL//EN" => 1,
1896              "-//W3C//DTD HTML 3.2//EN" => 1,
1897              "-//W3C//DTD HTML 3.2S DRAFT//EN" => 1,
1898              "-//W3C//DTD HTML 4.0 FRAMESET//EN" => 1,
1899              "-//W3C//DTD HTML 4.0 TRANSITIONAL//EN" => 1,
1900              "-//W3C//DTD HTML EXPERIMETNAL 19960712//EN" => 1,
1901              "-//W3C//DTD HTML EXPERIMENTAL 970421//EN" => 1,
1902              "-//W3C//DTD W3 HTML//EN" => 1,
1903              "-//W3O//DTD W3 HTML 3.0//EN" => 1,
1904              "-//W3O//DTD W3 HTML 3.0//EN//" => 1,
1905              "-//W3O//DTD W3 HTML STRICT 3.0//EN//" => 1,
1906              "-//WEBTECHS//DTD MOZILLA HTML 2.0//EN" => 1,
1907              "-//WEBTECHS//DTD MOZILLA HTML//EN" => 1,
1908              "-/W3C/DTD HTML 4.0 TRANSITIONAL/EN" => 1,
1909              "HTML" => 1,
1910            }->{$pubid}) {
1911              $self->{document}->manakai_compat_mode ('quirks');
1912            } elsif ($pubid eq "-//W3C//DTD HTML 4.01 FRAMESET//EN" or
1913                     $pubid eq "-//W3C//DTD HTML 4.01 TRANSITIONAL//EN") {
1914              if (defined $token->{system_identifier}) {
1915                $self->{document}->manakai_compat_mode ('quirks');
1916              } else {
1917                $self->{document}->manakai_compat_mode ('limited quirks');
1918            }            }
1919            } elsif ($pubid eq "-//W3C//DTD XHTML 1.0 Frameset//EN" or
1920                     $pubid eq "-//W3C//DTD XHTML 1.0 Transitional//EN") {
1921              $self->{document}->manakai_compat_mode ('limited quirks');
1922            }
1923          }
1924          if (defined $token->{system_identifier}) {
1925            my $sysid = $token->{system_identifier};
1926            $sysid =~ tr/A-Z/a-z/;
1927            if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {
1928              $self->{document}->manakai_compat_mode ('quirks');
1929          }          }
         ## ISSUE: Spec currently left this case undefined.  
         !!!parse-error (type => 'missing DOCTYPE');  
         #$phase = 'root element';  
         ## reprocess  
         #redo B;  
         return;  
       } else {  
         die "$0: $token->{type}: Unknown token";  
1930        }        }
1931      } # B        
1932          ## Go to the root element phase.
1933          !!!next-token;
1934          return;
1935        } elsif ({
1936                  'start tag' => 1,
1937                  'end tag' => 1,
1938                  'end-of-file' => 1,
1939                 }->{$token->{type}}) {
1940          !!!parse-error (type => 'no DOCTYPE');
1941          $self->{document}->manakai_compat_mode ('quirks');
1942          ## Go to the root element phase
1943          ## reprocess
1944          return;
1945        } elsif ($token->{type} eq 'character') {
1946          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D
1947            ## Ignore the token
1948    
1949            unless (length $token->{data}) {
1950              ## Stay in the phase
1951              !!!next-token;
1952              redo INITIAL;
1953            }
1954          }
1955    
1956          !!!parse-error (type => 'no DOCTYPE');
1957          $self->{document}->manakai_compat_mode ('quirks');
1958          ## Go to the root element phase
1959          ## reprocess
1960          return;
1961        } elsif ($token->{type} eq 'comment') {
1962          my $comment = $self->{document}->create_comment ($token->{data});
1963          $self->{document}->append_child ($comment);
1964          
1965          ## Stay in the phase.
1966          !!!next-token;
1967          redo INITIAL;
1968        } else {
1969          die "$0: $token->{type}: Unknown token";
1970        }
1971      } # INITIAL
1972  } # _tree_construction_initial  } # _tree_construction_initial
1973    
1974  sub _tree_construction_root_element ($) {  sub _tree_construction_root_element ($) {
# Line 1738  sub _tree_construction_root_element ($) Line 1988  sub _tree_construction_root_element ($)
1988          !!!next-token;          !!!next-token;
1989          redo B;          redo B;
1990        } elsif ($token->{type} eq 'character') {        } elsif ($token->{type} eq 'character') {
1991          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D
1992            $self->{document}->manakai_append_text ($1);            ## Ignore the token.
1993            ## ISSUE: DOM3 Core does not allow Document > Text  
1994            unless (length $token->{data}) {            unless (length $token->{data}) {
1995              ## Stay in the phase              ## Stay in the phase
1996              !!!next-token;              !!!next-token;
# Line 1780  sub _reset_insertion_mode ($) { Line 2030  sub _reset_insertion_mode ($) {
2030            
2031      ## Step 3      ## Step 3
2032      S3: {      S3: {
2033        $last = 1 if $self->{open_elements}->[0]->[0] eq $node->[0];        ## ISSUE: Oops! "If node is the first node in the stack of open
2034        if (defined $self->{inner_html_node}) {        ## elements, then set last to true. If the context element of the
2035          if ($self->{inner_html_node}->[1] eq 'td' or        ## HTML fragment parsing algorithm is neither a td element nor a
2036              $self->{inner_html_node}->[1] eq 'th') {        ## th element, then set node to the context element. (fragment case)":
2037            #        ## The second "if" is in the scope of the first "if"!?
2038          } else {        if ($self->{open_elements}->[0]->[0] eq $node->[0]) {
2039            $node = $self->{inner_html_node};          $last = 1;
2040            if (defined $self->{inner_html_node}) {
2041              if ($self->{inner_html_node}->[1] eq 'td' or
2042                  $self->{inner_html_node}->[1] eq 'th') {
2043                #
2044              } else {
2045                $node = $self->{inner_html_node};
2046              }
2047          }          }
2048        }        }
2049            
# Line 1917  sub _tree_construction_main ($) { Line 2174  sub _tree_construction_main ($) {
2174      }      }
2175    }; # $clear_up_to_marker    }; # $clear_up_to_marker
2176    
2177    my $style_start_tag = sub {    my $parse_rcdata = sub ($$) {
2178      my $style_el; !!!create-element ($style_el, 'style', $token->{attributes});      my ($content_model_flag, $insert) = @_;
2179      ## $self->{insertion_mode} eq 'in head' and ... (always true)  
2180      (($self->{insertion_mode} eq 'in head' and defined $self->{head_element})      ## Step 1
2181       ? $self->{head_element} : $self->{open_elements}->[-1]->[0])      my $start_tag_name = $token->{tag_name};
2182        ->append_child ($style_el);      my $el;
2183      $self->{content_model_flag} = 'CDATA';      !!!create-element ($el, $start_tag_name, $token->{attributes});
2184                  
2185        ## Step 2
2186        $insert->($el); # /context node/->append_child ($el)
2187    
2188        ## Step 3
2189        $self->{content_model_flag} = $content_model_flag; # CDATA or RCDATA
2190        delete $self->{escape}; # MUST
2191    
2192        ## Step 4
2193      my $text = '';      my $text = '';
2194      !!!next-token;      !!!next-token;
2195      while ($token->{type} eq 'character') {      while ($token->{type} eq 'character') { # or until stop tokenizing
2196        $text .= $token->{data};        $text .= $token->{data};
2197        !!!next-token;        !!!next-token;
2198      } # stop if non-character token or tokenizer stops tokenising      }
2199    
2200        ## Step 5
2201      if (length $text) {      if (length $text) {
2202        $style_el->manakai_append_text ($text);        my $text = $self->{document}->create_text_node ($text);
2203          $el->append_child ($text);
2204      }      }
2205        
2206        ## Step 6
2207      $self->{content_model_flag} = 'PCDATA';      $self->{content_model_flag} = 'PCDATA';
2208                  
2209      if ($token->{type} eq 'end tag' and $token->{tag_name} eq 'style') {      ## Step 7
2210        if ($token->{type} eq 'end tag' and $token->{tag_name} eq $start_tag_name) {
2211        ## Ignore the token        ## Ignore the token
2212      } else {      } else {
2213        !!!parse-error (type => 'in CDATA:#'.$token->{type});        !!!parse-error (type => 'in '.$content_model_flag.':#'.$token->{type});
       ## ISSUE: And ignore?  
2214      }      }
2215      !!!next-token;      !!!next-token;
2216    }; # $style_start_tag    }; # $parse_rcdata
2217    
2218    my $script_start_tag = sub {    my $script_start_tag = sub ($) {
2219        my $insert = $_[0];
2220      my $script_el;      my $script_el;
2221      !!!create-element ($script_el, 'script', $token->{attributes});      !!!create-element ($script_el, 'script', $token->{attributes});
2222      ## TODO: mark as "parser-inserted"      ## TODO: mark as "parser-inserted"
2223    
2224      $self->{content_model_flag} = 'CDATA';      $self->{content_model_flag} = 'CDATA';
2225        delete $self->{escape}; # MUST
2226            
2227      my $text = '';      my $text = '';
2228      !!!next-token;      !!!next-token;
# Line 1979  sub _tree_construction_main ($) { Line 2250  sub _tree_construction_main ($) {
2250      } else {      } else {
2251        ## TODO: $old_insertion_point = current insertion point        ## TODO: $old_insertion_point = current insertion point
2252        ## TODO: insertion point = just before the next input character        ## TODO: insertion point = just before the next input character
2253          
2254        (($self->{insertion_mode} eq 'in head' and defined $self->{head_element})        $insert->($script_el);
        ? $self->{head_element} : $self->{open_elements}->[-1]->[0])->append_child ($script_el);  
2255                
2256        ## TODO: insertion point = $old_insertion_point (might be "undefined")        ## TODO: insertion point = $old_insertion_point (might be "undefined")
2257                
# Line 2175  sub _tree_construction_main ($) { Line 2445  sub _tree_construction_main ($) {
2445    }; # $formatting_end_tag    }; # $formatting_end_tag
2446    
2447    my $insert_to_current = sub {    my $insert_to_current = sub {
2448      $self->{open_elements}->[-1]->[0]->append_child (shift);      $self->{open_elements}->[-1]->[0]->append_child ($_[0]);
2449    }; # $insert_to_current    }; # $insert_to_current
2450    
2451    my $insert_to_foster = sub {    my $insert_to_foster = sub {
# Line 2213  sub _tree_construction_main ($) { Line 2483  sub _tree_construction_main ($) {
2483      my $insert = shift;      my $insert = shift;
2484      if ($token->{type} eq 'start tag') {      if ($token->{type} eq 'start tag') {
2485        if ($token->{tag_name} eq 'script') {        if ($token->{tag_name} eq 'script') {
2486          $script_start_tag->();          ## NOTE: This is an "as if in head" code clone
2487            $script_start_tag->($insert);
2488          return;          return;
2489        } elsif ($token->{tag_name} eq 'style') {        } elsif ($token->{tag_name} eq 'style') {
2490          $style_start_tag->();          ## NOTE: This is an "as if in head" code clone
2491            $parse_rcdata->('CDATA', $insert);
2492          return;          return;
2493        } elsif ({        } elsif ({
2494                  base => 1, link => 1, meta => 1,                  base => 1, link => 1, meta => 1,
2495                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
2496          !!!parse-error (type => 'in body:'.$token->{tag_name});          ## NOTE: This is an "as if in head" code clone, only "-t" differs
2497          ## NOTE: This is an "as if in head" code clone          !!!insert-element-t ($token->{tag_name}, $token->{attributes});
2498          my $el;          pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
         !!!create-element ($el, $token->{tag_name}, $token->{attributes});  
         if (defined $self->{head_element}) {  
           $self->{head_element}->append_child ($el);  
         } else {  
           $insert->($el);  
         }  
           
2499          !!!next-token;          !!!next-token;
2500            ## TODO: Extracting |charset| from |meta|.
2501          return;          return;
2502        } elsif ($token->{tag_name} eq 'title') {        } elsif ($token->{tag_name} eq 'title') {
2503          !!!parse-error (type => 'in body:title');          !!!parse-error (type => 'in body:title');
2504          ## NOTE: There is an "as if in head" code clone          ## NOTE: This is an "as if in head" code clone
2505          my $title_el;          $parse_rcdata->('RCDATA', $insert);
         !!!create-element ($title_el, 'title', $token->{attributes});  
         (defined $self->{head_element} ? $self->{head_element} : $self->{open_elements}->[-1]->[0])  
           ->append_child ($title_el);  
         $self->{content_model_flag} = 'RCDATA';  
           
         my $text = '';  
         !!!next-token;  
         while ($token->{type} eq 'character') {  
           $text .= $token->{data};  
           !!!next-token;  
         }  
         if (length $text) {  
           $title_el->manakai_append_text ($text);  
         }  
           
         $self->{content_model_flag} = 'PCDATA';  
           
         if ($token->{type} eq 'end tag' and  
             $token->{tag_name} eq 'title') {  
           ## Ignore the token  
         } else {  
           !!!parse-error (type => 'in RCDATA:#'.$token->{type});  
           ## ISSUE: And ignore?  
         }  
         !!!next-token;  
2506          return;          return;
2507        } elsif ($token->{tag_name} eq 'body') {        } elsif ($token->{tag_name} eq 'body') {
2508          !!!parse-error (type => 'in body:body');          !!!parse-error (type => 'in body:body');
# Line 2364  sub _tree_construction_main ($) { Line 2605  sub _tree_construction_main ($) {
2605              if ($i != -1) {              if ($i != -1) {
2606                !!!parse-error (type => 'end tag missing:'.                !!!parse-error (type => 'end tag missing:'.
2607                                $self->{open_elements}->[-1]->[1]);                                $self->{open_elements}->[-1]->[1]);
               ## TODO: test  
2608              }              }
2609              splice @{$self->{open_elements}}, $i;              splice @{$self->{open_elements}}, $i;
2610              last LI;              last LI;
# Line 2412  sub _tree_construction_main ($) { Line 2652  sub _tree_construction_main ($) {
2652              if ($i != -1) {              if ($i != -1) {
2653                !!!parse-error (type => 'end tag missing:'.                !!!parse-error (type => 'end tag missing:'.
2654                                $self->{open_elements}->[-1]->[1]);                                $self->{open_elements}->[-1]->[1]);
               ## TODO: test  
2655              }              }
2656              splice @{$self->{open_elements}}, $i;              splice @{$self->{open_elements}}, $i;
2657              last LI;              last LI;
# Line 2475  sub _tree_construction_main ($) { Line 2714  sub _tree_construction_main ($) {
2714            }            }
2715          } # INSCOPE          } # INSCOPE
2716                        
2717            ## NOTE: See <http://html5.org/tools/web-apps-tracker?from=925&to=926>
2718          ## has an element in scope          ## has an element in scope
2719          my $i;          #my $i;
2720          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          #INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
2721            my $node = $self->{open_elements}->[$_];          #  my $node = $self->{open_elements}->[$_];
2722            if ({          #  if ({
2723                 h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,          #       h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
2724                }->{$node->[1]}) {          #      }->{$node->[1]}) {
2725              $i = $_;          #    $i = $_;
2726              last INSCOPE;          #    last INSCOPE;
2727            } elsif ({          #  } elsif ({
2728                      table => 1, caption => 1, td => 1, th => 1,          #            table => 1, caption => 1, td => 1, th => 1,
2729                      button => 1, marquee => 1, object => 1, html => 1,          #            button => 1, marquee => 1, object => 1, html => 1,
2730                     }->{$node->[1]}) {          #           }->{$node->[1]}) {
2731              last INSCOPE;          #    last INSCOPE;
2732            }          #  }
2733          } # INSCOPE          #} # INSCOPE
2734                      #  
2735          if (defined $i) {          #if (defined $i) {
2736            !!!parse-error (type => 'in hn:hn');          #  !!! parse-error (type => 'in hn:hn');
2737            splice @{$self->{open_elements}}, $i;          #  splice @{$self->{open_elements}}, $i;
2738          }          #}
2739                        
2740          !!!insert-element-t ($token->{tag_name}, $token->{attributes});          !!!insert-element-t ($token->{tag_name}, $token->{attributes});
2741                        
# Line 2538  sub _tree_construction_main ($) { Line 2778  sub _tree_construction_main ($) {
2778          return;          return;
2779        } elsif ({        } elsif ({
2780                  b => 1, big => 1, em => 1, font => 1, i => 1,                  b => 1, big => 1, em => 1, font => 1, i => 1,
2781                  nobr => 1, s => 1, small => 1, strile => 1,                  s => 1, small => 1, strile => 1,
2782                  strong => 1, tt => 1, u => 1,                  strong => 1, tt => 1, u => 1,
2783                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
2784          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
# Line 2548  sub _tree_construction_main ($) { Line 2788  sub _tree_construction_main ($) {
2788                    
2789          !!!next-token;          !!!next-token;
2790          return;          return;
2791          } elsif ($token->{tag_name} eq 'nobr') {
2792            $reconstruct_active_formatting_elements->($insert_to_current);
2793    
2794            ## has a |nobr| element in scope
2795            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
2796              my $node = $self->{open_elements}->[$_];
2797              if ($node->[1] eq 'nobr') {
2798                !!!back-token;
2799                $token = {type => 'end tag', tag_name => 'nobr'};
2800                return;
2801              } elsif ({
2802                        table => 1, caption => 1, td => 1, th => 1,
2803                        button => 1, marquee => 1, object => 1, html => 1,
2804                       }->{$node->[1]}) {
2805                last INSCOPE;
2806              }
2807            } # INSCOPE
2808            
2809            !!!insert-element-t ($token->{tag_name}, $token->{attributes});
2810            push @$active_formatting_elements, $self->{open_elements}->[-1];
2811            
2812            !!!next-token;
2813            return;
2814        } elsif ($token->{tag_name} eq 'button') {        } elsif ($token->{tag_name} eq 'button') {
2815          ## has a button element in scope          ## has a button element in scope
2816          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
# Line 2583  sub _tree_construction_main ($) { Line 2846  sub _tree_construction_main ($) {
2846          return;          return;
2847        } elsif ($token->{tag_name} eq 'xmp') {        } elsif ($token->{tag_name} eq 'xmp') {
2848          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
2849                    $parse_rcdata->('CDATA', $insert);
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
           
         $self->{content_model_flag} = 'CDATA';  
           
         !!!next-token;  
2850          return;          return;
2851        } elsif ($token->{tag_name} eq 'table') {        } elsif ($token->{tag_name} eq 'table') {
2852          ## has a p element in scope          ## has a p element in scope
# Line 2666  sub _tree_construction_main ($) { Line 2924  sub _tree_construction_main ($) {
2924            return;            return;
2925          } else {          } else {
2926            my $at = $token->{attributes};            my $at = $token->{attributes};
2927              my $form_attrs;
2928              $form_attrs->{action} = $at->{action} if $at->{action};
2929              my $prompt_attr = $at->{prompt};
2930            $at->{name} = {name => 'name', value => 'isindex'};            $at->{name} = {name => 'name', value => 'isindex'};
2931              delete $at->{action};
2932              delete $at->{prompt};
2933            my @tokens = (            my @tokens = (
2934                          {type => 'start tag', tag_name => 'form'},                          {type => 'start tag', tag_name => 'form',
2935                             attributes => $form_attrs},
2936                          {type => 'start tag', tag_name => 'hr'},                          {type => 'start tag', tag_name => 'hr'},
2937                          {type => 'start tag', tag_name => 'p'},                          {type => 'start tag', tag_name => 'p'},
2938                          {type => 'start tag', tag_name => 'label'},                          {type => 'start tag', tag_name => 'label'},
2939                          {type => 'character',                         );
2940                           data => 'This is a searchable index. Insert your search keywords here: '}, # SHOULD            if ($prompt_attr) {
2941                          ## TODO: make this configurable              push @tokens, {type => 'character', data => $prompt_attr->{value}};
2942              } else {
2943                push @tokens, {type => 'character',
2944                               data => 'This is a searchable index. Insert your search keywords here: '}; # SHOULD
2945                ## TODO: make this configurable
2946              }
2947              push @tokens,
2948                          {type => 'start tag', tag_name => 'input', attributes => $at},                          {type => 'start tag', tag_name => 'input', attributes => $at},
2949                          #{type => 'character', data => ''}, # SHOULD                          #{type => 'character', data => ''}, # SHOULD
2950                          {type => 'end tag', tag_name => 'label'},                          {type => 'end tag', tag_name => 'label'},
2951                          {type => 'end tag', tag_name => 'p'},                          {type => 'end tag', tag_name => 'p'},
2952                          {type => 'start tag', tag_name => 'hr'},                          {type => 'start tag', tag_name => 'hr'},
2953                          {type => 'end tag', tag_name => 'form'},                          {type => 'end tag', tag_name => 'form'};
                        );  
2954            $token = shift @tokens;            $token = shift @tokens;
2955            !!!back-token (@tokens);            !!!back-token (@tokens);
2956            return;            return;
2957          }          }
2958        } elsif ({        } elsif ($token->{tag_name} eq 'textarea') {
                 textarea => 1,  
                 iframe => 1,  
                 noembed => 1,  
                 noframes => 1,  
                 noscript => 0, ## TODO: 1 if scripting is enabled  
                }->{$token->{tag_name}}) {  
2959          my $tag_name = $token->{tag_name};          my $tag_name = $token->{tag_name};
2960          my $el;          my $el;
2961          !!!create-element ($el, $token->{tag_name}, $token->{attributes});          !!!create-element ($el, $token->{tag_name}, $token->{attributes});
2962                    
2963          if ($token->{tag_name} eq 'textarea') {          ## TODO: $self->{form_element} if defined
2964            ## TODO: $self->{form_element} if defined          $self->{content_model_flag} = 'RCDATA';
2965            $self->{content_model_flag} = 'RCDATA';          delete $self->{escape}; # MUST
         } else {  
           $self->{content_model_flag} = 'CDATA';  
         }  
2966                    
2967          $insert->($el);          $insert->($el);
2968                    
2969          my $text = '';          my $text = '';
2970          if ($token->{tag_name} eq 'textarea') {          !!!next-token;
2971            !!!next-token;          if ($token->{type} eq 'character') {
2972            if ($token->{type} eq 'character') {            $token->{data} =~ s/^\x0A//;
2973              $token->{data} =~ s/^\x0A//;            unless (length $token->{data}) {
2974              unless (length $token->{data}) {              !!!next-token;
               !!!next-token;  
             }  
2975            }            }
         } else {  
           !!!next-token;  
2976          }          }
2977          while ($token->{type} eq 'character') {          while ($token->{type} eq 'character') {
2978            $text .= $token->{data};            $text .= $token->{data};
# Line 2732  sub _tree_construction_main ($) { Line 2988  sub _tree_construction_main ($) {
2988              $token->{tag_name} eq $tag_name) {              $token->{tag_name} eq $tag_name) {
2989            ## Ignore the token            ## Ignore the token
2990          } else {          } else {
2991            if ($token->{tag_name} eq 'textarea') {            !!!parse-error (type => 'in RCDATA:#'.$token->{type});
             !!!parse-error (type => 'in RCDATA:#'.$token->{type});  
           } else {  
             !!!parse-error (type => 'in CDATA:#'.$token->{type});  
           }  
           ## ISSUE: And ignore?  
2992          }          }
2993          !!!next-token;          !!!next-token;
2994          return;          return;
2995          } elsif ({
2996                    iframe => 1,
2997                    noembed => 1,
2998                    noframes => 1,
2999                    noscript => 0, ## TODO: 1 if scripting is enabled
3000                   }->{$token->{tag_name}}) {
3001            $parse_rcdata->('CDATA', $insert);
3002            return;
3003        } elsif ($token->{tag_name} eq 'select') {        } elsif ($token->{tag_name} eq 'select') {
3004          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
3005                    
# Line 2771  sub _tree_construction_main ($) { Line 3030  sub _tree_construction_main ($) {
3030        }        }
3031      } elsif ($token->{type} eq 'end tag') {      } elsif ($token->{type} eq 'end tag') {
3032        if ($token->{tag_name} eq 'body') {        if ($token->{tag_name} eq 'body') {
3033          if (@{$self->{open_elements}} > 1 and $self->{open_elements}->[1]->[1] eq 'body') {          if (@{$self->{open_elements}} > 1 and
3034            ## ISSUE: There is an issue in the spec.              $self->{open_elements}->[1]->[1] eq 'body') {
3035            if ($self->{open_elements}->[-1]->[1] ne 'body') {            for (@{$self->{open_elements}}) {
3036              !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);              unless ({
3037                           dd => 1, dt => 1, li => 1, p => 1, td => 1,
3038                           th => 1, tr => 1, body => 1, html => 1,
3039                        }->{$_->[1]}) {
3040                  !!!parse-error (type => 'not closed:'.$_->[1]);
3041                }
3042            }            }
3043    
3044            $self->{insertion_mode} = 'after body';            $self->{insertion_mode} = 'after body';
3045            !!!next-token;            !!!next-token;
3046            return;            return;
# Line 2804  sub _tree_construction_main ($) { Line 3069  sub _tree_construction_main ($) {
3069                  address => 1, blockquote => 1, center => 1, dir => 1,                  address => 1, blockquote => 1, center => 1, dir => 1,
3070                  div => 1, dl => 1, fieldset => 1, listing => 1,                  div => 1, dl => 1, fieldset => 1, listing => 1,
3071                  menu => 1, ol => 1, pre => 1, ul => 1,                  menu => 1, ol => 1, pre => 1, ul => 1,
                 form => 1,  
3072                  p => 1,                  p => 1,
3073                  dd => 1, dt => 1, li => 1,                  dd => 1, dt => 1, li => 1,
3074                  button => 1, marquee => 1, object => 1,                  button => 1, marquee => 1, object => 1,
# Line 2842  sub _tree_construction_main ($) { Line 3106  sub _tree_construction_main ($) {
3106          }          }
3107                    
3108          splice @{$self->{open_elements}}, $i if defined $i;          splice @{$self->{open_elements}}, $i if defined $i;
         undef $self->{form_element} if $token->{tag_name} eq 'form';  
3109          $clear_up_to_marker->()          $clear_up_to_marker->()
3110            if {            if {
3111              button => 1, marquee => 1, object => 1,              button => 1, marquee => 1, object => 1,
3112            }->{$token->{tag_name}};            }->{$token->{tag_name}};
3113          !!!next-token;          !!!next-token;
3114          return;          return;
3115          } elsif ($token->{tag_name} eq 'form') {
3116            ## has an element in scope
3117            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3118              my $node = $self->{open_elements}->[$_];
3119              if ($node->[1] eq $token->{tag_name}) {
3120                ## generate implied end tags
3121                if ({
3122                     dd => 1, dt => 1, li => 1, p => 1,
3123                     td => 1, th => 1, tr => 1,
3124                    }->{$self->{open_elements}->[-1]->[1]}) {
3125                  !!!back-token;
3126                  $token = {type => 'end tag',
3127                            tag_name => $self->{open_elements}->[-1]->[1]}; # MUST
3128                  return;
3129                }
3130                last INSCOPE;
3131              } elsif ({
3132                        table => 1, caption => 1, td => 1, th => 1,
3133                        button => 1, marquee => 1, object => 1, html => 1,
3134                       }->{$node->[1]}) {
3135                last INSCOPE;
3136              }
3137            } # INSCOPE
3138            
3139            if ($self->{open_elements}->[-1]->[1] eq $token->{tag_name}) {
3140              pop @{$self->{open_elements}};
3141            } else {
3142              !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
3143            }
3144    
3145            undef $self->{form_element};
3146            !!!next-token;
3147            return;
3148        } elsif ({        } elsif ({
3149                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
3150                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
# Line 2950  sub _tree_construction_main ($) { Line 3246  sub _tree_construction_main ($) {
3246                  #not $phrasing_category->{$node->[1]} and                  #not $phrasing_category->{$node->[1]} and
3247                  ($special_category->{$node->[1]} or                  ($special_category->{$node->[1]} or
3248                   $scoping_category->{$node->[1]})) {                   $scoping_category->{$node->[1]})) {
3249                !!!parse-error (type => 'not closed:'.$node->[1]);                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3250                ## Ignore the token                ## Ignore the token
3251                !!!next-token;                !!!next-token;
3252                last S2;                last S2;
# Line 2979  sub _tree_construction_main ($) { Line 3275  sub _tree_construction_main ($) {
3275          redo B;          redo B;
3276        } elsif ($token->{type} eq 'start tag' and        } elsif ($token->{type} eq 'start tag' and
3277                 $token->{tag_name} eq 'html') {                 $token->{tag_name} eq 'html') {
3278          ## TODO: unless it is the first start tag token, parse-error  ## ISSUE: "aa<html>" is not a parse error.
3279    ## ISSUE: "<html>" in fragment is not a parse error.
3280            unless ($token->{first_start_tag}) {
3281              !!!parse-error (type => 'not first start tag');
3282            }
3283          my $top_el = $self->{open_elements}->[0]->[0];          my $top_el = $self->{open_elements}->[0]->[0];
3284          for my $attr_name (keys %{$token->{attributes}}) {          for my $attr_name (keys %{$token->{attributes}}) {
3285            unless ($top_el->has_attribute_ns (undef, $attr_name)) {            unless ($top_el->has_attribute_ns (undef, $attr_name)) {
# Line 3053  sub _tree_construction_main ($) { Line 3353  sub _tree_construction_main ($) {
3353              }              }
3354              redo B;              redo B;
3355            } elsif ($token->{type} eq 'end tag') {            } elsif ($token->{type} eq 'end tag') {
3356              if ($token->{tag_name} eq 'html') {              if ({head => 1, body => 1, html => 1}->{$token->{tag_name}}) {
3357                ## As if <head>                ## As if <head>
3358                !!!create-element ($self->{head_element}, 'head');                !!!create-element ($self->{head_element}, 'head');
3359                $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});                $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
# Line 3063  sub _tree_construction_main ($) { Line 3363  sub _tree_construction_main ($) {
3363                redo B;                redo B;
3364              } else {              } else {
3365                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3366                ## Ignore the token                ## Ignore the token ## ISSUE: An issue in the spec.
3367                !!!next-token;                !!!next-token;
3368                redo B;                redo B;
3369              }              }
3370            } else {            } else {
3371              die "$0: $token->{type}: Unknown type";              die "$0: $token->{type}: Unknown type";
3372            }            }
3373          } elsif ($self->{insertion_mode} eq 'in head') {          } elsif ($self->{insertion_mode} eq 'in head' or
3374                     $self->{insertion_mode} eq 'in head noscript' or
3375                     $self->{insertion_mode} eq 'after head') {
3376            if ($token->{type} eq 'character') {            if ($token->{type} eq 'character') {
3377              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {
3378                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
# Line 3087  sub _tree_construction_main ($) { Line 3389  sub _tree_construction_main ($) {
3389              !!!next-token;              !!!next-token;
3390              redo B;              redo B;
3391            } elsif ($token->{type} eq 'start tag') {            } elsif ($token->{type} eq 'start tag') {
3392              if ($token->{tag_name} eq 'title') {              if ({base => ($self->{insertion_mode} eq 'in head' or
3393                ## NOTE: There is an "as if in head" code clone                            $self->{insertion_mode} eq 'after head'),
3394                my $title_el;                   link => 1, meta => 1}->{$token->{tag_name}}) {
3395                !!!create-element ($title_el, 'title', $token->{attributes});                ## NOTE: There is a "as if in head" code clone.
3396                (defined $self->{head_element} ? $self->{head_element} : $self->{open_elements}->[-1]->[0])                if ($self->{insertion_mode} eq 'after head') {
3397                  ->append_child ($title_el);                  !!!parse-error (type => 'after head:'.$token->{tag_name});
3398                $self->{content_model_flag} = 'RCDATA';                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
3399                  }
3400                my $text = '';                !!!insert-element ($token->{tag_name}, $token->{attributes});
3401                  pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
3402                  ## TODO: Extracting |charset| from |meta|.
3403                  pop @{$self->{open_elements}}
3404                      if $self->{insertion_mode} eq 'after head';
3405                !!!next-token;                !!!next-token;
3406                while ($token->{type} eq 'character') {                redo B;
3407                  $text .= $token->{data};              } elsif ($token->{tag_name} eq 'title' and
3408                         $self->{insertion_mode} eq 'in head') {
3409                  ## NOTE: There is a "as if in head" code clone.
3410                  if ($self->{insertion_mode} eq 'after head') {
3411                    !!!parse-error (type => 'after head:'.$token->{tag_name});
3412                    push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
3413                  }
3414                  $parse_rcdata->('RCDATA', $insert_to_current);
3415                  pop @{$self->{open_elements}}
3416                      if $self->{insertion_mode} eq 'after head';
3417                  redo B;
3418                } elsif ($token->{tag_name} eq 'style') {
3419                  ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and
3420                  ## insertion mode 'in head')
3421                  ## NOTE: There is a "as if in head" code clone.
3422                  if ($self->{insertion_mode} eq 'after head') {
3423                    !!!parse-error (type => 'after head:'.$token->{tag_name});
3424                    push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
3425                  }
3426                  $parse_rcdata->('CDATA', $insert_to_current);
3427                  pop @{$self->{open_elements}}
3428                      if $self->{insertion_mode} eq 'after head';
3429                  redo B;
3430                } elsif ($token->{tag_name} eq 'noscript') {
3431                  if ($self->{insertion_mode} eq 'in head') {
3432                    ## NOTE: and scripting is disalbed
3433                    !!!insert-element ($token->{tag_name}, $token->{attributes});
3434                    $self->{insertion_mode} = 'in head noscript';
3435                  !!!next-token;                  !!!next-token;
3436                }                  redo B;
3437                if (length $text) {                } elsif ($self->{insertion_mode} eq 'in head noscript') {
3438                  $title_el->manakai_append_text ($text);                  !!!parse-error (type => 'in noscript:noscript');
               }  
                 
               $self->{content_model_flag} = 'PCDATA';  
                 
               if ($token->{type} eq 'end tag' and  
                   $token->{tag_name} eq 'title') {  
3439                  ## Ignore the token                  ## Ignore the token
3440                    redo B;
3441                } else {                } else {
3442                  !!!parse-error (type => 'in RCDATA:#'.$token->{type});                  #
                 ## ISSUE: And ignore?  
3443                }                }
3444                } elsif ($token->{tag_name} eq 'head' and
3445                         $self->{insertion_mode} ne 'after head') {
3446                  !!!parse-error (type => 'in head:head'); # or in head noscript
3447                  ## Ignore the token
3448                !!!next-token;                !!!next-token;
3449                redo B;                redo B;
3450              } elsif ($token->{tag_name} eq 'style') {              } elsif ($self->{insertion_mode} ne 'in head noscript' and
3451                $style_start_tag->();                       $token->{tag_name} eq 'script') {
3452                  if ($self->{insertion_mode} eq 'after head') {
3453                    !!!parse-error (type => 'after head:'.$token->{tag_name});
3454                    push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
3455                  }
3456                  ## NOTE: There is a "as if in head" code clone.
3457                  $script_start_tag->($insert_to_current);
3458                  pop @{$self->{open_elements}}
3459                      if $self->{insertion_mode} eq 'after head';
3460                redo B;                redo B;
3461              } elsif ($token->{tag_name} eq 'script') {              } elsif ($self->{insertion_mode} eq 'after head' and
3462                $script_start_tag->();                       $token->{tag_name} eq 'body') {
3463                redo B;                !!!insert-element ('body', $token->{attributes});
3464              } elsif ({base => 1, link => 1, meta => 1}->{$token->{tag_name}}) {                $self->{insertion_mode} = 'in body';
               ## NOTE: There are "as if in head" code clones  
               my $el;  
               !!!create-element ($el, $token->{tag_name}, $token->{attributes});  
               (defined $self->{head_element} ? $self->{head_element} : $self->{open_elements}->[-1]->[0])  
                 ->append_child ($el);  
   
3465                !!!next-token;                !!!next-token;
3466                redo B;                redo B;
3467              } elsif ($token->{tag_name} eq 'head') {              } elsif ($self->{insertion_mode} eq 'after head' and
3468                !!!parse-error (type => 'in head:head');                       $token->{tag_name} eq 'frameset') {
3469                ## Ignore the token                !!!insert-element ('frameset', $token->{attributes});
3470                  $self->{insertion_mode} = 'in frameset';
3471                !!!next-token;                !!!next-token;
3472                redo B;                redo B;
3473              } else {              } else {
3474                #                #
3475              }              }
3476            } elsif ($token->{type} eq 'end tag') {            } elsif ($token->{type} eq 'end tag') {
3477              if ($token->{tag_name} eq 'head') {              if ($self->{insertion_mode} eq 'in head' and
3478                if ($self->{open_elements}->[-1]->[1] eq 'head') {                  $token->{tag_name} eq 'head') {
3479                  pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
               } else {  
                 !!!parse-error (type => 'unmatched end tag:head');  
               }  
3480                $self->{insertion_mode} = 'after head';                $self->{insertion_mode} = 'after head';
3481                !!!next-token;                !!!next-token;
3482                redo B;                redo B;
3483              } elsif ($token->{tag_name} eq 'html') {              } elsif ($self->{insertion_mode} eq 'in head noscript' and
3484                    $token->{tag_name} eq 'noscript') {
3485                  pop @{$self->{open_elements}};
3486                  $self->{insertion_mode} = 'in head';
3487                  !!!next-token;
3488                  redo B;
3489                } elsif ($self->{insertion_mode} eq 'in head' and
3490                         ($token->{tag_name} eq 'body' or
3491                          $token->{tag_name} eq 'html')) {
3492                #                #
3493              } else {              } elsif ($self->{insertion_mode} ne 'after head') {
3494                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3495                ## Ignore the token                ## Ignore the token
3496                !!!next-token;                !!!next-token;
3497                redo B;                redo B;
3498                } else {
3499                  #
3500              }              }
3501            } else {            } else {
3502              #              #
3503            }            }
3504    
3505            if ($self->{open_elements}->[-1]->[1] eq 'head') {            ## As if </head> or </noscript> or <body>
3506              ## As if </head>            if ($self->{insertion_mode} eq 'in head') {
3507                pop @{$self->{open_elements}};
3508                $self->{insertion_mode} = 'after head';
3509              } elsif ($self->{insertion_mode} eq 'in head noscript') {
3510              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
3511                !!!parse-error (type => 'in noscript:'.(defined $token->{tag_name} ? ($token->{type} eq 'end tag' ? '/' : '') . $token->{tag_name} : '#' . $token->{type}));
3512                $self->{insertion_mode} = 'in head';
3513              } else { # 'after head'
3514                !!!insert-element ('body');
3515                $self->{insertion_mode} = 'in body';
3516            }            }
           $self->{insertion_mode} = 'after head';  
3517            ## reprocess            ## reprocess
3518            redo B;            redo B;
3519    
3520            ## ISSUE: An issue in the spec.            ## ISSUE: An issue in the spec.
         } elsif ($self->{insertion_mode} eq 'after head') {  
           if ($token->{type} eq 'character') {  
             if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {  
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);  
               unless (length $token->{data}) {  
                 !!!next-token;  
                 redo B;  
               }  
             }  
               
             #  
           } elsif ($token->{type} eq 'comment') {  
             my $comment = $self->{document}->create_comment ($token->{data});  
             $self->{open_elements}->[-1]->[0]->append_child ($comment);  
             !!!next-token;  
             redo B;  
           } elsif ($token->{type} eq 'start tag') {  
             if ($token->{tag_name} eq 'body') {  
               !!!insert-element ('body', $token->{attributes});  
               $self->{insertion_mode} = 'in body';  
               !!!next-token;  
               redo B;  
             } elsif ($token->{tag_name} eq 'frameset') {  
               !!!insert-element ('frameset', $token->{attributes});  
               $self->{insertion_mode} = 'in frameset';  
               !!!next-token;  
               redo B;  
             } elsif ({  
                       base => 1, link => 1, meta => 1,  
                       script => 1, style => 1, title => 1,  
                      }->{$token->{tag_name}}) {  
               !!!parse-error (type => 'after head:'.$token->{tag_name});  
               $self->{insertion_mode} = 'in head';  
               ## reprocess  
               redo B;  
             } else {  
               #  
             }  
           } else {  
             #  
           }  
             
           ## As if <body>  
           !!!insert-element ('body');  
           $self->{insertion_mode} = 'in body';  
           ## reprocess  
           redo B;  
3521          } elsif ($self->{insertion_mode} eq 'in body') {          } elsif ($self->{insertion_mode} eq 'in body') {
3522            if ($token->{type} eq 'character') {            if ($token->{type} eq 'character') {
3523              ## NOTE: There is a code clone of "character in body".              ## NOTE: There is a code clone of "character in body".
# Line 4725  sub _tree_construction_main ($) { Line 5026  sub _tree_construction_main ($) {
5026            }            }
5027                        
5028            if (defined $token->{tag_name}) {            if (defined $token->{tag_name}) {
5029              !!!parse-error (type => 'in frameset:'.$token->{tag_name});              !!!parse-error (type => 'in frameset:'.($token->{type} eq 'end tag' ? '/' : '').$token->{tag_name});
5030            } else {            } else {
5031              !!!parse-error (type => 'in frameset:#'.$token->{type});              !!!parse-error (type => 'in frameset:#'.$token->{type});
5032            }            }
# Line 4769  sub _tree_construction_main ($) { Line 5070  sub _tree_construction_main ($) {
5070            }            }
5071                        
5072            if (defined $token->{tag_name}) {            if (defined $token->{tag_name}) {
5073              !!!parse-error (type => 'after frameset:'.$token->{tag_name});              !!!parse-error (type => 'after frameset:'.($token->{tag_name} eq 'end tag' ? '/' : '').$token->{tag_name});
5074            } else {            } else {
5075              !!!parse-error (type => 'after frameset:#'.$token->{type});              !!!parse-error (type => 'after frameset:#'.$token->{type});
5076            }            }
# Line 4819  sub _tree_construction_main ($) { Line 5120  sub _tree_construction_main ($) {
5120          redo B;          redo B;
5121        } elsif ($token->{type} eq 'start tag' or        } elsif ($token->{type} eq 'start tag' or
5122                 $token->{type} eq 'end tag') {                 $token->{type} eq 'end tag') {
5123          !!!parse-error (type => 'after html:'.$token->{tag_name});          !!!parse-error (type => 'after html:'.($token->{type} eq 'end tag' ? '/' : '').$token->{tag_name});
5124          $phase = 'main';          $phase = 'main';
5125          ## reprocess          ## reprocess
5126          redo B;          redo B;
# Line 4865  sub set_inner_html ($$$) { Line 5166  sub set_inner_html ($$$) {
5166      ## NOTE: Most of this code is copied from |parse_string|      ## NOTE: Most of this code is copied from |parse_string|
5167    
5168      ## Step 1 # MUST      ## Step 1 # MUST
5169      my $doc = $node->owner_document->implementation->create_document;      my $this_doc = $node->owner_document;
5170      ## TODO: Mark as HTML document      my $doc = $this_doc->implementation->create_document;
5171        $doc->manakai_is_html (1);
5172      my $p = $class->new;      my $p = $class->new;
5173      $p->{document} = $doc;      $p->{document} = $doc;
5174    
# Line 4876  sub set_inner_html ($$$) { Line 5178  sub set_inner_html ($$$) {
5178      my $column = 0;      my $column = 0;
5179      $p->{set_next_input_character} = sub {      $p->{set_next_input_character} = sub {
5180        my $self = shift;        my $self = shift;
5181    
5182          pop @{$self->{prev_input_character}};
5183          unshift @{$self->{prev_input_character}}, $self->{next_input_character};
5184    
5185        $self->{next_input_character} = -1 and return if $i >= length $$s;        $self->{next_input_character} = -1 and return if $i >= length $$s;
5186        $self->{next_input_character} = ord substr $$s, $i++, 1;        $self->{next_input_character} = ord substr $$s, $i++, 1;
5187        $column++;        $column++;
# Line 4884  sub set_inner_html ($$$) { Line 5190  sub set_inner_html ($$$) {
5190          $line++;          $line++;
5191          $column = 0;          $column = 0;
5192        } elsif ($self->{next_input_character} == 0x000D) { # CR        } elsif ($self->{next_input_character} == 0x000D) { # CR
5193          if ($i >= length $$s) {          $i++ if substr ($$s, $i, 1) eq "\x0A";
           #  
         } else {  
           my $next_char = ord substr $$s, $i++, 1;  
           if ($next_char == 0x000A) { # LF  
             #  
           } else {  
             push @{$self->{char}}, $next_char;  
           }  
         }  
5194          $self->{next_input_character} = 0x000A; # LF # MUST          $self->{next_input_character} = 0x000A; # LF # MUST
5195          $line++;          $line++;
5196          $column = 0;          $column = 0;
5197        } elsif ($self->{next_input_character} > 0x10FFFF) {        } elsif ($self->{next_input_character} > 0x10FFFF) {
5198          $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST          $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
5199        } elsif ($self->{next_input_character} == 0x0000) { # NULL        } elsif ($self->{next_input_character} == 0x0000) { # NULL
5200            !!!parse-error (type => 'NULL');
5201          $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST          $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
5202        }        }
5203      };      };
5204        $p->{prev_input_character} = [-1, -1, -1];
5205        $p->{next_input_character} = -1;
5206            
5207      my $ponerror = $onerror || sub {      my $ponerror = $onerror || sub {
5208        my (%opt) = @_;        my (%opt) = @_;
# Line 4981  sub set_inner_html ($$$) { Line 5281  sub set_inner_html ($$$) {
5281      ## Step 12 # MUST      ## Step 12 # MUST
5282      @cn = @{$root->child_nodes};      @cn = @{$root->child_nodes};
5283      for (@cn) {      for (@cn) {
5284          $this_doc->adopt_node ($_);
5285        $node->append_child ($_);        $node->append_child ($_);
5286      }      }
5287      ## ISSUE: adopt_node? mutation events?      ## ISSUE: mutation events?
5288    
5289      $p->_terminate_tree_constructor;      $p->_terminate_tree_constructor;
5290    } else {    } else {
# Line 5028  sub get_inner_html ($$$) { Line 5329  sub get_inner_html ($$$) {
5329            
5330      my $nt = $child->node_type;      my $nt = $child->node_type;
5331      if ($nt == 1) { # Element      if ($nt == 1) { # Element
5332        my $tag_name = lc $child->tag_name; ## ISSUE: Definition of "lowercase"        my $tag_name = $child->tag_name; ## TODO: manakai_tag_name
5333        $s .= '<' . $tag_name;        $s .= '<' . $tag_name;
5334          ## NOTE: Non-HTML case:
5335        ## ISSUE: Non-html elements        ## <http://permalink.gmane.org/gmane.org.w3c.whatwg.discuss/11191>
5336    
5337        my @attrs = @{$child->attributes}; # sort order MUST be stable        my @attrs = @{$child->attributes}; # sort order MUST be stable
5338        for my $attr (@attrs) { # order is implementation dependent        for my $attr (@attrs) { # order is implementation dependent
5339          my $attr_name = lc $attr->name; ## ISSUE: Definition of "lowercase"          my $attr_name = $attr->name; ## TODO: manakai_name
5340          $s .= ' ' . $attr_name . '="';          $s .= ' ' . $attr_name . '="';
5341          my $attr_value = $attr->value;          my $attr_value = $attr->value;
5342          ## escape          ## escape
# Line 5054  sub get_inner_html ($$$) { Line 5355  sub get_inner_html ($$$) {
5355          spacer => 1, wbr => 1,          spacer => 1, wbr => 1,
5356        }->{$tag_name};        }->{$tag_name};
5357    
5358          $s .= "\x0A" if $tag_name eq 'pre' or $tag_name eq 'textarea';
5359    
5360        if (not $in_cdata and {        if (not $in_cdata and {
5361          style => 1, script => 1, xmp => 1, iframe => 1,          style => 1, script => 1, xmp => 1, iframe => 1,
5362          noembed => 1, noframes => 1, noscript => 1,          noembed => 1, noframes => 1, noscript => 1,
5363            plaintext => 1,
5364        }->{$tag_name}) {        }->{$tag_name}) {
5365          unshift @node, 'cdata-out';          unshift @node, 'cdata-out';
5366          $in_cdata = 1;          $in_cdata = 1;

Legend:
Removed from v.1.11  
changed lines
  Added in v.1.30

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24