/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.8 by wakaba, Sat Jun 23 02:26:51 2007 UTC revision 1.31 by wakaba, Sat Jun 30 14:13:19 2007 UTC
# Line 2  package Whatpm::HTML; Line 2  package Whatpm::HTML;
2  use strict;  use strict;
3  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};
4    
5  ## This is an early version of an HTML parser.  ## ISSUE:
6    ## var doc = implementation.createDocument (null, null, null);
7    ## doc.write ('');
8    ## alert (doc.compatMode);
9    
10    ## ISSUE: HTML5 revision 967 says that the encoding layer MUST NOT
11    ## strip BOM and the HTML layer MUST ignore it.  Whether we can do it
12    ## is not yet clear.
13    ## "{U+FEFF}..." in UTF-16BE/UTF-16LE is three or four characters?
14    ## "{U+FEFF}..." in GB18030?
15    
16  my $permitted_slash_tag_name = {  my $permitted_slash_tag_name = {
17    base => 1,    base => 1,
# Line 18  my $permitted_slash_tag_name = { Line 27  my $permitted_slash_tag_name = {
27    input => 1,    input => 1,
28  };  };
29    
 my $entity_char = {  
   AElig => "\x{00C6}",  
   Aacute => "\x{00C1}",  
   Acirc => "\x{00C2}",  
   Agrave => "\x{00C0}",  
   Alpha => "\x{0391}",  
   Aring => "\x{00C5}",  
   Atilde => "\x{00C3}",  
   Auml => "\x{00C4}",  
   Beta => "\x{0392}",  
   Ccedil => "\x{00C7}",  
   Chi => "\x{03A7}",  
   Dagger => "\x{2021}",  
   Delta => "\x{0394}",  
   ETH => "\x{00D0}",  
   Eacute => "\x{00C9}",  
   Ecirc => "\x{00CA}",  
   Egrave => "\x{00C8}",  
   Epsilon => "\x{0395}",  
   Eta => "\x{0397}",  
   Euml => "\x{00CB}",  
   Gamma => "\x{0393}",  
   Iacute => "\x{00CD}",  
   Icirc => "\x{00CE}",  
   Igrave => "\x{00CC}",  
   Iota => "\x{0399}",  
   Iuml => "\x{00CF}",  
   Kappa => "\x{039A}",  
   Lambda => "\x{039B}",  
   Mu => "\x{039C}",  
   Ntilde => "\x{00D1}",  
   Nu => "\x{039D}",  
   OElig => "\x{0152}",  
   Oacute => "\x{00D3}",  
   Ocirc => "\x{00D4}",  
   Ograve => "\x{00D2}",  
   Omega => "\x{03A9}",  
   Omicron => "\x{039F}",  
   Oslash => "\x{00D8}",  
   Otilde => "\x{00D5}",  
   Ouml => "\x{00D6}",  
   Phi => "\x{03A6}",  
   Pi => "\x{03A0}",  
   Prime => "\x{2033}",  
   Psi => "\x{03A8}",  
   Rho => "\x{03A1}",  
   Scaron => "\x{0160}",  
   Sigma => "\x{03A3}",  
   THORN => "\x{00DE}",  
   Tau => "\x{03A4}",  
   Theta => "\x{0398}",  
   Uacute => "\x{00DA}",  
   Ucirc => "\x{00DB}",  
   Ugrave => "\x{00D9}",  
   Upsilon => "\x{03A5}",  
   Uuml => "\x{00DC}",  
   Xi => "\x{039E}",  
   Yacute => "\x{00DD}",  
   Yuml => "\x{0178}",  
   Zeta => "\x{0396}",  
   aacute => "\x{00E1}",  
   acirc => "\x{00E2}",  
   acute => "\x{00B4}",  
   aelig => "\x{00E6}",  
   agrave => "\x{00E0}",  
   alefsym => "\x{2135}",  
   alpha => "\x{03B1}",  
   amp => "\x{0026}",  
   AMP => "\x{0026}",  
   and => "\x{2227}",  
   ang => "\x{2220}",  
   apos => "\x{0027}",  
   aring => "\x{00E5}",  
   asymp => "\x{2248}",  
   atilde => "\x{00E3}",  
   auml => "\x{00E4}",  
   bdquo => "\x{201E}",  
   beta => "\x{03B2}",  
   brvbar => "\x{00A6}",  
   bull => "\x{2022}",  
   cap => "\x{2229}",  
   ccedil => "\x{00E7}",  
   cedil => "\x{00B8}",  
   cent => "\x{00A2}",  
   chi => "\x{03C7}",  
   circ => "\x{02C6}",  
   clubs => "\x{2663}",  
   cong => "\x{2245}",  
   copy => "\x{00A9}",  
   COPY => "\x{00A9}",  
   crarr => "\x{21B5}",  
   cup => "\x{222A}",  
   curren => "\x{00A4}",  
   dArr => "\x{21D3}",  
   dagger => "\x{2020}",  
   darr => "\x{2193}",  
   deg => "\x{00B0}",  
   delta => "\x{03B4}",  
   diams => "\x{2666}",  
   divide => "\x{00F7}",  
   eacute => "\x{00E9}",  
   ecirc => "\x{00EA}",  
   egrave => "\x{00E8}",  
   empty => "\x{2205}",  
   emsp => "\x{2003}",  
   ensp => "\x{2002}",  
   epsilon => "\x{03B5}",  
   equiv => "\x{2261}",  
   eta => "\x{03B7}",  
   eth => "\x{00F0}",  
   euml => "\x{00EB}",  
   euro => "\x{20AC}",  
   exist => "\x{2203}",  
   fnof => "\x{0192}",  
   forall => "\x{2200}",  
   frac12 => "\x{00BD}",  
   frac14 => "\x{00BC}",  
   frac34 => "\x{00BE}",  
   frasl => "\x{2044}",  
   gamma => "\x{03B3}",  
   ge => "\x{2265}",  
   gt => "\x{003E}",  
   GT => "\x{003E}",  
   hArr => "\x{21D4}",  
   harr => "\x{2194}",  
   hearts => "\x{2665}",  
   hellip => "\x{2026}",  
   iacute => "\x{00ED}",  
   icirc => "\x{00EE}",  
   iexcl => "\x{00A1}",  
   igrave => "\x{00EC}",  
   image => "\x{2111}",  
   infin => "\x{221E}",  
   int => "\x{222B}",  
   iota => "\x{03B9}",  
   iquest => "\x{00BF}",  
   isin => "\x{2208}",  
   iuml => "\x{00EF}",  
   kappa => "\x{03BA}",  
   lArr => "\x{21D0}",  
   lambda => "\x{03BB}",  
   lang => "\x{2329}",  
   laquo => "\x{00AB}",  
   larr => "\x{2190}",  
   lceil => "\x{2308}",  
   ldquo => "\x{201C}",  
   le => "\x{2264}",  
   lfloor => "\x{230A}",  
   lowast => "\x{2217}",  
   loz => "\x{25CA}",  
   lrm => "\x{200E}",  
   lsaquo => "\x{2039}",  
   lsquo => "\x{2018}",  
   lt => "\x{003C}",  
   LT => "\x{003C}",  
   macr => "\x{00AF}",  
   mdash => "\x{2014}",  
   micro => "\x{00B5}",  
   middot => "\x{00B7}",  
   minus => "\x{2212}",  
   mu => "\x{03BC}",  
   nabla => "\x{2207}",  
   nbsp => "\x{00A0}",  
   ndash => "\x{2013}",  
   ne => "\x{2260}",  
   ni => "\x{220B}",  
   not => "\x{00AC}",  
   notin => "\x{2209}",  
   nsub => "\x{2284}",  
   ntilde => "\x{00F1}",  
   nu => "\x{03BD}",  
   oacute => "\x{00F3}",  
   ocirc => "\x{00F4}",  
   oelig => "\x{0153}",  
   ograve => "\x{00F2}",  
   oline => "\x{203E}",  
   omega => "\x{03C9}",  
   omicron => "\x{03BF}",  
   oplus => "\x{2295}",  
   or => "\x{2228}",  
   ordf => "\x{00AA}",  
   ordm => "\x{00BA}",  
   oslash => "\x{00F8}",  
   otilde => "\x{00F5}",  
   otimes => "\x{2297}",  
   ouml => "\x{00F6}",  
   para => "\x{00B6}",  
   part => "\x{2202}",  
   permil => "\x{2030}",  
   perp => "\x{22A5}",  
   phi => "\x{03C6}",  
   pi => "\x{03C0}",  
   piv => "\x{03D6}",  
   plusmn => "\x{00B1}",  
   pound => "\x{00A3}",  
   prime => "\x{2032}",  
   prod => "\x{220F}",  
   prop => "\x{221D}",  
   psi => "\x{03C8}",  
   quot => "\x{0022}",  
   QUOT => "\x{0022}",  
   rArr => "\x{21D2}",  
   radic => "\x{221A}",  
   rang => "\x{232A}",  
   raquo => "\x{00BB}",  
   rarr => "\x{2192}",  
   rceil => "\x{2309}",  
   rdquo => "\x{201D}",  
   real => "\x{211C}",  
   reg => "\x{00AE}",  
   REG => "\x{00AE}",  
   rfloor => "\x{230B}",  
   rho => "\x{03C1}",  
   rlm => "\x{200F}",  
   rsaquo => "\x{203A}",  
   rsquo => "\x{2019}",  
   sbquo => "\x{201A}",  
   scaron => "\x{0161}",  
   sdot => "\x{22C5}",  
   sect => "\x{00A7}",  
   shy => "\x{00AD}",  
   sigma => "\x{03C3}",  
   sigmaf => "\x{03C2}",  
   sim => "\x{223C}",  
   spades => "\x{2660}",  
   sub => "\x{2282}",  
   sube => "\x{2286}",  
   sum => "\x{2211}",  
   sup => "\x{2283}",  
   sup1 => "\x{00B9}",  
   sup2 => "\x{00B2}",  
   sup3 => "\x{00B3}",  
   supe => "\x{2287}",  
   szlig => "\x{00DF}",  
   tau => "\x{03C4}",  
   there4 => "\x{2234}",  
   theta => "\x{03B8}",  
   thetasym => "\x{03D1}",  
   thinsp => "\x{2009}",  
   thorn => "\x{00FE}",  
   tilde => "\x{02DC}",  
   times => "\x{00D7}",  
   trade => "\x{2122}",  
   uArr => "\x{21D1}",  
   uacute => "\x{00FA}",  
   uarr => "\x{2191}",  
   ucirc => "\x{00FB}",  
   ugrave => "\x{00F9}",  
   uml => "\x{00A8}",  
   upsih => "\x{03D2}",  
   upsilon => "\x{03C5}",  
   uuml => "\x{00FC}",  
   weierp => "\x{2118}",  
   xi => "\x{03BE}",  
   yacute => "\x{00FD}",  
   yen => "\x{00A5}",  
   yuml => "\x{00FF}",  
   zeta => "\x{03B6}",  
   zwj => "\x{200D}",  
   zwnj => "\x{200C}",  
 }; # $entity_char  
   
 ## TODO: Ensure that this table match to <http://html5.org/tools/web-apps-tracker?from=868&to=869>.  
 ## <http://lists.whatwg.org/pipermail/whatwg-whatwg.org/2006-December/thread.html#8562>  
30  my $c1_entity_char = {  my $c1_entity_char = {
31       128, 8364,    0x80 => 0x20AC,
32       129, 65533,    0x81 => 0xFFFD,
33       130, 8218,    0x82 => 0x201A,
34       131, 402,    0x83 => 0x0192,
35       132, 8222,    0x84 => 0x201E,
36       133, 8230,    0x85 => 0x2026,
37       134, 8224,    0x86 => 0x2020,
38       135, 8225,    0x87 => 0x2021,
39       136, 710,    0x88 => 0x02C6,
40       137, 8240,    0x89 => 0x2030,
41       138, 352,    0x8A => 0x0160,
42       139, 8249,    0x8B => 0x2039,
43       140, 338,    0x8C => 0x0152,
44       141, 65533,    0x8D => 0xFFFD,
45       142, 381,    0x8E => 0x017D,
46       143, 65533,    0x8F => 0xFFFD,
47       144, 65533,    0x90 => 0xFFFD,
48       145, 8216,    0x91 => 0x2018,
49       146, 8217,    0x92 => 0x2019,
50       147, 8220,    0x93 => 0x201C,
51       148, 8221,    0x94 => 0x201D,
52       149, 8226,    0x95 => 0x2022,
53       150, 8211,    0x96 => 0x2013,
54       151, 8212,    0x97 => 0x2014,
55       152, 732,    0x98 => 0x02DC,
56       153, 8482,    0x99 => 0x2122,
57       154, 353,    0x9A => 0x0161,
58       155, 8250,    0x9B => 0x203A,
59       156, 339,    0x9C => 0x0153,
60       157, 65533,    0x9D => 0xFFFD,
61       158, 382,    0x9E => 0x017E,
62       159, 376,    0x9F => 0x0178,
63  }; # $c1_entity_char  }; # $c1_entity_char
64    
65  my $special_category = {  my $special_category = {
# Line 351  sub parse_string ($$$;$) { Line 96  sub parse_string ($$$;$) {
96    my $column = 0;    my $column = 0;
97    $self->{set_next_input_character} = sub {    $self->{set_next_input_character} = sub {
98      my $self = shift;      my $self = shift;
99    
100        pop @{$self->{prev_input_character}};
101        unshift @{$self->{prev_input_character}}, $self->{next_input_character};
102    
103      $self->{next_input_character} = -1 and return if $i >= length $$s;      $self->{next_input_character} = -1 and return if $i >= length $$s;
104      $self->{next_input_character} = ord substr $$s, $i++, 1;      $self->{next_input_character} = ord substr $$s, $i++, 1;
105      $column++;      $column++;
# Line 359  sub parse_string ($$$;$) { Line 108  sub parse_string ($$$;$) {
108        $line++;        $line++;
109        $column = 0;        $column = 0;
110      } elsif ($self->{next_input_character} == 0x000D) { # CR      } elsif ($self->{next_input_character} == 0x000D) { # CR
111        if ($i >= length $$s) {        $i++ if substr ($$s, $i, 1) eq "\x0A";
         #  
       } else {  
         my $next_char = ord substr $$s, $i++, 1;  
         if ($next_char == 0x000A) { # LF  
           #  
         } else {  
           push @{$self->{char}}, $next_char;  
         }  
       }  
112        $self->{next_input_character} = 0x000A; # LF # MUST        $self->{next_input_character} = 0x000A; # LF # MUST
113        $line++;        $line++;
114        $column = 0;        $column = 0;
# Line 376  sub parse_string ($$$;$) { Line 116  sub parse_string ($$$;$) {
116        $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST        $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
117      } elsif ($self->{next_input_character} == 0x0000) { # NULL      } elsif ($self->{next_input_character} == 0x0000) { # NULL
118        !!!parse-error (type => 'NULL');        !!!parse-error (type => 'NULL');
 ## TODO: test  
119        $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST        $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
120      }      }
121    };    };
122      $self->{prev_input_character} = [-1, -1, -1];
123      $self->{next_input_character} = -1;
124    
125    my $onerror = $_[2] || sub {    my $onerror = $_[2] || sub {
126      my (%opt) = @_;      my (%opt) = @_;
# Line 423  sub _initialize_tokenizer ($) { Line 164  sub _initialize_tokenizer ($) {
164    # $self->{next_input_character}    # $self->{next_input_character}
165    !!!next-input-character;    !!!next-input-character;
166    $self->{token} = [];    $self->{token} = [];
167      # $self->{escape}
168  } # _initialize_tokenizer  } # _initialize_tokenizer
169    
170  ## A token has:  ## A token has:
171  ##   ->{type} eq 'DOCTYPE', 'start tag', 'end tag', 'comment',  ##   ->{type} eq 'DOCTYPE', 'start tag', 'end tag', 'comment',
172  ##       'character', or 'end-of-file'  ##       'character', or 'end-of-file'
173  ##   ->{name} (DOCTYPE, start tag (tagname), end tag (tagname))  ##   ->{name} (DOCTYPE, start tag (tag name), end tag (tag name))
174      ## ISSUE: the spec need s/tagname/tag name/  ##   ->{public_identifier} (DOCTYPE)
175  ##   ->{error} == 1 or 0 (DOCTYPE)  ##   ->{system_identifier} (DOCTYPE)
176    ##   ->{correct} == 1 or 0 (DOCTYPE)
177  ##   ->{attributes} isa HASH (start tag, end tag)  ##   ->{attributes} isa HASH (start tag, end tag)
178  ##   ->{data} (comment, character)  ##   ->{data} (comment, character)
179    
 ## Macros  
 ##   Macros MUST be preceded by three EXCLAMATION MARKs.  
 ##   emit ($token)  
 ##     Emits the specified token.  
   
180  ## Emitted token MUST immediately be handled by the tree construction state.  ## Emitted token MUST immediately be handled by the tree construction state.
181    
182  ## Before each step, UA MAY check to see if either one of the scripts in  ## Before each step, UA MAY check to see if either one of the scripts in
# Line 447  sub _initialize_tokenizer ($) { Line 185  sub _initialize_tokenizer ($) {
185  ## has completed loading.  If one has, then it MUST be executed  ## has completed loading.  If one has, then it MUST be executed
186  ## and removed from the list.  ## and removed from the list.
187    
 ## ISSUE: <http://html5.org/tools/web-apps-tracker?from=874&to=876>  
   
188  sub _get_next_token ($) {  sub _get_next_token ($) {
189    my $self = shift;    my $self = shift;
190    if (@{$self->{token}}) {    if (@{$self->{token}}) {
# Line 466  sub _get_next_token ($) { Line 202  sub _get_next_token ($) {
202          } else {          } else {
203            #            #
204          }          }
205          } elsif ($self->{next_input_character} == 0x002D) { # -
206            if ($self->{content_model_flag} eq 'RCDATA' or
207                $self->{content_model_flag} eq 'CDATA') {
208              unless ($self->{escape}) {
209                if ($self->{prev_input_character}->[0] == 0x002D and # -
210                    $self->{prev_input_character}->[1] == 0x0021 and # !
211                    $self->{prev_input_character}->[2] == 0x003C) { # <
212                  $self->{escape} = 1;
213                }
214              }
215            }
216            
217            #
218        } elsif ($self->{next_input_character} == 0x003C) { # <        } elsif ($self->{next_input_character} == 0x003C) { # <
219          if ($self->{content_model_flag} ne 'PLAINTEXT') {          if ($self->{content_model_flag} eq 'PCDATA' or
220                (($self->{content_model_flag} eq 'CDATA' or
221                  $self->{content_model_flag} eq 'RCDATA') and
222                 not $self->{escape})) {
223            $self->{state} = 'tag open';            $self->{state} = 'tag open';
224            !!!next-input-character;            !!!next-input-character;
225            redo A;            redo A;
226          } else {          } else {
227            #            #
228          }          }
229          } elsif ($self->{next_input_character} == 0x003E) { # >
230            if ($self->{escape} and
231                ($self->{content_model_flag} eq 'RCDATA' or
232                 $self->{content_model_flag} eq 'CDATA')) {
233              if ($self->{prev_input_character}->[0] == 0x002D and # -
234                  $self->{prev_input_character}->[1] == 0x002D) { # -
235                delete $self->{escape};
236              }
237            }
238            
239            #
240        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
241          !!!emit ({type => 'end-of-file'});          !!!emit ({type => 'end-of-file'});
242          last A; ## TODO: ok?          last A; ## TODO: ok?
# Line 490  sub _get_next_token ($) { Line 253  sub _get_next_token ($) {
253      } elsif ($self->{state} eq 'entity data') {      } elsif ($self->{state} eq 'entity data') {
254        ## (cannot happen in CDATA state)        ## (cannot happen in CDATA state)
255                
256        my $token = $self->_tokenize_attempt_to_consume_an_entity;        my $token = $self->_tokenize_attempt_to_consume_an_entity (0);
257    
258        $self->{state} = 'data';        $self->{state} = 'data';
259        # next-input-character is already done        # next-input-character is already done
# Line 569  sub _get_next_token ($) { Line 332  sub _get_next_token ($) {
332      } elsif ($self->{state} eq 'close tag open') {      } elsif ($self->{state} eq 'close tag open') {
333        if ($self->{content_model_flag} eq 'RCDATA' or        if ($self->{content_model_flag} eq 'RCDATA' or
334            $self->{content_model_flag} eq 'CDATA') {            $self->{content_model_flag} eq 'CDATA') {
335          my @next_char;          if (defined $self->{last_emitted_start_tag_name}) {
336          TAGNAME: for (my $i = 0; $i < length $self->{last_emitted_start_tag_name}; $i++) {            ## NOTE: <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>
337              my @next_char;
338              TAGNAME: for (my $i = 0; $i < length $self->{last_emitted_start_tag_name}; $i++) {
339                push @next_char, $self->{next_input_character};
340                my $c = ord substr ($self->{last_emitted_start_tag_name}, $i, 1);
341                my $C = 0x0061 <= $c && $c <= 0x007A ? $c - 0x0020 : $c;
342                if ($self->{next_input_character} == $c or $self->{next_input_character} == $C) {
343                  !!!next-input-character;
344                  next TAGNAME;
345                } else {
346                  $self->{next_input_character} = shift @next_char; # reconsume
347                  !!!back-next-input-character (@next_char);
348                  $self->{state} = 'data';
349    
350                  !!!emit ({type => 'character', data => '</'});
351      
352                  redo A;
353                }
354              }
355            push @next_char, $self->{next_input_character};            push @next_char, $self->{next_input_character};
356            my $c = ord substr ($self->{last_emitted_start_tag_name}, $i, 1);        
357            my $C = 0x0061 <= $c && $c <= 0x007A ? $c - 0x0020 : $c;            unless ($self->{next_input_character} == 0x0009 or # HT
358            if ($self->{next_input_character} == $c or $self->{next_input_character} == $C) {                    $self->{next_input_character} == 0x000A or # LF
359              !!!next-input-character;                    $self->{next_input_character} == 0x000B or # VT
360              next TAGNAME;                    $self->{next_input_character} == 0x000C or # FF
361            } else {                    $self->{next_input_character} == 0x0020 or # SP
362              !!!parse-error (type => 'unmatched end tag');                    $self->{next_input_character} == 0x003E or # >
363                      $self->{next_input_character} == 0x002F or # /
364                      $self->{next_input_character} == -1) {
365              $self->{next_input_character} = shift @next_char; # reconsume              $self->{next_input_character} = shift @next_char; # reconsume
366              !!!back-next-input-character (@next_char);              !!!back-next-input-character (@next_char);
367              $self->{state} = 'data';              $self->{state} = 'data';
   
368              !!!emit ({type => 'character', data => '</'});              !!!emit ({type => 'character', data => '</'});
   
369              redo A;              redo A;
370              } else {
371                $self->{next_input_character} = shift @next_char;
372                !!!back-next-input-character (@next_char);
373                # and consume...
374            }            }
375          }          } else {
376          push @next_char, $self->{next_input_character};            ## No start tag token has ever been emitted
377                  # next-input-character is already done
         unless ($self->{next_input_character} == 0x0009 or # HT  
                 $self->{next_input_character} == 0x000A or # LF  
                 $self->{next_input_character} == 0x000B or # VT  
                 $self->{next_input_character} == 0x000C or # FF  
                 $self->{next_input_character} == 0x0020 or # SP  
                 $self->{next_input_character} == 0x003E or # >  
                 $self->{next_input_character} == 0x002F or # /  
                 $self->{next_input_character} == 0x003C or # <  
                 $self->{next_input_character} == -1) {  
           !!!parse-error (type => 'unmatched end tag');  
           $self->{next_input_character} = shift @next_char; # reconsume  
           !!!back-next-input-character (@next_char);  
378            $self->{state} = 'data';            $self->{state} = 'data';
   
379            !!!emit ({type => 'character', data => '</'});            !!!emit ({type => 'character', data => '</'});
   
380            redo A;            redo A;
         } else {  
           $self->{next_input_character} = shift @next_char;  
           !!!back-next-input-character (@next_char);  
           # and consume...  
381          }          }
382        }        }
383                
# Line 658  sub _get_next_token ($) { Line 425  sub _get_next_token ($) {
425          redo A;          redo A;
426        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
427          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
428              $self->{current_token}->{first_start_tag}
429                  = not defined $self->{last_emitted_start_tag_name};
430            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
431          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
432            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 671  sub _get_next_token ($) { Line 440  sub _get_next_token ($) {
440          !!!next-input-character;          !!!next-input-character;
441    
442          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
443    
444          redo A;          redo A;
445        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{next_input_character} and
# Line 681  sub _get_next_token ($) { Line 449  sub _get_next_token ($) {
449          ## Stay in this state          ## Stay in this state
450          !!!next-input-character;          !!!next-input-character;
451          redo A;          redo A;
452        } elsif ($self->{next_input_character} == 0x003C or # <        } elsif ($self->{next_input_character} == -1) {
                $self->{next_input_character} == -1) {  
453          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
454          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
455              $self->{current_token}->{first_start_tag}
456                  = not defined $self->{last_emitted_start_tag_name};
457            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
458          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
459            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 698  sub _get_next_token ($) { Line 467  sub _get_next_token ($) {
467          # reconsume          # reconsume
468    
469          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
470    
471          redo A;          redo A;
472        } elsif ($self->{next_input_character} == 0x002F) { # /        } elsif ($self->{next_input_character} == 0x002F) { # /
# Line 732  sub _get_next_token ($) { Line 500  sub _get_next_token ($) {
500          redo A;          redo A;
501        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
502          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
503              $self->{current_token}->{first_start_tag}
504                  = not defined $self->{last_emitted_start_tag_name};
505            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
506          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
507            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 745  sub _get_next_token ($) { Line 515  sub _get_next_token ($) {
515          !!!next-input-character;          !!!next-input-character;
516    
517          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
518    
519          redo A;          redo A;
520        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{next_input_character} and
# Line 768  sub _get_next_token ($) { Line 537  sub _get_next_token ($) {
537          ## Stay in the state          ## Stay in the state
538          # next-input-character is already done          # next-input-character is already done
539          redo A;          redo A;
540        } elsif ($self->{next_input_character} == 0x003C or # <        } elsif ($self->{next_input_character} == -1) {
                $self->{next_input_character} == -1) {  
541          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
542          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
543              $self->{current_token}->{first_start_tag}
544                  = not defined $self->{last_emitted_start_tag_name};
545            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
546          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
547            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 785  sub _get_next_token ($) { Line 555  sub _get_next_token ($) {
555          # reconsume          # reconsume
556    
557          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
558    
559          redo A;          redo A;
560        } else {        } else {
# Line 824  sub _get_next_token ($) { Line 593  sub _get_next_token ($) {
593        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
594          $before_leave->();          $before_leave->();
595          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
596              $self->{current_token}->{first_start_tag}
597                  = not defined $self->{last_emitted_start_tag_name};
598            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
599          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
600            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 837  sub _get_next_token ($) { Line 608  sub _get_next_token ($) {
608          !!!next-input-character;          !!!next-input-character;
609    
610          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
611    
612          redo A;          redo A;
613        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{next_input_character} and
# Line 860  sub _get_next_token ($) { Line 630  sub _get_next_token ($) {
630          $self->{state} = 'before attribute name';          $self->{state} = 'before attribute name';
631          # next-input-character is already done          # next-input-character is already done
632          redo A;          redo A;
633        } elsif ($self->{next_input_character} == 0x003C or # <        } elsif ($self->{next_input_character} == -1) {
                $self->{next_input_character} == -1) {  
634          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
635          $before_leave->();          $before_leave->();
636          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
637              $self->{current_token}->{first_start_tag}
638                  = not defined $self->{last_emitted_start_tag_name};
639            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
640          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
641            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 878  sub _get_next_token ($) { Line 649  sub _get_next_token ($) {
649          # reconsume          # reconsume
650    
651          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
652    
653          redo A;          redo A;
654        } else {        } else {
# Line 902  sub _get_next_token ($) { Line 672  sub _get_next_token ($) {
672          redo A;          redo A;
673        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
674          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
675              $self->{current_token}->{first_start_tag}
676                  = not defined $self->{last_emitted_start_tag_name};
677            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
678          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
679            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 915  sub _get_next_token ($) { Line 687  sub _get_next_token ($) {
687          !!!next-input-character;          !!!next-input-character;
688    
689          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
690    
691          redo A;          redo A;
692        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{next_input_character} and
# Line 938  sub _get_next_token ($) { Line 709  sub _get_next_token ($) {
709          $self->{state} = 'before attribute name';          $self->{state} = 'before attribute name';
710          # next-input-character is already done          # next-input-character is already done
711          redo A;          redo A;
712        } elsif ($self->{next_input_character} == 0x003C or # <        } elsif ($self->{next_input_character} == -1) {
                $self->{next_input_character} == -1) {  
713          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
714          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
715              $self->{current_token}->{first_start_tag}
716                  = not defined $self->{last_emitted_start_tag_name};
717            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
718          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
719            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 955  sub _get_next_token ($) { Line 727  sub _get_next_token ($) {
727          # reconsume          # reconsume
728    
729          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
730    
731          redo A;          redo A;
732        } else {        } else {
# Line 988  sub _get_next_token ($) { Line 759  sub _get_next_token ($) {
759          redo A;          redo A;
760        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
761          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
762              $self->{current_token}->{first_start_tag}
763                  = not defined $self->{last_emitted_start_tag_name};
764            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
765          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
766            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 1001  sub _get_next_token ($) { Line 774  sub _get_next_token ($) {
774          !!!next-input-character;          !!!next-input-character;
775    
776          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
777    
778          redo A;          redo A;
779        } elsif ($self->{next_input_character} == 0x003C or # <        } elsif ($self->{next_input_character} == -1) {
                $self->{next_input_character} == -1) {  
780          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
781          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
782              $self->{current_token}->{first_start_tag}
783                  = not defined $self->{last_emitted_start_tag_name};
784            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
785          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
786            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 1021  sub _get_next_token ($) { Line 794  sub _get_next_token ($) {
794          ## reconsume          ## reconsume
795    
796          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
797    
798          redo A;          redo A;
799        } else {        } else {
# Line 1043  sub _get_next_token ($) { Line 815  sub _get_next_token ($) {
815        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
816          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
817          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
818              $self->{current_token}->{first_start_tag}
819                  = not defined $self->{last_emitted_start_tag_name};
820            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
821          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
822            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 1056  sub _get_next_token ($) { Line 830  sub _get_next_token ($) {
830          ## reconsume          ## reconsume
831    
832          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
833    
834          redo A;          redo A;
835        } else {        } else {
# Line 1078  sub _get_next_token ($) { Line 851  sub _get_next_token ($) {
851        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
852          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
853          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
854              $self->{current_token}->{first_start_tag}
855                  = not defined $self->{last_emitted_start_tag_name};
856            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
857          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
858            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 1091  sub _get_next_token ($) { Line 866  sub _get_next_token ($) {
866          ## reconsume          ## reconsume
867    
868          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
869    
870          redo A;          redo A;
871        } else {        } else {
# Line 1116  sub _get_next_token ($) { Line 890  sub _get_next_token ($) {
890          redo A;          redo A;
891        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
892          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
893              $self->{current_token}->{first_start_tag}
894                  = not defined $self->{last_emitted_start_tag_name};
895            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
896          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
897            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 1129  sub _get_next_token ($) { Line 905  sub _get_next_token ($) {
905          !!!next-input-character;          !!!next-input-character;
906    
907          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
908    
909          redo A;          redo A;
910        } elsif ($self->{next_input_character} == 0x003C or # <        } elsif ($self->{next_input_character} == -1) {
                $self->{next_input_character} == -1) {  
911          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
912          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
913              $self->{current_token}->{first_start_tag}
914                  = not defined $self->{last_emitted_start_tag_name};
915            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
916          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
917            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 1149  sub _get_next_token ($) { Line 925  sub _get_next_token ($) {
925          ## reconsume          ## reconsume
926    
927          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
928    
929          redo A;          redo A;
930        } else {        } else {
# Line 1159  sub _get_next_token ($) { Line 934  sub _get_next_token ($) {
934          redo A;          redo A;
935        }        }
936      } elsif ($self->{state} eq 'entity in attribute value') {      } elsif ($self->{state} eq 'entity in attribute value') {
937        my $token = $self->_tokenize_attempt_to_consume_an_entity;        my $token = $self->_tokenize_attempt_to_consume_an_entity (1);
938    
939        unless (defined $token) {        unless (defined $token) {
940          $self->{current_attribute}->{value} .= '&';          $self->{current_attribute}->{value} .= '&';
# Line 1208  sub _get_next_token ($) { Line 983  sub _get_next_token ($) {
983          push @next_char, $self->{next_input_character};          push @next_char, $self->{next_input_character};
984          if ($self->{next_input_character} == 0x002D) { # -          if ($self->{next_input_character} == 0x002D) { # -
985            $self->{current_token} = {type => 'comment', data => ''};            $self->{current_token} = {type => 'comment', data => ''};
986            $self->{state} = 'comment';            $self->{state} = 'comment start';
987            !!!next-input-character;            !!!next-input-character;
988            redo A;            redo A;
989          }          }
# Line 1250  sub _get_next_token ($) { Line 1025  sub _get_next_token ($) {
1025          }          }
1026        }        }
1027    
1028        !!!parse-error (type => 'bogus comment open');        !!!parse-error (type => 'bogus comment');
1029        $self->{next_input_character} = shift @next_char;        $self->{next_input_character} = shift @next_char;
1030        !!!back-next-input-character (@next_char);        !!!back-next-input-character (@next_char);
1031        $self->{state} = 'bogus comment';        $self->{state} = 'bogus comment';
# Line 1258  sub _get_next_token ($) { Line 1033  sub _get_next_token ($) {
1033                
1034        ## ISSUE: typos in spec: chacacters, is is a parse error        ## ISSUE: typos in spec: chacacters, is is a parse error
1035        ## ISSUE: spec is somewhat unclear on "is the first character that will be in the comment"; what is "that will be in the comment" is what the algorithm defines, isn't it?        ## ISSUE: spec is somewhat unclear on "is the first character that will be in the comment"; what is "that will be in the comment" is what the algorithm defines, isn't it?
1036        } elsif ($self->{state} eq 'comment start') {
1037          if ($self->{next_input_character} == 0x002D) { # -
1038            $self->{state} = 'comment start dash';
1039            !!!next-input-character;
1040            redo A;
1041          } elsif ($self->{next_input_character} == 0x003E) { # >
1042            !!!parse-error (type => 'bogus comment');
1043            $self->{state} = 'data';
1044            !!!next-input-character;
1045    
1046            !!!emit ($self->{current_token}); # comment
1047    
1048            redo A;
1049          } elsif ($self->{next_input_character} == -1) {
1050            !!!parse-error (type => 'unclosed comment');
1051            $self->{state} = 'data';
1052            ## reconsume
1053    
1054            !!!emit ($self->{current_token}); # comment
1055    
1056            redo A;
1057          } else {
1058            $self->{current_token}->{data} # comment
1059                .= chr ($self->{next_input_character});
1060            $self->{state} = 'comment';
1061            !!!next-input-character;
1062            redo A;
1063          }
1064        } elsif ($self->{state} eq 'comment start dash') {
1065          if ($self->{next_input_character} == 0x002D) { # -
1066            $self->{state} = 'comment end';
1067            !!!next-input-character;
1068            redo A;
1069          } elsif ($self->{next_input_character} == 0x003E) { # >
1070            !!!parse-error (type => 'bogus comment');
1071            $self->{state} = 'data';
1072            !!!next-input-character;
1073    
1074            !!!emit ($self->{current_token}); # comment
1075    
1076            redo A;
1077          } elsif ($self->{next_input_character} == -1) {
1078            !!!parse-error (type => 'unclosed comment');
1079            $self->{state} = 'data';
1080            ## reconsume
1081    
1082            !!!emit ($self->{current_token}); # comment
1083    
1084            redo A;
1085          } else {
1086            $self->{current_token}->{data} # comment
1087                .= chr ($self->{next_input_character});
1088            $self->{state} = 'comment';
1089            !!!next-input-character;
1090            redo A;
1091          }
1092      } elsif ($self->{state} eq 'comment') {      } elsif ($self->{state} eq 'comment') {
1093        if ($self->{next_input_character} == 0x002D) { # -        if ($self->{next_input_character} == 0x002D) { # -
1094          $self->{state} = 'comment dash';          $self->{state} = 'comment end dash';
1095          !!!next-input-character;          !!!next-input-character;
1096          redo A;          redo A;
1097        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
# Line 1269  sub _get_next_token ($) { Line 1100  sub _get_next_token ($) {
1100          ## reconsume          ## reconsume
1101    
1102          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{current_token}); # comment
         undef $self->{current_token};  
1103    
1104          redo A;          redo A;
1105        } else {        } else {
# Line 1278  sub _get_next_token ($) { Line 1108  sub _get_next_token ($) {
1108          !!!next-input-character;          !!!next-input-character;
1109          redo A;          redo A;
1110        }        }
1111      } elsif ($self->{state} eq 'comment dash') {      } elsif ($self->{state} eq 'comment end dash') {
1112        if ($self->{next_input_character} == 0x002D) { # -        if ($self->{next_input_character} == 0x002D) { # -
1113          $self->{state} = 'comment end';          $self->{state} = 'comment end';
1114          !!!next-input-character;          !!!next-input-character;
# Line 1289  sub _get_next_token ($) { Line 1119  sub _get_next_token ($) {
1119          ## reconsume          ## reconsume
1120    
1121          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{current_token}); # comment
         undef $self->{current_token};  
1122    
1123          redo A;          redo A;
1124        } else {        } else {
# Line 1304  sub _get_next_token ($) { Line 1133  sub _get_next_token ($) {
1133          !!!next-input-character;          !!!next-input-character;
1134    
1135          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{current_token}); # comment
         undef $self->{current_token};  
1136    
1137          redo A;          redo A;
1138        } elsif ($self->{next_input_character} == 0x002D) { # -        } elsif ($self->{next_input_character} == 0x002D) { # -
# Line 1319  sub _get_next_token ($) { Line 1147  sub _get_next_token ($) {
1147          ## reconsume          ## reconsume
1148    
1149          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{current_token}); # comment
         undef $self->{current_token};  
1150    
1151          redo A;          redo A;
1152        } else {        } else {
# Line 1353  sub _get_next_token ($) { Line 1180  sub _get_next_token ($) {
1180          ## Stay in the state          ## Stay in the state
1181          !!!next-input-character;          !!!next-input-character;
1182          redo A;          redo A;
       } elsif (0x0061 <= $self->{next_input_character} and  
                $self->{next_input_character} <= 0x007A) { # a..z  
 ## ISSUE: "Set the token's name name to the" in the spec  
         $self->{current_token} = {type => 'DOCTYPE',  
                           name => chr ($self->{next_input_character} - 0x0020),  
                           error => 1};  
         $self->{state} = 'DOCTYPE name';  
         !!!next-input-character;  
         redo A;  
1183        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
1184          !!!parse-error (type => 'no DOCTYPE name');          !!!parse-error (type => 'no DOCTYPE name');
1185          $self->{state} = 'data';          $self->{state} = 'data';
1186          !!!next-input-character;          !!!next-input-character;
1187    
1188          !!!emit ({type => 'DOCTYPE', name => '', error => 1});          !!!emit ({type => 'DOCTYPE'}); # incorrect
1189    
1190          redo A;          redo A;
1191        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
# Line 1375  sub _get_next_token ($) { Line 1193  sub _get_next_token ($) {
1193          $self->{state} = 'data';          $self->{state} = 'data';
1194          ## reconsume          ## reconsume
1195    
1196          !!!emit ({type => 'DOCTYPE', name => '', error => 1});          !!!emit ({type => 'DOCTYPE'}); # incorrect
1197    
1198          redo A;          redo A;
1199        } else {        } else {
1200          $self->{current_token} = {type => 'DOCTYPE',          $self->{current_token}
1201                            name => chr ($self->{next_input_character}),              = {type => 'DOCTYPE',
1202                            error => 1};                 name => chr ($self->{next_input_character}),
1203                   correct => 1};
1204  ## ISSUE: "Set the token's name name to the" in the spec  ## ISSUE: "Set the token's name name to the" in the spec
1205          $self->{state} = 'DOCTYPE name';          $self->{state} = 'DOCTYPE name';
1206          !!!next-input-character;          !!!next-input-character;
1207          redo A;          redo A;
1208        }        }
1209      } elsif ($self->{state} eq 'DOCTYPE name') {      } elsif ($self->{state} eq 'DOCTYPE name') {
1210    ## ISSUE: Redundant "First," in the spec.
1211        if ($self->{next_input_character} == 0x0009 or # HT        if ($self->{next_input_character} == 0x0009 or # HT
1212            $self->{next_input_character} == 0x000A or # LF            $self->{next_input_character} == 0x000A or # LF
1213            $self->{next_input_character} == 0x000B or # VT            $self->{next_input_character} == 0x000B or # VT
1214            $self->{next_input_character} == 0x000C or # FF            $self->{next_input_character} == 0x000C or # FF
1215            $self->{next_input_character} == 0x0020) { # SP            $self->{next_input_character} == 0x0020) { # SP
         $self->{current_token}->{error} = ($self->{current_token}->{name} ne 'HTML'); # DOCTYPE  
1216          $self->{state} = 'after DOCTYPE name';          $self->{state} = 'after DOCTYPE name';
1217          !!!next-input-character;          !!!next-input-character;
1218          redo A;          redo A;
1219        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
         $self->{current_token}->{error} = ($self->{current_token}->{name} ne 'HTML'); # DOCTYPE  
1220          $self->{state} = 'data';          $self->{state} = 'data';
1221          !!!next-input-character;          !!!next-input-character;
1222    
1223          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1224    
1225          redo A;          redo A;
       } elsif (0x0061 <= $self->{next_input_character} and  
                $self->{next_input_character} <= 0x007A) { # a..z  
         $self->{current_token}->{name} .= chr ($self->{next_input_character} - 0x0020); # DOCTYPE  
         #$self->{current_token}->{error} = ($self->{current_token}->{name} ne 'HTML');  
         ## Stay in the state  
         !!!next-input-character;  
         redo A;  
1226        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
1227          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
         $self->{current_token}->{error} = ($self->{current_token}->{name} ne 'HTML'); # DOCTYPE  
1228          $self->{state} = 'data';          $self->{state} = 'data';
1229          ## reconsume          ## reconsume
1230    
1231          !!!emit ($self->{current_token});          delete $self->{current_token}->{correct};
1232          undef $self->{current_token};          !!!emit ($self->{current_token}); # DOCTYPE
1233    
1234          redo A;          redo A;
1235        } else {        } else {
1236          $self->{current_token}->{name}          $self->{current_token}->{name}
1237            .= chr ($self->{next_input_character}); # DOCTYPE            .= chr ($self->{next_input_character}); # DOCTYPE
         #$self->{current_token}->{error} = ($self->{current_token}->{name} ne 'HTML');  
1238          ## Stay in the state          ## Stay in the state
1239          !!!next-input-character;          !!!next-input-character;
1240          redo A;          redo A;
# Line 1445  sub _get_next_token ($) { Line 1253  sub _get_next_token ($) {
1253          !!!next-input-character;          !!!next-input-character;
1254    
1255          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1256    
1257          redo A;          redo A;
1258        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
# Line 1453  sub _get_next_token ($) { Line 1260  sub _get_next_token ($) {
1260          $self->{state} = 'data';          $self->{state} = 'data';
1261          ## reconsume          ## reconsume
1262    
1263            delete $self->{current_token}->{correct};
1264            !!!emit ($self->{current_token}); # DOCTYPE
1265    
1266            redo A;
1267          } elsif ($self->{next_input_character} == 0x0050 or # P
1268                   $self->{next_input_character} == 0x0070) { # p
1269            !!!next-input-character;
1270            if ($self->{next_input_character} == 0x0055 or # U
1271                $self->{next_input_character} == 0x0075) { # u
1272              !!!next-input-character;
1273              if ($self->{next_input_character} == 0x0042 or # B
1274                  $self->{next_input_character} == 0x0062) { # b
1275                !!!next-input-character;
1276                if ($self->{next_input_character} == 0x004C or # L
1277                    $self->{next_input_character} == 0x006C) { # l
1278                  !!!next-input-character;
1279                  if ($self->{next_input_character} == 0x0049 or # I
1280                      $self->{next_input_character} == 0x0069) { # i
1281                    !!!next-input-character;
1282                    if ($self->{next_input_character} == 0x0043 or # C
1283                        $self->{next_input_character} == 0x0063) { # c
1284                      $self->{state} = 'before DOCTYPE public identifier';
1285                      !!!next-input-character;
1286                      redo A;
1287                    }
1288                  }
1289                }
1290              }
1291            }
1292    
1293            #
1294          } elsif ($self->{next_input_character} == 0x0053 or # S
1295                   $self->{next_input_character} == 0x0073) { # s
1296            !!!next-input-character;
1297            if ($self->{next_input_character} == 0x0059 or # Y
1298                $self->{next_input_character} == 0x0079) { # y
1299              !!!next-input-character;
1300              if ($self->{next_input_character} == 0x0053 or # S
1301                  $self->{next_input_character} == 0x0073) { # s
1302                !!!next-input-character;
1303                if ($self->{next_input_character} == 0x0054 or # T
1304                    $self->{next_input_character} == 0x0074) { # t
1305                  !!!next-input-character;
1306                  if ($self->{next_input_character} == 0x0045 or # E
1307                      $self->{next_input_character} == 0x0065) { # e
1308                    !!!next-input-character;
1309                    if ($self->{next_input_character} == 0x004D or # M
1310                        $self->{next_input_character} == 0x006D) { # m
1311                      $self->{state} = 'before DOCTYPE system identifier';
1312                      !!!next-input-character;
1313                      redo A;
1314                    }
1315                  }
1316                }
1317              }
1318            }
1319    
1320            #
1321          } else {
1322            !!!next-input-character;
1323            #
1324          }
1325    
1326          !!!parse-error (type => 'string after DOCTYPE name');
1327          $self->{state} = 'bogus DOCTYPE';
1328          # next-input-character is already done
1329          redo A;
1330        } elsif ($self->{state} eq 'before DOCTYPE public identifier') {
1331          if ({
1332                0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
1333                #0x000D => 1, # HT, LF, VT, FF, SP, CR
1334              }->{$self->{next_input_character}}) {
1335            ## Stay in the state
1336            !!!next-input-character;
1337            redo A;
1338          } elsif ($self->{next_input_character} eq 0x0022) { # "
1339            $self->{current_token}->{public_identifier} = ''; # DOCTYPE
1340            $self->{state} = 'DOCTYPE public identifier (double-quoted)';
1341            !!!next-input-character;
1342            redo A;
1343          } elsif ($self->{next_input_character} eq 0x0027) { # '
1344            $self->{current_token}->{public_identifier} = ''; # DOCTYPE
1345            $self->{state} = 'DOCTYPE public identifier (single-quoted)';
1346            !!!next-input-character;
1347            redo A;
1348          } elsif ($self->{next_input_character} eq 0x003E) { # >
1349            !!!parse-error (type => 'no PUBLIC literal');
1350    
1351            $self->{state} = 'data';
1352            !!!next-input-character;
1353    
1354            delete $self->{current_token}->{correct};
1355            !!!emit ($self->{current_token}); # DOCTYPE
1356    
1357            redo A;
1358          } elsif ($self->{next_input_character} == -1) {
1359            !!!parse-error (type => 'unclosed DOCTYPE');
1360    
1361            $self->{state} = 'data';
1362            ## reconsume
1363    
1364            delete $self->{current_token}->{correct};
1365            !!!emit ($self->{current_token}); # DOCTYPE
1366    
1367            redo A;
1368          } else {
1369            !!!parse-error (type => 'string after PUBLIC');
1370            $self->{state} = 'bogus DOCTYPE';
1371            !!!next-input-character;
1372            redo A;
1373          }
1374        } elsif ($self->{state} eq 'DOCTYPE public identifier (double-quoted)') {
1375          if ($self->{next_input_character} == 0x0022) { # "
1376            $self->{state} = 'after DOCTYPE public identifier';
1377            !!!next-input-character;
1378            redo A;
1379          } elsif ($self->{next_input_character} == -1) {
1380            !!!parse-error (type => 'unclosed PUBLIC literal');
1381    
1382            $self->{state} = 'data';
1383            ## reconsume
1384    
1385            delete $self->{current_token}->{correct};
1386            !!!emit ($self->{current_token}); # DOCTYPE
1387    
1388            redo A;
1389          } else {
1390            $self->{current_token}->{public_identifier} # DOCTYPE
1391                .= chr $self->{next_input_character};
1392            ## Stay in the state
1393            !!!next-input-character;
1394            redo A;
1395          }
1396        } elsif ($self->{state} eq 'DOCTYPE public identifier (single-quoted)') {
1397          if ($self->{next_input_character} == 0x0027) { # '
1398            $self->{state} = 'after DOCTYPE public identifier';
1399            !!!next-input-character;
1400            redo A;
1401          } elsif ($self->{next_input_character} == -1) {
1402            !!!parse-error (type => 'unclosed PUBLIC literal');
1403    
1404            $self->{state} = 'data';
1405            ## reconsume
1406    
1407            delete $self->{current_token}->{correct};
1408            !!!emit ($self->{current_token}); # DOCTYPE
1409    
1410            redo A;
1411          } else {
1412            $self->{current_token}->{public_identifier} # DOCTYPE
1413                .= chr $self->{next_input_character};
1414            ## Stay in the state
1415            !!!next-input-character;
1416            redo A;
1417          }
1418        } elsif ($self->{state} eq 'after DOCTYPE public identifier') {
1419          if ({
1420                0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
1421                #0x000D => 1, # HT, LF, VT, FF, SP, CR
1422              }->{$self->{next_input_character}}) {
1423            ## Stay in the state
1424            !!!next-input-character;
1425            redo A;
1426          } elsif ($self->{next_input_character} == 0x0022) { # "
1427            $self->{current_token}->{system_identifier} = ''; # DOCTYPE
1428            $self->{state} = 'DOCTYPE system identifier (double-quoted)';
1429            !!!next-input-character;
1430            redo A;
1431          } elsif ($self->{next_input_character} == 0x0027) { # '
1432            $self->{current_token}->{system_identifier} = ''; # DOCTYPE
1433            $self->{state} = 'DOCTYPE system identifier (single-quoted)';
1434            !!!next-input-character;
1435            redo A;
1436          } elsif ($self->{next_input_character} == 0x003E) { # >
1437            $self->{state} = 'data';
1438            !!!next-input-character;
1439    
1440            !!!emit ($self->{current_token}); # DOCTYPE
1441    
1442            redo A;
1443          } elsif ($self->{next_input_character} == -1) {
1444            !!!parse-error (type => 'unclosed DOCTYPE');
1445    
1446            $self->{state} = 'data';
1447            ## reconsume
1448    
1449            delete $self->{current_token}->{correct};
1450          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1451    
1452          redo A;          redo A;
1453        } else {        } else {
1454          !!!parse-error (type => 'string after DOCTYPE name');          !!!parse-error (type => 'string after PUBLIC literal');
1455          $self->{current_token}->{error} = 1; # DOCTYPE          $self->{state} = 'bogus DOCTYPE';
1456            !!!next-input-character;
1457            redo A;
1458          }
1459        } elsif ($self->{state} eq 'before DOCTYPE system identifier') {
1460          if ({
1461                0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
1462                #0x000D => 1, # HT, LF, VT, FF, SP, CR
1463              }->{$self->{next_input_character}}) {
1464            ## Stay in the state
1465            !!!next-input-character;
1466            redo A;
1467          } elsif ($self->{next_input_character} == 0x0022) { # "
1468            $self->{current_token}->{system_identifier} = ''; # DOCTYPE
1469            $self->{state} = 'DOCTYPE system identifier (double-quoted)';
1470            !!!next-input-character;
1471            redo A;
1472          } elsif ($self->{next_input_character} == 0x0027) { # '
1473            $self->{current_token}->{system_identifier} = ''; # DOCTYPE
1474            $self->{state} = 'DOCTYPE system identifier (single-quoted)';
1475            !!!next-input-character;
1476            redo A;
1477          } elsif ($self->{next_input_character} == 0x003E) { # >
1478            !!!parse-error (type => 'no SYSTEM literal');
1479            $self->{state} = 'data';
1480            !!!next-input-character;
1481    
1482            delete $self->{current_token}->{correct};
1483            !!!emit ($self->{current_token}); # DOCTYPE
1484    
1485            redo A;
1486          } elsif ($self->{next_input_character} == -1) {
1487            !!!parse-error (type => 'unclosed DOCTYPE');
1488    
1489            $self->{state} = 'data';
1490            ## reconsume
1491    
1492            delete $self->{current_token}->{correct};
1493            !!!emit ($self->{current_token}); # DOCTYPE
1494    
1495            redo A;
1496          } else {
1497            !!!parse-error (type => 'string after SYSTEM');
1498            $self->{state} = 'bogus DOCTYPE';
1499            !!!next-input-character;
1500            redo A;
1501          }
1502        } elsif ($self->{state} eq 'DOCTYPE system identifier (double-quoted)') {
1503          if ($self->{next_input_character} == 0x0022) { # "
1504            $self->{state} = 'after DOCTYPE system identifier';
1505            !!!next-input-character;
1506            redo A;
1507          } elsif ($self->{next_input_character} == -1) {
1508            !!!parse-error (type => 'unclosed SYSTEM literal');
1509    
1510            $self->{state} = 'data';
1511            ## reconsume
1512    
1513            delete $self->{current_token}->{correct};
1514            !!!emit ($self->{current_token}); # DOCTYPE
1515    
1516            redo A;
1517          } else {
1518            $self->{current_token}->{system_identifier} # DOCTYPE
1519                .= chr $self->{next_input_character};
1520            ## Stay in the state
1521            !!!next-input-character;
1522            redo A;
1523          }
1524        } elsif ($self->{state} eq 'DOCTYPE system identifier (single-quoted)') {
1525          if ($self->{next_input_character} == 0x0027) { # '
1526            $self->{state} = 'after DOCTYPE system identifier';
1527            !!!next-input-character;
1528            redo A;
1529          } elsif ($self->{next_input_character} == -1) {
1530            !!!parse-error (type => 'unclosed SYSTEM literal');
1531    
1532            $self->{state} = 'data';
1533            ## reconsume
1534    
1535            delete $self->{current_token}->{correct};
1536            !!!emit ($self->{current_token}); # DOCTYPE
1537    
1538            redo A;
1539          } else {
1540            $self->{current_token}->{system_identifier} # DOCTYPE
1541                .= chr $self->{next_input_character};
1542            ## Stay in the state
1543            !!!next-input-character;
1544            redo A;
1545          }
1546        } elsif ($self->{state} eq 'after DOCTYPE system identifier') {
1547          if ({
1548                0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
1549                #0x000D => 1, # HT, LF, VT, FF, SP, CR
1550              }->{$self->{next_input_character}}) {
1551            ## Stay in the state
1552            !!!next-input-character;
1553            redo A;
1554          } elsif ($self->{next_input_character} == 0x003E) { # >
1555            $self->{state} = 'data';
1556            !!!next-input-character;
1557    
1558            !!!emit ($self->{current_token}); # DOCTYPE
1559    
1560            redo A;
1561          } elsif ($self->{next_input_character} == -1) {
1562            !!!parse-error (type => 'unclosed DOCTYPE');
1563    
1564            $self->{state} = 'data';
1565            ## reconsume
1566    
1567            delete $self->{current_token}->{correct};
1568            !!!emit ($self->{current_token}); # DOCTYPE
1569    
1570            redo A;
1571          } else {
1572            !!!parse-error (type => 'string after SYSTEM literal');
1573          $self->{state} = 'bogus DOCTYPE';          $self->{state} = 'bogus DOCTYPE';
1574          !!!next-input-character;          !!!next-input-character;
1575          redo A;          redo A;
# Line 1469  sub _get_next_token ($) { Line 1579  sub _get_next_token ($) {
1579          $self->{state} = 'data';          $self->{state} = 'data';
1580          !!!next-input-character;          !!!next-input-character;
1581    
1582            delete $self->{current_token}->{correct};
1583          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1584    
1585          redo A;          redo A;
1586        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
# Line 1478  sub _get_next_token ($) { Line 1588  sub _get_next_token ($) {
1588          $self->{state} = 'data';          $self->{state} = 'data';
1589          ## reconsume          ## reconsume
1590    
1591            delete $self->{current_token}->{correct};
1592          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1593    
1594          redo A;          redo A;
1595        } else {        } else {
# Line 1495  sub _get_next_token ($) { Line 1605  sub _get_next_token ($) {
1605    die "$0: _get_next_token: unexpected case";    die "$0: _get_next_token: unexpected case";
1606  } # _get_next_token  } # _get_next_token
1607    
1608  sub _tokenize_attempt_to_consume_an_entity ($) {  sub _tokenize_attempt_to_consume_an_entity ($$) {
1609    my $self = shift;    my ($self, $in_attr) = @_;
1610      
1611    if ($self->{next_input_character} == 0x0023) { # #    if ({
1612           0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, # HT, LF, VT, FF,
1613           0x0020 => 1, 0x003C => 1, 0x0026 => 1, -1 => 1, # SP, <, & # 0x000D # CR
1614          }->{$self->{next_input_character}}) {
1615        ## Don't consume
1616        ## No error
1617        return undef;
1618      } elsif ($self->{next_input_character} == 0x0023) { # #
1619      !!!next-input-character;      !!!next-input-character;
1620      if ($self->{next_input_character} == 0x0078 or # x      if ($self->{next_input_character} == 0x0078 or # x
1621          $self->{next_input_character} == 0x0058) { # X          $self->{next_input_character} == 0x0058) { # X
1622        my $num;        my $code;
1623        X: {        X: {
1624          my $x_char = $self->{next_input_character};          my $x_char = $self->{next_input_character};
1625          !!!next-input-character;          !!!next-input-character;
1626          if (0x0030 <= $self->{next_input_character} and          if (0x0030 <= $self->{next_input_character} and
1627              $self->{next_input_character} <= 0x0039) { # 0..9              $self->{next_input_character} <= 0x0039) { # 0..9
1628            $num ||= 0;            $code ||= 0;
1629            $num *= 0x10;            $code *= 0x10;
1630            $num += $self->{next_input_character} - 0x0030;            $code += $self->{next_input_character} - 0x0030;
1631            redo X;            redo X;
1632          } elsif (0x0061 <= $self->{next_input_character} and          } elsif (0x0061 <= $self->{next_input_character} and
1633                   $self->{next_input_character} <= 0x0066) { # a..f                   $self->{next_input_character} <= 0x0066) { # a..f
1634            ## ISSUE: the spec says U+0078, which is apparently incorrect            $code ||= 0;
1635            $num ||= 0;            $code *= 0x10;
1636            $num *= 0x10;            $code += $self->{next_input_character} - 0x0060 + 9;
           $num += $self->{next_input_character} - 0x0060 + 9;  
1637            redo X;            redo X;
1638          } elsif (0x0041 <= $self->{next_input_character} and          } elsif (0x0041 <= $self->{next_input_character} and
1639                   $self->{next_input_character} <= 0x0046) { # A..F                   $self->{next_input_character} <= 0x0046) { # A..F
1640            ## ISSUE: the spec says U+0058, which is apparently incorrect            $code ||= 0;
1641            $num ||= 0;            $code *= 0x10;
1642            $num *= 0x10;            $code += $self->{next_input_character} - 0x0040 + 9;
           $num += $self->{next_input_character} - 0x0040 + 9;  
1643            redo X;            redo X;
1644          } elsif (not defined $num) { # no hexadecimal digit          } elsif (not defined $code) { # no hexadecimal digit
1645            !!!parse-error (type => 'bare hcro');            !!!parse-error (type => 'bare hcro');
1646            $self->{next_input_character} = 0x0023; # #            $self->{next_input_character} = 0x0023; # #
1647            !!!back-next-input-character ($x_char);            !!!back-next-input-character ($x_char);
# Line 1537  sub _tokenize_attempt_to_consume_an_enti Line 1652  sub _tokenize_attempt_to_consume_an_enti
1652            !!!parse-error (type => 'no refc');            !!!parse-error (type => 'no refc');
1653          }          }
1654    
1655          ## TODO: check the definition for |a valid Unicode character|.          if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {
1656          ## <http://lists.whatwg.org/pipermail/whatwg-whatwg.org/2006-December/thread.html#8189>            !!!parse-error (type => sprintf 'invalid character reference:U+%04X', $code);
1657          if ($num > 1114111 or $num == 0) {            $code = 0xFFFD;
1658            $num = 0xFFFD; # REPLACEMENT CHARACTER          } elsif ($code > 0x10FFFF) {
1659            ## ISSUE: Why this is not an error?            !!!parse-error (type => sprintf 'invalid character reference:U-%08X', $code);
1660          } elsif (0x80 <= $num and $num <= 0x9F) {            $code = 0xFFFD;
1661            !!!parse-error (type => sprintf 'c1 entity:U+%04X', $num);          } elsif ($code == 0x000D) {
1662            $num = $c1_entity_char->{$num};            !!!parse-error (type => 'CR character reference');
1663              $code = 0x000A;
1664            } elsif (0x80 <= $code and $code <= 0x9F) {
1665              !!!parse-error (type => sprintf 'C1 character reference:U+%04X', $code);
1666              $code = $c1_entity_char->{$code};
1667          }          }
1668    
1669          return {type => 'character', data => chr $num};          return {type => 'character', data => chr $code};
1670        } # X        } # X
1671      } elsif (0x0030 <= $self->{next_input_character} and      } elsif (0x0030 <= $self->{next_input_character} and
1672               $self->{next_input_character} <= 0x0039) { # 0..9               $self->{next_input_character} <= 0x0039) { # 0..9
# Line 1568  sub _tokenize_attempt_to_consume_an_enti Line 1687  sub _tokenize_attempt_to_consume_an_enti
1687          !!!parse-error (type => 'no refc');          !!!parse-error (type => 'no refc');
1688        }        }
1689    
1690        ## TODO: check the definition for |a valid Unicode character|.        if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {
1691        if ($code > 1114111 or $code == 0) {          !!!parse-error (type => sprintf 'invalid character reference:U+%04X', $code);
1692          $code = 0xFFFD; # REPLACEMENT CHARACTER          $code = 0xFFFD;
1693          ## ISSUE: Why this is not an error?        } elsif ($code > 0x10FFFF) {
1694            !!!parse-error (type => sprintf 'invalid character reference:U-%08X', $code);
1695            $code = 0xFFFD;
1696          } elsif ($code == 0x000D) {
1697            !!!parse-error (type => 'CR character reference');
1698            $code = 0x000A;
1699        } elsif (0x80 <= $code and $code <= 0x9F) {        } elsif (0x80 <= $code and $code <= 0x9F) {
1700          !!!parse-error (type => sprintf 'c1 entity:U+%04X', $code);          !!!parse-error (type => sprintf 'C1 character reference:U+%04X', $code);
1701          $code = $c1_entity_char->{$code};          $code = $c1_entity_char->{$code};
1702        }        }
1703                
# Line 1593  sub _tokenize_attempt_to_consume_an_enti Line 1717  sub _tokenize_attempt_to_consume_an_enti
1717    
1718      my $value = $entity_name;      my $value = $entity_name;
1719      my $match;      my $match;
1720        require Whatpm::_NamedEntityList;
1721        our $EntityChar;
1722    
1723      while (length $entity_name < 10 and      while (length $entity_name < 10 and
1724             ## NOTE: Some number greater than the maximum length of entity name             ## NOTE: Some number greater than the maximum length of entity name
1725             ((0x0041 <= $self->{next_input_character} and             ((0x0041 <= $self->{next_input_character} and # a
1726               $self->{next_input_character} <= 0x005A) or               $self->{next_input_character} <= 0x005A) or # x
1727              (0x0061 <= $self->{next_input_character} and              (0x0061 <= $self->{next_input_character} and # a
1728               $self->{next_input_character} <= 0x007A) or               $self->{next_input_character} <= 0x007A) or # z
1729              (0x0030 <= $self->{next_input_character} and              (0x0030 <= $self->{next_input_character} and # 0
1730               $self->{next_input_character} <= 0x0039))) {               $self->{next_input_character} <= 0x0039) or # 9
1731                $self->{next_input_character} == 0x003B)) { # ;
1732        $entity_name .= chr $self->{next_input_character};        $entity_name .= chr $self->{next_input_character};
1733        if (defined $entity_char->{$entity_name}) {        if (defined $EntityChar->{$entity_name}) {
1734          $value = $entity_char->{$entity_name};          if ($self->{next_input_character} == 0x003B) { # ;
1735          $match = 1;            $value = $EntityChar->{$entity_name};
1736              $match = 1;
1737              !!!next-input-character;
1738              last;
1739            } elsif (not $in_attr) {
1740              $value = $EntityChar->{$entity_name};
1741              $match = -1;
1742            } else {
1743              $value .= chr $self->{next_input_character};
1744            }
1745        } else {        } else {
1746          $value .= chr $self->{next_input_character};          $value .= chr $self->{next_input_character};
1747        }        }
1748        !!!next-input-character;        !!!next-input-character;
1749      }      }
1750            
1751      if ($match) {      if ($match > 0) {
1752        if ($self->{next_input_character} == 0x003B) { # ;        return {type => 'character', data => $value};
1753          !!!next-input-character;      } elsif ($match < 0) {
1754        } else {        !!!parse-error (type => 'no refc');
         !!!parse-error (type => 'refc');  
       }  
   
1755        return {type => 'character', data => $value};        return {type => 'character', data => $value};
1756      } else {      } else {
1757        !!!parse-error (type => 'bare ero');        !!!parse-error (type => 'bare ero');
1758        ## NOTE: No characters are consumed in the spec.        ## NOTE: No characters are consumed in the spec.
1759        !!!back-token ({type => 'character', data => $value});        return {type => 'character', data => '&'.$value};
       return undef;  
1760      }      }
1761    } else {    } else {
1762      ## no characters are consumed      ## no characters are consumed
# Line 1639  sub _initialize_tree_constructor ($) { Line 1771  sub _initialize_tree_constructor ($) {
1771    $self->{document}->strict_error_checking (0);    $self->{document}->strict_error_checking (0);
1772    ## TODO: Turn mutation events off # MUST    ## TODO: Turn mutation events off # MUST
1773    ## TODO: Turn loose Document option (manakai extension) on    ## TODO: Turn loose Document option (manakai extension) on
1774    ## TODO: Mark the Document as an HTML document # MUST    $self->{document}->manakai_is_html (1); # MUST
1775  } # _initialize_tree_constructor  } # _initialize_tree_constructor
1776    
1777  sub _terminate_tree_constructor ($) {  sub _terminate_tree_constructor ($) {
# Line 1679  sub _construct_tree ($) { Line 1811  sub _construct_tree ($) {
1811    
1812  sub _tree_construction_initial ($) {  sub _tree_construction_initial ($) {
1813    my $self = shift;    my $self = shift;
1814    B: {    INITIAL: {
1815        if ($token->{type} eq 'DOCTYPE') {      if ($token->{type} eq 'DOCTYPE') {
1816          if ($token->{error}) {        ## NOTE: Conformance checkers MAY, instead of reporting "not HTML5"
1817            ## ISSUE: Spec currently left this case undefined.        ## error, switch to a conformance checking mode for another
1818            !!!parse-error (type => 'bogus DOCTYPE');        ## language.
1819          }        my $doctype_name = $token->{name};
1820          my $doctype = $self->{document}->create_document_type_definition        $doctype_name = '' unless defined $doctype_name;
1821            ($token->{name});        $doctype_name =~ tr/a-z/A-Z/;
1822          $self->{document}->append_child ($doctype);        if (not defined $token->{name} or # <!DOCTYPE>
1823          #$phase = 'root element';            defined $token->{public_identifier} or
1824          !!!next-token;            defined $token->{system_identifier}) {
1825          #redo B;          !!!parse-error (type => 'not HTML5');
1826          return;        } elsif ($doctype_name ne 'HTML') {
1827        } elsif ({          ## ISSUE: ASCII case-insensitive? (in fact it does not matter)
1828                  comment => 1,          !!!parse-error (type => 'not HTML5');
1829                  'start tag' => 1,        }
1830                  'end tag' => 1,        
1831                  'end-of-file' => 1,        my $doctype = $self->{document}->create_document_type_definition
1832                 }->{$token->{type}}) {          ($token->{name}); ## ISSUE: If name is missing (e.g. <!DOCTYPE>)?
1833          ## ISSUE: Spec currently left this case undefined.        $doctype->public_id ($token->{public_identifier})
1834          !!!parse-error (type => 'missing DOCTYPE');            if defined $token->{public_identifier};
1835          #$phase = 'root element';        $doctype->system_id ($token->{system_identifier})
1836          ## reprocess            if defined $token->{system_identifier};
1837          #redo B;        ## NOTE: Other DocumentType attributes are null or empty lists.
1838          return;        ## ISSUE: internalSubset = null??
1839        } elsif ($token->{type} eq 'character') {        $self->{document}->append_child ($doctype);
1840          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {        
1841            $self->{document}->manakai_append_text ($1);        if (not $token->{correct} or $doctype_name ne 'HTML') {
1842            ## ISSUE: DOM3 Core does not allow Document > Text          $self->{document}->manakai_compat_mode ('quirks');
1843            unless (length $token->{data}) {        } elsif (defined $token->{public_identifier}) {
1844              ## Stay in the phase          my $pubid = $token->{public_identifier};
1845              !!!next-token;          $pubid =~ tr/a-z/A-z/;
1846              redo B;          if ({
1847              "+//SILMARIL//DTD HTML PRO V0R11 19970101//EN" => 1,
1848              "-//ADVASOFT LTD//DTD HTML 3.0 ASWEDIT + EXTENSIONS//EN" => 1,
1849              "-//AS//DTD HTML 3.0 ASWEDIT + EXTENSIONS//EN" => 1,
1850              "-//IETF//DTD HTML 2.0 LEVEL 1//EN" => 1,
1851              "-//IETF//DTD HTML 2.0 LEVEL 2//EN" => 1,
1852              "-//IETF//DTD HTML 2.0 STRICT LEVEL 1//EN" => 1,
1853              "-//IETF//DTD HTML 2.0 STRICT LEVEL 2//EN" => 1,
1854              "-//IETF//DTD HTML 2.0 STRICT//EN" => 1,
1855              "-//IETF//DTD HTML 2.0//EN" => 1,
1856              "-//IETF//DTD HTML 2.1E//EN" => 1,
1857              "-//IETF//DTD HTML 3.0//EN" => 1,
1858              "-//IETF//DTD HTML 3.0//EN//" => 1,
1859              "-//IETF//DTD HTML 3.2 FINAL//EN" => 1,
1860              "-//IETF//DTD HTML 3.2//EN" => 1,
1861              "-//IETF//DTD HTML 3//EN" => 1,
1862              "-//IETF//DTD HTML LEVEL 0//EN" => 1,
1863              "-//IETF//DTD HTML LEVEL 0//EN//2.0" => 1,
1864              "-//IETF//DTD HTML LEVEL 1//EN" => 1,
1865              "-//IETF//DTD HTML LEVEL 1//EN//2.0" => 1,
1866              "-//IETF//DTD HTML LEVEL 2//EN" => 1,
1867              "-//IETF//DTD HTML LEVEL 2//EN//2.0" => 1,
1868              "-//IETF//DTD HTML LEVEL 3//EN" => 1,
1869              "-//IETF//DTD HTML LEVEL 3//EN//3.0" => 1,
1870              "-//IETF//DTD HTML STRICT LEVEL 0//EN" => 1,
1871              "-//IETF//DTD HTML STRICT LEVEL 0//EN//2.0" => 1,
1872              "-//IETF//DTD HTML STRICT LEVEL 1//EN" => 1,
1873              "-//IETF//DTD HTML STRICT LEVEL 1//EN//2.0" => 1,
1874              "-//IETF//DTD HTML STRICT LEVEL 2//EN" => 1,
1875              "-//IETF//DTD HTML STRICT LEVEL 2//EN//2.0" => 1,
1876              "-//IETF//DTD HTML STRICT LEVEL 3//EN" => 1,
1877              "-//IETF//DTD HTML STRICT LEVEL 3//EN//3.0" => 1,
1878              "-//IETF//DTD HTML STRICT//EN" => 1,
1879              "-//IETF//DTD HTML STRICT//EN//2.0" => 1,
1880              "-//IETF//DTD HTML STRICT//EN//3.0" => 1,
1881              "-//IETF//DTD HTML//EN" => 1,
1882              "-//IETF//DTD HTML//EN//2.0" => 1,
1883              "-//IETF//DTD HTML//EN//3.0" => 1,
1884              "-//METRIUS//DTD METRIUS PRESENTATIONAL//EN" => 1,
1885              "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML STRICT//EN" => 1,
1886              "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 HTML//EN" => 1,
1887              "-//MICROSOFT//DTD INTERNET EXPLORER 2.0 TABLES//EN" => 1,
1888              "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML STRICT//EN" => 1,
1889              "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 HTML//EN" => 1,
1890              "-//MICROSOFT//DTD INTERNET EXPLORER 3.0 TABLES//EN" => 1,
1891              "-//NETSCAPE COMM. CORP.//DTD HTML//EN" => 1,
1892              "-//NETSCAPE COMM. CORP.//DTD STRICT HTML//EN" => 1,
1893              "-//O'REILLY AND ASSOCIATES//DTD HTML 2.0//EN" => 1,
1894              "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED 1.0//EN" => 1,
1895              "-//SPYGLASS//DTD HTML 2.0 EXTENDED//EN" => 1,
1896              "-//SQ//DTD HTML 2.0 HOTMETAL + EXTENSIONS//EN" => 1,
1897              "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA HTML//EN" => 1,
1898              "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA STRICT HTML//EN" => 1,
1899              "-//W3C//DTD HTML 3 1995-03-24//EN" => 1,
1900              "-//W3C//DTD HTML 3.2 DRAFT//EN" => 1,
1901              "-//W3C//DTD HTML 3.2 FINAL//EN" => 1,
1902              "-//W3C//DTD HTML 3.2//EN" => 1,
1903              "-//W3C//DTD HTML 3.2S DRAFT//EN" => 1,
1904              "-//W3C//DTD HTML 4.0 FRAMESET//EN" => 1,
1905              "-//W3C//DTD HTML 4.0 TRANSITIONAL//EN" => 1,
1906              "-//W3C//DTD HTML EXPERIMETNAL 19960712//EN" => 1,
1907              "-//W3C//DTD HTML EXPERIMENTAL 970421//EN" => 1,
1908              "-//W3C//DTD W3 HTML//EN" => 1,
1909              "-//W3O//DTD W3 HTML 3.0//EN" => 1,
1910              "-//W3O//DTD W3 HTML 3.0//EN//" => 1,
1911              "-//W3O//DTD W3 HTML STRICT 3.0//EN//" => 1,
1912              "-//WEBTECHS//DTD MOZILLA HTML 2.0//EN" => 1,
1913              "-//WEBTECHS//DTD MOZILLA HTML//EN" => 1,
1914              "-/W3C/DTD HTML 4.0 TRANSITIONAL/EN" => 1,
1915              "HTML" => 1,
1916            }->{$pubid}) {
1917              $self->{document}->manakai_compat_mode ('quirks');
1918            } elsif ($pubid eq "-//W3C//DTD HTML 4.01 FRAMESET//EN" or
1919                     $pubid eq "-//W3C//DTD HTML 4.01 TRANSITIONAL//EN") {
1920              if (defined $token->{system_identifier}) {
1921                $self->{document}->manakai_compat_mode ('quirks');
1922              } else {
1923                $self->{document}->manakai_compat_mode ('limited quirks');
1924            }            }
1925            } elsif ($pubid eq "-//W3C//DTD XHTML 1.0 Frameset//EN" or
1926                     $pubid eq "-//W3C//DTD XHTML 1.0 Transitional//EN") {
1927              $self->{document}->manakai_compat_mode ('limited quirks');
1928            }
1929          }
1930          if (defined $token->{system_identifier}) {
1931            my $sysid = $token->{system_identifier};
1932            $sysid =~ tr/A-Z/a-z/;
1933            if ($sysid eq "http://www.ibm.com/data/dtd/v11/ibmxhtml1-transitional.dtd") {
1934              $self->{document}->manakai_compat_mode ('quirks');
1935            }
1936          }
1937          
1938          ## Go to the root element phase.
1939          !!!next-token;
1940          return;
1941        } elsif ({
1942                  'start tag' => 1,
1943                  'end tag' => 1,
1944                  'end-of-file' => 1,
1945                 }->{$token->{type}}) {
1946          !!!parse-error (type => 'no DOCTYPE');
1947          $self->{document}->manakai_compat_mode ('quirks');
1948          ## Go to the root element phase
1949          ## reprocess
1950          return;
1951        } elsif ($token->{type} eq 'character') {
1952          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D
1953            ## Ignore the token
1954    
1955            unless (length $token->{data}) {
1956              ## Stay in the phase
1957              !!!next-token;
1958              redo INITIAL;
1959          }          }
         ## ISSUE: Spec currently left this case undefined.  
         !!!parse-error (type => 'missing DOCTYPE');  
         #$phase = 'root element';  
         ## reprocess  
         #redo B;  
         return;  
       } else {  
         die "$0: $token->{type}: Unknown token";  
1960        }        }
1961      } # B  
1962          !!!parse-error (type => 'no DOCTYPE');
1963          $self->{document}->manakai_compat_mode ('quirks');
1964          ## Go to the root element phase
1965          ## reprocess
1966          return;
1967        } elsif ($token->{type} eq 'comment') {
1968          my $comment = $self->{document}->create_comment ($token->{data});
1969          $self->{document}->append_child ($comment);
1970          
1971          ## Stay in the phase.
1972          !!!next-token;
1973          redo INITIAL;
1974        } else {
1975          die "$0: $token->{type}: Unknown token";
1976        }
1977      } # INITIAL
1978  } # _tree_construction_initial  } # _tree_construction_initial
1979    
1980  sub _tree_construction_root_element ($) {  sub _tree_construction_root_element ($) {
# Line 1743  sub _tree_construction_root_element ($) Line 1994  sub _tree_construction_root_element ($)
1994          !!!next-token;          !!!next-token;
1995          redo B;          redo B;
1996        } elsif ($token->{type} eq 'character') {        } elsif ($token->{type} eq 'character') {
1997          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D
1998            $self->{document}->manakai_append_text ($1);            ## Ignore the token.
1999            ## ISSUE: DOM3 Core does not allow Document > Text  
2000            unless (length $token->{data}) {            unless (length $token->{data}) {
2001              ## Stay in the phase              ## Stay in the phase
2002              !!!next-token;              !!!next-token;
# Line 1785  sub _reset_insertion_mode ($) { Line 2036  sub _reset_insertion_mode ($) {
2036            
2037      ## Step 3      ## Step 3
2038      S3: {      S3: {
2039        $last = 1 if $self->{open_elements}->[0]->[0] eq $node->[0];        ## ISSUE: Oops! "If node is the first node in the stack of open
2040        if (defined $self->{inner_html_node}) {        ## elements, then set last to true. If the context element of the
2041          if ($self->{inner_html_node}->[1] eq 'td' or        ## HTML fragment parsing algorithm is neither a td element nor a
2042              $self->{inner_html_node}->[1] eq 'th') {        ## th element, then set node to the context element. (fragment case)":
2043            #        ## The second "if" is in the scope of the first "if"!?
2044          } else {        if ($self->{open_elements}->[0]->[0] eq $node->[0]) {
2045            $node = $self->{inner_html_node};          $last = 1;
2046            if (defined $self->{inner_html_node}) {
2047              if ($self->{inner_html_node}->[1] eq 'td' or
2048                  $self->{inner_html_node}->[1] eq 'th') {
2049                #
2050              } else {
2051                $node = $self->{inner_html_node};
2052              }
2053          }          }
2054        }        }
2055            
# Line 1922  sub _tree_construction_main ($) { Line 2180  sub _tree_construction_main ($) {
2180      }      }
2181    }; # $clear_up_to_marker    }; # $clear_up_to_marker
2182    
2183    my $style_start_tag = sub {    my $parse_rcdata = sub ($$) {
2184      my $style_el; !!!create-element ($style_el, 'style', $token->{attributes});      my ($content_model_flag, $insert) = @_;
2185      ## $self->{insertion_mode} eq 'in head' and ... (always true)  
2186      (($self->{insertion_mode} eq 'in head' and defined $self->{head_element})      ## Step 1
2187       ? $self->{head_element} : $self->{open_elements}->[-1]->[0])      my $start_tag_name = $token->{tag_name};
2188        ->append_child ($style_el);      my $el;
2189      $self->{content_model_flag} = 'CDATA';      !!!create-element ($el, $start_tag_name, $token->{attributes});
2190                  
2191        ## Step 2
2192        $insert->($el); # /context node/->append_child ($el)
2193    
2194        ## Step 3
2195        $self->{content_model_flag} = $content_model_flag; # CDATA or RCDATA
2196        delete $self->{escape}; # MUST
2197    
2198        ## Step 4
2199      my $text = '';      my $text = '';
2200      !!!next-token;      !!!next-token;
2201      while ($token->{type} eq 'character') {      while ($token->{type} eq 'character') { # or until stop tokenizing
2202        $text .= $token->{data};        $text .= $token->{data};
2203        !!!next-token;        !!!next-token;
2204      } # stop if non-character token or tokenizer stops tokenising      }
2205    
2206        ## Step 5
2207      if (length $text) {      if (length $text) {
2208        $style_el->manakai_append_text ($text);        my $text = $self->{document}->create_text_node ($text);
2209          $el->append_child ($text);
2210      }      }
2211        
2212        ## Step 6
2213      $self->{content_model_flag} = 'PCDATA';      $self->{content_model_flag} = 'PCDATA';
2214                  
2215      if ($token->{type} eq 'end tag' and $token->{tag_name} eq 'style') {      ## Step 7
2216        if ($token->{type} eq 'end tag' and $token->{tag_name} eq $start_tag_name) {
2217        ## Ignore the token        ## Ignore the token
2218      } else {      } else {
2219        !!!parse-error (type => 'in CDATA:#'.$token->{type});        !!!parse-error (type => 'in '.$content_model_flag.':#'.$token->{type});
       ## ISSUE: And ignore?  
2220      }      }
2221      !!!next-token;      !!!next-token;
2222    }; # $style_start_tag    }; # $parse_rcdata
2223    
2224    my $script_start_tag = sub {    my $script_start_tag = sub ($) {
2225        my $insert = $_[0];
2226      my $script_el;      my $script_el;
2227      !!!create-element ($script_el, 'script', $token->{attributes});      !!!create-element ($script_el, 'script', $token->{attributes});
2228      ## TODO: mark as "parser-inserted"      ## TODO: mark as "parser-inserted"
2229    
2230      $self->{content_model_flag} = 'CDATA';      $self->{content_model_flag} = 'CDATA';
2231        delete $self->{escape}; # MUST
2232            
2233      my $text = '';      my $text = '';
2234      !!!next-token;      !!!next-token;
# Line 1984  sub _tree_construction_main ($) { Line 2256  sub _tree_construction_main ($) {
2256      } else {      } else {
2257        ## TODO: $old_insertion_point = current insertion point        ## TODO: $old_insertion_point = current insertion point
2258        ## TODO: insertion point = just before the next input character        ## TODO: insertion point = just before the next input character
2259          
2260        (($self->{insertion_mode} eq 'in head' and defined $self->{head_element})        $insert->($script_el);
        ? $self->{head_element} : $self->{open_elements}->[-1]->[0])->append_child ($script_el);  
2261                
2262        ## TODO: insertion point = $old_insertion_point (might be "undefined")        ## TODO: insertion point = $old_insertion_point (might be "undefined")
2263                
# Line 2180  sub _tree_construction_main ($) { Line 2451  sub _tree_construction_main ($) {
2451    }; # $formatting_end_tag    }; # $formatting_end_tag
2452    
2453    my $insert_to_current = sub {    my $insert_to_current = sub {
2454      $self->{open_elements}->[-1]->[0]->append_child (shift);      $self->{open_elements}->[-1]->[0]->append_child ($_[0]);
2455    }; # $insert_to_current    }; # $insert_to_current
2456    
2457    my $insert_to_foster = sub {    my $insert_to_foster = sub {
# Line 2218  sub _tree_construction_main ($) { Line 2489  sub _tree_construction_main ($) {
2489      my $insert = shift;      my $insert = shift;
2490      if ($token->{type} eq 'start tag') {      if ($token->{type} eq 'start tag') {
2491        if ($token->{tag_name} eq 'script') {        if ($token->{tag_name} eq 'script') {
2492          $script_start_tag->();          ## NOTE: This is an "as if in head" code clone
2493            $script_start_tag->($insert);
2494          return;          return;
2495        } elsif ($token->{tag_name} eq 'style') {        } elsif ($token->{tag_name} eq 'style') {
2496          $style_start_tag->();          ## NOTE: This is an "as if in head" code clone
2497            $parse_rcdata->('CDATA', $insert);
2498          return;          return;
2499        } elsif ({        } elsif ({
2500                  base => 1, link => 1, meta => 1,                  base => 1, link => 1, meta => 1,
2501                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
2502          !!!parse-error (type => 'in body:'.$token->{tag_name});          ## NOTE: This is an "as if in head" code clone, only "-t" differs
2503          ## NOTE: This is an "as if in head" code clone          !!!insert-element-t ($token->{tag_name}, $token->{attributes});
2504          my $el;          pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
         !!!create-element ($el, $token->{tag_name}, $token->{attributes});  
         if (defined $self->{head_element}) {  
           $self->{head_element}->append_child ($el);  
         } else {  
           $insert->($el);  
         }  
           
2505          !!!next-token;          !!!next-token;
2506            ## TODO: Extracting |charset| from |meta|.
2507          return;          return;
2508        } elsif ($token->{tag_name} eq 'title') {        } elsif ($token->{tag_name} eq 'title') {
2509          !!!parse-error (type => 'in body:title');          !!!parse-error (type => 'in body:title');
2510          ## NOTE: There is an "as if in head" code clone          ## NOTE: This is an "as if in head" code clone
2511          my $title_el;          $parse_rcdata->('RCDATA', sub {
2512          !!!create-element ($title_el, 'title', $token->{attributes});            if (defined $self->{head_element}) {
2513          (defined $self->{head_element} ? $self->{head_element} : $self->{open_elements}->[-1]->[0])              $self->{head_element}->append_child ($_[0]);
2514            ->append_child ($title_el);            } else {
2515          $self->{content_model_flag} = 'RCDATA';              $insert->($_[0]);
2516                      }
2517          my $text = '';          });
         !!!next-token;  
         while ($token->{type} eq 'character') {  
           $text .= $token->{data};  
           !!!next-token;  
         }  
         if (length $text) {  
           $title_el->manakai_append_text ($text);  
         }  
           
         $self->{content_model_flag} = 'PCDATA';  
           
         if ($token->{type} eq 'end tag' and  
             $token->{tag_name} eq 'title') {  
           ## Ignore the token  
         } else {  
           !!!parse-error (type => 'in RCDATA:#'.$token->{type});  
           ## ISSUE: And ignore?  
         }  
         !!!next-token;  
2518          return;          return;
2519        } elsif ($token->{tag_name} eq 'body') {        } elsif ($token->{tag_name} eq 'body') {
2520          !!!parse-error (type => 'in body:body');          !!!parse-error (type => 'in body:body');
# Line 2369  sub _tree_construction_main ($) { Line 2617  sub _tree_construction_main ($) {
2617              if ($i != -1) {              if ($i != -1) {
2618                !!!parse-error (type => 'end tag missing:'.                !!!parse-error (type => 'end tag missing:'.
2619                                $self->{open_elements}->[-1]->[1]);                                $self->{open_elements}->[-1]->[1]);
               ## TODO: test  
2620              }              }
2621              splice @{$self->{open_elements}}, $i;              splice @{$self->{open_elements}}, $i;
2622              last LI;              last LI;
# Line 2417  sub _tree_construction_main ($) { Line 2664  sub _tree_construction_main ($) {
2664              if ($i != -1) {              if ($i != -1) {
2665                !!!parse-error (type => 'end tag missing:'.                !!!parse-error (type => 'end tag missing:'.
2666                                $self->{open_elements}->[-1]->[1]);                                $self->{open_elements}->[-1]->[1]);
               ## TODO: test  
2667              }              }
2668              splice @{$self->{open_elements}}, $i;              splice @{$self->{open_elements}}, $i;
2669              last LI;              last LI;
# Line 2480  sub _tree_construction_main ($) { Line 2726  sub _tree_construction_main ($) {
2726            }            }
2727          } # INSCOPE          } # INSCOPE
2728                        
2729            ## NOTE: See <http://html5.org/tools/web-apps-tracker?from=925&to=926>
2730          ## has an element in scope          ## has an element in scope
2731          my $i;          #my $i;
2732          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          #INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
2733            my $node = $self->{open_elements}->[$_];          #  my $node = $self->{open_elements}->[$_];
2734            if ({          #  if ({
2735                 h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,          #       h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
2736                }->{$node->[1]}) {          #      }->{$node->[1]}) {
2737              $i = $_;          #    $i = $_;
2738              last INSCOPE;          #    last INSCOPE;
2739            } elsif ({          #  } elsif ({
2740                      table => 1, caption => 1, td => 1, th => 1,          #            table => 1, caption => 1, td => 1, th => 1,
2741                      button => 1, marquee => 1, object => 1, html => 1,          #            button => 1, marquee => 1, object => 1, html => 1,
2742                     }->{$node->[1]}) {          #           }->{$node->[1]}) {
2743              last INSCOPE;          #    last INSCOPE;
2744            }          #  }
2745          } # INSCOPE          #} # INSCOPE
2746                      #  
2747          if (defined $i) {          #if (defined $i) {
2748            !!!parse-error (type => 'in hn:hn');          #  !!! parse-error (type => 'in hn:hn');
2749            splice @{$self->{open_elements}}, $i;          #  splice @{$self->{open_elements}}, $i;
2750          }          #}
2751                        
2752          !!!insert-element-t ($token->{tag_name}, $token->{attributes});          !!!insert-element-t ($token->{tag_name}, $token->{attributes});
2753                        
# Line 2543  sub _tree_construction_main ($) { Line 2790  sub _tree_construction_main ($) {
2790          return;          return;
2791        } elsif ({        } elsif ({
2792                  b => 1, big => 1, em => 1, font => 1, i => 1,                  b => 1, big => 1, em => 1, font => 1, i => 1,
2793                  nobr => 1, s => 1, small => 1, strile => 1,                  s => 1, small => 1, strile => 1,
2794                  strong => 1, tt => 1, u => 1,                  strong => 1, tt => 1, u => 1,
2795                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
2796          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
# Line 2553  sub _tree_construction_main ($) { Line 2800  sub _tree_construction_main ($) {
2800                    
2801          !!!next-token;          !!!next-token;
2802          return;          return;
2803          } elsif ($token->{tag_name} eq 'nobr') {
2804            $reconstruct_active_formatting_elements->($insert_to_current);
2805    
2806            ## has a |nobr| element in scope
2807            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
2808              my $node = $self->{open_elements}->[$_];
2809              if ($node->[1] eq 'nobr') {
2810                !!!parse-error (type => 'not closed:nobr');
2811                !!!back-token;
2812                $token = {type => 'end tag', tag_name => 'nobr'};
2813                return;
2814              } elsif ({
2815                        table => 1, caption => 1, td => 1, th => 1,
2816                        button => 1, marquee => 1, object => 1, html => 1,
2817                       }->{$node->[1]}) {
2818                last INSCOPE;
2819              }
2820            } # INSCOPE
2821            
2822            !!!insert-element-t ($token->{tag_name}, $token->{attributes});
2823            push @$active_formatting_elements, $self->{open_elements}->[-1];
2824            
2825            !!!next-token;
2826            return;
2827        } elsif ($token->{tag_name} eq 'button') {        } elsif ($token->{tag_name} eq 'button') {
2828          ## has a button element in scope          ## has a button element in scope
2829          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
# Line 2588  sub _tree_construction_main ($) { Line 2859  sub _tree_construction_main ($) {
2859          return;          return;
2860        } elsif ($token->{tag_name} eq 'xmp') {        } elsif ($token->{tag_name} eq 'xmp') {
2861          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
2862                    $parse_rcdata->('CDATA', $insert);
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
           
         $self->{content_model_flag} = 'CDATA';  
           
         !!!next-token;  
2863          return;          return;
2864        } elsif ($token->{tag_name} eq 'table') {        } elsif ($token->{tag_name} eq 'table') {
2865          ## has a p element in scope          ## has a p element in scope
# Line 2625  sub _tree_construction_main ($) { Line 2891  sub _tree_construction_main ($) {
2891            !!!parse-error (type => 'image');            !!!parse-error (type => 'image');
2892            $token->{tag_name} = 'img';            $token->{tag_name} = 'img';
2893          }          }
2894            
2895            ## NOTE: There is an "as if <br>" code clone.
2896          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
2897                    
2898          !!!insert-element-t ($token->{tag_name}, $token->{attributes});          !!!insert-element-t ($token->{tag_name}, $token->{attributes});
# Line 2671  sub _tree_construction_main ($) { Line 2938  sub _tree_construction_main ($) {
2938            return;            return;
2939          } else {          } else {
2940            my $at = $token->{attributes};            my $at = $token->{attributes};
2941              my $form_attrs;
2942              $form_attrs->{action} = $at->{action} if $at->{action};
2943              my $prompt_attr = $at->{prompt};
2944            $at->{name} = {name => 'name', value => 'isindex'};            $at->{name} = {name => 'name', value => 'isindex'};
2945              delete $at->{action};
2946              delete $at->{prompt};
2947            my @tokens = (            my @tokens = (
2948                          {type => 'start tag', tag_name => 'form'},                          {type => 'start tag', tag_name => 'form',
2949                             attributes => $form_attrs},
2950                          {type => 'start tag', tag_name => 'hr'},                          {type => 'start tag', tag_name => 'hr'},
2951                          {type => 'start tag', tag_name => 'p'},                          {type => 'start tag', tag_name => 'p'},
2952                          {type => 'start tag', tag_name => 'label'},                          {type => 'start tag', tag_name => 'label'},
2953                          {type => 'character',                         );
2954                           data => 'This is a searchable index. Insert your search keywords here: '}, # SHOULD            if ($prompt_attr) {
2955                          ## TODO: make this configurable              push @tokens, {type => 'character', data => $prompt_attr->{value}};
2956              } else {
2957                push @tokens, {type => 'character',
2958                               data => 'This is a searchable index. Insert your search keywords here: '}; # SHOULD
2959                ## TODO: make this configurable
2960              }
2961              push @tokens,
2962                          {type => 'start tag', tag_name => 'input', attributes => $at},                          {type => 'start tag', tag_name => 'input', attributes => $at},
2963                          #{type => 'character', data => ''}, # SHOULD                          #{type => 'character', data => ''}, # SHOULD
2964                          {type => 'end tag', tag_name => 'label'},                          {type => 'end tag', tag_name => 'label'},
2965                          {type => 'end tag', tag_name => 'p'},                          {type => 'end tag', tag_name => 'p'},
2966                          {type => 'start tag', tag_name => 'hr'},                          {type => 'start tag', tag_name => 'hr'},
2967                          {type => 'end tag', tag_name => 'form'},                          {type => 'end tag', tag_name => 'form'};
                        );  
2968            $token = shift @tokens;            $token = shift @tokens;
2969            !!!back-token (@tokens);            !!!back-token (@tokens);
2970            return;            return;
2971          }          }
2972        } elsif ({        } elsif ($token->{tag_name} eq 'textarea') {
                 textarea => 1,  
                 iframe => 1,  
                 noembed => 1,  
                 noframes => 1,  
                 noscript => 0, ## TODO: 1 if scripting is enabled  
                }->{$token->{tag_name}}) {  
2973          my $tag_name = $token->{tag_name};          my $tag_name = $token->{tag_name};
2974          my $el;          my $el;
2975          !!!create-element ($el, $token->{tag_name}, $token->{attributes});          !!!create-element ($el, $token->{tag_name}, $token->{attributes});
2976                    
2977          if ($token->{tag_name} eq 'textarea') {          ## TODO: $self->{form_element} if defined
2978            ## TODO: $self->{form_element} if defined          $self->{content_model_flag} = 'RCDATA';
2979            ## TODO: ignore first LF <http://html5.org/tools/web-apps-tracker?from=866&to=867>          delete $self->{escape}; # MUST
           $self->{content_model_flag} = 'RCDATA';  
         } else {  
           $self->{content_model_flag} = 'CDATA';  
         }  
2980                    
2981          $insert->($el);          $insert->($el);
2982                    
2983          my $text = '';          my $text = '';
2984          !!!next-token;          !!!next-token;
2985            if ($token->{type} eq 'character') {
2986              $token->{data} =~ s/^\x0A//;
2987              unless (length $token->{data}) {
2988                !!!next-token;
2989              }
2990            }
2991          while ($token->{type} eq 'character') {          while ($token->{type} eq 'character') {
2992            $text .= $token->{data};            $text .= $token->{data};
2993            !!!next-token;            !!!next-token;
# Line 2728  sub _tree_construction_main ($) { Line 3002  sub _tree_construction_main ($) {
3002              $token->{tag_name} eq $tag_name) {              $token->{tag_name} eq $tag_name) {
3003            ## Ignore the token            ## Ignore the token
3004          } else {          } else {
3005            if ($token->{tag_name} eq 'textarea') { ## TODO: This is incorrect maybe            !!!parse-error (type => 'in RCDATA:#'.$token->{type});
 ## TODO: <http://html5.org/tools/web-apps-tracker?from=866&to=867>  
             !!!parse-error (type => 'in CDATA:#'.$token->{type});  
           } else {  
             !!!parse-error (type => 'in RCDATA:#'.$token->{type});  
           }  
           ## ISSUE: And ignore?  
3006          }          }
3007          !!!next-token;          !!!next-token;
3008          return;          return;
3009          } elsif ({
3010                    iframe => 1,
3011                    noembed => 1,
3012                    noframes => 1,
3013                    noscript => 0, ## TODO: 1 if scripting is enabled
3014                   }->{$token->{tag_name}}) {
3015            $parse_rcdata->('CDATA', $insert);
3016            return;
3017        } elsif ($token->{tag_name} eq 'select') {        } elsif ($token->{tag_name} eq 'select') {
3018          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
3019                    
# Line 2768  sub _tree_construction_main ($) { Line 3044  sub _tree_construction_main ($) {
3044        }        }
3045      } elsif ($token->{type} eq 'end tag') {      } elsif ($token->{type} eq 'end tag') {
3046        if ($token->{tag_name} eq 'body') {        if ($token->{tag_name} eq 'body') {
3047          if (@{$self->{open_elements}} > 1 and $self->{open_elements}->[1]->[1] eq 'body') {          if (@{$self->{open_elements}} > 1 and
3048            ## ISSUE: There is an issue in the spec.              $self->{open_elements}->[1]->[1] eq 'body') {
3049            if ($self->{open_elements}->[-1]->[1] ne 'body') {            for (@{$self->{open_elements}}) {
3050              !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);              unless ({
3051                           dd => 1, dt => 1, li => 1, p => 1, td => 1,
3052                           th => 1, tr => 1, body => 1, html => 1,
3053                         tbody => 1, tfoot => 1, thead => 1,
3054                        }->{$_->[1]}) {
3055                  !!!parse-error (type => 'not closed:'.$_->[1]);
3056                }
3057            }            }
3058    
3059            $self->{insertion_mode} = 'after body';            $self->{insertion_mode} = 'after body';
3060            !!!next-token;            !!!next-token;
3061            return;            return;
# Line 2801  sub _tree_construction_main ($) { Line 3084  sub _tree_construction_main ($) {
3084                  address => 1, blockquote => 1, center => 1, dir => 1,                  address => 1, blockquote => 1, center => 1, dir => 1,
3085                  div => 1, dl => 1, fieldset => 1, listing => 1,                  div => 1, dl => 1, fieldset => 1, listing => 1,
3086                  menu => 1, ol => 1, pre => 1, ul => 1,                  menu => 1, ol => 1, pre => 1, ul => 1,
                 form => 1,  
3087                  p => 1,                  p => 1,
3088                  dd => 1, dt => 1, li => 1,                  dd => 1, dt => 1, li => 1,
3089                  button => 1, marquee => 1, object => 1,                  button => 1, marquee => 1, object => 1,
# Line 2818  sub _tree_construction_main ($) { Line 3100  sub _tree_construction_main ($) {
3100                   li => ($token->{tag_name} ne 'li'),                   li => ($token->{tag_name} ne 'li'),
3101                   p => ($token->{tag_name} ne 'p'),                   p => ($token->{tag_name} ne 'p'),
3102                   td => 1, th => 1, tr => 1,                   td => 1, th => 1, tr => 1,
3103                     tbody => 1, tfoot=> 1, thead => 1,
3104                  }->{$self->{open_elements}->[-1]->[1]}) {                  }->{$self->{open_elements}->[-1]->[1]}) {
3105                !!!back-token;                !!!back-token;
3106                $token = {type => 'end tag',                $token = {type => 'end tag',
# Line 2838  sub _tree_construction_main ($) { Line 3121  sub _tree_construction_main ($) {
3121            !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);            !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
3122          }          }
3123                    
3124          splice @{$self->{open_elements}}, $i if defined $i;          if (defined $i) {
3125          undef $self->{form_element} if $token->{tag_name} eq 'form';            splice @{$self->{open_elements}}, $i;
3126            } elsif ($token->{tag_name} eq 'p') {
3127              ## As if <p>, then reprocess the current token
3128              my $el;
3129              !!!create-element ($el, 'p');
3130              $insert->($el);
3131            }
3132          $clear_up_to_marker->()          $clear_up_to_marker->()
3133            if {            if {
3134              button => 1, marquee => 1, object => 1,              button => 1, marquee => 1, object => 1,
3135            }->{$token->{tag_name}};            }->{$token->{tag_name}};
3136          !!!next-token;          !!!next-token;
3137          return;          return;
3138          } elsif ($token->{tag_name} eq 'form') {
3139            ## has an element in scope
3140            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
3141              my $node = $self->{open_elements}->[$_];
3142              if ($node->[1] eq $token->{tag_name}) {
3143                ## generate implied end tags
3144                if ({
3145                     dd => 1, dt => 1, li => 1, p => 1,
3146                     td => 1, th => 1, tr => 1,
3147                     tbody => 1, tfoot=> 1, thead => 1,
3148                    }->{$self->{open_elements}->[-1]->[1]}) {
3149                  !!!back-token;
3150                  $token = {type => 'end tag',
3151                            tag_name => $self->{open_elements}->[-1]->[1]}; # MUST
3152                  return;
3153                }
3154                last INSCOPE;
3155              } elsif ({
3156                        table => 1, caption => 1, td => 1, th => 1,
3157                        button => 1, marquee => 1, object => 1, html => 1,
3158                       }->{$node->[1]}) {
3159                last INSCOPE;
3160              }
3161            } # INSCOPE
3162            
3163            if ($self->{open_elements}->[-1]->[1] eq $token->{tag_name}) {
3164              pop @{$self->{open_elements}};
3165            } else {
3166              !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
3167            }
3168    
3169            undef $self->{form_element};
3170            !!!next-token;
3171            return;
3172        } elsif ({        } elsif ({
3173                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,                  h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
3174                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
# Line 2860  sub _tree_construction_main ($) { Line 3183  sub _tree_construction_main ($) {
3183              if ({              if ({
3184                   dd => 1, dt => 1, li => 1, p => 1,                   dd => 1, dt => 1, li => 1, p => 1,
3185                   td => 1, th => 1, tr => 1,                   td => 1, th => 1, tr => 1,
3186                     tbody => 1, tfoot=> 1, thead => 1,
3187                  }->{$self->{open_elements}->[-1]->[1]}) {                  }->{$self->{open_elements}->[-1]->[1]}) {
3188                !!!back-token;                !!!back-token;
3189                $token = {type => 'end tag',                $token = {type => 'end tag',
# Line 2890  sub _tree_construction_main ($) { Line 3214  sub _tree_construction_main ($) {
3214                  strong => 1, tt => 1, u => 1,                  strong => 1, tt => 1, u => 1,
3215                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
3216          $formatting_end_tag->($token->{tag_name});          $formatting_end_tag->($token->{tag_name});
3217  ## TODO: <http://html5.org/tools/web-apps-tracker?from=883&to=884>          return;
3218          } elsif ($token->{tag_name} eq 'br') {
3219            !!!parse-error (type => 'unmatched end tag:br');
3220    
3221            ## As if <br>
3222            $reconstruct_active_formatting_elements->($insert_to_current);
3223            
3224            my $el;
3225            !!!create-element ($el, 'br');
3226            $insert->($el);
3227            
3228            ## Ignore the token.
3229            !!!next-token;
3230          return;          return;
3231        } elsif ({        } elsif ({
3232                  caption => 1, col => 1, colgroup => 1, frame => 1,                  caption => 1, col => 1, colgroup => 1, frame => 1,
3233                  frameset => 1, head => 1, option => 1, optgroup => 1,                  frameset => 1, head => 1, option => 1, optgroup => 1,
3234                  tbody => 1, td => 1, tfoot => 1, th => 1,                  tbody => 1, td => 1, tfoot => 1, th => 1,
3235                  thead => 1, tr => 1,                  thead => 1, tr => 1,
3236                  area => 1, basefont => 1, bgsound => 1, br => 1,                  area => 1, basefont => 1, bgsound => 1,
3237                  embed => 1, hr => 1, iframe => 1, image => 1,                  embed => 1, hr => 1, iframe => 1, image => 1,
3238                  img => 1, input => 1, isindex => 1, noembed => 1,                  img => 1, input => 1, isindex => 1, noembed => 1,
3239                  noframes => 1, param => 1, select => 1, spacer => 1,                  noframes => 1, param => 1, select => 1, spacer => 1,
# Line 2924  sub _tree_construction_main ($) { Line 3260  sub _tree_construction_main ($) {
3260              if ({              if ({
3261                   dd => 1, dt => 1, li => 1, p => 1,                   dd => 1, dt => 1, li => 1, p => 1,
3262                   td => 1, th => 1, tr => 1,                   td => 1, th => 1, tr => 1,
3263                     tbody => 1, tfoot=> 1, thead => 1,
3264                  }->{$self->{open_elements}->[-1]->[1]}) {                  }->{$self->{open_elements}->[-1]->[1]}) {
3265                !!!back-token;                !!!back-token;
3266                $token = {type => 'end tag',                $token = {type => 'end tag',
# Line 2947  sub _tree_construction_main ($) { Line 3284  sub _tree_construction_main ($) {
3284                  #not $phrasing_category->{$node->[1]} and                  #not $phrasing_category->{$node->[1]} and
3285                  ($special_category->{$node->[1]} or                  ($special_category->{$node->[1]} or
3286                   $scoping_category->{$node->[1]})) {                   $scoping_category->{$node->[1]})) {
3287                !!!parse-error (type => 'not closed:'.$node->[1]);                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3288                ## Ignore the token                ## Ignore the token
3289                !!!next-token;                !!!next-token;
3290                last S2;                last S2;
# Line 2976  sub _tree_construction_main ($) { Line 3313  sub _tree_construction_main ($) {
3313          redo B;          redo B;
3314        } elsif ($token->{type} eq 'start tag' and        } elsif ($token->{type} eq 'start tag' and
3315                 $token->{tag_name} eq 'html') {                 $token->{tag_name} eq 'html') {
3316          ## TODO: unless it is the first start tag token, parse-error  ## ISSUE: "aa<html>" is not a parse error.
3317    ## ISSUE: "<html>" in fragment is not a parse error.
3318            unless ($token->{first_start_tag}) {
3319              !!!parse-error (type => 'not first start tag');
3320            }
3321          my $top_el = $self->{open_elements}->[0]->[0];          my $top_el = $self->{open_elements}->[0]->[0];
3322          for my $attr_name (keys %{$token->{attributes}}) {          for my $attr_name (keys %{$token->{attributes}}) {
3323            unless ($top_el->has_attribute_ns (undef, $attr_name)) {            unless ($top_el->has_attribute_ns (undef, $attr_name)) {
# Line 2991  sub _tree_construction_main ($) { Line 3332  sub _tree_construction_main ($) {
3332          ## Generate implied end tags          ## Generate implied end tags
3333          if ({          if ({
3334               dd => 1, dt => 1, li => 1, p => 1, td => 1, th => 1, tr => 1,               dd => 1, dt => 1, li => 1, p => 1, td => 1, th => 1, tr => 1,
3335                 tbody => 1, tfoot=> 1, thead => 1,
3336              }->{$self->{open_elements}->[-1]->[1]}) {              }->{$self->{open_elements}->[-1]->[1]}) {
3337            !!!back-token;            !!!back-token;
3338            $token = {type => 'end tag', tag_name => $self->{open_elements}->[-1]->[1]};            $token = {type => 'end tag', tag_name => $self->{open_elements}->[-1]->[1]};
# Line 3050  sub _tree_construction_main ($) { Line 3392  sub _tree_construction_main ($) {
3392              }              }
3393              redo B;              redo B;
3394            } elsif ($token->{type} eq 'end tag') {            } elsif ($token->{type} eq 'end tag') {
3395              if ($token->{tag_name} eq 'html') {              if ({
3396                     head => 1, body => 1, html => 1,
3397                     p => 1, br => 1,
3398                    }->{$token->{tag_name}}) {
3399                ## As if <head>                ## As if <head>
3400                !!!create-element ($self->{head_element}, 'head');                !!!create-element ($self->{head_element}, 'head');
3401                $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});                $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
# Line 3060  sub _tree_construction_main ($) { Line 3405  sub _tree_construction_main ($) {
3405                redo B;                redo B;
3406              } else {              } else {
3407                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3408                ## Ignore the token                ## Ignore the token ## ISSUE: An issue in the spec.
3409                !!!next-token;                !!!next-token;
3410                redo B;                redo B;
3411              }              }
3412            } else {            } else {
3413              die "$0: $token->{type}: Unknown type";              die "$0: $token->{type}: Unknown type";
3414            }            }
3415          } elsif ($self->{insertion_mode} eq 'in head') {          } elsif ($self->{insertion_mode} eq 'in head' or
3416                     $self->{insertion_mode} eq 'in head noscript' or
3417                     $self->{insertion_mode} eq 'after head') {
3418            if ($token->{type} eq 'character') {            if ($token->{type} eq 'character') {
3419              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {
3420                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
# Line 3084  sub _tree_construction_main ($) { Line 3431  sub _tree_construction_main ($) {
3431              !!!next-token;              !!!next-token;
3432              redo B;              redo B;
3433            } elsif ($token->{type} eq 'start tag') {            } elsif ($token->{type} eq 'start tag') {
3434              if ($token->{tag_name} eq 'title') {              if ({base => ($self->{insertion_mode} eq 'in head' or
3435                ## NOTE: There is an "as if in head" code clone                            $self->{insertion_mode} eq 'after head'),
3436                my $title_el;                   link => 1, meta => 1}->{$token->{tag_name}}) {
3437                !!!create-element ($title_el, 'title', $token->{attributes});                ## NOTE: There is a "as if in head" code clone.
3438                (defined $self->{head_element} ? $self->{head_element} : $self->{open_elements}->[-1]->[0])                if ($self->{insertion_mode} eq 'after head') {
3439                  ->append_child ($title_el);                  !!!parse-error (type => 'after head:'.$token->{tag_name});
3440                $self->{content_model_flag} = 'RCDATA';                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
3441                  }
3442                my $text = '';                !!!insert-element ($token->{tag_name}, $token->{attributes});
3443                  pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
3444                  ## TODO: Extracting |charset| from |meta|.
3445                  pop @{$self->{open_elements}}
3446                      if $self->{insertion_mode} eq 'after head';
3447                !!!next-token;                !!!next-token;
3448                while ($token->{type} eq 'character') {                redo B;
3449                  $text .= $token->{data};              } elsif ($token->{tag_name} eq 'title' and
3450                         $self->{insertion_mode} eq 'in head') {
3451                  ## NOTE: There is a "as if in head" code clone.
3452                  if ($self->{insertion_mode} eq 'after head') {
3453                    !!!parse-error (type => 'after head:'.$token->{tag_name});
3454                    push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
3455                  }
3456                  my $parent = defined $self->{head_element} ? $self->{head_element}
3457                      : $self->{open_elements}->[-1]->[0];
3458                  $parse_rcdata->('RCDATA', sub { $parent->append_child ($_[0]) });
3459                  pop @{$self->{open_elements}}
3460                      if $self->{insertion_mode} eq 'after head';
3461                  redo B;
3462                } elsif ($token->{tag_name} eq 'style') {
3463                  ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and
3464                  ## insertion mode 'in head')
3465                  ## NOTE: There is a "as if in head" code clone.
3466                  if ($self->{insertion_mode} eq 'after head') {
3467                    !!!parse-error (type => 'after head:'.$token->{tag_name});
3468                    push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
3469                  }
3470                  $parse_rcdata->('CDATA', $insert_to_current);
3471                  pop @{$self->{open_elements}}
3472                      if $self->{insertion_mode} eq 'after head';
3473                  redo B;
3474                } elsif ($token->{tag_name} eq 'noscript') {
3475                  if ($self->{insertion_mode} eq 'in head') {
3476                    ## NOTE: and scripting is disalbed
3477                    !!!insert-element ($token->{tag_name}, $token->{attributes});
3478                    $self->{insertion_mode} = 'in head noscript';
3479                  !!!next-token;                  !!!next-token;
3480                }                  redo B;
3481                if (length $text) {                } elsif ($self->{insertion_mode} eq 'in head noscript') {
3482                  $title_el->manakai_append_text ($text);                  !!!parse-error (type => 'in noscript:noscript');
               }  
                 
               $self->{content_model_flag} = 'PCDATA';  
                 
               if ($token->{type} eq 'end tag' and  
                   $token->{tag_name} eq 'title') {  
3483                  ## Ignore the token                  ## Ignore the token
3484                    redo B;
3485                } else {                } else {
3486                  !!!parse-error (type => 'in RCDATA:#'.$token->{type});                  #
                 ## ISSUE: And ignore?  
3487                }                }
3488                } elsif ($token->{tag_name} eq 'head' and
3489                         $self->{insertion_mode} ne 'after head') {
3490                  !!!parse-error (type => 'in head:head'); # or in head noscript
3491                  ## Ignore the token
3492                !!!next-token;                !!!next-token;
3493                redo B;                redo B;
3494              } elsif ($token->{tag_name} eq 'style') {              } elsif ($self->{insertion_mode} ne 'in head noscript' and
3495                $style_start_tag->();                       $token->{tag_name} eq 'script') {
3496                redo B;                if ($self->{insertion_mode} eq 'after head') {
3497              } elsif ($token->{tag_name} eq 'script') {                  !!!parse-error (type => 'after head:'.$token->{tag_name});
3498                $script_start_tag->();                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
3499                  }
3500                  ## NOTE: There is a "as if in head" code clone.
3501                  $script_start_tag->($insert_to_current);
3502                  pop @{$self->{open_elements}}
3503                      if $self->{insertion_mode} eq 'after head';
3504                redo B;                redo B;
3505              } elsif ({base => 1, link => 1, meta => 1}->{$token->{tag_name}}) {              } elsif ($self->{insertion_mode} eq 'after head' and
3506                ## NOTE: There are "as if in head" code clones                       $token->{tag_name} eq 'body') {
3507                my $el;                !!!insert-element ('body', $token->{attributes});
3508                !!!create-element ($el, $token->{tag_name}, $token->{attributes});                $self->{insertion_mode} = 'in body';
               (defined $self->{head_element} ? $self->{head_element} : $self->{open_elements}->[-1]->[0])  
                 ->append_child ($el);  
   
3509                !!!next-token;                !!!next-token;
3510                redo B;                redo B;
3511              } elsif ($token->{tag_name} eq 'head') {              } elsif ($self->{insertion_mode} eq 'after head' and
3512                !!!parse-error (type => 'in head:head');                       $token->{tag_name} eq 'frameset') {
3513                ## Ignore the token                !!!insert-element ('frameset', $token->{attributes});
3514                  $self->{insertion_mode} = 'in frameset';
3515                !!!next-token;                !!!next-token;
3516                redo B;                redo B;
3517              } else {              } else {
3518                #                #
3519              }              }
3520            } elsif ($token->{type} eq 'end tag') {            } elsif ($token->{type} eq 'end tag') {
3521              if ($token->{tag_name} eq 'head') {              if ($self->{insertion_mode} eq 'in head' and
3522                if ($self->{open_elements}->[-1]->[1] eq 'head') {                  $token->{tag_name} eq 'head') {
3523                  pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
               } else {  
                 !!!parse-error (type => 'unmatched end tag:head');  
               }  
3524                $self->{insertion_mode} = 'after head';                $self->{insertion_mode} = 'after head';
3525                !!!next-token;                !!!next-token;
3526                redo B;                redo B;
3527              } elsif ($token->{tag_name} eq 'html') {              } elsif ($self->{insertion_mode} eq 'in head noscript' and
3528                    $token->{tag_name} eq 'noscript') {
3529                  pop @{$self->{open_elements}};
3530                  $self->{insertion_mode} = 'in head';
3531                  !!!next-token;
3532                  redo B;
3533                } elsif ($self->{insertion_mode} eq 'in head' and
3534                         {
3535                          body => 1, html => 1,
3536                          p => 1, br => 1,
3537                         }->{$token->{tag_name}}) {
3538                #                #
3539              } else {              } elsif ($self->{insertion_mode} eq 'in head noscript' and
3540                         {
3541                          p => 1, br => 1,
3542                         }->{$token->{tag_name}}) {
3543                  #
3544                } elsif ($self->{insertion_mode} ne 'after head') {
3545                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3546                ## Ignore the token                ## Ignore the token
3547                !!!next-token;                !!!next-token;
3548                redo B;                redo B;
3549                } else {
3550                  #
3551              }              }
3552            } else {            } else {
3553              #              #
3554            }            }
3555    
3556            if ($self->{open_elements}->[-1]->[1] eq 'head') {            ## As if </head> or </noscript> or <body>
3557              ## As if </head>            if ($self->{insertion_mode} eq 'in head') {
3558                pop @{$self->{open_elements}};
3559                $self->{insertion_mode} = 'after head';
3560              } elsif ($self->{insertion_mode} eq 'in head noscript') {
3561              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
3562                !!!parse-error (type => 'in noscript:'.(defined $token->{tag_name} ? ($token->{type} eq 'end tag' ? '/' : '') . $token->{tag_name} : '#' . $token->{type}));
3563                $self->{insertion_mode} = 'in head';
3564              } else { # 'after head'
3565                !!!insert-element ('body');
3566                $self->{insertion_mode} = 'in body';
3567            }            }
           $self->{insertion_mode} = 'after head';  
3568            ## reprocess            ## reprocess
3569            redo B;            redo B;
3570    
3571            ## ISSUE: An issue in the spec.            ## ISSUE: An issue in the spec.
         } elsif ($self->{insertion_mode} eq 'after head') {  
           if ($token->{type} eq 'character') {  
             if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {  
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);  
               unless (length $token->{data}) {  
                 !!!next-token;  
                 redo B;  
               }  
             }  
               
             #  
           } elsif ($token->{type} eq 'comment') {  
             my $comment = $self->{document}->create_comment ($token->{data});  
             $self->{open_elements}->[-1]->[0]->append_child ($comment);  
             !!!next-token;  
             redo B;  
           } elsif ($token->{type} eq 'start tag') {  
             if ($token->{tag_name} eq 'body') {  
               !!!insert-element ('body', $token->{attributes});  
               $self->{insertion_mode} = 'in body';  
               !!!next-token;  
               redo B;  
             } elsif ($token->{tag_name} eq 'frameset') {  
               !!!insert-element ('frameset', $token->{attributes});  
               $self->{insertion_mode} = 'in frameset';  
               !!!next-token;  
               redo B;  
             } elsif ({  
                       base => 1, link => 1, meta => 1,  
                       script => 1, style => 1, title => 1,  
                      }->{$token->{tag_name}}) {  
               !!!parse-error (type => 'after head:'.$token->{tag_name});  
               $self->{insertion_mode} = 'in head';  
               ## reprocess  
               redo B;  
             } else {  
               #  
             }  
           } else {  
             #  
           }  
             
           ## As if <body>  
           !!!insert-element ('body');  
           $self->{insertion_mode} = 'in body';  
           ## reprocess  
           redo B;  
3572          } elsif ($self->{insertion_mode} eq 'in body') {          } elsif ($self->{insertion_mode} eq 'in body') {
3573            if ($token->{type} eq 'character') {            if ($token->{type} eq 'character') {
3574              ## NOTE: There is a code clone of "character in body".              ## NOTE: There is a code clone of "character in body".
# Line 3368  sub _tree_construction_main ($) { Line 3723  sub _tree_construction_main ($) {
3723                if ({                if ({
3724                     dd => 1, dt => 1, li => 1, p => 1,                     dd => 1, dt => 1, li => 1, p => 1,
3725                     td => 1, th => 1, tr => 1,                     td => 1, th => 1, tr => 1,
3726                       tbody => 1, tfoot=> 1, thead => 1,
3727                    }->{$self->{open_elements}->[-1]->[1]}) {                    }->{$self->{open_elements}->[-1]->[1]}) {
3728                  !!!back-token; # <table>                  !!!back-token; # <table>
3729                  $token = {type => 'end tag', tag_name => 'table'};                  $token = {type => 'end tag', tag_name => 'table'};
# Line 3416  sub _tree_construction_main ($) { Line 3772  sub _tree_construction_main ($) {
3772                if ({                if ({
3773                     dd => 1, dt => 1, li => 1, p => 1,                     dd => 1, dt => 1, li => 1, p => 1,
3774                     td => 1, th => 1, tr => 1,                     td => 1, th => 1, tr => 1,
3775                       tbody => 1, tfoot=> 1, thead => 1,
3776                    }->{$self->{open_elements}->[-1]->[1]}) {                    }->{$self->{open_elements}->[-1]->[1]}) {
3777                  !!!back-token;                  !!!back-token;
3778                  $token = {type => 'end tag',                  $token = {type => 'end tag',
# Line 3499  sub _tree_construction_main ($) { Line 3856  sub _tree_construction_main ($) {
3856                if ({                if ({
3857                     dd => 1, dt => 1, li => 1, p => 1,                     dd => 1, dt => 1, li => 1, p => 1,
3858                     td => 1, th => 1, tr => 1,                     td => 1, th => 1, tr => 1,
3859                       tbody => 1, tfoot=> 1, thead => 1,
3860                    }->{$self->{open_elements}->[-1]->[1]}) {                    }->{$self->{open_elements}->[-1]->[1]}) {
3861                  !!!back-token; # <?>                  !!!back-token; # <?>
3862                  $token = {type => 'end tag', tag_name => 'caption'};                  $token = {type => 'end tag', tag_name => 'caption'};
# Line 3549  sub _tree_construction_main ($) { Line 3907  sub _tree_construction_main ($) {
3907                if ({                if ({
3908                     dd => 1, dt => 1, li => 1, p => 1,                     dd => 1, dt => 1, li => 1, p => 1,
3909                     td => 1, th => 1, tr => 1,                     td => 1, th => 1, tr => 1,
3910                       tbody => 1, tfoot=> 1, thead => 1,
3911                    }->{$self->{open_elements}->[-1]->[1]}) {                    }->{$self->{open_elements}->[-1]->[1]}) {
3912                  !!!back-token;                  !!!back-token;
3913                  $token = {type => 'end tag',                  $token = {type => 'end tag',
# Line 3596  sub _tree_construction_main ($) { Line 3955  sub _tree_construction_main ($) {
3955                if ({                if ({
3956                     dd => 1, dt => 1, li => 1, p => 1,                     dd => 1, dt => 1, li => 1, p => 1,
3957                     td => 1, th => 1, tr => 1,                     td => 1, th => 1, tr => 1,
3958                       tbody => 1, tfoot=> 1, thead => 1,
3959                    }->{$self->{open_elements}->[-1]->[1]}) {                    }->{$self->{open_elements}->[-1]->[1]}) {
3960                  !!!back-token; # </table>                  !!!back-token; # </table>
3961                  $token = {type => 'end tag', tag_name => 'caption'};                  $token = {type => 'end tag', tag_name => 'caption'};
# Line 3861  sub _tree_construction_main ($) { Line 4221  sub _tree_construction_main ($) {
4221                if ({                if ({
4222                     dd => 1, dt => 1, li => 1, p => 1,                     dd => 1, dt => 1, li => 1, p => 1,
4223                     td => 1, th => 1, tr => 1,                     td => 1, th => 1, tr => 1,
4224                       tbody => 1, tfoot=> 1, thead => 1,
4225                    }->{$self->{open_elements}->[-1]->[1]}) {                    }->{$self->{open_elements}->[-1]->[1]}) {
4226                  !!!back-token; # <table>                  !!!back-token; # <table>
4227                  $token = {type => 'end tag', tag_name => 'table'};                  $token = {type => 'end tag', tag_name => 'table'};
# Line 4129  sub _tree_construction_main ($) { Line 4490  sub _tree_construction_main ($) {
4490                if ({                if ({
4491                     dd => 1, dt => 1, li => 1, p => 1,                     dd => 1, dt => 1, li => 1, p => 1,
4492                     td => 1, th => 1, tr => 1,                     td => 1, th => 1, tr => 1,
4493                       tbody => 1, tfoot=> 1, thead => 1,
4494                    }->{$self->{open_elements}->[-1]->[1]}) {                    }->{$self->{open_elements}->[-1]->[1]}) {
4495                  !!!back-token; # <table>                  !!!back-token; # <table>
4496                  $token = {type => 'end tag', tag_name => 'table'};                  $token = {type => 'end tag', tag_name => 'table'};
# Line 4370  sub _tree_construction_main ($) { Line 4732  sub _tree_construction_main ($) {
4732                     td => ($token->{tag_name} eq 'th'),                     td => ($token->{tag_name} eq 'th'),
4733                     th => ($token->{tag_name} eq 'td'),                     th => ($token->{tag_name} eq 'td'),
4734                     tr => 1,                     tr => 1,
4735                       tbody => 1, tfoot=> 1, thead => 1,
4736                    }->{$self->{open_elements}->[-1]->[1]}) {                    }->{$self->{open_elements}->[-1]->[1]}) {
4737                  !!!back-token;                  !!!back-token;
4738                  $token = {type => 'end tag',                  $token = {type => 'end tag',
# Line 4722  sub _tree_construction_main ($) { Line 5085  sub _tree_construction_main ($) {
5085            }            }
5086                        
5087            if (defined $token->{tag_name}) {            if (defined $token->{tag_name}) {
5088              !!!parse-error (type => 'in frameset:'.$token->{tag_name});              !!!parse-error (type => 'in frameset:'.($token->{type} eq 'end tag' ? '/' : '').$token->{tag_name});
5089            } else {            } else {
5090              !!!parse-error (type => 'in frameset:#'.$token->{type});              !!!parse-error (type => 'in frameset:#'.$token->{type});
5091            }            }
# Line 4766  sub _tree_construction_main ($) { Line 5129  sub _tree_construction_main ($) {
5129            }            }
5130                        
5131            if (defined $token->{tag_name}) {            if (defined $token->{tag_name}) {
5132              !!!parse-error (type => 'after frameset:'.$token->{tag_name});              !!!parse-error (type => 'after frameset:'.($token->{tag_name} eq 'end tag' ? '/' : '').$token->{tag_name});
5133            } else {            } else {
5134              !!!parse-error (type => 'after frameset:#'.$token->{type});              !!!parse-error (type => 'after frameset:#'.$token->{type});
5135            }            }
# Line 4816  sub _tree_construction_main ($) { Line 5179  sub _tree_construction_main ($) {
5179          redo B;          redo B;
5180        } elsif ($token->{type} eq 'start tag' or        } elsif ($token->{type} eq 'start tag' or
5181                 $token->{type} eq 'end tag') {                 $token->{type} eq 'end tag') {
5182          !!!parse-error (type => 'after html:'.$token->{tag_name});          !!!parse-error (type => 'after html:'.($token->{type} eq 'end tag' ? '/' : '').$token->{tag_name});
5183          $phase = 'main';          $phase = 'main';
5184          ## reprocess          ## reprocess
5185          redo B;          redo B;
# Line 4862  sub set_inner_html ($$$) { Line 5225  sub set_inner_html ($$$) {
5225      ## NOTE: Most of this code is copied from |parse_string|      ## NOTE: Most of this code is copied from |parse_string|
5226    
5227      ## Step 1 # MUST      ## Step 1 # MUST
5228      my $doc = $node->owner_document->implementation->create_document;      my $this_doc = $node->owner_document;
5229      ## TODO: Mark as HTML document      my $doc = $this_doc->implementation->create_document;
5230        $doc->manakai_is_html (1);
5231      my $p = $class->new;      my $p = $class->new;
5232      $p->{document} = $doc;      $p->{document} = $doc;
5233    
# Line 4873  sub set_inner_html ($$$) { Line 5237  sub set_inner_html ($$$) {
5237      my $column = 0;      my $column = 0;
5238      $p->{set_next_input_character} = sub {      $p->{set_next_input_character} = sub {
5239        my $self = shift;        my $self = shift;
5240    
5241          pop @{$self->{prev_input_character}};
5242          unshift @{$self->{prev_input_character}}, $self->{next_input_character};
5243    
5244        $self->{next_input_character} = -1 and return if $i >= length $$s;        $self->{next_input_character} = -1 and return if $i >= length $$s;
5245        $self->{next_input_character} = ord substr $$s, $i++, 1;        $self->{next_input_character} = ord substr $$s, $i++, 1;
5246        $column++;        $column++;
# Line 4881  sub set_inner_html ($$$) { Line 5249  sub set_inner_html ($$$) {
5249          $line++;          $line++;
5250          $column = 0;          $column = 0;
5251        } elsif ($self->{next_input_character} == 0x000D) { # CR        } elsif ($self->{next_input_character} == 0x000D) { # CR
5252          if ($i >= length $$s) {          $i++ if substr ($$s, $i, 1) eq "\x0A";
           #  
         } else {  
           my $next_char = ord substr $$s, $i++, 1;  
           if ($next_char == 0x000A) { # LF  
             #  
           } else {  
             push @{$self->{char}}, $next_char;  
           }  
         }  
5253          $self->{next_input_character} = 0x000A; # LF # MUST          $self->{next_input_character} = 0x000A; # LF # MUST
5254          $line++;          $line++;
5255          $column = 0;          $column = 0;
5256        } elsif ($self->{next_input_character} > 0x10FFFF) {        } elsif ($self->{next_input_character} > 0x10FFFF) {
5257          $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST          $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
5258        } elsif ($self->{next_input_character} == 0x0000) { # NULL        } elsif ($self->{next_input_character} == 0x0000) { # NULL
5259            !!!parse-error (type => 'NULL');
5260          $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST          $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
5261        }        }
5262      };      };
5263        $p->{prev_input_character} = [-1, -1, -1];
5264        $p->{next_input_character} = -1;
5265            
5266      my $ponerror = $onerror || sub {      my $ponerror = $onerror || sub {
5267        my (%opt) = @_;        my (%opt) = @_;
# Line 4978  sub set_inner_html ($$$) { Line 5340  sub set_inner_html ($$$) {
5340      ## Step 12 # MUST      ## Step 12 # MUST
5341      @cn = @{$root->child_nodes};      @cn = @{$root->child_nodes};
5342      for (@cn) {      for (@cn) {
5343          $this_doc->adopt_node ($_);
5344        $node->append_child ($_);        $node->append_child ($_);
5345      }      }
5346      ## ISSUE: adopt_node? mutation events?      ## ISSUE: mutation events?
5347    
5348      $p->_terminate_tree_constructor;      $p->_terminate_tree_constructor;
5349    } else {    } else {
# Line 5025  sub get_inner_html ($$$) { Line 5388  sub get_inner_html ($$$) {
5388            
5389      my $nt = $child->node_type;      my $nt = $child->node_type;
5390      if ($nt == 1) { # Element      if ($nt == 1) { # Element
5391        my $tag_name = lc $child->tag_name; ## ISSUE: Definition of "lowercase"        my $tag_name = $child->tag_name; ## TODO: manakai_tag_name
5392        $s .= '<' . $tag_name;        $s .= '<' . $tag_name;
5393          ## NOTE: Non-HTML case:
5394        ## ISSUE: Non-html elements        ## <http://permalink.gmane.org/gmane.org.w3c.whatwg.discuss/11191>
5395    
5396        my @attrs = @{$child->attributes}; # sort order MUST be stable        my @attrs = @{$child->attributes}; # sort order MUST be stable
5397        for my $attr (@attrs) { # order is implementation dependent        for my $attr (@attrs) { # order is implementation dependent
5398          my $attr_name = lc $attr->name; ## ISSUE: Definition of "lowercase"          my $attr_name = $attr->name; ## TODO: manakai_name
5399          $s .= ' ' . $attr_name . '="';          $s .= ' ' . $attr_name . '="';
5400          my $attr_value = $attr->value;          my $attr_value = $attr->value;
5401          ## escape          ## escape
# Line 5051  sub get_inner_html ($$$) { Line 5414  sub get_inner_html ($$$) {
5414          spacer => 1, wbr => 1,          spacer => 1, wbr => 1,
5415        }->{$tag_name};        }->{$tag_name};
5416    
5417          $s .= "\x0A" if $tag_name eq 'pre' or $tag_name eq 'textarea';
5418    
5419        if (not $in_cdata and {        if (not $in_cdata and {
5420          style => 1, script => 1, xmp => 1, iframe => 1,          style => 1, script => 1, xmp => 1, iframe => 1,
5421          noembed => 1, noframes => 1, noscript => 1,          noembed => 1, noframes => 1, noscript => 1,
5422            plaintext => 1,
5423        }->{$tag_name}) {        }->{$tag_name}) {
5424          unshift @node, 'cdata-out';          unshift @node, 'cdata-out';
5425          $in_cdata = 1;          $in_cdata = 1;

Legend:
Removed from v.1.8  
changed lines
  Added in v.1.31

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24