/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.8 by wakaba, Sat Jun 23 02:26:51 2007 UTC revision 1.113 by wakaba, Sun Mar 16 07:07:59 2008 UTC
# Line 1  Line 1 
1  package Whatpm::HTML;  package Whatpm::HTML;
2  use strict;  use strict;
3  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};
4    use Error qw(:try);
5    
6  ## This is an early version of an HTML parser.  ## ISSUE:
7    ## var doc = implementation.createDocument (null, null, null);
8    ## doc.write ('');
9    ## alert (doc.compatMode);
10    
11    ## TODO: Control charcters and noncharacters are not allowed (HTML5 revision 1263)
12    ## TODO: 1252 parse error (revision 1264)
13    ## TODO: 8859-11 = 874 (revision 1271)
14    
15  my $permitted_slash_tag_name = {  my $permitted_slash_tag_name = {
16    base => 1,    base => 1,
# Line 10  my $permitted_slash_tag_name = { Line 18  my $permitted_slash_tag_name = {
18    meta => 1,    meta => 1,
19    hr => 1,    hr => 1,
20    br => 1,    br => 1,
21    img=> 1,    img => 1,
22    embed => 1,    embed => 1,
23    param => 1,    param => 1,
24    area => 1,    area => 1,
# Line 18  my $permitted_slash_tag_name = { Line 26  my $permitted_slash_tag_name = {
26    input => 1,    input => 1,
27  };  };
28    
 my $entity_char = {  
   AElig => "\x{00C6}",  
   Aacute => "\x{00C1}",  
   Acirc => "\x{00C2}",  
   Agrave => "\x{00C0}",  
   Alpha => "\x{0391}",  
   Aring => "\x{00C5}",  
   Atilde => "\x{00C3}",  
   Auml => "\x{00C4}",  
   Beta => "\x{0392}",  
   Ccedil => "\x{00C7}",  
   Chi => "\x{03A7}",  
   Dagger => "\x{2021}",  
   Delta => "\x{0394}",  
   ETH => "\x{00D0}",  
   Eacute => "\x{00C9}",  
   Ecirc => "\x{00CA}",  
   Egrave => "\x{00C8}",  
   Epsilon => "\x{0395}",  
   Eta => "\x{0397}",  
   Euml => "\x{00CB}",  
   Gamma => "\x{0393}",  
   Iacute => "\x{00CD}",  
   Icirc => "\x{00CE}",  
   Igrave => "\x{00CC}",  
   Iota => "\x{0399}",  
   Iuml => "\x{00CF}",  
   Kappa => "\x{039A}",  
   Lambda => "\x{039B}",  
   Mu => "\x{039C}",  
   Ntilde => "\x{00D1}",  
   Nu => "\x{039D}",  
   OElig => "\x{0152}",  
   Oacute => "\x{00D3}",  
   Ocirc => "\x{00D4}",  
   Ograve => "\x{00D2}",  
   Omega => "\x{03A9}",  
   Omicron => "\x{039F}",  
   Oslash => "\x{00D8}",  
   Otilde => "\x{00D5}",  
   Ouml => "\x{00D6}",  
   Phi => "\x{03A6}",  
   Pi => "\x{03A0}",  
   Prime => "\x{2033}",  
   Psi => "\x{03A8}",  
   Rho => "\x{03A1}",  
   Scaron => "\x{0160}",  
   Sigma => "\x{03A3}",  
   THORN => "\x{00DE}",  
   Tau => "\x{03A4}",  
   Theta => "\x{0398}",  
   Uacute => "\x{00DA}",  
   Ucirc => "\x{00DB}",  
   Ugrave => "\x{00D9}",  
   Upsilon => "\x{03A5}",  
   Uuml => "\x{00DC}",  
   Xi => "\x{039E}",  
   Yacute => "\x{00DD}",  
   Yuml => "\x{0178}",  
   Zeta => "\x{0396}",  
   aacute => "\x{00E1}",  
   acirc => "\x{00E2}",  
   acute => "\x{00B4}",  
   aelig => "\x{00E6}",  
   agrave => "\x{00E0}",  
   alefsym => "\x{2135}",  
   alpha => "\x{03B1}",  
   amp => "\x{0026}",  
   AMP => "\x{0026}",  
   and => "\x{2227}",  
   ang => "\x{2220}",  
   apos => "\x{0027}",  
   aring => "\x{00E5}",  
   asymp => "\x{2248}",  
   atilde => "\x{00E3}",  
   auml => "\x{00E4}",  
   bdquo => "\x{201E}",  
   beta => "\x{03B2}",  
   brvbar => "\x{00A6}",  
   bull => "\x{2022}",  
   cap => "\x{2229}",  
   ccedil => "\x{00E7}",  
   cedil => "\x{00B8}",  
   cent => "\x{00A2}",  
   chi => "\x{03C7}",  
   circ => "\x{02C6}",  
   clubs => "\x{2663}",  
   cong => "\x{2245}",  
   copy => "\x{00A9}",  
   COPY => "\x{00A9}",  
   crarr => "\x{21B5}",  
   cup => "\x{222A}",  
   curren => "\x{00A4}",  
   dArr => "\x{21D3}",  
   dagger => "\x{2020}",  
   darr => "\x{2193}",  
   deg => "\x{00B0}",  
   delta => "\x{03B4}",  
   diams => "\x{2666}",  
   divide => "\x{00F7}",  
   eacute => "\x{00E9}",  
   ecirc => "\x{00EA}",  
   egrave => "\x{00E8}",  
   empty => "\x{2205}",  
   emsp => "\x{2003}",  
   ensp => "\x{2002}",  
   epsilon => "\x{03B5}",  
   equiv => "\x{2261}",  
   eta => "\x{03B7}",  
   eth => "\x{00F0}",  
   euml => "\x{00EB}",  
   euro => "\x{20AC}",  
   exist => "\x{2203}",  
   fnof => "\x{0192}",  
   forall => "\x{2200}",  
   frac12 => "\x{00BD}",  
   frac14 => "\x{00BC}",  
   frac34 => "\x{00BE}",  
   frasl => "\x{2044}",  
   gamma => "\x{03B3}",  
   ge => "\x{2265}",  
   gt => "\x{003E}",  
   GT => "\x{003E}",  
   hArr => "\x{21D4}",  
   harr => "\x{2194}",  
   hearts => "\x{2665}",  
   hellip => "\x{2026}",  
   iacute => "\x{00ED}",  
   icirc => "\x{00EE}",  
   iexcl => "\x{00A1}",  
   igrave => "\x{00EC}",  
   image => "\x{2111}",  
   infin => "\x{221E}",  
   int => "\x{222B}",  
   iota => "\x{03B9}",  
   iquest => "\x{00BF}",  
   isin => "\x{2208}",  
   iuml => "\x{00EF}",  
   kappa => "\x{03BA}",  
   lArr => "\x{21D0}",  
   lambda => "\x{03BB}",  
   lang => "\x{2329}",  
   laquo => "\x{00AB}",  
   larr => "\x{2190}",  
   lceil => "\x{2308}",  
   ldquo => "\x{201C}",  
   le => "\x{2264}",  
   lfloor => "\x{230A}",  
   lowast => "\x{2217}",  
   loz => "\x{25CA}",  
   lrm => "\x{200E}",  
   lsaquo => "\x{2039}",  
   lsquo => "\x{2018}",  
   lt => "\x{003C}",  
   LT => "\x{003C}",  
   macr => "\x{00AF}",  
   mdash => "\x{2014}",  
   micro => "\x{00B5}",  
   middot => "\x{00B7}",  
   minus => "\x{2212}",  
   mu => "\x{03BC}",  
   nabla => "\x{2207}",  
   nbsp => "\x{00A0}",  
   ndash => "\x{2013}",  
   ne => "\x{2260}",  
   ni => "\x{220B}",  
   not => "\x{00AC}",  
   notin => "\x{2209}",  
   nsub => "\x{2284}",  
   ntilde => "\x{00F1}",  
   nu => "\x{03BD}",  
   oacute => "\x{00F3}",  
   ocirc => "\x{00F4}",  
   oelig => "\x{0153}",  
   ograve => "\x{00F2}",  
   oline => "\x{203E}",  
   omega => "\x{03C9}",  
   omicron => "\x{03BF}",  
   oplus => "\x{2295}",  
   or => "\x{2228}",  
   ordf => "\x{00AA}",  
   ordm => "\x{00BA}",  
   oslash => "\x{00F8}",  
   otilde => "\x{00F5}",  
   otimes => "\x{2297}",  
   ouml => "\x{00F6}",  
   para => "\x{00B6}",  
   part => "\x{2202}",  
   permil => "\x{2030}",  
   perp => "\x{22A5}",  
   phi => "\x{03C6}",  
   pi => "\x{03C0}",  
   piv => "\x{03D6}",  
   plusmn => "\x{00B1}",  
   pound => "\x{00A3}",  
   prime => "\x{2032}",  
   prod => "\x{220F}",  
   prop => "\x{221D}",  
   psi => "\x{03C8}",  
   quot => "\x{0022}",  
   QUOT => "\x{0022}",  
   rArr => "\x{21D2}",  
   radic => "\x{221A}",  
   rang => "\x{232A}",  
   raquo => "\x{00BB}",  
   rarr => "\x{2192}",  
   rceil => "\x{2309}",  
   rdquo => "\x{201D}",  
   real => "\x{211C}",  
   reg => "\x{00AE}",  
   REG => "\x{00AE}",  
   rfloor => "\x{230B}",  
   rho => "\x{03C1}",  
   rlm => "\x{200F}",  
   rsaquo => "\x{203A}",  
   rsquo => "\x{2019}",  
   sbquo => "\x{201A}",  
   scaron => "\x{0161}",  
   sdot => "\x{22C5}",  
   sect => "\x{00A7}",  
   shy => "\x{00AD}",  
   sigma => "\x{03C3}",  
   sigmaf => "\x{03C2}",  
   sim => "\x{223C}",  
   spades => "\x{2660}",  
   sub => "\x{2282}",  
   sube => "\x{2286}",  
   sum => "\x{2211}",  
   sup => "\x{2283}",  
   sup1 => "\x{00B9}",  
   sup2 => "\x{00B2}",  
   sup3 => "\x{00B3}",  
   supe => "\x{2287}",  
   szlig => "\x{00DF}",  
   tau => "\x{03C4}",  
   there4 => "\x{2234}",  
   theta => "\x{03B8}",  
   thetasym => "\x{03D1}",  
   thinsp => "\x{2009}",  
   thorn => "\x{00FE}",  
   tilde => "\x{02DC}",  
   times => "\x{00D7}",  
   trade => "\x{2122}",  
   uArr => "\x{21D1}",  
   uacute => "\x{00FA}",  
   uarr => "\x{2191}",  
   ucirc => "\x{00FB}",  
   ugrave => "\x{00F9}",  
   uml => "\x{00A8}",  
   upsih => "\x{03D2}",  
   upsilon => "\x{03C5}",  
   uuml => "\x{00FC}",  
   weierp => "\x{2118}",  
   xi => "\x{03BE}",  
   yacute => "\x{00FD}",  
   yen => "\x{00A5}",  
   yuml => "\x{00FF}",  
   zeta => "\x{03B6}",  
   zwj => "\x{200D}",  
   zwnj => "\x{200C}",  
 }; # $entity_char  
   
 ## TODO: Ensure that this table match to <http://html5.org/tools/web-apps-tracker?from=868&to=869>.  
 ## <http://lists.whatwg.org/pipermail/whatwg-whatwg.org/2006-December/thread.html#8562>  
29  my $c1_entity_char = {  my $c1_entity_char = {
30       128, 8364,    0x80 => 0x20AC,
31       129, 65533,    0x81 => 0xFFFD,
32       130, 8218,    0x82 => 0x201A,
33       131, 402,    0x83 => 0x0192,
34       132, 8222,    0x84 => 0x201E,
35       133, 8230,    0x85 => 0x2026,
36       134, 8224,    0x86 => 0x2020,
37       135, 8225,    0x87 => 0x2021,
38       136, 710,    0x88 => 0x02C6,
39       137, 8240,    0x89 => 0x2030,
40       138, 352,    0x8A => 0x0160,
41       139, 8249,    0x8B => 0x2039,
42       140, 338,    0x8C => 0x0152,
43       141, 65533,    0x8D => 0xFFFD,
44       142, 381,    0x8E => 0x017D,
45       143, 65533,    0x8F => 0xFFFD,
46       144, 65533,    0x90 => 0xFFFD,
47       145, 8216,    0x91 => 0x2018,
48       146, 8217,    0x92 => 0x2019,
49       147, 8220,    0x93 => 0x201C,
50       148, 8221,    0x94 => 0x201D,
51       149, 8226,    0x95 => 0x2022,
52       150, 8211,    0x96 => 0x2013,
53       151, 8212,    0x97 => 0x2014,
54       152, 732,    0x98 => 0x02DC,
55       153, 8482,    0x99 => 0x2122,
56       154, 353,    0x9A => 0x0161,
57       155, 8250,    0x9B => 0x203A,
58       156, 339,    0x9C => 0x0153,
59       157, 65533,    0x9D => 0xFFFD,
60       158, 382,    0x9E => 0x017E,
61       159, 376,    0x9F => 0x0178,
62  }; # $c1_entity_char  }; # $c1_entity_char
63    
64  my $special_category = {  my $special_category = {
# Line 330  my $special_category = { Line 74  my $special_category = {
74    textarea => 1, tfoot => 1, thead => 1, title => 1, tr => 1, ul => 1, wbr => 1,    textarea => 1, tfoot => 1, thead => 1, title => 1, tr => 1, ul => 1, wbr => 1,
75  };  };
76  my $scoping_category = {  my $scoping_category = {
77    button => 1, caption => 1, html => 1, marquee => 1, object => 1,    applet => 1, button => 1, caption => 1, html => 1, marquee => 1, object => 1,
78    table => 1, td => 1, th => 1,    table => 1, td => 1, th => 1,
79  };  };
80  my $formatting_category = {  my $formatting_category = {
# Line 339  my $formatting_category = { Line 83  my $formatting_category = {
83  };  };
84  # $phrasing_category: all other elements  # $phrasing_category: all other elements
85    
86    sub parse_byte_string ($$$$;$) {
87      my $self = ref $_[0] ? shift : shift->new;
88      my $charset = shift;
89      my $bytes_s = ref $_[0] ? $_[0] : \($_[0]);
90      my $s;
91      
92      if (defined $charset) {
93        require Encode; ## TODO: decode(utf8) don't delete BOM
94        $s = \ (Encode::decode ($charset, $$bytes_s));
95        $self->{input_encoding} = lc $charset; ## TODO: normalize name
96        $self->{confident} = 1;
97      } else {
98        ## TODO: Implement HTML5 detection algorithm
99        require Whatpm::Charset::UniversalCharDet;
100        $charset = Whatpm::Charset::UniversalCharDet->detect_byte_string
101            (substr ($$bytes_s, 0, 1024));
102        $charset ||= 'windows-1252';
103        $s = \ (Encode::decode ($charset, $$bytes_s));
104        $self->{input_encoding} = $charset;
105        $self->{confident} = 0;
106      }
107    
108      $self->{change_encoding} = sub {
109        my $self = shift;
110        my $charset = lc shift;
111        ## TODO: if $charset is supported
112        ## TODO: normalize charset name
113    
114        ## "Change the encoding" algorithm:
115    
116        ## Step 1    
117        if ($charset eq 'utf-16') { ## ISSUE: UTF-16BE -> UTF-8? UTF-16LE -> UTF-8?
118          $charset = 'utf-8';
119        }
120    
121        ## Step 2
122        if (defined $self->{input_encoding} and
123            $self->{input_encoding} eq $charset) {
124          $self->{confident} = 1;
125          return;
126        }
127    
128        !!!parse-error (type => 'charset label detected:'.$self->{input_encoding}.
129            ':'.$charset, level => 'w');
130    
131        ## Step 3
132        # if (can) {
133          ## change the encoding on the fly.
134          #$self->{confident} = 1;
135          #return;
136        # }
137    
138        ## Step 4
139        throw Whatpm::HTML::RestartParser (charset => $charset);
140      }; # $self->{change_encoding}
141    
142      my @args = @_; shift @args; # $s
143      my $return;
144      try {
145        $return = $self->parse_char_string ($s, @args);  
146      } catch Whatpm::HTML::RestartParser with {
147        my $charset = shift->{charset};
148        $s = \ (Encode::decode ($charset, $$bytes_s));    
149        $self->{input_encoding} = $charset; ## TODO: normalize
150        $self->{confident} = 1;
151        $return = $self->parse_char_string ($s, @args);
152      };
153      return $return;
154    } # parse_byte_string
155    
156    ## NOTE: HTML5 spec says that the encoding layer MUST NOT strip BOM
157    ## and the HTML layer MUST ignore it.  However, we does strip BOM in
158    ## the encoding layer and the HTML layer does not ignore any U+FEFF,
159    ## because the core part of our HTML parser expects a string of character,
160    ## not a string of bytes or code units or anything which might contain a BOM.
161    ## Therefore, any parser interface that accepts a string of bytes,
162    ## such as |parse_byte_string| in this module, must ensure that it does
163    ## strip the BOM and never strip any ZWNBSP.
164    
165    *parse_char_string = \&parse_string;
166    
167  sub parse_string ($$$;$) {  sub parse_string ($$$;$) {
168    my $self = shift->new;    my $self = ref $_[0] ? shift : shift->new;
169    my $s = \$_[0];    my $s = ref $_[0] ? $_[0] : \($_[0]);
170    $self->{document} = $_[1];    $self->{document} = $_[1];
171      @{$self->{document}->child_nodes} = ();
172    
173    ## NOTE: |set_inner_html| copies most of this method's code    ## NOTE: |set_inner_html| copies most of this method's code
174    
175      $self->{confident} = 1 unless exists $self->{confident};
176      $self->{document}->input_encoding ($self->{input_encoding})
177          if defined $self->{input_encoding};
178    
179    my $i = 0;    my $i = 0;
180    my $line = 1;    $self->{line_prev} = $self->{line} = 1;
181    my $column = 0;    $self->{column_prev} = $self->{column} = 0;
182    $self->{set_next_input_character} = sub {    $self->{set_next_char} = sub {
183      my $self = shift;      my $self = shift;
184      $self->{next_input_character} = -1 and return if $i >= length $$s;  
185      $self->{next_input_character} = ord substr $$s, $i++, 1;      pop @{$self->{prev_char}};
186      $column++;      unshift @{$self->{prev_char}}, $self->{next_char};
187    
188        $self->{next_char} = -1 and return if $i >= length $$s;
189        $self->{next_char} = ord substr $$s, $i++, 1;
190    
191        ($self->{line_prev}, $self->{column_prev})
192            = ($self->{line}, $self->{column});
193        $self->{column}++;
194            
195      if ($self->{next_input_character} == 0x000A) { # LF      if ($self->{next_char} == 0x000A) { # LF
196        $line++;        $self->{line}++;
197        $column = 0;        $self->{column} = 0;
198      } elsif ($self->{next_input_character} == 0x000D) { # CR      } elsif ($self->{next_char} == 0x000D) { # CR
199        if ($i >= length $$s) {        $i++ if substr ($$s, $i, 1) eq "\x0A";
200          #        $self->{next_char} = 0x000A; # LF # MUST
201        } else {        $self->{line}++;
202          my $next_char = ord substr $$s, $i++, 1;        $self->{column} = 0;
203          if ($next_char == 0x000A) { # LF      } elsif ($self->{next_char} > 0x10FFFF) {
204            #        $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
205          } else {      } elsif ($self->{next_char} == 0x0000) { # NULL
           push @{$self->{char}}, $next_char;  
         }  
       }  
       $self->{next_input_character} = 0x000A; # LF # MUST  
       $line++;  
       $column = 0;  
     } elsif ($self->{next_input_character} > 0x10FFFF) {  
       $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST  
     } elsif ($self->{next_input_character} == 0x0000) { # NULL  
206        !!!parse-error (type => 'NULL');        !!!parse-error (type => 'NULL');
207  ## TODO: test        $self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST
       $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST  
208      }      }
209    };    };
210      $self->{prev_char} = [-1, -1, -1];
211      $self->{next_char} = -1;
212    
213    my $onerror = $_[2] || sub {    my $onerror = $_[2] || sub {
214      my (%opt) = @_;      my (%opt) = @_;
215      warn "Parse error ($opt{type}) at line $opt{line} column $opt{column}\n";      my $line = $opt{token} ? $opt{token}->{line} : $opt{line};
216        my $column = $opt{token} ? $opt{token}->{column} : $opt{column};
217        warn "Parse error ($opt{type}) at line $line column $column\n";
218    };    };
219    $self->{parse_error} = sub {    $self->{parse_error} = sub {
220      $onerror->(@_, line => $line, column => $column);      $onerror->(line => $self->{line}, column => $self->{column}, @_);
221    };    };
222    
223    $self->_initialize_tokenizer;    $self->_initialize_tokenizer;
# Line 394  sub parse_string ($$$;$) { Line 225  sub parse_string ($$$;$) {
225    $self->_construct_tree;    $self->_construct_tree;
226    $self->_terminate_tree_constructor;    $self->_terminate_tree_constructor;
227    
228      delete $self->{parse_error}; # remove loop
229    
230    return $self->{document};    return $self->{document};
231  } # parse_string  } # parse_string
232    
233  sub new ($) {  sub new ($) {
234    my $class = shift;    my $class = shift;
235    my $self = bless {}, $class;    my $self = bless {}, $class;
236    $self->{set_next_input_character} = sub {    $self->{set_next_char} = sub {
237      $self->{next_input_character} = -1;      $self->{next_char} = -1;
238    };    };
239    $self->{parse_error} = sub {    $self->{parse_error} = sub {
240      #      #
241    };    };
242      $self->{change_encoding} = sub {
243        # if ($_[0] is a supported encoding) {
244        #   run "change the encoding" algorithm;
245        #   throw Whatpm::HTML::RestartParser (charset => $new_encoding);
246        # }
247      };
248      $self->{application_cache_selection} = sub {
249        #
250      };
251    return $self;    return $self;
252  } # new  } # new
253    
254    sub CM_ENTITY () { 0b001 } # & markup in data
255    sub CM_LIMITED_MARKUP () { 0b010 } # < markup in data (limited)
256    sub CM_FULL_MARKUP () { 0b100 } # < markup in data (any)
257    
258    sub PLAINTEXT_CONTENT_MODEL () { 0 }
259    sub CDATA_CONTENT_MODEL () { CM_LIMITED_MARKUP }
260    sub RCDATA_CONTENT_MODEL () { CM_ENTITY | CM_LIMITED_MARKUP }
261    sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP }
262    
263    sub DATA_STATE () { 0 }
264    sub ENTITY_DATA_STATE () { 1 }
265    sub TAG_OPEN_STATE () { 2 }
266    sub CLOSE_TAG_OPEN_STATE () { 3 }
267    sub TAG_NAME_STATE () { 4 }
268    sub BEFORE_ATTRIBUTE_NAME_STATE () { 5 }
269    sub ATTRIBUTE_NAME_STATE () { 6 }
270    sub AFTER_ATTRIBUTE_NAME_STATE () { 7 }
271    sub BEFORE_ATTRIBUTE_VALUE_STATE () { 8 }
272    sub ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE () { 9 }
273    sub ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE () { 10 }
274    sub ATTRIBUTE_VALUE_UNQUOTED_STATE () { 11 }
275    sub ENTITY_IN_ATTRIBUTE_VALUE_STATE () { 12 }
276    sub MARKUP_DECLARATION_OPEN_STATE () { 13 }
277    sub COMMENT_START_STATE () { 14 }
278    sub COMMENT_START_DASH_STATE () { 15 }
279    sub COMMENT_STATE () { 16 }
280    sub COMMENT_END_STATE () { 17 }
281    sub COMMENT_END_DASH_STATE () { 18 }
282    sub BOGUS_COMMENT_STATE () { 19 }
283    sub DOCTYPE_STATE () { 20 }
284    sub BEFORE_DOCTYPE_NAME_STATE () { 21 }
285    sub DOCTYPE_NAME_STATE () { 22 }
286    sub AFTER_DOCTYPE_NAME_STATE () { 23 }
287    sub BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE () { 24 }
288    sub DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE () { 25 }
289    sub DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE () { 26 }
290    sub AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE () { 27 }
291    sub BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 28 }
292    sub DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE () { 29 }
293    sub DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE () { 30 }
294    sub AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 31 }
295    sub BOGUS_DOCTYPE_STATE () { 32 }
296    sub AFTER_ATTRIBUTE_VALUE_QUOTED_STATE () { 33 }
297    
298    sub DOCTYPE_TOKEN () { 1 }
299    sub COMMENT_TOKEN () { 2 }
300    sub START_TAG_TOKEN () { 3 }
301    sub END_TAG_TOKEN () { 4 }
302    sub END_OF_FILE_TOKEN () { 5 }
303    sub CHARACTER_TOKEN () { 6 }
304    
305    sub AFTER_HTML_IMS () { 0b100 }
306    sub HEAD_IMS ()       { 0b1000 }
307    sub BODY_IMS ()       { 0b10000 }
308    sub BODY_TABLE_IMS () { 0b100000 }
309    sub TABLE_IMS ()      { 0b1000000 }
310    sub ROW_IMS ()        { 0b10000000 }
311    sub BODY_AFTER_IMS () { 0b100000000 }
312    sub FRAME_IMS ()      { 0b1000000000 }
313    sub SELECT_IMS ()     { 0b10000000000 }
314    
315    ## NOTE: "initial" and "before html" insertion modes have no constants.
316    
317    ## NOTE: "after after body" insertion mode.
318    sub AFTER_HTML_BODY_IM () { AFTER_HTML_IMS | BODY_AFTER_IMS }
319    
320    ## NOTE: "after after frameset" insertion mode.
321    sub AFTER_HTML_FRAMESET_IM () { AFTER_HTML_IMS | FRAME_IMS }
322    
323    sub IN_HEAD_IM () { HEAD_IMS | 0b00 }
324    sub IN_HEAD_NOSCRIPT_IM () { HEAD_IMS | 0b01 }
325    sub AFTER_HEAD_IM () { HEAD_IMS | 0b10 }
326    sub BEFORE_HEAD_IM () { HEAD_IMS | 0b11 }
327    sub IN_BODY_IM () { BODY_IMS }
328    sub IN_CELL_IM () { BODY_IMS | BODY_TABLE_IMS | 0b01 }
329    sub IN_CAPTION_IM () { BODY_IMS | BODY_TABLE_IMS | 0b10 }
330    sub IN_ROW_IM () { TABLE_IMS | ROW_IMS | 0b01 }
331    sub IN_TABLE_BODY_IM () { TABLE_IMS | ROW_IMS | 0b10 }
332    sub IN_TABLE_IM () { TABLE_IMS }
333    sub AFTER_BODY_IM () { BODY_AFTER_IMS }
334    sub IN_FRAMESET_IM () { FRAME_IMS | 0b01 }
335    sub AFTER_FRAMESET_IM () { FRAME_IMS | 0b10 }
336    sub IN_SELECT_IM () { SELECT_IMS | 0b01 }
337    sub IN_SELECT_IN_TABLE_IM () { SELECT_IMS | 0b10 }
338    sub IN_COLUMN_GROUP_IM () { 0b10 }
339    
340  ## Implementations MUST act as if state machine in the spec  ## Implementations MUST act as if state machine in the spec
341    
342  sub _initialize_tokenizer ($) {  sub _initialize_tokenizer ($) {
343    my $self = shift;    my $self = shift;
344    $self->{state} = 'data'; # MUST    $self->{state} = DATA_STATE; # MUST
345    $self->{content_model_flag} = 'PCDATA'; # be    $self->{content_model} = PCDATA_CONTENT_MODEL; # be
346    undef $self->{current_token}; # start tag, end tag, comment, or DOCTYPE    undef $self->{current_token}; # start tag, end tag, comment, or DOCTYPE
347    undef $self->{current_attribute};    undef $self->{current_attribute};
348    undef $self->{last_emitted_start_tag_name};    undef $self->{last_emitted_start_tag_name};
349    undef $self->{last_attribute_value_state};    undef $self->{last_attribute_value_state};
350    $self->{char} = [];    $self->{char} = [];
351    # $self->{next_input_character}    # $self->{next_char}
352    !!!next-input-character;    !!!next-input-character;
353    $self->{token} = [];    $self->{token} = [];
354      # $self->{escape}
355  } # _initialize_tokenizer  } # _initialize_tokenizer
356    
357  ## A token has:  ## A token has:
358  ##   ->{type} eq 'DOCTYPE', 'start tag', 'end tag', 'comment',  ##   ->{type} == DOCTYPE_TOKEN, START_TAG_TOKEN, END_TAG_TOKEN, COMMENT_TOKEN,
359  ##       'character', or 'end-of-file'  ##       CHARACTER_TOKEN, or END_OF_FILE_TOKEN
360  ##   ->{name} (DOCTYPE, start tag (tagname), end tag (tagname))  ##   ->{name} (DOCTYPE_TOKEN)
361      ## ISSUE: the spec need s/tagname/tag name/  ##   ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN)
362  ##   ->{error} == 1 or 0 (DOCTYPE)  ##   ->{public_identifier} (DOCTYPE_TOKEN)
363  ##   ->{attributes} isa HASH (start tag, end tag)  ##   ->{system_identifier} (DOCTYPE_TOKEN)
364  ##   ->{data} (comment, character)  ##   ->{quirks} == 1 or 0 (DOCTYPE_TOKEN): "force-quirks" flag
365    ##   ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)
366  ## Macros  ##        ->{name}
367  ##   Macros MUST be preceded by three EXCLAMATION MARKs.  ##        ->{value}
368  ##   emit ($token)  ##        ->{has_reference} == 1 or 0
369  ##     Emits the specified token.  ##   ->{data} (COMMENT_TOKEN, CHARACTER_TOKEN)
370    
371  ## Emitted token MUST immediately be handled by the tree construction state.  ## Emitted token MUST immediately be handled by the tree construction state.
372    
# Line 447  sub _initialize_tokenizer ($) { Line 376  sub _initialize_tokenizer ($) {
376  ## has completed loading.  If one has, then it MUST be executed  ## has completed loading.  If one has, then it MUST be executed
377  ## and removed from the list.  ## and removed from the list.
378    
379  ## ISSUE: <http://html5.org/tools/web-apps-tracker?from=874&to=876>  ## NOTE: HTML5 "Writing HTML documents" section, applied to
380    ## documents and not to user agents and conformance checkers,
381    ## contains some requirements that are not detected by the
382    ## parsing algorithm:
383    ## - Some requirements on character encoding declarations. ## TODO
384    ## - "Elements MUST NOT contain content that their content model disallows."
385    ##   ... Some are parse error, some are not (will be reported by c.c.).
386    ## - Polytheistic slash SHOULD NOT be used. (Applied only to atheists.) ## TODO
387    ## - Text (in elements, attributes, and comments) SHOULD NOT contain
388    ##   control characters other than space characters. ## TODO: (what is control character? C0, C1 and DEL?  Unicode control character?)
389    
390    ## TODO: HTML5 poses authors two SHOULD-level requirements that cannot
391