Parent Directory
|
Revision Log
|
Patch
| revision 1.31 by wakaba, Sat Jun 30 14:13:19 2007 UTC | revision 1.65 by wakaba, Mon Nov 19 12:18:26 2007 UTC | |
|---|---|---|
| # | Line 1 | Line 1 |
| 1 | package Whatpm::HTML; | package Whatpm::HTML; |
| 2 | use strict; | use strict; |
| 3 | our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r}; | our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r}; |
| 4 | use Error qw(:try); | |
| 5 | ||
| 6 | ## ISSUE: | ## ISSUE: |
| 7 | ## var doc = implementation.createDocument (null, null, null); | ## var doc = implementation.createDocument (null, null, null); |
| # | Line 84 my $formatting_category = { | Line 85 my $formatting_category = { |
| 85 | }; | }; |
| 86 | # $phrasing_category: all other elements | # $phrasing_category: all other elements |
| 87 | ||
| 88 | sub parse_byte_string ($$$$;$) { | |
| 89 | my $self = ref $_[0] ? shift : shift->new; | |
| 90 | my $charset = shift; | |
| 91 | my $bytes_s = ref $_[0] ? $_[0] : \($_[0]); | |
| 92 | my $s; | |
| 93 | ||
| 94 | if (defined $charset) { | |
| 95 | require Encode; ## TODO: decode(utf8) don't delete BOM | |
| 96 | $s = \ (Encode::decode ($charset, $$bytes_s)); | |
| 97 | $self->{input_encoding} = lc $charset; ## TODO: normalize name | |
| 98 | $self->{confident} = 1; | |
| 99 | } else { | |
| 100 | ## TODO: Implement HTML5 detection algorithm | |
| 101 | require Whatpm::Charset::UniversalCharDet; | |
| 102 | $charset = Whatpm::Charset::UniversalCharDet->detect_byte_string | |
| 103 | (substr ($$bytes_s, 0, 1024)); | |
| 104 | $charset ||= 'windows-1252'; | |
| 105 | $s = \ (Encode::decode ($charset, $$bytes_s)); | |
| 106 | $self->{input_encoding} = $charset; | |
| 107 | $self->{confident} = 0; | |
| 108 | } | |
| 109 | ||
| 110 | $self->{change_encoding} = sub { | |
| 111 | my $self = shift; | |
| 112 | my $charset = lc shift; | |
| 113 | ## TODO: if $charset is supported | |
| 114 | ## TODO: normalize charset name | |
| 115 | ||
| 116 | ## "Change the encoding" algorithm: | |
| 117 | ||
| 118 | ## Step 1 | |
| 119 | if ($charset eq 'utf-16') { ## ISSUE: UTF-16BE -> UTF-8? UTF-16LE -> UTF-8? | |
| 120 | $charset = 'utf-8'; | |
| 121 | } | |
| 122 | ||
| 123 | ## Step 2 | |
| 124 | if (defined $self->{input_encoding} and | |
| 125 | $self->{input_encoding} eq $charset) { | |
| 126 | $self->{confident} = 1; | |
| 127 | return; | |
| 128 | } | |
| 129 | ||
| 130 | !!!parse-error (type => 'charset label detected:'.$self->{input_encoding}. | |
| 131 | ':'.$charset, level => 'w'); | |
| 132 | ||
| 133 | ## Step 3 | |
| 134 | # if (can) { | |
| 135 | ## change the encoding on the fly. | |
| 136 | #$self->{confident} = 1; | |
| 137 | #return; | |
| 138 | # } | |
| 139 | ||
| 140 | ## Step 4 | |
| 141 | throw Whatpm::HTML::RestartParser (charset => $charset); | |
| 142 | }; # $self->{change_encoding} | |
| 143 | ||
| 144 | my @args = @_; shift @args; # $s | |
| 145 | my $return; | |
| 146 | try { | |
| 147 | $return = $self->parse_char_string ($s, @args); | |
| 148 | } catch Whatpm::HTML::RestartParser with { | |
| 149 | my $charset = shift->{charset}; | |
| 150 | $s = \ (Encode::decode ($charset, $$bytes_s)); | |
| 151 | $self->{input_encoding} = $charset; ## TODO: normalize | |
| 152 | $self->{confident} = 1; | |
| 153 | $return = $self->parse_char_string ($s, @args); | |
| 154 | }; | |
| 155 | return $return; | |
| 156 | } # parse_byte_string | |
| 157 | ||
| 158 | *parse_char_string = \&parse_string; | |
| 159 | ||
| 160 | sub parse_string ($$$;$) { | sub parse_string ($$$;$) { |
| 161 | my $self = shift->new; | my $self = ref $_[0] ? shift : shift->new; |
| 162 | my $s = \$_[0]; | my $s = ref $_[0] ? $_[0] : \($_[0]); |
| 163 | $self->{document} = $_[1]; | $self->{document} = $_[1]; |
| 164 | @{$self->{document}->child_nodes} = (); | |
| 165 | ||
| 166 | ## NOTE: |set_inner_html| copies most of this method's code | ## NOTE: |set_inner_html| copies most of this method's code |
| 167 | ||
| 168 | $self->{confident} = 1 unless exists $self->{confident}; | |
| 169 | $self->{document}->input_encoding ($self->{input_encoding}) | |
| 170 | if defined $self->{input_encoding}; | |
| 171 | ||
| 172 | my $i = 0; | my $i = 0; |
| 173 | my $line = 1; | my $line = 1; |
| 174 | my $column = 0; | my $column = 0; |
| # | Line 147 sub new ($) { | Line 225 sub new ($) { |
| 225 | $self->{parse_error} = sub { | $self->{parse_error} = sub { |
| 226 | # | # |
| 227 | }; | }; |
| 228 | $self->{change_encoding} = sub { | |
| 229 | # if ($_[0] is a supported encoding) { | |
| 230 | # run "change the encoding" algorithm; | |
| 231 | # throw Whatpm::HTML::RestartParser (charset => $new_encoding); | |
| 232 | # } | |
| 233 | }; | |
| 234 | $self->{application_cache_selection} = sub { | |
| 235 | # | |
| 236 | }; | |
| 237 | return $self; | return $self; |
| 238 | } # new | } # new |
| 239 | ||
| 240 | sub CM_ENTITY () { 0b001 } # & markup in data | |
| 241 | sub CM_LIMITED_MARKUP () { 0b010 } # < markup in data (limited) | |
| 242 | sub CM_FULL_MARKUP () { 0b100 } # < markup in data (any) | |
| 243 | ||
| 244 | sub PLAINTEXT_CONTENT_MODEL () { 0 } | |
| 245 | sub CDATA_CONTENT_MODEL () { CM_LIMITED_MARKUP } | |
| 246 | sub RCDATA_CONTENT_MODEL () { CM_ENTITY | CM_LIMITED_MARKUP } | |
| 247 | sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP } | |
| 248 | ||
| 249 | sub DATA_STATE () { 0 } | |
| 250 | sub ENTITY_DATA_STATE () { 1 } | |
| 251 | sub TAG_OPEN_STATE () { 2 } | |
| 252 | sub CLOSE_TAG_OPEN_STATE () { 3 } | |
| 253 | sub TAG_NAME_STATE () { 4 } | |
| 254 | sub BEFORE_ATTRIBUTE_NAME_STATE () { 5 } | |
| 255 | sub ATTRIBUTE_NAME_STATE () { 6 } | |
| 256 | sub AFTER_ATTRIBUTE_NAME_STATE () { 7 } | |
| 257 | sub BEFORE_ATTRIBUTE_VALUE_STATE () { 8 } | |
| 258 | sub ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE () { 9 } | |
| 259 | sub ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE () { 10 } | |
| 260 | sub ATTRIBUTE_VALUE_UNQUOTED_STATE () { 11 } | |
| 261 | sub ENTITY_IN_ATTRIBUTE_VALUE_STATE () { 12 } | |
| 262 | sub MARKUP_DECLARATION_OPEN_STATE () { 13 } | |
| 263 | sub COMMENT_START_STATE () { 14 } | |
| 264 | sub COMMENT_START_DASH_STATE () { 15 } | |
| 265 | sub COMMENT_STATE () { 16 } | |
| 266 | sub COMMENT_END_STATE () { 17 } | |
| 267 | sub COMMENT_END_DASH_STATE () { 18 } | |
| 268 | sub BOGUS_COMMENT_STATE () { 19 } | |
| 269 | sub DOCTYPE_STATE () { 20 } | |
| 270 | sub BEFORE_DOCTYPE_NAME_STATE () { 21 } | |
| 271 | sub DOCTYPE_NAME_STATE () { 22 } | |
| 272 | sub AFTER_DOCTYPE_NAME_STATE () { 23 } | |
| 273 | sub BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE () { 24 } | |
| 274 | sub DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE () { 25 } | |
| 275 | sub DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE () { 26 } | |
| 276 | sub AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE () { 27 } | |
| 277 | sub BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 28 } | |
| 278 | sub DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE () { 29 } | |
| 279 | sub DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE () { 30 } | |
| 280 | sub AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 31 } | |
| 281 | sub BOGUS_DOCTYPE_STATE () { 32 } | |
| 282 | ||
| 283 | sub DOCTYPE_TOKEN () { 1 } | |
| 284 | sub COMMENT_TOKEN () { 2 } | |
| 285 | sub START_TAG_TOKEN () { 3 } | |
| 286 | sub END_TAG_TOKEN () { 4 } | |
| 287 | sub END_OF_FILE_TOKEN () { 5 } | |
| 288 | sub CHARACTER_TOKEN () { 6 } | |
| 289 | ||
| 290 | sub AFTER_HTML_IMS () { 0b100 } | |
| 291 | sub HEAD_IMS () { 0b1000 } | |
| 292 | sub BODY_IMS () { 0b10000 } | |
| 293 | sub BODY_TABLE_IMS () { 0b100000 } | |
| 294 | sub TABLE_IMS () { 0b1000000 } | |
| 295 | sub ROW_IMS () { 0b10000000 } | |
| 296 | sub BODY_AFTER_IMS () { 0b100000000 } | |
| 297 | sub FRAME_IMS () { 0b1000000000 } | |
| 298 | ||
| 299 | sub AFTER_HTML_BODY_IM () { AFTER_HTML_IMS | BODY_AFTER_IMS } | |
| 300 | sub AFTER_HTML_FRAMESET_IM () { AFTER_HTML_IMS | FRAME_IMS } | |
| 301 | sub IN_HEAD_IM () { HEAD_IMS | 0b00 } | |
| 302 | sub IN_HEAD_NOSCRIPT_IM () { HEAD_IMS | 0b01 } | |
| 303 | sub AFTER_HEAD_IM () { HEAD_IMS | 0b10 } | |
| 304 | sub BEFORE_HEAD_IM () { HEAD_IMS | 0b11 } | |
| 305 | sub IN_BODY_IM () { BODY_IMS } | |
| 306 | sub IN_CELL_IM () { BODY_IMS | BODY_TABLE_IMS | 0b01 } | |
| 307 | sub IN_CAPTION_IM () { BODY_IMS | BODY_TABLE_IMS | 0b10 } | |
| 308 | sub IN_ROW_IM () { TABLE_IMS | ROW_IMS | 0b01 } | |
| 309 | sub IN_TABLE_BODY_IM () { TABLE_IMS | ROW_IMS | 0b10 } | |
| 310 | sub IN_TABLE_IM () { TABLE_IMS } | |
| 311 | sub AFTER_BODY_IM () { BODY_AFTER_IMS } | |
| 312 | sub IN_FRAMESET_IM () { FRAME_IMS | 0b01 } | |
| 313 | sub AFTER_FRAMESET_IM () { FRAME_IMS | 0b10 } | |
| 314 | sub IN_SELECT_IM () { 0b01 } | |
| 315 | sub IN_COLUMN_GROUP_IM () { 0b10 } | |
| 316 | ||
| 317 | ## Implementations MUST act as if state machine in the spec | ## Implementations MUST act as if state machine in the spec |
| 318 | ||
| 319 | sub _initialize_tokenizer ($) { | sub _initialize_tokenizer ($) { |
| 320 | my $self = shift; | my $self = shift; |
| 321 | $self->{state} = 'data'; # MUST | $self->{state} = DATA_STATE; # MUST |
| 322 | $self->{content_model_flag} = 'PCDATA'; # be | $self->{content_model} = PCDATA_CONTENT_MODEL; # be |
| 323 | undef $self->{current_token}; # start tag, end tag, comment, or DOCTYPE | undef $self->{current_token}; # start tag, end tag, comment, or DOCTYPE |
| 324 | undef $self->{current_attribute}; | undef $self->{current_attribute}; |
| 325 | undef $self->{last_emitted_start_tag_name}; | undef $self->{last_emitted_start_tag_name}; |
| # | Line 168 sub _initialize_tokenizer ($) { | Line 332 sub _initialize_tokenizer ($) { |
| 332 | } # _initialize_tokenizer | } # _initialize_tokenizer |
| 333 | ||
| 334 | ## A token has: | ## A token has: |
| 335 | ## ->{type} eq 'DOCTYPE', 'start tag', 'end tag', 'comment', | ## ->{type} == DOCTYPE_TOKEN, START_TAG_TOKEN, END_TAG_TOKEN, COMMENT_TOKEN, |
| 336 | ## 'character', or 'end-of-file' | ## CHARACTER_TOKEN, or END_OF_FILE_TOKEN |
| 337 | ## ->{name} (DOCTYPE, start tag (tag name), end tag (tag name)) | ## ->{name} (DOCTYPE_TOKEN) |
| 338 | ## ->{public_identifier} (DOCTYPE) | ## ->{tag_name} (START_TAG_TOKEN, END_TAG_TOKEN) |
| 339 | ## ->{system_identifier} (DOCTYPE) | ## ->{public_identifier} (DOCTYPE_TOKEN) |
| 340 | ## ->{correct} == 1 or 0 (DOCTYPE) | ## ->{system_identifier} (DOCTYPE_TOKEN) |
| 341 | ## ->{attributes} isa HASH (start tag, end tag) | ## ->{correct} == 1 or 0 (DOCTYPE_TOKEN) |
| 342 | ## ->{data} (comment, character) | ## ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN) |
| 343 | ## ->{data} (COMMENT_TOKEN, CHARACTER_TOKEN) | |
| 344 | ||
| 345 | ## Emitted token MUST immediately be handled by the tree construction state. | ## Emitted token MUST immediately be handled by the tree construction state. |
| 346 | ||
| # | Line 185 sub _initialize_tokenizer ($) { | Line 350 sub _initialize_tokenizer ($) { |
| 350 | ## has completed loading. If one has, then it MUST be executed | ## has completed loading. If one has, then it MUST be executed |
| 351 | ## and removed from the list. | ## and removed from the list. |
| 352 | ||
| 353 | ## NOTE: HTML5 "Writing HTML documents" section, applied to | |
| 354 | ## documents and not to user agents and conformance checkers, | |
| 355 | ## contains some requirements that are not detected by the | |
| 356 | ## parsing algorithm: | |
| 357 | ## - Some requirements on character encoding declarations. ## TODO | |
| 358 | ## - "Elements MUST NOT contain content that their content model disallows." | |
| 359 | ## ... Some are parse error, some are not (will be reported by c.c.). | |
| 360 | ## - Polytheistic slash SHOULD NOT be used. (Applied only to atheists.) ## TODO | |
| 361 | ## - Text (in elements, attributes, and comments) SHOULD NOT contain | |
| 362 | ## control characters other than space characters. ## TODO: (what is control character? C0, C1 and DEL? Unicode control character?) | |
| 363 | ||
| 364 | ## TODO: HTML5 poses authors two SHOULD-level requirements that cannot | |