/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.156 by wakaba, Sat Aug 30 13:43:50 2008 UTC revision 1.160 by wakaba, Wed Sep 10 10:27:08 2008 UTC
# Line 372  sub parse_byte_stream ($$$$;$) { Line 372  sub parse_byte_stream ($$$$;$) {
372    my ($char_stream, $e_status);    my ($char_stream, $e_status);
373    
374    SNIFFING: {    SNIFFING: {
375        ## NOTE: By setting |allow_fallback| option true when the
376        ## |get_decode_handle| method is invoked, we ignore what the HTML5
377        ## spec requires, i.e. unsupported encoding should be ignored.
378          ## TODO: We should not do this unless the parser is invoked
379          ## in the conformance checking mode, in which this behavior
380          ## would be useful.
381    
382      ## Step 1      ## Step 1
383      if (defined $charset_name) {      if (defined $charset_name) {
# Line 475  sub parse_byte_stream ($$$$;$) { Line 481  sub parse_byte_stream ($$$$;$) {
481      $self->{confident} = 0;      $self->{confident} = 0;
482    } # SNIFFING    } # SNIFFING
483    
   $self->{input_encoding} = $charset->get_iana_name;  
484    if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {    if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
485        $self->{input_encoding} = $charset->get_iana_name; ## TODO: Should we set actual charset decoder's encoding name?
486      !!!parse-error (type => 'chardecode:fallback',      !!!parse-error (type => 'chardecode:fallback',
487                      text => $self->{input_encoding},                      #text => $self->{input_encoding},
488                      level => $self->{level}->{uncertain},                      level => $self->{level}->{uncertain},
489                      line => 1, column => 1,                      line => 1, column => 1,
490                      layer => 'encode');                      layer => 'encode');
491    } elsif (not ($e_status &    } elsif (not ($e_status &
492                  Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) {                  Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) {
493        $self->{input_encoding} = $charset->get_iana_name;
494      !!!parse-error (type => 'chardecode:no error',      !!!parse-error (type => 'chardecode:no error',
495                      text => $self->{input_encoding},                      text => $self->{input_encoding},
496                      level => $self->{level}->{uncertain},                      level => $self->{level}->{uncertain},
497                      line => 1, column => 1,                      line => 1, column => 1,
498                      layer => 'encode');                      layer => 'encode');
499      } else {
500        $self->{input_encoding} = $charset->get_iana_name;
501    }    }
502    
503    $self->{change_encoding} = sub {    $self->{change_encoding} = sub {
# Line 560  sub parse_byte_stream ($$$$;$) { Line 569  sub parse_byte_stream ($$$$;$) {
569    } catch Whatpm::HTML::RestartParser with {    } catch Whatpm::HTML::RestartParser with {
570      ## NOTE: Invoked after {change_encoding}.      ## NOTE: Invoked after {change_encoding}.
571    
     $self->{input_encoding} = $charset->get_iana_name;  
572      if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {      if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
573          $self->{input_encoding} = $charset->get_iana_name; ## TODO: Should we set actual charset decoder's encoding name?
574        !!!parse-error (type => 'chardecode:fallback',        !!!parse-error (type => 'chardecode:fallback',
                       text => $self->{input_encoding},  
575                        level => $self->{level}->{uncertain},                        level => $self->{level}->{uncertain},
576                          #text => $self->{input_encoding},
577                        line => 1, column => 1,                        line => 1, column => 1,
578                        layer => 'encode');                        layer => 'encode');
579      } elsif (not ($e_status &      } elsif (not ($e_status &
580                    Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) {                    Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) {
581          $self->{input_encoding} = $charset->get_iana_name;
582        !!!parse-error (type => 'chardecode:no error',        !!!parse-error (type => 'chardecode:no error',
583                        text => $self->{input_encoding},                        text => $self->{input_encoding},
584                        level => $self->{level}->{uncertain},                        level => $self->{level}->{uncertain},
585                        line => 1, column => 1,                        line => 1, column => 1,
586                        layer => 'encode');                        layer => 'encode');
587        } else {
588          $self->{input_encoding} = $charset->get_iana_name;
589      }      }
590      $self->{confident} = 1;      $self->{confident} = 1;
591      $char_stream->onerror ($char_onerror);      $char_stream->onerror ($char_onerror);
# Line 708  sub new ($) { Line 720  sub new ($) {
720    my $class = shift;    my $class = shift;
721    my $self = bless {    my $self = bless {
722      level => {must => 'm',      level => {must => 'm',
723                  should => 's',
724                warn => 'w',                warn => 'w',
725                info => 'i',                info => 'i',
726                uncertain => 'u'},                uncertain => 'u'},
# Line 3154  sub _tree_construction_initial ($) { Line 3167  sub _tree_construction_initial ($) {
3167        ## language.        ## language.
3168        my $doctype_name = $token->{name};        my $doctype_name = $token->{name};
3169        $doctype_name = '' unless defined $doctype_name;        $doctype_name = '' unless defined $doctype_name;
3170        $doctype_name =~ tr/a-z/A-Z/;        $doctype_name =~ tr/a-z/A-Z/; # ASCII case-insensitive
3171        if (not defined $token->{name} or # <!DOCTYPE>        if (not defined $token->{name} or # <!DOCTYPE>
           defined $token->{public_identifier} or  
3172            defined $token->{system_identifier}) {            defined $token->{system_identifier}) {
3173          !!!cp ('t1');          !!!cp ('t1');
3174          !!!parse-error (type => 'not HTML5', token => $token);          !!!parse-error (type => 'not HTML5', token => $token);
3175        } elsif ($doctype_name ne 'HTML') {        } elsif ($doctype_name ne 'HTML') {
3176          !!!cp ('t2');          !!!cp ('t2');
         ## ISSUE: ASCII case-insensitive? (in fact it does not matter)  
3177          !!!parse-error (type => 'not HTML5', token => $token);          !!!parse-error (type => 'not HTML5', token => $token);
3178          } elsif (defined $token->{public_identifier}) {
3179            if ($token->{public_identifier} eq 'XSLT-compat') {
3180              !!!cp ('t1.2');
3181              !!!parse-error (type => 'XSLT-compat', token => $token,
3182                              level => $self->{level}->{should});
3183            } else {
3184              !!!parse-error (type => 'not HTML5', token => $token);
3185            }
3186        } else {        } else {
3187          !!!cp ('t3');          !!!cp ('t3');
3188            #
3189        }        }
3190                
3191        my $doctype = $self->{document}->create_document_type_definition        my $doctype = $self->{document}->create_document_type_definition
# Line 6301  sub _tree_construction_main ($) { Line 6321  sub _tree_construction_main ($) {
6321            } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {            } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6322              !!!cp ('t312');              !!!cp ('t312');
6323              !!!parse-error (type => 'after frameset:#text', token => $token);              !!!parse-error (type => 'after frameset:#text', token => $token);
6324            } else { # "after html frameset"            } else { # "after after frameset"
6325              !!!cp ('t313');              !!!cp ('t313');
6326              !!!parse-error (type => 'after html:#text', token => $token);              !!!parse-error (type => 'after html:#text', token => $token);
   
             $self->{insertion_mode} = AFTER_FRAMESET_IM;  
             ## Reprocess in the "after frameset" insertion mode.  
             !!!parse-error (type => 'after frameset:#text', token => $token);  
6327            }            }
6328                        
6329            ## Ignore the token.            ## Ignore the token.
# Line 6323  sub _tree_construction_main ($) { Line 6339  sub _tree_construction_main ($) {
6339                    
6340          die qq[$0: Character "$token->{data}"];          die qq[$0: Character "$token->{data}"];
6341        } elsif ($token->{type} == START_TAG_TOKEN) {        } elsif ($token->{type} == START_TAG_TOKEN) {
         if ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {  
           !!!cp ('t316');  
           !!!parse-error (type => 'after html',  
                           text => $token->{tag_name}, token => $token);  
   
           $self->{insertion_mode} = AFTER_FRAMESET_IM;  
           ## Process in the "after frameset" insertion mode.  
         } else {  
           !!!cp ('t317');  
         }  
   
6342          if ($token->{tag_name} eq 'frameset' and          if ($token->{tag_name} eq 'frameset' and
6343              $self->{insertion_mode} == IN_FRAMESET_IM) {              $self->{insertion_mode} == IN_FRAMESET_IM) {
6344            !!!cp ('t318');            !!!cp ('t318');
# Line 6354  sub _tree_construction_main ($) { Line 6359  sub _tree_construction_main ($) {
6359            ## NOTE: As if in head.            ## NOTE: As if in head.
6360            $parse_rcdata->(CDATA_CONTENT_MODEL);            $parse_rcdata->(CDATA_CONTENT_MODEL);
6361            next B;            next B;
6362    
6363              ## NOTE: |<!DOCTYPE HTML><frameset></frameset></html><noframes></noframes>|
6364              ## has no parse error.
6365          } else {          } else {
6366            if ($self->{insertion_mode} == IN_FRAMESET_IM) {            if ($self->{insertion_mode} == IN_FRAMESET_IM) {
6367              !!!cp ('t321');              !!!cp ('t321');
6368              !!!parse-error (type => 'in frameset',              !!!parse-error (type => 'in frameset',
6369                              text => $token->{tag_name}, token => $token);                              text => $token->{tag_name}, token => $token);
6370            } else {            } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6371              !!!cp ('t322');              !!!cp ('t322');
6372              !!!parse-error (type => 'after frameset',              !!!parse-error (type => 'after frameset',
6373                              text => $token->{tag_name}, token => $token);                              text => $token->{tag_name}, token => $token);
6374              } else { # "after after frameset"
6375                !!!cp ('t322.2');
6376                !!!parse-error (type => 'after after frameset',
6377                                text => $token->{tag_name}, token => $token);
6378            }            }
6379            ## Ignore the token            ## Ignore the token
6380            !!!nack ('t322.1');            !!!nack ('t322.1');
# Line 6370  sub _tree_construction_main ($) { Line 6382  sub _tree_construction_main ($) {
6382            next B;            next B;
6383          }          }
6384        } elsif ($token->{type} == END_TAG_TOKEN) {        } elsif ($token->{type} == END_TAG_TOKEN) {
         if ($self->{insertion_mode} == AFTER_HTML_FRAMESET_IM) {  
           !!!cp ('t323');  
           !!!parse-error (type => 'after html:/',  
                           text => $token->{tag_name}, token => $token);  
   
           $self->{insertion_mode} = AFTER_FRAMESET_IM;  
           ## Process in the "after frameset" insertion mode.  
         } else {  
           !!!cp ('t324');  
         }  
   
6385          if ($token->{tag_name} eq 'frameset' and          if ($token->{tag_name} eq 'frameset' and
6386              $self->{insertion_mode} == IN_FRAMESET_IM) {              $self->{insertion_mode} == IN_FRAMESET_IM) {
6387            if ($self->{open_elements}->[-1]->[1] & HTML_EL and            if ($self->{open_elements}->[-1]->[1] & HTML_EL and
# Line 6415  sub _tree_construction_main ($) { Line 6416  sub _tree_construction_main ($) {
6416              !!!cp ('t330');              !!!cp ('t330');
6417              !!!parse-error (type => 'in frameset:/',              !!!parse-error (type => 'in frameset:/',
6418                              text => $token->{tag_name}, token => $token);                              text => $token->{tag_name}, token => $token);
6419            } else {            } elsif ($self->{insertion_mode} == AFTER_FRAMESET_IM) {
6420              !!!cp ('t331');              !!!cp ('t330.1');
6421              !!!parse-error (type => 'after frameset:/',              !!!parse-error (type => 'after frameset:/',
6422                              text => $token->{tag_name}, token => $token);                              text => $token->{tag_name}, token => $token);
6423              } else { # "after after html"
6424                !!!cp ('t331');
6425                !!!parse-error (type => 'after after frameset:/',
6426                                text => $token->{tag_name}, token => $token);
6427            }            }
6428            ## Ignore the token            ## Ignore the token
6429            !!!next-token;            !!!next-token;
# Line 7141  sub _tree_construction_main ($) { Line 7146  sub _tree_construction_main ($) {
7146            !!!cp ('t413');            !!!cp ('t413');
7147            !!!parse-error (type => 'unmatched end tag',            !!!parse-error (type => 'unmatched end tag',
7148                            text => $token->{tag_name}, token => $token);                            text => $token->{tag_name}, token => $token);
7149              ## NOTE: Ignore the token.
7150          } else {          } else {
7151            ## Step 1. generate implied end tags            ## Step 1. generate implied end tags
7152            while ({            while ({
# Line 7200  sub _tree_construction_main ($) { Line 7206  sub _tree_construction_main ($) {
7206            !!!cp ('t421');            !!!cp ('t421');
7207            !!!parse-error (type => 'unmatched end tag',            !!!parse-error (type => 'unmatched end tag',
7208                            text => $token->{tag_name}, token => $token);                            text => $token->{tag_name}, token => $token);
7209              ## NOTE: Ignore the token.
7210          } else {          } else {
7211            ## Step 1. generate implied end tags            ## Step 1. generate implied end tags
7212            while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {            while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {
# Line 7246  sub _tree_construction_main ($) { Line 7253  sub _tree_construction_main ($) {
7253            !!!cp ('t425.1');            !!!cp ('t425.1');
7254            !!!parse-error (type => 'unmatched end tag',            !!!parse-error (type => 'unmatched end tag',
7255                            text => $token->{tag_name}, token => $token);                            text => $token->{tag_name}, token => $token);
7256              ## NOTE: Ignore the token.
7257          } else {          } else {
7258            ## Step 1. generate implied end tags            ## Step 1. generate implied end tags
7259            while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {            while ($self->{open_elements}->[-1]->[1] & END_TAG_OPTIONAL_EL) {

Legend:
Removed from v.1.156  
changed lines
  Added in v.1.160

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24