/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.136 by wakaba, Sat May 17 12:29:24 2008 UTC revision 1.141 by wakaba, Sat May 24 10:18:26 2008 UTC
# Line 11  use Error qw(:try); Line 11  use Error qw(:try);
11  ## TODO: 1252 parse error (revision 1264)  ## TODO: 1252 parse error (revision 1264)
12  ## TODO: 8859-11 = 874 (revision 1271)  ## TODO: 8859-11 = 874 (revision 1271)
13    
14    require IO::Handle;
15    
16  my $HTML_NS = q<http://www.w3.org/1999/xhtml>;  my $HTML_NS = q<http://www.w3.org/1999/xhtml>;
17  my $MML_NS = q<http://www.w3.org/1998/Math/MathML>;  my $MML_NS = q<http://www.w3.org/1998/Math/MathML>;
18  my $SVG_NS = q<http://www.w3.org/2000/svg>;  my $SVG_NS = q<http://www.w3.org/2000/svg>;
# Line 332  my $c1_entity_char = { Line 334  my $c1_entity_char = {
334  }; # $c1_entity_char  }; # $c1_entity_char
335    
336  sub parse_byte_string ($$$$;$) {  sub parse_byte_string ($$$$;$) {
337      my $self = shift;
338      my $charset_name = shift;
339      open my $input, '<', ref $_[0] ? $_[0] : \($_[0]);
340      return $self->parse_byte_stream ($charset_name, $input, @_[1..$#_]);
341    } # parse_byte_string
342    
343    sub parse_byte_stream ($$$$;$) {
344    my $self = ref $_[0] ? shift : shift->new;    my $self = ref $_[0] ? shift : shift->new;
345    my $charset_name = shift;    my $charset_name = shift;
346    open my $byte_stream, '<', ref $_[0] ? $_[0] : \($_[0]);    my $byte_stream = $_[0];
347    
348    my $onerror = $_[2] || sub {    my $onerror = $_[2] || sub {
349      my (%opt) = @_;      my (%opt) = @_;
# Line 513  sub parse_byte_string ($$$$;$) { Line 522  sub parse_byte_string ($$$$;$) {
522    
523    my $char_onerror = sub {    my $char_onerror = sub {
524      my (undef, $type, %opt) = @_;      my (undef, $type, %opt) = @_;
525      !!!parse-error (%opt, type => $type);      !!!parse-error (%opt, type => $type,
526                        line => $self->{line}, column => $self->{column} + 1);
527      if ($opt{octets}) {      if ($opt{octets}) {
528        ${$opt{octets}} = "\x{FFFD}"; # relacement character        ${$opt{octets}} = "\x{FFFD}"; # relacement character
529      }      }
# Line 545  sub parse_byte_string ($$$$;$) { Line 555  sub parse_byte_string ($$$$;$) {
555      $return = $self->parse_char_stream ($char_stream, @args);      $return = $self->parse_char_stream ($char_stream, @args);
556    };    };
557    return $return;    return $return;
558  } # parse_byte_string  } # parse_byte_stream
559    
560  ## NOTE: HTML5 spec says that the encoding layer MUST NOT strip BOM  ## NOTE: HTML5 spec says that the encoding layer MUST NOT strip BOM
561  ## and the HTML layer MUST ignore it.  However, we does strip BOM in  ## and the HTML layer MUST ignore it.  However, we does strip BOM in
# Line 558  sub parse_byte_string ($$$$;$) { Line 568  sub parse_byte_string ($$$$;$) {
568    
569  sub parse_char_string ($$$;$) {  sub parse_char_string ($$$;$) {
570    my $self = shift;    my $self = shift;
571    open my $input, '<:utf8', ref $_[0] ? $_[0] : \($_[0]);    require utf8;
572      my $s = ref $_[0] ? $_[0] : \($_[0]);
573      open my $input, '<' . (utf8::is_utf8 ($$s) ? ':utf8' : ''), $s;
574    return $self->parse_char_stream ($input, @_[1..$#_]);    return $self->parse_char_stream ($input, @_[1..$#_]);
575  } # parse_char_string  } # parse_char_string
576  *parse_string = \&parse_char_string;  *parse_string = \&parse_char_string;
# Line 584  sub parse_char_stream ($$$;$) { Line 596  sub parse_char_stream ($$$;$) {
596      pop @{$self->{prev_char}};      pop @{$self->{prev_char}};
597      unshift @{$self->{prev_char}}, $self->{next_char};      unshift @{$self->{prev_char}}, $self->{next_char};
598    
599      my $char = $input->getc;      my $char;
600        if (defined $self->{next_next_char}) {
601          $char = $self->{next_next_char};
602          delete $self->{next_next_char};
603        } else {
604          $char = $input->getc;
605        }
606      $self->{next_char} = -1 and return unless defined $char;      $self->{next_char} = -1 and return unless defined $char;
607      $self->{next_char} = ord $char;      $self->{next_char} = ord $char;
608    
# Line 599  sub parse_char_stream ($$$;$) { Line 617  sub parse_char_stream ($$$;$) {
617      } elsif ($self->{next_char} == 0x000D) { # CR      } elsif ($self->{next_char} == 0x000D) { # CR
618        !!!cp ('j2');        !!!cp ('j2');
619        my $next = $input->getc;        my $next = $input->getc;
620        if ($next ne "\x0A") {        if (defined $next and $next ne "\x0A") {
621          $input->ungetc ($next);          $self->{next_next_char} = $next;
622        }        }
623        $self->{next_char} = 0x000A; # LF # MUST        $self->{next_char} = 0x000A; # LF # MUST
624        $self->{line}++;        $self->{line}++;
# Line 1809  sub _get_next_token ($) { Line 1827  sub _get_next_token ($) {
1827          $self->{state} = SELF_CLOSING_START_TAG_STATE;          $self->{state} = SELF_CLOSING_START_TAG_STATE;
1828          !!!next-input-character;          !!!next-input-character;
1829          redo A;          redo A;
1830          } elsif ($self->{next_char} == -1) {
1831            !!!parse-error (type => 'unclosed tag');
1832            if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1833              !!!cp (122.3);
1834              $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1835            } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1836              if ($self->{current_token}->{attributes}) {
1837                !!!cp (122.1);
1838                !!!parse-error (type => 'end tag attribute');
1839              } else {
1840                ## NOTE: This state should never be reached.
1841                !!!cp (122.2);
1842              }
1843            } else {
1844              die "$0: $self->{current_token}->{type}: Unknown token type";
1845            }
1846            $self->{state} = DATA_STATE;
1847            ## Reconsume.
1848            !!!emit ($self->{current_token}); # start tag or end tag
1849            redo A;
1850        } else {        } else {
1851          !!!cp ('124.1');          !!!cp ('124.1');
1852          !!!parse-error (type => 'no space between attributes');          !!!parse-error (type => 'no space between attributes');
# Line 1841  sub _get_next_token ($) { Line 1879  sub _get_next_token ($) {
1879          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
1880    
1881          redo A;          redo A;
1882          } elsif ($self->{next_char} == -1) {
1883            !!!parse-error (type => 'unclosed tag');
1884            if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1885              !!!cp (124.7);
1886              $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1887            } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1888              if ($self->{current_token}->{attributes}) {
1889                !!!cp (124.5);
1890                !!!parse-error (type => 'end tag attribute');
1891              } else {
1892                ## NOTE: This state should never be reached.
1893                !!!cp (124.6);
1894              }
1895            } else {
1896              die "$0: $self->{current_token}->{type}: Unknown token type";
1897            }
1898            $self->{state} = DATA_STATE;
1899            ## Reconsume.
1900            !!!emit ($self->{current_token}); # start tag or end tag
1901            redo A;
1902        } else {        } else {
1903          !!!cp ('124.4');          !!!cp ('124.4');
1904          !!!parse-error (type => 'nestc');          !!!parse-error (type => 'nestc');
# Line 3356  sub _reset_insertion_mode ($) { Line 3414  sub _reset_insertion_mode ($) {
3414        if ($self->{open_elements}->[0]->[0] eq $node->[0]) {        if ($self->{open_elements}->[0]->[0] eq $node->[0]) {
3415          $last = 1;          $last = 1;
3416          if (defined $self->{inner_html_node}) {          if (defined $self->{inner_html_node}) {
3417            if ($self->{inner_html_node}->[1] & TABLE_CELL_EL) {            !!!cp ('t28');
3418              !!!cp ('t27');            $node = $self->{inner_html_node};
3419              #          } else {
3420            } else {            die "_reset_insertion_mode: t27";
             !!!cp ('t28');  
             $node = $self->{inner_html_node};  
           }  
3421          }          }
3422        }        }
3423              
3424      ## Step 4..14        ## Step 4..14
3425      my $new_mode;        my $new_mode;
3426      if ($node->[1] & FOREIGN_EL) {        if ($node->[1] & FOREIGN_EL) {
3427        ## NOTE: Strictly spaking, the line below only applies to MathML and          !!!cp ('t28.1');
3428        ## SVG elements.  Currently the HTML syntax supports only MathML and          ## NOTE: Strictly spaking, the line below only applies to MathML and
3429        ## SVG elements as foreigners.          ## SVG elements.  Currently the HTML syntax supports only MathML and
3430        $new_mode = $self->{insertion_mode} | IN_FOREIGN_CONTENT_IM;          ## SVG elements as foreigners.
3431        ## ISSUE: What is set as the secondary insertion mode?          $new_mode = $self->{insertion_mode} | IN_FOREIGN_CONTENT_IM;
3432      } else {          ## ISSUE: What is set as the secondary insertion mode?
3433        $new_mode = {        } elsif ($node->[1] & TABLE_CELL_EL) {
3434            if ($last) {
3435              !!!cp ('t28.2');
3436              #
3437            } else {
3438              !!!cp ('t28.3');
3439              $new_mode = IN_CELL_IM;
3440            }
3441          } else {
3442            !!!cp ('t28.4');
3443            $new_mode = {
3444                        select => IN_SELECT_IM,                        select => IN_SELECT_IM,
3445                        ## NOTE: |option| and |optgroup| do not set                        ## NOTE: |option| and |optgroup| do not set
3446                        ## insertion mode to "in select" by themselves.                        ## insertion mode to "in select" by themselves.
                       td => IN_CELL_IM,  
                       th => IN_CELL_IM,  
3447                        tr => IN_ROW_IM,                        tr => IN_ROW_IM,
3448                        tbody => IN_TABLE_BODY_IM,                        tbody => IN_TABLE_BODY_IM,
3449                        thead => IN_TABLE_BODY_IM,                        thead => IN_TABLE_BODY_IM,
# Line 3392  sub _reset_insertion_mode ($) { Line 3455  sub _reset_insertion_mode ($) {
3455                        body => IN_BODY_IM,                        body => IN_BODY_IM,
3456                        frameset => IN_FRAMESET_IM,                        frameset => IN_FRAMESET_IM,
3457                       }->{$node->[0]->manakai_local_name};                       }->{$node->[0]->manakai_local_name};
3458      }        }
3459      $self->{insertion_mode} = $new_mode and return if defined $new_mode;        $self->{insertion_mode} = $new_mode and return if defined $new_mode;
3460                
3461        ## Step 15        ## Step 15
3462        if ($node->[1] & HTML_EL) {        if ($node->[1] & HTML_EL) {
# Line 4136  sub _tree_construction_main ($) { Line 4199  sub _tree_construction_main ($) {
4199              !!!next-token;              !!!next-token;
4200              next B;              next B;
4201            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {            } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4202              !!!cp ('t94');              !!!cp ('t93.2');
4203              #              !!!parse-error (type => 'after head:head', token => $token); ## TODO: error type
4204                ## Ignore the token
4205                !!!nack ('t93.3');
4206                !!!next-token;
4207                next B;
4208            } else {            } else {
4209              !!!cp ('t95');              !!!cp ('t95');
4210              !!!parse-error (type => 'in head:head', token => $token); # or in head noscript              !!!parse-error (type => 'in head:head', token => $token); # or in head noscript
# Line 4459  sub _tree_construction_main ($) { Line 4526  sub _tree_construction_main ($) {
4526                  $self->{insertion_mode} = AFTER_HEAD_IM;                  $self->{insertion_mode} = AFTER_HEAD_IM;
4527                  !!!next-token;                  !!!next-token;
4528                  next B;                  next B;
4529                  } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4530                    !!!cp ('t134.1');
4531                    !!!parse-error (type => 'unmatched end tag:head', token => $token);
4532                    ## Ignore the token
4533                    !!!next-token;
4534                    next B;
4535                } else {                } else {
4536                  !!!cp ('t135');                  die "$0: $self->{insertion_mode}: Unknown insertion mode";
                 #  
4537                }                }
4538              } elsif ($token->{tag_name} eq 'noscript') {              } elsif ($token->{tag_name} eq 'noscript') {
4539                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {                if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
# Line 4470  sub _tree_construction_main ($) { Line 4542  sub _tree_construction_main ($) {
4542                  $self->{insertion_mode} = IN_HEAD_IM;                  $self->{insertion_mode} = IN_HEAD_IM;
4543                  !!!next-token;                  !!!next-token;
4544                  next B;                  next B;
4545                } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM) {                } elsif ($self->{insertion_mode} == BEFORE_HEAD_IM or
4546                           $self->{insertion_mode} == AFTER_HEAD_IM) {
4547                  !!!cp ('t137');                  !!!cp ('t137');
4548                  !!!parse-error (type => 'unmatched end tag:noscript', token => $token);                  !!!parse-error (type => 'unmatched end tag:noscript', token => $token);
4549                  ## Ignore the token ## ISSUE: An issue in the spec.                  ## Ignore the token ## ISSUE: An issue in the spec.
# Line 4483  sub _tree_construction_main ($) { Line 4556  sub _tree_construction_main ($) {
4556              } elsif ({              } elsif ({
4557                        body => 1, html => 1,                        body => 1, html => 1,
4558                       }->{$token->{tag_name}}) {                       }->{$token->{tag_name}}) {
4559                if ($self->{insertion_mode} == BEFORE_HEAD_IM) {                if ($self->{insertion_mode} == BEFORE_HEAD_IM or
4560                  !!!cp ('t139');                    $self->{insertion_mode} == IN_HEAD_IM or
4561                  ## As if <head>                    $self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
                 !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);  
                 $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});  
                 push @{$self->{open_elements}},  
                     [$self->{head_element}, $el_category->{head}];  
   
                 $self->{insertion_mode} = IN_HEAD_IM;  
                 ## Reprocess in the "in head" insertion mode...  
               } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {  
4562                  !!!cp ('t140');                  !!!cp ('t140');
4563                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
4564                  ## Ignore the token                  ## Ignore the token
4565                  !!!next-token;                  !!!next-token;
4566                  next B;                  next B;
4567                  } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4568                    !!!cp ('t140.1');
4569                    !!!parse-error (type => 'unmatched end tag:' . $token->{tag_name}, token => $token);
4570                    ## Ignore the token
4571                    !!!next-token;
4572                    next B;
4573                } else {                } else {
4574                  !!!cp ('t141');                  die "$0: $self->{insertion_mode}: Unknown insertion mode";
4575                }                }
4576                              } elsif ($token->{tag_name} eq 'p') {
4577                #                !!!cp ('t142');
4578              } elsif ({                !!!parse-error (type => 'unmatched end tag:p', token => $token);
4579                        p => 1, br => 1,                ## Ignore the token
4580                       }->{$token->{tag_name}}) {                !!!next-token;
4581                  next B;
4582                } elsif ($token->{tag_name} eq 'br') {
4583                if ($self->{insertion_mode} == BEFORE_HEAD_IM) {                if ($self->{insertion_mode} == BEFORE_HEAD_IM) {
4584                  !!!cp ('t142');                  !!!cp ('t142.2');
4585                  ## As if <head>                  ## (before head) as if <head>, (in head) as if </head>
4586                  !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);                  !!!create-element ($self->{head_element}, $HTML_NS, 'head',, $token);
4587                  $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});                  $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
4588                  push @{$self->{open_elements}},                  $self->{insertion_mode} = AFTER_HEAD_IM;
4589                      [$self->{head_element}, $el_category->{head}];    
4590                    ## Reprocess in the "after head" insertion mode...
4591                  } elsif ($self->{insertion_mode} == IN_HEAD_IM) {
4592                    !!!cp ('t143.2');
4593                    ## As if </head>
4594                    pop @{$self->{open_elements}};
4595                    $self->{insertion_mode} = AFTER_HEAD_IM;
4596      
4597                    ## Reprocess in the "after head" insertion mode...
4598                  } elsif ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {
4599                    !!!cp ('t143.3');
4600                    ## ISSUE: Two parse errors for <head><noscript></br>
4601                    !!!parse-error (type => 'unmatched end tag:br', token => $token);
4602                    ## As if </noscript>
4603                    pop @{$self->{open_elements}};
4604                  $self->{insertion_mode} = IN_HEAD_IM;                  $self->{insertion_mode} = IN_HEAD_IM;
4605    
4606                  ## Reprocess in the "in head" insertion mode...                  ## Reprocess in the "in head" insertion mode...
4607                } else {                  ## As if </head>
4608                  !!!cp ('t143');                  pop @{$self->{open_elements}};
4609                }                  $self->{insertion_mode} = AFTER_HEAD_IM;
4610    
4611                #                  ## Reprocess in the "after head" insertion mode...
4612              } else {                } elsif ($self->{insertion_mode} == AFTER_HEAD_IM) {
4613                if ($self->{insertion_mode} == AFTER_HEAD_IM) {                  !!!cp ('t143.4');
                 !!!cp ('t144');  
4614                  #                  #
4615                } else {                } else {
4616                  !!!cp ('t145');                  die "$0: $self->{insertion_mode}: Unknown insertion mode";
                 !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);  
                 ## Ignore the token  
                 !!!next-token;  
                 next B;  
4617                }                }
4618    
4619                  ## ISSUE: does not agree with IE7 - it doesn't ignore </br>.
4620                  !!!parse-error (type => 'unmatched end tag:br', token => $token);
4621                  ## Ignore the token
4622                  !!!next-token;
4623                  next B;
4624                } else {
4625                  !!!cp ('t145');
4626                  !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name}, token => $token);
4627                  ## Ignore the token
4628                  !!!next-token;
4629                  next B;
4630              }              }
4631    
4632              if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {              if ($self->{insertion_mode} == IN_HEAD_NOSCRIPT_IM) {

Legend:
Removed from v.1.136  
changed lines
  Added in v.1.141

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24