/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.166 by wakaba, Sat Sep 13 08:21:35 2008 UTC revision 1.167 by wakaba, Sat Sep 13 09:02:28 2008 UTC
# Line 812  sub CDATA_SECTION_MSE1_STATE () { 40 } # Line 812  sub CDATA_SECTION_MSE1_STATE () { 40 } #
812  sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec  sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec
813  sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec  sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec
814  sub SYSTEM_STATE () { 43 } # "after DOCTYPE name state" in the spec  sub SYSTEM_STATE () { 43 } # "after DOCTYPE name state" in the spec
815    sub ENTITY_STATE () { 44 } # "consume a character reference" in the spec
816    
817  sub DOCTYPE_TOKEN () { 1 }  sub DOCTYPE_TOKEN () { 1 }
818  sub COMMENT_TOKEN () { 2 }  sub COMMENT_TOKEN () { 2 }
# Line 940  sub _get_next_token ($) { Line 941  sub _get_next_token ($) {
941          if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA          if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA
942              not $self->{escape}) {              not $self->{escape}) {
943            !!!cp (1);            !!!cp (1);
944            $self->{state} = ENTITY_DATA_STATE;            ## NOTE: In the spec, the tokenizer is switched to the
945              ## "entity data state".  In this implementation, the tokenizer
946              ## is switched to the |ENTITY_STATE|, which is an implementation
947              ## of the "consume a character reference" algorithm.
948              #$self->{state} = ENTITY_DATA_STATE;
949              $self->{entity_in_attr} = 0;
950              $self->{entity_additional} = -1;
951              $self->{state} = ENTITY_STATE;
952            !!!next-input-character;            !!!next-input-character;
953            redo A;            redo A;
954          } else {          } else {
# Line 1011  sub _get_next_token ($) { Line 1019  sub _get_next_token ($) {
1019    
1020        redo A;        redo A;
1021      } elsif ($self->{state} == ENTITY_DATA_STATE) {      } elsif ($self->{state} == ENTITY_DATA_STATE) {
       ## (cannot happen in CDATA state)  
   
1022        my ($l, $c) = ($self->{line_prev}, $self->{column_prev});        my ($l, $c) = ($self->{line_prev}, $self->{column_prev});
1023          
1024        my $token = $self->_tokenize_attempt_to_consume_an_entity (0, -1);        my $token = $self->{entity_return};
1025    
1026        $self->{state} = DATA_STATE;        $self->{state} = DATA_STATE;
1027        # next-input-character is already done        # next-input-character is already done
# Line 1705  sub _get_next_token ($) { Line 1711  sub _get_next_token ($) {
1711        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{next_char} == 0x0026) { # &
1712          !!!cp (96);          !!!cp (96);
1713          $self->{last_attribute_value_state} = $self->{state};          $self->{last_attribute_value_state} = $self->{state};
1714          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          ## NOTE: In the spec, the tokenizer is switched to the
1715            ## "entity in attribute value state".  In this implementation, the
1716            ## tokenizer is switched to the |ENTITY_STATE|, which is an
1717            ## implementation of the "consume a character reference" algorithm.
1718            #$self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;
1719            $self->{entity_in_attr} = 1;
1720            $self->{entity_additional} = 0x0022; # "
1721            $self->{state} = ENTITY_STATE;
1722          !!!next-input-character;          !!!next-input-character;
1723          redo A;          redo A;
1724        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
# Line 1747  sub _get_next_token ($) { Line 1760  sub _get_next_token ($) {
1760        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{next_char} == 0x0026) { # &
1761          !!!cp (102);          !!!cp (102);
1762          $self->{last_attribute_value_state} = $self->{state};          $self->{last_attribute_value_state} = $self->{state};
1763          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          ## NOTE: In the spec, the tokenizer is switched to the
1764            ## "entity in attribute value state".  In this implementation, the
1765            ## tokenizer is switched to the |ENTITY_STATE|, which is an
1766            ## implementation of the "consume a character reference" algorithm.
1767            #$self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;
1768            $self->{entity_in_attr} = 1;
1769            $self->{entity_additional} = 0x0027; # '
1770            $self->{state} = ENTITY_STATE;
1771          !!!next-input-character;          !!!next-input-character;
1772          redo A;          redo A;
1773        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
# Line 1793  sub _get_next_token ($) { Line 1813  sub _get_next_token ($) {
1813        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{next_char} == 0x0026) { # &
1814          !!!cp (108);          !!!cp (108);
1815          $self->{last_attribute_value_state} = $self->{state};          $self->{last_attribute_value_state} = $self->{state};
1816          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          ## NOTE: In the spec, the tokenizer is switched to the
1817            ## "entity in attribute value state".  In this implementation, the
1818            ## tokenizer is switched to the |ENTITY_STATE|, which is an
1819            ## implementation of the "consume a character reference" algorithm.
1820            #$self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;
1821            $self->{entity_in_attr} = 1;
1822            $self->{entity_additional} = -1;
1823            $self->{state} = ENTITY_STATE;
1824          !!!next-input-character;          !!!next-input-character;
1825          redo A;          redo A;
1826        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
# Line 1858  sub _get_next_token ($) { Line 1885  sub _get_next_token ($) {
1885          redo A;          redo A;
1886        }        }
1887      } elsif ($self->{state} == ENTITY_IN_ATTRIBUTE_VALUE_STATE) {      } elsif ($self->{state} == ENTITY_IN_ATTRIBUTE_VALUE_STATE) {
1888        my $token = $self->_tokenize_attempt_to_consume_an_entity        my $token = $self->{entity_return};
           (1,  
            $self->{last_attribute_value_state}  
              == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE ? 0x0022 : # "  
            $self->{last_attribute_value_state}  
              == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE ? 0x0027 : # '  
            -1);  
1889    
1890        unless (defined $token) {        unless (defined $token) {
1891          !!!cp (117);          !!!cp (117);
# Line 1998  sub _get_next_token ($) { Line 2019  sub _get_next_token ($) {
2019        }        }
2020      } elsif ($self->{state} == BOGUS_COMMENT_STATE) {      } elsif ($self->{state} == BOGUS_COMMENT_STATE) {
2021        ## (only happen if PCDATA state)        ## (only happen if PCDATA state)
         
       ## NOTE: Set by the previous state  
       #my $token = {type => COMMENT_TOKEN, data => ''};  
   
       BC: {  
         if ($self->{next_char} == 0x003E) { # >  
           !!!cp (124);  
           $self->{state} = DATA_STATE;  
           !!!next-input-character;  
2022    
2023            !!!emit ($self->{current_token}); # comment        ## NOTE: Unlike spec's "bogus comment state", this implementation
2024          ## consumes characters one-by-one basis.
2025            redo A;        
2026          } elsif ($self->{next_char} == -1) {        if ($self->{next_char} == 0x003E) { # >
2027            !!!cp (125);          !!!cp (124);
2028            $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2029            ## reconsume          !!!next-input-character;
   
           !!!emit ($self->{current_token}); # comment  
2030    
2031            redo A;          !!!emit ($self->{current_token}); # comment
2032          } else {          redo A;
2033            !!!cp (126);        } elsif ($self->{next_char} == -1) {
2034            $self->{current_token}->{data} .= chr ($self->{next_char}); # comment          !!!cp (125);
2035            !!!next-input-character;          $self->{state} = DATA_STATE;
2036            redo BC;          ## reconsume
         }  
       } # BC  
2037    
2038        die "$0: _get_next_token: unexpected case [BC]";          !!!emit ($self->{current_token}); # comment
2039            redo A;
2040          } else {
2041            !!!cp (126);
2042            $self->{current_token}->{data} .= chr ($self->{next_char}); # comment
2043            ## Stay in the state.
2044            !!!next-input-character;
2045            redo A;
2046          }
2047      } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {      } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {
2048        ## (only happen if PCDATA state)        ## (only happen if PCDATA state)
2049                
# Line 2963  sub _get_next_token ($) { Line 2979  sub _get_next_token ($) {
2979          ## Reconsume.          ## Reconsume.
2980          redo A;          redo A;
2981        }        }
     } else {  
       die "$0: $self->{state}: Unknown state";  
     }  
   } # A    
   
   die "$0: _get_next_token: unexpected case";  
 } # _get_next_token  
2982    
2983  sub _tokenize_attempt_to_consume_an_entity ($$$) {      } elsif ($self->{state} == ENTITY_STATE) {
2984    my ($self, $in_attr, $additional) = @_;        my $in_attr = $self->{entity_in_attr};
2985          my $additional = $self->{entity_additional};
2986    
2987    my ($l, $c) = ($self->{line_prev}, $self->{column_prev});    my ($l, $c) = ($self->{line_prev}, $self->{column_prev});
2988    
# Line 2984  sub _tokenize_attempt_to_consume_an_enti Line 2994  sub _tokenize_attempt_to_consume_an_enti
2994      !!!cp (1001);      !!!cp (1001);
2995      ## Don't consume      ## Don't consume
2996      ## No error      ## No error
2997      return undef;      $self->{entity_return} = undef;
2998        $self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE;
2999        redo A;
3000    } elsif ($self->{next_char} == 0x0023) { # #    } elsif ($self->{next_char} == 0x0023) { # #
3001      !!!next-input-character;      !!!next-input-character;
3002      if ($self->{next_char} == 0x0078 or # x      if ($self->{next_char} == 0x0078 or # x
# Line 3019  sub _tokenize_attempt_to_consume_an_enti Line 3031  sub _tokenize_attempt_to_consume_an_enti
3031            !!!parse-error (type => 'bare hcro', line => $l, column => $c);            !!!parse-error (type => 'bare hcro', line => $l, column => $c);
3032            !!!back-next-input-character ($x_char, $self->{next_char});            !!!back-next-input-character ($x_char, $self->{next_char});
3033            $self->{next_char} = 0x0023; # #            $self->{next_char} = 0x0023; # #
3034            return undef;            $self->{entity_return} = undef;
3035              $self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE;
3036              redo A;
3037          } elsif ($self->{next_char} == 0x003B) { # ;          } elsif ($self->{next_char} == 0x003B) { # ;
3038            !!!cp (1006);            !!!cp (1006);
3039            !!!next-input-character;            !!!next-input-character;
# Line 3050  sub _tokenize_attempt_to_consume_an_enti Line 3064  sub _tokenize_attempt_to_consume_an_enti
3064            $code = $c1_entity_char->{$code};            $code = $c1_entity_char->{$code};
3065          }          }
3066    
3067          return {type => CHARACTER_TOKEN, data => chr $code,          $self->{entity_return} = {type => CHARACTER_TOKEN, data => chr $code,
3068                  has_reference => 1,                  has_reference => 1,
3069                  line => $l, column => $c,                  line => $l, column => $c,
3070                 };                 };
3071            $self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE;
3072            redo A;
3073        } # X        } # X
3074      } elsif (0x0030 <= $self->{next_char} and      } elsif (0x0030 <= $self->{next_char} and
3075               $self->{next_char} <= 0x0039) { # 0..9               $self->{next_char} <= 0x0039) { # 0..9
# Line 3102  sub _tokenize_attempt_to_consume_an_enti Line 3118  sub _tokenize_attempt_to_consume_an_enti
3118          $code = $c1_entity_char->{$code};          $code = $c1_entity_char->{$code};
3119        }        }
3120                
3121        return {type => CHARACTER_TOKEN, data => chr $code, has_reference => 1,        $self->{entity_return} = {type => CHARACTER_TOKEN, data => chr $code, has_reference => 1,
3122                line => $l, column => $c,                line => $l, column => $c,
3123               };               };
3124          $self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE;
3125          redo A;
3126      } else {      } else {
3127        !!!cp (1019);        !!!cp (1019);
3128        !!!parse-error (type => 'bare nero', line => $l, column => $c);        !!!parse-error (type => 'bare nero', line => $l, column => $c);
3129        !!!back-next-input-character ($self->{next_char});        !!!back-next-input-character ($self->{next_char});
3130        $self->{next_char} = 0x0023; # #        $self->{next_char} = 0x0023; # #
3131        return undef;        $self->{entity_return} = undef;
3132          $self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE;
3133          redo A;
3134      }      }
3135    } elsif ((0x0041 <= $self->{next_char} and    } elsif ((0x0041 <= $self->{next_char} and
3136              $self->{next_char} <= 0x005A) or              $self->{next_char} <= 0x005A) or
# Line 3157  sub _tokenize_attempt_to_consume_an_enti Line 3177  sub _tokenize_attempt_to_consume_an_enti
3177            
3178      if ($match > 0) {      if ($match > 0) {
3179        !!!cp (1023);        !!!cp (1023);
3180        return {type => CHARACTER_TOKEN, data => $value, has_reference => 1,        $self->{entity_return} = {type => CHARACTER_TOKEN, data => $value, has_reference => 1,
3181                line => $l, column => $c,                line => $l, column => $c,
3182               };               };
3183          $self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE;
3184          redo A;
3185      } elsif ($match < 0) {      } elsif ($match < 0) {
3186        !!!parse-error (type => 'no refc', line => $l, column => $c);        !!!parse-error (type => 'no refc', line => $l, column => $c);
3187        if ($in_attr and $match < -1) {        if ($in_attr and $match < -1) {
3188          !!!cp (1024);          !!!cp (1024);
3189          return {type => CHARACTER_TOKEN, data => '&'.$entity_name,          $self->{entity_return} = {type => CHARACTER_TOKEN, data => '&'.$entity_name,
3190                  line => $l, column => $c,                  line => $l, column => $c,
3191                 };                 };
3192            $self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE;
3193            redo A;
3194        } else {        } else {
3195          !!!cp (1025);          !!!cp (1025);
3196          return {type => CHARACTER_TOKEN, data => $value, has_reference => 1,          $self->{entity_return} = {type => CHARACTER_TOKEN, data => $value, has_reference => 1,
3197                  line => $l, column => $c,                  line => $l, column => $c,
3198                 };                 };
3199            $self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE;
3200            redo A;
3201        }        }
3202      } else {      } else {
3203        !!!cp (1026);        !!!cp (1026);
3204        !!!parse-error (type => 'bare ero', line => $l, column => $c);        !!!parse-error (type => 'bare ero', line => $l, column => $c);
3205        ## NOTE: "No characters are consumed" in the spec.        ## NOTE: "No characters are consumed" in the spec.
3206        return {type => CHARACTER_TOKEN, data => '&'.$value,        $self->{entity_return} = {type => CHARACTER_TOKEN, data => '&'.$value,
3207                line => $l, column => $c,                line => $l, column => $c,
3208               };               };
3209          $self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE;
3210          redo A;
3211      }      }
3212    } else {    } else {
3213      !!!cp (1027);      !!!cp (1027);
3214      ## no characters are consumed      ## no characters are consumed
3215      !!!parse-error (type => 'bare ero', line => $l, column => $c);      !!!parse-error (type => 'bare ero', line => $l, column => $c);
3216      return undef;      $self->{entity_return} = undef;
3217        $self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE;
3218        redo A;
3219    }    }
3220  } # _tokenize_attempt_to_consume_an_entity  
3221        } else {
3222          die "$0: $self->{state}: Unknown state";
3223        }
3224      } # A  
3225    
3226      die "$0: _get_next_token: unexpected case";
3227    } # _get_next_token
3228    
3229  sub _initialize_tree_constructor ($) {  sub _initialize_tree_constructor ($) {
3230    my $self = shift;    my $self = shift;

Legend:
Removed from v.1.166  
changed lines
  Added in v.1.167

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24