/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.184 by wakaba, Mon Sep 15 09:02:27 2008 UTC revision 1.186 by wakaba, Sat Sep 20 07:54:47 2008 UTC
# Line 853  sub CDATA_SECTION_STATE () { 35 } Line 853  sub CDATA_SECTION_STATE () { 35 }
853  sub MD_HYPHEN_STATE () { 36 } # "markup declaration open state" in the spec  sub MD_HYPHEN_STATE () { 36 } # "markup declaration open state" in the spec
854  sub MD_DOCTYPE_STATE () { 37 } # "markup declaration open state" in the spec  sub MD_DOCTYPE_STATE () { 37 } # "markup declaration open state" in the spec
855  sub MD_CDATA_STATE () { 38 } # "markup declaration open state" in the spec  sub MD_CDATA_STATE () { 38 } # "markup declaration open state" in the spec
856  sub CDATA_PCDATA_CLOSE_TAG_STATE () { 39 } # "close tag open state" in the spec  sub CDATA_RCDATA_CLOSE_TAG_STATE () { 39 } # "close tag open state" in the spec
857  sub CDATA_SECTION_MSE1_STATE () { 40 } # "CDATA section state" in the spec  sub CDATA_SECTION_MSE1_STATE () { 40 } # "CDATA section state" in the spec
858  sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec  sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec
859  sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec  sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec
# Line 867  sub NCR_NUM_STATE () { 46 } Line 867  sub NCR_NUM_STATE () { 46 }
867  sub HEXREF_X_STATE () { 47 }  sub HEXREF_X_STATE () { 47 }
868  sub HEXREF_HEX_STATE () { 48 }  sub HEXREF_HEX_STATE () { 48 }
869  sub ENTITY_NAME_STATE () { 49 }  sub ENTITY_NAME_STATE () { 49 }
870    sub PCDATA_STATE () { 50 } # "data state" in the spec
871    
872  sub DOCTYPE_TOKEN () { 1 }  sub DOCTYPE_TOKEN () { 1 }
873  sub COMMENT_TOKEN () { 2 }  sub COMMENT_TOKEN () { 2 }
# Line 982  sub _get_next_token ($) { Line 983  sub _get_next_token ($) {
983    }    }
984    
985    A: {    A: {
986      if ($self->{state} == DATA_STATE) {      if ($self->{state} == PCDATA_STATE) {
987          ## NOTE: Same as |DATA_STATE|, but only for |PCDATA| content model.
988    
989        if ($self->{nc} == 0x0026) { # &        if ($self->{nc} == 0x0026) { # &
990          delete $self->{s_kwd};          !!!cp (0.1);
991            ## NOTE: In the spec, the tokenizer is switched to the
992            ## "entity data state".  In this implementation, the tokenizer
993            ## is switched to the |ENTITY_STATE|, which is an implementation
994            ## of the "consume a character reference" algorithm.
995            $self->{entity_add} = -1;
996            $self->{prev_state} = DATA_STATE;
997            $self->{state} = ENTITY_STATE;
998            !!!next-input-character;
999            redo A;
1000          } elsif ($self->{nc} == 0x003C) { # <
1001            !!!cp (0.2);
1002            $self->{state} = TAG_OPEN_STATE;
1003            !!!next-input-character;
1004            redo A;
1005          } elsif ($self->{nc} == -1) {
1006            !!!cp (0.3);
1007            !!!emit ({type => END_OF_FILE_TOKEN,
1008                      line => $self->{line}, column => $self->{column}});
1009            last A; ## TODO: ok?
1010          } else {
1011            !!!cp (0.4);
1012            #
1013          }
1014    
1015          # Anything else
1016          my $token = {type => CHARACTER_TOKEN,
1017                       data => chr $self->{nc},
1018                       line => $self->{line}, column => $self->{column},
1019                      };
1020          $self->{read_until}->($token->{data}, q[<&], length $token->{data});
1021    
1022          ## Stay in the state.
1023          !!!next-input-character;
1024          !!!emit ($token);
1025          redo A;
1026        } elsif ($self->{state} == DATA_STATE) {
1027          $self->{s_kwd} = '' unless defined $self->{s_kwd};
1028          if ($self->{nc} == 0x0026) { # &
1029            $self->{s_kwd} = '';
1030          if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA          if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA
1031              not $self->{escape}) {              not $self->{escape}) {
1032            !!!cp (1);            !!!cp (1);
# Line 1003  sub _get_next_token ($) { Line 1045  sub _get_next_token ($) {
1045          }          }
1046        } elsif ($self->{nc} == 0x002D) { # -        } elsif ($self->{nc} == 0x002D) { # -
1047          if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA          if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1048            if (defined $self->{s_kwd}) {            $self->{s_kwd} .= '-';
1049              !!!cp (2.1);            
             $self->{s_kwd} .= '-';  
           } else {  
             !!!cp (2.2);  
             $self->{s_kwd} = '-';  
           }  
   
1050            if ($self->{s_kwd} eq '<!--') {            if ($self->{s_kwd} eq '<!--') {
1051              !!!cp (3);              !!!cp (3);
1052              $self->{escape} = 1; # unless $self->{escape};              $self->{escape} = 1; # unless $self->{escape};
# Line 1028  sub _get_next_token ($) { Line 1064  sub _get_next_token ($) {
1064                    
1065          #          #
1066        } elsif ($self->{nc} == 0x0021) { # !        } elsif ($self->{nc} == 0x0021) { # !
1067          if (defined $self->{s_kwd}) {          if (length $self->{s_kwd}) {
1068            !!!cp (5.1);            !!!cp (5.1);
1069            $self->{s_kwd} .= '!';            $self->{s_kwd} .= '!';
1070            #            #
1071          } else {          } else {
1072            !!!cp (5.2);            !!!cp (5.2);
1073              #$self->{s_kwd} = '';
1074            #            #
1075          }          }
1076          #          #
1077        } elsif ($self->{nc} == 0x003C) { # <        } elsif ($self->{nc} == 0x003C) { # <
         delete $self->{s_kwd};  
1078          if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA          if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA
1079              (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA              (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA
1080               not $self->{escape})) {               not $self->{escape})) {
# Line 1048  sub _get_next_token ($) { Line 1084  sub _get_next_token ($) {
1084            redo A;            redo A;
1085          } else {          } else {
1086            !!!cp (7);            !!!cp (7);
1087              $self->{s_kwd} = '';
1088            #            #
1089          }          }
1090        } elsif ($self->{nc} == 0x003E) { # >        } elsif ($self->{nc} == 0x003E) { # >
1091          if ($self->{escape} and          if ($self->{escape} and
1092              ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA              ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA
1093            if (defined $self->{s_kwd} and $self->{s_kwd} eq '--') {            if ($self->{s_kwd} eq '--') {
1094              !!!cp (8);              !!!cp (8);
1095              delete $self->{escape};              delete $self->{escape};
1096            } else {            } else {
# Line 1063  sub _get_next_token ($) { Line 1100  sub _get_next_token ($) {
1100            !!!cp (10);            !!!cp (10);
1101          }          }
1102                    
1103          delete $self->{s_kwd};          $self->{s_kwd} = '';
1104          #          #
1105        } elsif ($self->{nc} == -1) {        } elsif ($self->{nc} == -1) {
1106          !!!cp (11);          !!!cp (11);
1107          delete $self->{s_kwd};          $self->{s_kwd} = '';
1108          !!!emit ({type => END_OF_FILE_TOKEN,          !!!emit ({type => END_OF_FILE_TOKEN,
1109                    line => $self->{line}, column => $self->{column}});                    line => $self->{line}, column => $self->{column}});
1110          last A; ## TODO: ok?          last A; ## TODO: ok?
1111        } else {        } else {
1112          !!!cp (12);          !!!cp (12);
1113          delete $self->{s_kwd};          $self->{s_kwd} = '';
1114          #          #
1115        }        }
1116    
# Line 1084  sub _get_next_token ($) { Line 1121  sub _get_next_token ($) {
1121                    };                    };
1122        if ($self->{read_until}->($token->{data}, q[-!<>&],        if ($self->{read_until}->($token->{data}, q[-!<>&],
1123                                  length $token->{data})) {                                  length $token->{data})) {
1124          delete $self->{s_kwd};          $self->{s_kwd} = '';
1125        }        }
1126    
1127        ## Stay in the data state        ## Stay in the data state.
1128          if ($self->{content_model} == PCDATA_CONTENT_MODEL) {
1129            !!!cp (13);
1130            $self->{state} = PCDATA_STATE;
1131          } else {
1132            !!!cp (14);
1133            ## Stay in the state.
1134          }
1135        !!!next-input-character;        !!!next-input-character;
1136        !!!emit ($token);        !!!emit ($token);
1137        redo A;        redo A;
# Line 1192  sub _get_next_token ($) { Line 1236  sub _get_next_token ($) {
1236        }        }
1237      } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {      } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {
1238        ## NOTE: The "close tag open state" in the spec is implemented as        ## NOTE: The "close tag open state" in the spec is implemented as
1239        ## |CLOSE_TAG_OPEN_STATE| and |CDATA_PCDATA_CLOSE_TAG_STATE|.        ## |CLOSE_TAG_OPEN_STATE| and |CDATA_RCDATA_CLOSE_TAG_STATE|.
1240    
1241        my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"        my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"
1242        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
1243          if (defined $self->{last_stag_name}) {          if (defined $self->{last_stag_name}) {
1244            $self->{state} = CDATA_PCDATA_CLOSE_TAG_STATE;            $self->{state} = CDATA_RCDATA_CLOSE_TAG_STATE;
1245            $self->{s_kwd} = '';            $self->{s_kwd} = '';
1246            ## Reconsume.            ## Reconsume.
1247            redo A;            redo A;
# Line 1268  sub _get_next_token ($) { Line 1312  sub _get_next_token ($) {
1312          ## "bogus comment state" entry.          ## "bogus comment state" entry.
1313          redo A;          redo A;
1314        }        }
1315      } elsif ($self->{state} == CDATA_PCDATA_CLOSE_TAG_STATE) {      } elsif ($self->{state} == CDATA_RCDATA_CLOSE_TAG_STATE) {
1316        my $ch = substr $self->{last_stag_name}, length $self->{s_kwd}, 1;        my $ch = substr $self->{last_stag_name}, length $self->{s_kwd}, 1;
1317        if (length $ch) {        if (length $ch) {
1318          my $CH = $ch;          my $CH = $ch;
# Line 4707  sub _tree_construction_main ($) { Line 4751  sub _tree_construction_main ($) {
4751                  } elsif ($token->{attributes}->{content}) {                  } elsif ($token->{attributes}->{content}) {
4752                    if ($token->{attributes}->{content}->{value}                    if ($token->{attributes}->{content}->{value}
4753                        =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]                        =~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
4754                            [\x09-\x0D\x20]*=                            [\x09\x0A\x0C\x0D\x20]*=
4755                            [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|                            [\x09\x0A\x0C\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
4756                            ([^"'\x09-\x0D\x20][^\x09-\x0D\x20\x3B]*))/x) {                            ([^"'\x09\x0A\x0C\x0D\x20]
4757                               [^\x09\x0A\x0C\x0D\x20\x3B]*))/x) {
4758                      !!!cp ('t107');                      !!!cp ('t107');
4759                      ## NOTE: Whether the encoding is supported or not is handled                      ## NOTE: Whether the encoding is supported or not is handled
4760                      ## in the {change_encoding} callback.                      ## in the {change_encoding} callback.

Legend:
Removed from v.1.184  
changed lines
  Added in v.1.186

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24