| 853 |
sub MD_HYPHEN_STATE () { 36 } # "markup declaration open state" in the spec |
sub MD_HYPHEN_STATE () { 36 } # "markup declaration open state" in the spec |
| 854 |
sub MD_DOCTYPE_STATE () { 37 } # "markup declaration open state" in the spec |
sub MD_DOCTYPE_STATE () { 37 } # "markup declaration open state" in the spec |
| 855 |
sub MD_CDATA_STATE () { 38 } # "markup declaration open state" in the spec |
sub MD_CDATA_STATE () { 38 } # "markup declaration open state" in the spec |
| 856 |
sub CDATA_PCDATA_CLOSE_TAG_STATE () { 39 } # "close tag open state" in the spec |
sub CDATA_RCDATA_CLOSE_TAG_STATE () { 39 } # "close tag open state" in the spec |
| 857 |
sub CDATA_SECTION_MSE1_STATE () { 40 } # "CDATA section state" in the spec |
sub CDATA_SECTION_MSE1_STATE () { 40 } # "CDATA section state" in the spec |
| 858 |
sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec |
sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec |
| 859 |
sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec |
sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec |
| 867 |
sub HEXREF_X_STATE () { 47 } |
sub HEXREF_X_STATE () { 47 } |
| 868 |
sub HEXREF_HEX_STATE () { 48 } |
sub HEXREF_HEX_STATE () { 48 } |
| 869 |
sub ENTITY_NAME_STATE () { 49 } |
sub ENTITY_NAME_STATE () { 49 } |
| 870 |
|
sub PCDATA_STATE () { 50 } # "data state" in the spec |
| 871 |
|
|
| 872 |
sub DOCTYPE_TOKEN () { 1 } |
sub DOCTYPE_TOKEN () { 1 } |
| 873 |
sub COMMENT_TOKEN () { 2 } |
sub COMMENT_TOKEN () { 2 } |
| 983 |
} |
} |
| 984 |
|
|
| 985 |
A: { |
A: { |
| 986 |
if ($self->{state} == DATA_STATE) { |
if ($self->{state} == PCDATA_STATE) { |
| 987 |
|
## NOTE: Same as |DATA_STATE|, but only for |PCDATA| content model. |
| 988 |
|
|
| 989 |
if ($self->{nc} == 0x0026) { # & |
if ($self->{nc} == 0x0026) { # & |
| 990 |
delete $self->{s_kwd}; |
!!!cp (0.1); |
| 991 |
|
## NOTE: In the spec, the tokenizer is switched to the |
| 992 |
|
## "entity data state". In this implementation, the tokenizer |
| 993 |
|
## is switched to the |ENTITY_STATE|, which is an implementation |
| 994 |
|
## of the "consume a character reference" algorithm. |
| 995 |
|
$self->{entity_add} = -1; |
| 996 |
|
$self->{prev_state} = DATA_STATE; |
| 997 |
|
$self->{state} = ENTITY_STATE; |
| 998 |
|
!!!next-input-character; |
| 999 |
|
redo A; |
| 1000 |
|
} elsif ($self->{nc} == 0x003C) { # < |
| 1001 |
|
!!!cp (0.2); |
| 1002 |
|
$self->{state} = TAG_OPEN_STATE; |
| 1003 |
|
!!!next-input-character; |
| 1004 |
|
redo A; |
| 1005 |
|
} elsif ($self->{nc} == -1) { |
| 1006 |
|
!!!cp (0.3); |
| 1007 |
|
!!!emit ({type => END_OF_FILE_TOKEN, |
| 1008 |
|
line => $self->{line}, column => $self->{column}}); |
| 1009 |
|
last A; ## TODO: ok? |
| 1010 |
|
} else { |
| 1011 |
|
!!!cp (0.4); |
| 1012 |
|
# |
| 1013 |
|
} |
| 1014 |
|
|
| 1015 |
|
# Anything else |
| 1016 |
|
my $token = {type => CHARACTER_TOKEN, |
| 1017 |
|
data => chr $self->{nc}, |
| 1018 |
|
line => $self->{line}, column => $self->{column}, |
| 1019 |
|
}; |
| 1020 |
|
$self->{read_until}->($token->{data}, q[<&], length $token->{data}); |
| 1021 |
|
|
| 1022 |
|
## Stay in the state. |
| 1023 |
|
!!!next-input-character; |
| 1024 |
|
!!!emit ($token); |
| 1025 |
|
redo A; |
| 1026 |
|
} elsif ($self->{state} == DATA_STATE) { |
| 1027 |
|
$self->{s_kwd} = '' unless defined $self->{s_kwd}; |
| 1028 |
|
if ($self->{nc} == 0x0026) { # & |
| 1029 |
|
$self->{s_kwd} = ''; |
| 1030 |
if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA |
if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA |
| 1031 |
not $self->{escape}) { |
not $self->{escape}) { |
| 1032 |
!!!cp (1); |
!!!cp (1); |
| 1045 |
} |
} |
| 1046 |
} elsif ($self->{nc} == 0x002D) { # - |
} elsif ($self->{nc} == 0x002D) { # - |
| 1047 |
if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA |
if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA |
| 1048 |
if (defined $self->{s_kwd}) { |
$self->{s_kwd} .= '-'; |
| 1049 |
!!!cp (2.1); |
|
|
$self->{s_kwd} .= '-'; |
|
|
} else { |
|
|
!!!cp (2.2); |
|
|
$self->{s_kwd} = '-'; |
|
|
} |
|
|
|
|
| 1050 |
if ($self->{s_kwd} eq '<!--') { |
if ($self->{s_kwd} eq '<!--') { |
| 1051 |
!!!cp (3); |
!!!cp (3); |
| 1052 |
$self->{escape} = 1; # unless $self->{escape}; |
$self->{escape} = 1; # unless $self->{escape}; |
| 1064 |
|
|
| 1065 |
# |
# |
| 1066 |
} elsif ($self->{nc} == 0x0021) { # ! |
} elsif ($self->{nc} == 0x0021) { # ! |
| 1067 |
if (defined $self->{s_kwd}) { |
if (length $self->{s_kwd}) { |
| 1068 |
!!!cp (5.1); |
!!!cp (5.1); |
| 1069 |
$self->{s_kwd} .= '!'; |
$self->{s_kwd} .= '!'; |
| 1070 |
# |
# |
| 1071 |
} else { |
} else { |
| 1072 |
!!!cp (5.2); |
!!!cp (5.2); |
| 1073 |
|
#$self->{s_kwd} = ''; |
| 1074 |
# |
# |
| 1075 |
} |
} |
| 1076 |
# |
# |
| 1077 |
} elsif ($self->{nc} == 0x003C) { # < |
} elsif ($self->{nc} == 0x003C) { # < |
|
delete $self->{s_kwd}; |
|
| 1078 |
if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA |
if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA |
| 1079 |
(($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA |
(($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA |
| 1080 |
not $self->{escape})) { |
not $self->{escape})) { |
| 1084 |
redo A; |
redo A; |
| 1085 |
} else { |
} else { |
| 1086 |
!!!cp (7); |
!!!cp (7); |
| 1087 |
|
$self->{s_kwd} = ''; |
| 1088 |
# |
# |
| 1089 |
} |
} |
| 1090 |
} elsif ($self->{nc} == 0x003E) { # > |
} elsif ($self->{nc} == 0x003E) { # > |
| 1091 |
if ($self->{escape} and |
if ($self->{escape} and |
| 1092 |
($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA |
($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA |
| 1093 |
if (defined $self->{s_kwd} and $self->{s_kwd} eq '--') { |
if ($self->{s_kwd} eq '--') { |
| 1094 |
!!!cp (8); |
!!!cp (8); |
| 1095 |
delete $self->{escape}; |
delete $self->{escape}; |
| 1096 |
} else { |
} else { |
| 1100 |
!!!cp (10); |
!!!cp (10); |
| 1101 |
} |
} |
| 1102 |
|
|
| 1103 |
delete $self->{s_kwd}; |
$self->{s_kwd} = ''; |
| 1104 |
# |
# |
| 1105 |
} elsif ($self->{nc} == -1) { |
} elsif ($self->{nc} == -1) { |
| 1106 |
!!!cp (11); |
!!!cp (11); |
| 1107 |
delete $self->{s_kwd}; |
$self->{s_kwd} = ''; |
| 1108 |
!!!emit ({type => END_OF_FILE_TOKEN, |
!!!emit ({type => END_OF_FILE_TOKEN, |
| 1109 |
line => $self->{line}, column => $self->{column}}); |
line => $self->{line}, column => $self->{column}}); |
| 1110 |
last A; ## TODO: ok? |
last A; ## TODO: ok? |
| 1111 |
} else { |
} else { |
| 1112 |
!!!cp (12); |
!!!cp (12); |
| 1113 |
delete $self->{s_kwd}; |
$self->{s_kwd} = ''; |
| 1114 |
# |
# |
| 1115 |
} |
} |
| 1116 |
|
|
| 1121 |
}; |
}; |
| 1122 |
if ($self->{read_until}->($token->{data}, q[-!<>&], |
if ($self->{read_until}->($token->{data}, q[-!<>&], |
| 1123 |
length $token->{data})) { |
length $token->{data})) { |
| 1124 |
delete $self->{s_kwd}; |
$self->{s_kwd} = ''; |
| 1125 |
} |
} |
| 1126 |
|
|
| 1127 |
## Stay in the data state |
## Stay in the data state. |
| 1128 |
|
if ($self->{content_model} == PCDATA_CONTENT_MODEL) { |
| 1129 |
|
!!!cp (13); |
| 1130 |
|
$self->{state} = PCDATA_STATE; |
| 1131 |
|
} else { |
| 1132 |
|
!!!cp (14); |
| 1133 |
|
## Stay in the state. |
| 1134 |
|
} |
| 1135 |
!!!next-input-character; |
!!!next-input-character; |
| 1136 |
!!!emit ($token); |
!!!emit ($token); |
| 1137 |
redo A; |
redo A; |
| 1236 |
} |
} |
| 1237 |
} elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) { |
} elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) { |
| 1238 |
## NOTE: The "close tag open state" in the spec is implemented as |
## NOTE: The "close tag open state" in the spec is implemented as |
| 1239 |
## |CLOSE_TAG_OPEN_STATE| and |CDATA_PCDATA_CLOSE_TAG_STATE|. |
## |CLOSE_TAG_OPEN_STATE| and |CDATA_RCDATA_CLOSE_TAG_STATE|. |
| 1240 |
|
|
| 1241 |
my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</" |
my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</" |
| 1242 |
if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA |
if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA |
| 1243 |
if (defined $self->{last_stag_name}) { |
if (defined $self->{last_stag_name}) { |
| 1244 |
$self->{state} = CDATA_PCDATA_CLOSE_TAG_STATE; |
$self->{state} = CDATA_RCDATA_CLOSE_TAG_STATE; |
| 1245 |
$self->{s_kwd} = ''; |
$self->{s_kwd} = ''; |
| 1246 |
## Reconsume. |
## Reconsume. |
| 1247 |
redo A; |
redo A; |
| 1312 |
## "bogus comment state" entry. |
## "bogus comment state" entry. |
| 1313 |
redo A; |
redo A; |
| 1314 |
} |
} |
| 1315 |
} elsif ($self->{state} == CDATA_PCDATA_CLOSE_TAG_STATE) { |
} elsif ($self->{state} == CDATA_RCDATA_CLOSE_TAG_STATE) { |
| 1316 |
my $ch = substr $self->{last_stag_name}, length $self->{s_kwd}, 1; |
my $ch = substr $self->{last_stag_name}, length $self->{s_kwd}, 1; |
| 1317 |
if (length $ch) { |
if (length $ch) { |
| 1318 |
my $CH = $ch; |
my $CH = $ch; |
| 4751 |
} elsif ($token->{attributes}->{content}) { |
} elsif ($token->{attributes}->{content}) { |
| 4752 |
if ($token->{attributes}->{content}->{value} |
if ($token->{attributes}->{content}->{value} |
| 4753 |
=~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt] |
=~ /[Cc][Hh][Aa][Rr][Ss][Ee][Tt] |
| 4754 |
[\x09-\x0D\x20]*= |
[\x09\x0A\x0C\x0D\x20]*= |
| 4755 |
[\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'| |
[\x09\x0A\x0C\x0D\x20]*(?>"([^"]*)"|'([^']*)'| |
| 4756 |
([^"'\x09-\x0D\x20][^\x09-\x0D\x20\x3B]*))/x) { |
([^"'\x09\x0A\x0C\x0D\x20] |
| 4757 |
|
[^\x09\x0A\x0C\x0D\x20\x3B]*))/x) { |
| 4758 |
!!!cp ('t107'); |
!!!cp ('t107'); |
| 4759 |
## NOTE: Whether the encoding is supported or not is handled |
## NOTE: Whether the encoding is supported or not is handled |
| 4760 |
## in the {change_encoding} callback. |
## in the {change_encoding} callback. |