| 812 |
sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec |
sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec |
| 813 |
sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec |
sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec |
| 814 |
sub SYSTEM_STATE () { 43 } # "after DOCTYPE name state" in the spec |
sub SYSTEM_STATE () { 43 } # "after DOCTYPE name state" in the spec |
| 815 |
|
sub ENTITY_STATE () { 44 } # "consume a character reference" in the spec |
| 816 |
|
|
| 817 |
sub DOCTYPE_TOKEN () { 1 } |
sub DOCTYPE_TOKEN () { 1 } |
| 818 |
sub COMMENT_TOKEN () { 2 } |
sub COMMENT_TOKEN () { 2 } |
| 941 |
if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA |
if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA |
| 942 |
not $self->{escape}) { |
not $self->{escape}) { |
| 943 |
!!!cp (1); |
!!!cp (1); |
| 944 |
$self->{state} = ENTITY_DATA_STATE; |
## NOTE: In the spec, the tokenizer is switched to the |
| 945 |
|
## "entity data state". In this implementation, the tokenizer |
| 946 |
|
## is switched to the |ENTITY_STATE|, which is an implementation |
| 947 |
|
## of the "consume a character reference" algorithm. |
| 948 |
|
#$self->{state} = ENTITY_DATA_STATE; |
| 949 |
|
$self->{entity_in_attr} = 0; |
| 950 |
|
$self->{entity_additional} = -1; |
| 951 |
|
$self->{state} = ENTITY_STATE; |
| 952 |
!!!next-input-character; |
!!!next-input-character; |
| 953 |
redo A; |
redo A; |
| 954 |
} else { |
} else { |
| 1019 |
|
|
| 1020 |
redo A; |
redo A; |
| 1021 |
} elsif ($self->{state} == ENTITY_DATA_STATE) { |
} elsif ($self->{state} == ENTITY_DATA_STATE) { |
|
## (cannot happen in CDATA state) |
|
|
|
|
| 1022 |
my ($l, $c) = ($self->{line_prev}, $self->{column_prev}); |
my ($l, $c) = ($self->{line_prev}, $self->{column_prev}); |
| 1023 |
|
|
| 1024 |
my $token = $self->_tokenize_attempt_to_consume_an_entity (0, -1); |
my $token = $self->{entity_return}; |
| 1025 |
|
|
| 1026 |
$self->{state} = DATA_STATE; |
$self->{state} = DATA_STATE; |
| 1027 |
# next-input-character is already done |
# next-input-character is already done |
| 1711 |
} elsif ($self->{next_char} == 0x0026) { # & |
} elsif ($self->{next_char} == 0x0026) { # & |
| 1712 |
!!!cp (96); |
!!!cp (96); |
| 1713 |
$self->{last_attribute_value_state} = $self->{state}; |
$self->{last_attribute_value_state} = $self->{state}; |
| 1714 |
$self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE; |
## NOTE: In the spec, the tokenizer is switched to the |
| 1715 |
|
## "entity in attribute value state". In this implementation, the |
| 1716 |
|
## tokenizer is switched to the |ENTITY_STATE|, which is an |
| 1717 |
|
## implementation of the "consume a character reference" algorithm. |
| 1718 |
|
#$self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE; |
| 1719 |
|
$self->{entity_in_attr} = 1; |
| 1720 |
|
$self->{entity_additional} = 0x0022; # " |
| 1721 |
|
$self->{state} = ENTITY_STATE; |
| 1722 |
!!!next-input-character; |
!!!next-input-character; |
| 1723 |
redo A; |
redo A; |
| 1724 |
} elsif ($self->{next_char} == -1) { |
} elsif ($self->{next_char} == -1) { |
| 1760 |
} elsif ($self->{next_char} == 0x0026) { # & |
} elsif ($self->{next_char} == 0x0026) { # & |
| 1761 |
!!!cp (102); |
!!!cp (102); |
| 1762 |
$self->{last_attribute_value_state} = $self->{state}; |
$self->{last_attribute_value_state} = $self->{state}; |
| 1763 |
$self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE; |
## NOTE: In the spec, the tokenizer is switched to the |
| 1764 |
|
## "entity in attribute value state". In this implementation, the |
| 1765 |
|
## tokenizer is switched to the |ENTITY_STATE|, which is an |
| 1766 |
|
## implementation of the "consume a character reference" algorithm. |
| 1767 |
|
#$self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE; |
| 1768 |
|
$self->{entity_in_attr} = 1; |
| 1769 |
|
$self->{entity_additional} = 0x0027; # ' |
| 1770 |
|
$self->{state} = ENTITY_STATE; |
| 1771 |
!!!next-input-character; |
!!!next-input-character; |
| 1772 |
redo A; |
redo A; |
| 1773 |
} elsif ($self->{next_char} == -1) { |
} elsif ($self->{next_char} == -1) { |
| 1813 |
} elsif ($self->{next_char} == 0x0026) { # & |
} elsif ($self->{next_char} == 0x0026) { # & |
| 1814 |
!!!cp (108); |
!!!cp (108); |
| 1815 |
$self->{last_attribute_value_state} = $self->{state}; |
$self->{last_attribute_value_state} = $self->{state}; |
| 1816 |
$self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE; |
## NOTE: In the spec, the tokenizer is switched to the |
| 1817 |
|
## "entity in attribute value state". In this implementation, the |
| 1818 |
|
## tokenizer is switched to the |ENTITY_STATE|, which is an |
| 1819 |
|
## implementation of the "consume a character reference" algorithm. |
| 1820 |
|
#$self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE; |
| 1821 |
|
$self->{entity_in_attr} = 1; |
| 1822 |
|
$self->{entity_additional} = -1; |
| 1823 |
|
$self->{state} = ENTITY_STATE; |
| 1824 |
!!!next-input-character; |
!!!next-input-character; |
| 1825 |
redo A; |
redo A; |
| 1826 |
} elsif ($self->{next_char} == 0x003E) { # > |
} elsif ($self->{next_char} == 0x003E) { # > |
| 1885 |
redo A; |
redo A; |
| 1886 |
} |
} |
| 1887 |
} elsif ($self->{state} == ENTITY_IN_ATTRIBUTE_VALUE_STATE) { |
} elsif ($self->{state} == ENTITY_IN_ATTRIBUTE_VALUE_STATE) { |
| 1888 |
my $token = $self->_tokenize_attempt_to_consume_an_entity |
my $token = $self->{entity_return}; |
|
(1, |
|
|
$self->{last_attribute_value_state} |
|
|
== ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE ? 0x0022 : # " |
|
|
$self->{last_attribute_value_state} |
|
|
== ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE ? 0x0027 : # ' |
|
|
-1); |
|
| 1889 |
|
|
| 1890 |
unless (defined $token) { |
unless (defined $token) { |
| 1891 |
!!!cp (117); |
!!!cp (117); |
| 2019 |
} |
} |
| 2020 |
} elsif ($self->{state} == BOGUS_COMMENT_STATE) { |
} elsif ($self->{state} == BOGUS_COMMENT_STATE) { |
| 2021 |
## (only happen if PCDATA state) |
## (only happen if PCDATA state) |
|
|
|
|
## NOTE: Set by the previous state |
|
|
#my $token = {type => COMMENT_TOKEN, data => ''}; |
|
|
|
|
|
BC: { |
|
|
if ($self->{next_char} == 0x003E) { # > |
|
|
!!!cp (124); |
|
|
$self->{state} = DATA_STATE; |
|
|
!!!next-input-character; |
|
| 2022 |
|
|
| 2023 |
!!!emit ($self->{current_token}); # comment |
## NOTE: Unlike spec's "bogus comment state", this implementation |
| 2024 |
|
## consumes characters one-by-one basis. |
| 2025 |
redo A; |
|
| 2026 |
} elsif ($self->{next_char} == -1) { |
if ($self->{next_char} == 0x003E) { # > |
| 2027 |
!!!cp (125); |
!!!cp (124); |
| 2028 |
$self->{state} = DATA_STATE; |
$self->{state} = DATA_STATE; |
| 2029 |
## reconsume |
!!!next-input-character; |
|
|
|
|
!!!emit ($self->{current_token}); # comment |
|
| 2030 |
|
|
| 2031 |
redo A; |
!!!emit ($self->{current_token}); # comment |
| 2032 |
} else { |
redo A; |
| 2033 |
!!!cp (126); |
} elsif ($self->{next_char} == -1) { |
| 2034 |
$self->{current_token}->{data} .= chr ($self->{next_char}); # comment |
!!!cp (125); |
| 2035 |
!!!next-input-character; |
$self->{state} = DATA_STATE; |
| 2036 |
redo BC; |
## reconsume |
|
} |
|
|
} # BC |
|
| 2037 |
|
|
| 2038 |
die "$0: _get_next_token: unexpected case [BC]"; |
!!!emit ($self->{current_token}); # comment |
| 2039 |
|
redo A; |
| 2040 |
|
} else { |
| 2041 |
|
!!!cp (126); |
| 2042 |
|
$self->{current_token}->{data} .= chr ($self->{next_char}); # comment |
| 2043 |
|
## Stay in the state. |
| 2044 |
|
!!!next-input-character; |
| 2045 |
|
redo A; |
| 2046 |
|
} |
| 2047 |
} elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) { |
} elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) { |
| 2048 |
## (only happen if PCDATA state) |
## (only happen if PCDATA state) |
| 2049 |
|
|
| 2979 |
## Reconsume. |
## Reconsume. |
| 2980 |
redo A; |
redo A; |
| 2981 |
} |
} |
|
} else { |
|
|
die "$0: $self->{state}: Unknown state"; |
|
|
} |
|
|
} # A |
|
|
|
|
|
die "$0: _get_next_token: unexpected case"; |
|
|
} # _get_next_token |
|
| 2982 |
|
|
| 2983 |
sub _tokenize_attempt_to_consume_an_entity ($$$) { |
} elsif ($self->{state} == ENTITY_STATE) { |
| 2984 |
my ($self, $in_attr, $additional) = @_; |
my $in_attr = $self->{entity_in_attr}; |
| 2985 |
|
my $additional = $self->{entity_additional}; |
| 2986 |
|
|
| 2987 |
my ($l, $c) = ($self->{line_prev}, $self->{column_prev}); |
my ($l, $c) = ($self->{line_prev}, $self->{column_prev}); |
| 2988 |
|
|
| 2994 |
!!!cp (1001); |
!!!cp (1001); |
| 2995 |
## Don't consume |
## Don't consume |
| 2996 |
## No error |
## No error |
| 2997 |
return undef; |
$self->{entity_return} = undef; |
| 2998 |
|
$self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE; |
| 2999 |
|
redo A; |
| 3000 |
} elsif ($self->{next_char} == 0x0023) { # # |
} elsif ($self->{next_char} == 0x0023) { # # |
| 3001 |
!!!next-input-character; |
!!!next-input-character; |
| 3002 |
if ($self->{next_char} == 0x0078 or # x |
if ($self->{next_char} == 0x0078 or # x |
| 3031 |
!!!parse-error (type => 'bare hcro', line => $l, column => $c); |
!!!parse-error (type => 'bare hcro', line => $l, column => $c); |
| 3032 |
!!!back-next-input-character ($x_char, $self->{next_char}); |
!!!back-next-input-character ($x_char, $self->{next_char}); |
| 3033 |
$self->{next_char} = 0x0023; # # |
$self->{next_char} = 0x0023; # # |
| 3034 |
return undef; |
$self->{entity_return} = undef; |
| 3035 |
|
$self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE; |
| 3036 |
|
redo A; |
| 3037 |
} elsif ($self->{next_char} == 0x003B) { # ; |
} elsif ($self->{next_char} == 0x003B) { # ; |
| 3038 |
!!!cp (1006); |
!!!cp (1006); |
| 3039 |
!!!next-input-character; |
!!!next-input-character; |
| 3064 |
$code = $c1_entity_char->{$code}; |
$code = $c1_entity_char->{$code}; |
| 3065 |
} |
} |
| 3066 |
|
|
| 3067 |
return {type => CHARACTER_TOKEN, data => chr $code, |
$self->{entity_return} = {type => CHARACTER_TOKEN, data => chr $code, |
| 3068 |
has_reference => 1, |
has_reference => 1, |
| 3069 |
line => $l, column => $c, |
line => $l, column => $c, |
| 3070 |
}; |
}; |
| 3071 |
|
$self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE; |
| 3072 |
|
redo A; |
| 3073 |
} # X |
} # X |
| 3074 |
} elsif (0x0030 <= $self->{next_char} and |
} elsif (0x0030 <= $self->{next_char} and |
| 3075 |
$self->{next_char} <= 0x0039) { # 0..9 |
$self->{next_char} <= 0x0039) { # 0..9 |
| 3118 |
$code = $c1_entity_char->{$code}; |
$code = $c1_entity_char->{$code}; |
| 3119 |
} |
} |
| 3120 |
|
|
| 3121 |
return {type => CHARACTER_TOKEN, data => chr $code, has_reference => 1, |
$self->{entity_return} = {type => CHARACTER_TOKEN, data => chr $code, has_reference => 1, |
| 3122 |
line => $l, column => $c, |
line => $l, column => $c, |
| 3123 |
}; |
}; |
| 3124 |
|
$self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE; |
| 3125 |
|
redo A; |
| 3126 |
} else { |
} else { |
| 3127 |
!!!cp (1019); |
!!!cp (1019); |
| 3128 |
!!!parse-error (type => 'bare nero', line => $l, column => $c); |
!!!parse-error (type => 'bare nero', line => $l, column => $c); |
| 3129 |
!!!back-next-input-character ($self->{next_char}); |
!!!back-next-input-character ($self->{next_char}); |
| 3130 |
$self->{next_char} = 0x0023; # # |
$self->{next_char} = 0x0023; # # |
| 3131 |
return undef; |
$self->{entity_return} = undef; |
| 3132 |
|
$self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE; |
| 3133 |
|
redo A; |
| 3134 |
} |
} |
| 3135 |
} elsif ((0x0041 <= $self->{next_char} and |
} elsif ((0x0041 <= $self->{next_char} and |
| 3136 |
$self->{next_char} <= 0x005A) or |
$self->{next_char} <= 0x005A) or |
| 3177 |
|
|
| 3178 |
if ($match > 0) { |
if ($match > 0) { |
| 3179 |
!!!cp (1023); |
!!!cp (1023); |
| 3180 |
return {type => CHARACTER_TOKEN, data => $value, has_reference => 1, |
$self->{entity_return} = {type => CHARACTER_TOKEN, data => $value, has_reference => 1, |
| 3181 |
line => $l, column => $c, |
line => $l, column => $c, |
| 3182 |
}; |
}; |
| 3183 |
|
$self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE; |
| 3184 |
|
redo A; |
| 3185 |
} elsif ($match < 0) { |
} elsif ($match < 0) { |
| 3186 |
!!!parse-error (type => 'no refc', line => $l, column => $c); |
!!!parse-error (type => 'no refc', line => $l, column => $c); |
| 3187 |
if ($in_attr and $match < -1) { |
if ($in_attr and $match < -1) { |
| 3188 |
!!!cp (1024); |
!!!cp (1024); |
| 3189 |
return {type => CHARACTER_TOKEN, data => '&'.$entity_name, |
$self->{entity_return} = {type => CHARACTER_TOKEN, data => '&'.$entity_name, |
| 3190 |
line => $l, column => $c, |
line => $l, column => $c, |
| 3191 |
}; |
}; |
| 3192 |
|
$self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE; |
| 3193 |
|
redo A; |
| 3194 |
} else { |
} else { |
| 3195 |
!!!cp (1025); |
!!!cp (1025); |
| 3196 |
return {type => CHARACTER_TOKEN, data => $value, has_reference => 1, |
$self->{entity_return} = {type => CHARACTER_TOKEN, data => $value, has_reference => 1, |
| 3197 |
line => $l, column => $c, |
line => $l, column => $c, |
| 3198 |
}; |
}; |
| 3199 |
|
$self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE; |
| 3200 |
|
redo A; |
| 3201 |
} |
} |
| 3202 |
} else { |
} else { |
| 3203 |
!!!cp (1026); |
!!!cp (1026); |
| 3204 |
!!!parse-error (type => 'bare ero', line => $l, column => $c); |
!!!parse-error (type => 'bare ero', line => $l, column => $c); |
| 3205 |
## NOTE: "No characters are consumed" in the spec. |
## NOTE: "No characters are consumed" in the spec. |
| 3206 |
return {type => CHARACTER_TOKEN, data => '&'.$value, |
$self->{entity_return} = {type => CHARACTER_TOKEN, data => '&'.$value, |
| 3207 |
line => $l, column => $c, |
line => $l, column => $c, |
| 3208 |
}; |
}; |
| 3209 |
|
$self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE; |
| 3210 |
|
redo A; |
| 3211 |
} |
} |
| 3212 |
} else { |
} else { |
| 3213 |
!!!cp (1027); |
!!!cp (1027); |
| 3214 |
## no characters are consumed |
## no characters are consumed |
| 3215 |
!!!parse-error (type => 'bare ero', line => $l, column => $c); |
!!!parse-error (type => 'bare ero', line => $l, column => $c); |
| 3216 |
return undef; |
$self->{entity_return} = undef; |
| 3217 |
|
$self->{state} = $self->{entity_in_attr} ? ENTITY_IN_ATTRIBUTE_VALUE_STATE : ENTITY_DATA_STATE; |
| 3218 |
|
redo A; |
| 3219 |
} |
} |
| 3220 |
} # _tokenize_attempt_to_consume_an_entity |
|
| 3221 |
|
} else { |
| 3222 |
|
die "$0: $self->{state}: Unknown state"; |
| 3223 |
|
} |
| 3224 |
|
} # A |
| 3225 |
|
|
| 3226 |
|
die "$0: _get_next_token: unexpected case"; |
| 3227 |
|
} # _get_next_token |
| 3228 |
|
|
| 3229 |
sub _initialize_tree_constructor ($) { |
sub _initialize_tree_constructor ($) { |
| 3230 |
my $self = shift; |
my $self = shift; |