| 810 |
sub CDATA_PCDATA_CLOSE_TAG_STATE () { 39 } # "close tag open state" in the spec |
sub CDATA_PCDATA_CLOSE_TAG_STATE () { 39 } # "close tag open state" in the spec |
| 811 |
sub CDATA_SECTION_MSE1_STATE () { 40 } # "CDATA section state" in the spec |
sub CDATA_SECTION_MSE1_STATE () { 40 } # "CDATA section state" in the spec |
| 812 |
sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec |
sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec |
| 813 |
|
sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec |
| 814 |
|
sub SYSTEM_STATE () { 43 } # "after DOCTYPE name state" in the spec |
| 815 |
|
|
| 816 |
sub DOCTYPE_TOKEN () { 1 } |
sub DOCTYPE_TOKEN () { 1 } |
| 817 |
sub COMMENT_TOKEN () { 2 } |
sub COMMENT_TOKEN () { 2 } |
| 2442 |
redo A; |
redo A; |
| 2443 |
} elsif ($self->{next_char} == 0x0050 or # P |
} elsif ($self->{next_char} == 0x0050 or # P |
| 2444 |
$self->{next_char} == 0x0070) { # p |
$self->{next_char} == 0x0070) { # p |
| 2445 |
|
$self->{state} = PUBLIC_STATE; |
| 2446 |
|
$self->{state_keyword} = chr $self->{next_char}; |
| 2447 |
!!!next-input-character; |
!!!next-input-character; |
| 2448 |
if ($self->{next_char} == 0x0055 or # U |
redo A; |
|
$self->{next_char} == 0x0075) { # u |
|
|
!!!next-input-character; |
|
|
if ($self->{next_char} == 0x0042 or # B |
|
|
$self->{next_char} == 0x0062) { # b |
|
|
!!!next-input-character; |
|
|
if ($self->{next_char} == 0x004C or # L |
|
|
$self->{next_char} == 0x006C) { # l |
|
|
!!!next-input-character; |
|
|
if ($self->{next_char} == 0x0049 or # I |
|
|
$self->{next_char} == 0x0069) { # i |
|
|
!!!next-input-character; |
|
|
if ($self->{next_char} == 0x0043 or # C |
|
|
$self->{next_char} == 0x0063) { # c |
|
|
!!!cp (168); |
|
|
$self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE; |
|
|
!!!next-input-character; |
|
|
redo A; |
|
|
} else { |
|
|
!!!cp (169); |
|
|
} |
|
|
} else { |
|
|
!!!cp (170); |
|
|
} |
|
|
} else { |
|
|
!!!cp (171); |
|
|
} |
|
|
} else { |
|
|
!!!cp (172); |
|
|
} |
|
|
} else { |
|
|
!!!cp (173); |
|
|
} |
|
|
|
|
|
# |
|
| 2449 |
} elsif ($self->{next_char} == 0x0053 or # S |
} elsif ($self->{next_char} == 0x0053 or # S |
| 2450 |
$self->{next_char} == 0x0073) { # s |
$self->{next_char} == 0x0073) { # s |
| 2451 |
|
$self->{state} = SYSTEM_STATE; |
| 2452 |
|
$self->{state_keyword} = chr $self->{next_char}; |
| 2453 |
!!!next-input-character; |
!!!next-input-character; |
| 2454 |
if ($self->{next_char} == 0x0059 or # Y |
redo A; |
|
$self->{next_char} == 0x0079) { # y |
|
|
!!!next-input-character; |
|
|
if ($self->{next_char} == 0x0053 or # S |
|
|
$self->{next_char} == 0x0073) { # s |
|
|
!!!next-input-character; |
|
|
if ($self->{next_char} == 0x0054 or # T |
|
|
$self->{next_char} == 0x0074) { # t |
|
|
!!!next-input-character; |
|
|
if ($self->{next_char} == 0x0045 or # E |
|
|
$self->{next_char} == 0x0065) { # e |
|
|
!!!next-input-character; |
|
|
if ($self->{next_char} == 0x004D or # M |
|
|
$self->{next_char} == 0x006D) { # m |
|
|
!!!cp (174); |
|
|
$self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE; |
|
|
!!!next-input-character; |
|
|
redo A; |
|
|
} else { |
|
|
!!!cp (175); |
|
|
} |
|
|
} else { |
|
|
!!!cp (176); |
|
|
} |
|
|
} else { |
|
|
!!!cp (177); |
|
|
} |
|
|
} else { |
|
|
!!!cp (178); |
|
|
} |
|
|
} else { |
|
|
!!!cp (179); |
|
|
} |
|
|
|
|
|
# |
|
| 2455 |
} else { |
} else { |
| 2456 |
!!!cp (180); |
!!!cp (180); |
| 2457 |
|
!!!parse-error (type => 'string after DOCTYPE name'); |
| 2458 |
|
$self->{current_token}->{quirks} = 1; |
| 2459 |
|
|
| 2460 |
|
$self->{state} = BOGUS_DOCTYPE_STATE; |
| 2461 |
!!!next-input-character; |
!!!next-input-character; |
| 2462 |
# |
redo A; |
| 2463 |
} |
} |
| 2464 |
|
} elsif ($self->{state} == PUBLIC_STATE) { |
| 2465 |
|
## ASCII case-insensitive |
| 2466 |
|
if ($self->{next_char} == [ |
| 2467 |
|
undef, |
| 2468 |
|
0x0055, # U |
| 2469 |
|
0x0042, # B |
| 2470 |
|
0x004C, # L |
| 2471 |
|
0x0049, # I |
| 2472 |
|
]->[length $self->{state_keyword}] or |
| 2473 |
|
$self->{next_char} == [ |
| 2474 |
|
undef, |
| 2475 |
|
0x0075, # u |
| 2476 |
|
0x0062, # b |
| 2477 |
|
0x006C, # l |
| 2478 |
|
0x0069, # i |
| 2479 |
|
]->[length $self->{state_keyword}]) { |
| 2480 |
|
!!!cp (175); |
| 2481 |
|
## Stay in the state. |
| 2482 |
|
$self->{state_keyword} .= chr $self->{next_char}; |
| 2483 |
|
!!!next-input-character; |
| 2484 |
|
redo A; |
| 2485 |
|
} elsif ((length $self->{state_keyword}) == 5 and |
| 2486 |
|
($self->{next_char} == 0x0043 or # C |
| 2487 |
|
$self->{next_char} == 0x0063)) { # c |
| 2488 |
|
!!!cp (168); |
| 2489 |
|
$self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE; |
| 2490 |
|
!!!next-input-character; |
| 2491 |
|
redo A; |
| 2492 |
|
} else { |
| 2493 |
|
!!!cp (169); |
| 2494 |
|
!!!parse-error (type => 'string after DOCTYPE name', |
| 2495 |
|
line => $self->{line_prev}, |
| 2496 |
|
column => $self->{column_prev} + 1 - length $self->{state_keyword}); |
| 2497 |
|
$self->{current_token}->{quirks} = 1; |
| 2498 |
|
|
| 2499 |
!!!parse-error (type => 'string after DOCTYPE name'); |
$self->{state} = BOGUS_DOCTYPE_STATE; |
| 2500 |
$self->{current_token}->{quirks} = 1; |
## Reconsume. |
| 2501 |
|
redo A; |
| 2502 |
|
} |
| 2503 |
|
} elsif ($self->{state} == SYSTEM_STATE) { |
| 2504 |
|
## ASCII case-insensitive |
| 2505 |
|
if ($self->{next_char} == [ |
| 2506 |
|
undef, |
| 2507 |
|
0x0059, # Y |
| 2508 |
|
0x0053, # S |
| 2509 |
|
0x0054, # T |
| 2510 |
|
0x0045, # E |
| 2511 |
|
]->[length $self->{state_keyword}] or |
| 2512 |
|
$self->{next_char} == [ |
| 2513 |
|
undef, |
| 2514 |
|
0x0079, # y |
| 2515 |
|
0x0073, # s |
| 2516 |
|
0x0074, # t |
| 2517 |
|
0x0065, # e |
| 2518 |
|
]->[length $self->{state_keyword}]) { |
| 2519 |
|
!!!cp (170); |
| 2520 |
|
## Stay in the state. |
| 2521 |
|
$self->{state_keyword} .= chr $self->{next_char}; |
| 2522 |
|
!!!next-input-character; |
| 2523 |
|
redo A; |
| 2524 |
|
} elsif ((length $self->{state_keyword}) == 5 and |
| 2525 |
|
($self->{next_char} == 0x004D or # M |
| 2526 |
|
$self->{next_char} == 0x006D)) { # m |
| 2527 |
|
!!!cp (171); |
| 2528 |
|
$self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE; |
| 2529 |
|
!!!next-input-character; |
| 2530 |
|
redo A; |
| 2531 |
|
} else { |
| 2532 |
|
!!!cp (172); |
| 2533 |
|
!!!parse-error (type => 'string after DOCTYPE name', |
| 2534 |
|
line => $self->{line_prev}, |
| 2535 |
|
column => $self->{column_prev} + 1 - length $self->{state_keyword}); |
| 2536 |
|
$self->{current_token}->{quirks} = 1; |
| 2537 |
|
|
| 2538 |
$self->{state} = BOGUS_DOCTYPE_STATE; |
$self->{state} = BOGUS_DOCTYPE_STATE; |
| 2539 |
# next-input-character is already done |
## Reconsume. |
| 2540 |
redo A; |
redo A; |
| 2541 |
|
} |
| 2542 |
} elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) { |
} elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) { |
| 2543 |
if ({ |
if ({ |
| 2544 |
0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1, |
0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1, |