| 466 |
# Anything else |
# Anything else |
| 467 |
my $token = {type => CHARACTER_TOKEN, |
my $token = {type => CHARACTER_TOKEN, |
| 468 |
data => chr $self->{next_char}, |
data => chr $self->{next_char}, |
| 469 |
#line => $self->{line}, column => $self->{column}, |
line => $self->{line}, column => $self->{column}, |
| 470 |
}; |
}; |
| 471 |
## Stay in the data state |
## Stay in the data state |
| 472 |
!!!next-input-character; |
!!!next-input-character; |
| 477 |
} elsif ($self->{state} == ENTITY_DATA_STATE) { |
} elsif ($self->{state} == ENTITY_DATA_STATE) { |
| 478 |
## (cannot happen in CDATA state) |
## (cannot happen in CDATA state) |
| 479 |
|
|
| 480 |
#my ($l, $c) = ($self->{line_prev}, $self->{column_prev}); |
my ($l, $c) = ($self->{line_prev}, $self->{column_prev}); |
| 481 |
|
|
| 482 |
my $token = $self->_tokenize_attempt_to_consume_an_entity (0, -1); |
my $token = $self->_tokenize_attempt_to_consume_an_entity (0, -1); |
| 483 |
|
|
| 487 |
unless (defined $token) { |
unless (defined $token) { |
| 488 |
!!!cp (13); |
!!!cp (13); |
| 489 |
!!!emit ({type => CHARACTER_TOKEN, data => '&', |
!!!emit ({type => CHARACTER_TOKEN, data => '&', |
| 490 |
#line => $l, column => $c, |
line => $l, column => $c, |
| 491 |
}); |
}); |
| 492 |
} else { |
} else { |
| 493 |
!!!cp (14); |
!!!cp (14); |
| 508 |
$self->{state} = DATA_STATE; |
$self->{state} = DATA_STATE; |
| 509 |
|
|
| 510 |
!!!emit ({type => CHARACTER_TOKEN, data => '<', |
!!!emit ({type => CHARACTER_TOKEN, data => '<', |
| 511 |
#line => $self->{line_prev}, |
line => $self->{line_prev}, |
| 512 |
#column => $self->{column_prev}, |
column => $self->{column_prev}, |
| 513 |
}); |
}); |
| 514 |
|
|
| 515 |
redo A; |
redo A; |
| 555 |
!!!next-input-character; |
!!!next-input-character; |
| 556 |
|
|
| 557 |
!!!emit ({type => CHARACTER_TOKEN, data => '<>', |
!!!emit ({type => CHARACTER_TOKEN, data => '<>', |
| 558 |
#line => $self->{line_prev}, |
line => $self->{line_prev}, |
| 559 |
#column => $self->{column_prev}, |
column => $self->{column_prev}, |
| 560 |
}); |
}); |
| 561 |
|
|
| 562 |
redo A; |
redo A; |
| 567 |
column => $self->{column_prev}); |
column => $self->{column_prev}); |
| 568 |
$self->{state} = BOGUS_COMMENT_STATE; |
$self->{state} = BOGUS_COMMENT_STATE; |
| 569 |
$self->{current_token} = {type => COMMENT_TOKEN, data => '', |
$self->{current_token} = {type => COMMENT_TOKEN, data => '', |
| 570 |
#line => $self->{line_prev}, |
line => $self->{line_prev}, |
| 571 |
#column => $self->{column_prev}, |
column => $self->{column_prev}, |
| 572 |
}; |
}; |
| 573 |
## $self->{next_char} is intentionally left as is |
## $self->{next_char} is intentionally left as is |
| 574 |
redo A; |
redo A; |
| 579 |
## reconsume |
## reconsume |
| 580 |
|
|
| 581 |
!!!emit ({type => CHARACTER_TOKEN, data => '<', |
!!!emit ({type => CHARACTER_TOKEN, data => '<', |
| 582 |
#line => $self->{line_prev}, |
line => $self->{line_prev}, |
| 583 |
#column => $self->{column_prev}, |
column => $self->{column_prev}, |
| 584 |
}); |
}); |
| 585 |
|
|
| 586 |
redo A; |
redo A; |
| 610 |
$self->{state} = DATA_STATE; |
$self->{state} = DATA_STATE; |
| 611 |
|
|
| 612 |
!!!emit ({type => CHARACTER_TOKEN, data => '</', |
!!!emit ({type => CHARACTER_TOKEN, data => '</', |
| 613 |
#line => $l, column => $c, |
line => $l, column => $c, |
| 614 |
}); |
}); |
| 615 |
|
|
| 616 |
redo A; |
redo A; |
| 631 |
!!!back-next-input-character (@next_char); |
!!!back-next-input-character (@next_char); |
| 632 |
$self->{state} = DATA_STATE; |
$self->{state} = DATA_STATE; |
| 633 |
!!!emit ({type => CHARACTER_TOKEN, data => '</', |
!!!emit ({type => CHARACTER_TOKEN, data => '</', |
| 634 |
#line => $l, column => $c, |
line => $l, column => $c, |
| 635 |
}); |
}); |
| 636 |
redo A; |
redo A; |
| 637 |
} else { |
} else { |
| 646 |
# next-input-character is already done |
# next-input-character is already done |
| 647 |
$self->{state} = DATA_STATE; |
$self->{state} = DATA_STATE; |
| 648 |
!!!emit ({type => CHARACTER_TOKEN, data => '</', |
!!!emit ({type => CHARACTER_TOKEN, data => '</', |
| 649 |
#line => $l, column => $c, |
line => $l, column => $c, |
| 650 |
}); |
}); |
| 651 |
redo A; |
redo A; |
| 652 |
} |
} |
| 686 |
# reconsume |
# reconsume |
| 687 |
|
|
| 688 |
!!!emit ({type => CHARACTER_TOKEN, data => '</', |
!!!emit ({type => CHARACTER_TOKEN, data => '</', |
| 689 |
#line => $l, column => $c, |
line => $l, column => $c, |
| 690 |
}); |
}); |
| 691 |
|
|
| 692 |
redo A; |
redo A; |
| 695 |
!!!parse-error (type => 'bogus end tag'); |
!!!parse-error (type => 'bogus end tag'); |
| 696 |
$self->{state} = BOGUS_COMMENT_STATE; |
$self->{state} = BOGUS_COMMENT_STATE; |
| 697 |
$self->{current_token} = {type => COMMENT_TOKEN, data => '', |
$self->{current_token} = {type => COMMENT_TOKEN, data => '', |
| 698 |
#line => $self->{line_prev}, # "<" of "</" |
line => $self->{line_prev}, # "<" of "</" |
| 699 |
#column => $self->{column_prev} - 1, |
column => $self->{column_prev} - 1, |
| 700 |
}; |
}; |
| 701 |
## $self->{next_char} is intentionally left as is |
## $self->{next_char} is intentionally left as is |
| 702 |
redo A; |
redo A; |
| 889 |
if (exists $self->{current_token}->{attributes} # start tag or end tag |
if (exists $self->{current_token}->{attributes} # start tag or end tag |
| 890 |
->{$self->{current_attribute}->{name}}) { # MUST |
->{$self->{current_attribute}->{name}}) { # MUST |
| 891 |
!!!cp (57); |
!!!cp (57); |
| 892 |
!!!parse-error (type => 'duplicate attribute:'.$self->{current_attribute}->{name}); |
!!!parse-error (type => 'duplicate attribute:'.$self->{current_attribute}->{name}, line => $self->{current_attribute}->{line}, column => $self->{current_attribute}->{column}); |
| 893 |
## Discard $self->{current_attribute} # MUST |
## Discard $self->{current_attribute} # MUST |
| 894 |
} else { |
} else { |
| 895 |
!!!cp (58); |
!!!cp (58); |
| 1444 |
} elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) { |
} elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) { |
| 1445 |
## (only happen if PCDATA state) |
## (only happen if PCDATA state) |
| 1446 |
|
|
| 1447 |
#my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); |
my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); |
| 1448 |
|
|
| 1449 |
my @next_char; |
my @next_char; |
| 1450 |
push @next_char, $self->{next_char}; |
push @next_char, $self->{next_char}; |
| 1455 |
if ($self->{next_char} == 0x002D) { # - |
if ($self->{next_char} == 0x002D) { # - |
| 1456 |
!!!cp (127); |
!!!cp (127); |
| 1457 |
$self->{current_token} = {type => COMMENT_TOKEN, data => '', |
$self->{current_token} = {type => COMMENT_TOKEN, data => '', |
| 1458 |
#line => $l, column => $c, |
line => $l, column => $c, |
| 1459 |
}; |
}; |
| 1460 |
$self->{state} = COMMENT_START_STATE; |
$self->{state} = COMMENT_START_STATE; |
| 1461 |
!!!next-input-character; |
!!!next-input-character; |
| 1494 |
$self->{state} = DOCTYPE_STATE; |
$self->{state} = DOCTYPE_STATE; |
| 1495 |
$self->{current_token} = {type => DOCTYPE_TOKEN, |
$self->{current_token} = {type => DOCTYPE_TOKEN, |
| 1496 |
quirks => 1, |
quirks => 1, |
| 1497 |
#line => $l, column => $c, |
line => $l, column => $c, |
| 1498 |
}; |
}; |
| 1499 |
!!!next-input-character; |
!!!next-input-character; |
| 1500 |
redo A; |
redo A; |
| 1525 |
!!!back-next-input-character (@next_char); |
!!!back-next-input-character (@next_char); |
| 1526 |
$self->{state} = BOGUS_COMMENT_STATE; |
$self->{state} = BOGUS_COMMENT_STATE; |
| 1527 |
$self->{current_token} = {type => COMMENT_TOKEN, data => '', |
$self->{current_token} = {type => COMMENT_TOKEN, data => '', |
| 1528 |
#line => $l, column => $c, |
line => $l, column => $c, |
| 1529 |
}; |
}; |
| 1530 |
redo A; |
redo A; |
| 1531 |
|
|
| 2325 |
|
|
| 2326 |
return {type => CHARACTER_TOKEN, data => chr $code, |
return {type => CHARACTER_TOKEN, data => chr $code, |
| 2327 |
has_reference => 1, |
has_reference => 1, |
| 2328 |
#line => $l, column => $c, |
line => $l, column => $c, |
| 2329 |
}; |
}; |
| 2330 |
} # X |
} # X |
| 2331 |
} elsif (0x0030 <= $self->{next_char} and |
} elsif (0x0030 <= $self->{next_char} and |
| 2369 |
} |
} |
| 2370 |
|
|
| 2371 |
return {type => CHARACTER_TOKEN, data => chr $code, has_reference => 1, |
return {type => CHARACTER_TOKEN, data => chr $code, has_reference => 1, |
| 2372 |
#line => $l, column => $c, |
line => $l, column => $c, |
| 2373 |
}; |
}; |
| 2374 |
} else { |
} else { |
| 2375 |
!!!cp (1019); |
!!!cp (1019); |
| 2424 |
if ($match > 0) { |
if ($match > 0) { |
| 2425 |
!!!cp (1023); |
!!!cp (1023); |
| 2426 |
return {type => CHARACTER_TOKEN, data => $value, has_reference => 1, |
return {type => CHARACTER_TOKEN, data => $value, has_reference => 1, |
| 2427 |
#line => $l, column => $c, |
line => $l, column => $c, |
| 2428 |
}; |
}; |
| 2429 |
} elsif ($match < 0) { |
} elsif ($match < 0) { |
| 2430 |
!!!parse-error (type => 'no refc', line => $l, column => $c); |
!!!parse-error (type => 'no refc', line => $l, column => $c); |
| 2431 |
if ($in_attr and $match < -1) { |
if ($in_attr and $match < -1) { |
| 2432 |
!!!cp (1024); |
!!!cp (1024); |
| 2433 |
return {type => CHARACTER_TOKEN, data => '&'.$entity_name, |
return {type => CHARACTER_TOKEN, data => '&'.$entity_name, |
| 2434 |
#line => $l, column => $c, |
line => $l, column => $c, |
| 2435 |
}; |
}; |
| 2436 |
} else { |
} else { |
| 2437 |
!!!cp (1025); |
!!!cp (1025); |
| 2438 |
return {type => CHARACTER_TOKEN, data => $value, has_reference => 1, |
return {type => CHARACTER_TOKEN, data => $value, has_reference => 1, |
| 2439 |
#line => $l, column => $c, |
line => $l, column => $c, |
| 2440 |
}; |
}; |
| 2441 |
} |
} |
| 2442 |
} else { |
} else { |
| 2444 |
!!!parse-error (type => 'bare ero', line => $l, column => $c); |
!!!parse-error (type => 'bare ero', line => $l, column => $c); |
| 2445 |
## NOTE: "No characters are consumed" in the spec. |
## NOTE: "No characters are consumed" in the spec. |
| 2446 |
return {type => CHARACTER_TOKEN, data => '&'.$value, |
return {type => CHARACTER_TOKEN, data => '&'.$value, |
| 2447 |
#line => $l, column => $c, |
line => $l, column => $c, |
| 2448 |
}; |
}; |
| 2449 |
} |
} |
| 2450 |
} else { |
} else { |
| 6483 |
|
|
| 6484 |
## Step 8 # MUST |
## Step 8 # MUST |
| 6485 |
my $i = 0; |
my $i = 0; |
| 6486 |
my $line = 1; |
$p->{line_prev} = $p->{line} = 1; |
| 6487 |
my $column = 0; |
$p->{column_prev} = $p->{column} = 0; |
| 6488 |
$p->{set_next_char} = sub { |
$p->{set_next_char} = sub { |
| 6489 |
my $self = shift; |
my $self = shift; |
| 6490 |
|
|
| 6493 |
|
|
| 6494 |
$self->{next_char} = -1 and return if $i >= length $$s; |
$self->{next_char} = -1 and return if $i >= length $$s; |
| 6495 |
$self->{next_char} = ord substr $$s, $i++, 1; |
$self->{next_char} = ord substr $$s, $i++, 1; |
| 6496 |
$column++; |
|
| 6497 |
|
($p->{line_prev}, $p->{column_prev}) = ($p->{line}, $p->{column}); |
| 6498 |
|
$p->{column}++; |
| 6499 |
|
|
| 6500 |
if ($self->{next_char} == 0x000A) { # LF |
if ($self->{next_char} == 0x000A) { # LF |
| 6501 |
$line++; |
$p->{line}++; |
| 6502 |
$column = 0; |
$p->{column} = 0; |
| 6503 |
!!!cp ('i1'); |
!!!cp ('i1'); |
| 6504 |
} elsif ($self->{next_char} == 0x000D) { # CR |
} elsif ($self->{next_char} == 0x000D) { # CR |
| 6505 |
$i++ if substr ($$s, $i, 1) eq "\x0A"; |
$i++ if substr ($$s, $i, 1) eq "\x0A"; |
| 6506 |
$self->{next_char} = 0x000A; # LF # MUST |
$self->{next_char} = 0x000A; # LF # MUST |
| 6507 |
$line++; |
$p->{line}++; |
| 6508 |
$column = 0; |
$p->{column} = 0; |
| 6509 |
!!!cp ('i2'); |
!!!cp ('i2'); |
| 6510 |
} elsif ($self->{next_char} > 0x10FFFF) { |
} elsif ($self->{next_char} > 0x10FFFF) { |
| 6511 |
$self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST |
$self->{next_char} = 0xFFFD; # REPLACEMENT CHARACTER # MUST |
| 6521 |
|
|
| 6522 |
my $ponerror = $onerror || sub { |
my $ponerror = $onerror || sub { |
| 6523 |
my (%opt) = @_; |
my (%opt) = @_; |
| 6524 |
warn "Parse error ($opt{type}) at line $opt{line} column $opt{column}\n"; |
my $line = $opt{line}; |
| 6525 |
|
my $column = $opt{column}; |
| 6526 |
|
if (defined $opt{token} and defined $opt{token}->{line}) { |
| 6527 |
|
$line = $opt{token}->{line}; |
| 6528 |
|
$column = $opt{token}->{column}; |
| 6529 |
|
} |
| 6530 |
|
warn "Parse error ($opt{type}) at line $line column $column\n"; |
| 6531 |
}; |
}; |
| 6532 |
$p->{parse_error} = sub { |
$p->{parse_error} = sub { |
| 6533 |
$ponerror->(@_, line => $line, column => $column); |
$ponerror->(line => $p->{line}, column => $p->{column}, @_); |
| 6534 |
}; |
}; |
| 6535 |
|
|
| 6536 |
$p->_initialize_tokenizer; |
$p->_initialize_tokenizer; |
| 6610 |
## ISSUE: mutation events? |
## ISSUE: mutation events? |
| 6611 |
|
|
| 6612 |
$p->_terminate_tree_constructor; |
$p->_terminate_tree_constructor; |
| 6613 |
|
|
| 6614 |
|
delete $p->{parse_error}; # delete loop |
| 6615 |
} else { |
} else { |
| 6616 |
die "$0: |set_inner_html| is not defined for node of type $nt"; |
die "$0: |set_inner_html| is not defined for node of type $nt"; |
| 6617 |
} |
} |