/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.76 by wakaba, Mon Mar 3 09:17:10 2008 UTC revision 1.78 by wakaba, Mon Mar 3 11:56:18 2008 UTC
# Line 387  sub _get_next_token ($) { Line 387  sub _get_next_token ($) {
387        if ($self->{next_char} == 0x0026) { # &        if ($self->{next_char} == 0x0026) { # &
388          if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA          if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA
389              not $self->{escape}) {              not $self->{escape}) {
390              !!!cp (1);
391            $self->{state} = ENTITY_DATA_STATE;            $self->{state} = ENTITY_DATA_STATE;
392            !!!next-input-character;            !!!next-input-character;
393            redo A;            redo A;
394          } else {          } else {
395              !!!cp (2);
396            #            #
397          }          }
398        } elsif ($self->{next_char} == 0x002D) { # -        } elsif ($self->{next_char} == 0x002D) { # -
# Line 399  sub _get_next_token ($) { Line 401  sub _get_next_token ($) {
401              if ($self->{prev_char}->[0] == 0x002D and # -              if ($self->{prev_char}->[0] == 0x002D and # -
402                  $self->{prev_char}->[1] == 0x0021 and # !                  $self->{prev_char}->[1] == 0x0021 and # !
403                  $self->{prev_char}->[2] == 0x003C) { # <                  $self->{prev_char}->[2] == 0x003C) { # <
404                  !!!cp (3);
405                $self->{escape} = 1;                $self->{escape} = 1;
406                } else {
407                  !!!cp (4);
408              }              }
409              } else {
410                !!!cp (5);
411            }            }
412          }          }
413                    
# Line 409  sub _get_next_token ($) { Line 416  sub _get_next_token ($) {
416          if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA          if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA
417              (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA              (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA
418               not $self->{escape})) {               not $self->{escape})) {
419              !!!cp (6);
420            $self->{state} = TAG_OPEN_STATE;            $self->{state} = TAG_OPEN_STATE;
421            !!!next-input-character;            !!!next-input-character;
422            redo A;            redo A;
423          } else {          } else {
424              !!!cp (7);
425            #            #
426          }          }
427        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
# Line 420  sub _get_next_token ($) { Line 429  sub _get_next_token ($) {
429              ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA              ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA
430            if ($self->{prev_char}->[0] == 0x002D and # -            if ($self->{prev_char}->[0] == 0x002D and # -
431                $self->{prev_char}->[1] == 0x002D) { # -                $self->{prev_char}->[1] == 0x002D) { # -
432                !!!cp (8);
433              delete $self->{escape};              delete $self->{escape};
434              } else {
435                !!!cp (9);
436            }            }
437            } else {
438              !!!cp (10);
439          }          }
440                    
441          #          #
442        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
443            !!!cp (11);
444          !!!emit ({type => END_OF_FILE_TOKEN});          !!!emit ({type => END_OF_FILE_TOKEN});
445          last A; ## TODO: ok?          last A; ## TODO: ok?
446          } else {
447            !!!cp (12);
448        }        }
449        # Anything else        # Anything else
450        my $token = {type => CHARACTER_TOKEN,        my $token = {type => CHARACTER_TOKEN,
# Line 447  sub _get_next_token ($) { Line 464  sub _get_next_token ($) {
464        # next-input-character is already done        # next-input-character is already done
465    
466        unless (defined $token) {        unless (defined $token) {
467            !!!cp (13);
468          !!!emit ({type => CHARACTER_TOKEN, data => '&'});          !!!emit ({type => CHARACTER_TOKEN, data => '&'});
469        } else {        } else {
470            !!!cp (14);
471          !!!emit ($token);          !!!emit ($token);
472        }        }
473    
# Line 456  sub _get_next_token ($) { Line 475  sub _get_next_token ($) {
475      } elsif ($self->{state} == TAG_OPEN_STATE) {      } elsif ($self->{state} == TAG_OPEN_STATE) {
476        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
477          if ($self->{next_char} == 0x002F) { # /          if ($self->{next_char} == 0x002F) { # /
478              !!!cp (15);
479            !!!next-input-character;            !!!next-input-character;
480            $self->{state} = CLOSE_TAG_OPEN_STATE;            $self->{state} = CLOSE_TAG_OPEN_STATE;
481            redo A;            redo A;
482          } else {          } else {
483              !!!cp (16);
484            ## reconsume            ## reconsume
485            $self->{state} = DATA_STATE;            $self->{state} = DATA_STATE;
486    
# Line 469  sub _get_next_token ($) { Line 490  sub _get_next_token ($) {
490          }          }
491        } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA        } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA
492          if ($self->{next_char} == 0x0021) { # !          if ($self->{next_char} == 0x0021) { # !
493              !!!cp (17);
494            $self->{state} = MARKUP_DECLARATION_OPEN_STATE;            $self->{state} = MARKUP_DECLARATION_OPEN_STATE;
495            !!!next-input-character;            !!!next-input-character;
496            redo A;            redo A;
497          } elsif ($self->{next_char} == 0x002F) { # /          } elsif ($self->{next_char} == 0x002F) { # /
498              !!!cp (18);
499            $self->{state} = CLOSE_TAG_OPEN_STATE;            $self->{state} = CLOSE_TAG_OPEN_STATE;
500            !!!next-input-character;            !!!next-input-character;
501            redo A;            redo A;
502          } elsif (0x0041 <= $self->{next_char} and          } elsif (0x0041 <= $self->{next_char} and
503                   $self->{next_char} <= 0x005A) { # A..Z                   $self->{next_char} <= 0x005A) { # A..Z
504              !!!cp (19);
505            $self->{current_token}            $self->{current_token}
506              = {type => START_TAG_TOKEN,              = {type => START_TAG_TOKEN,
507                 tag_name => chr ($self->{next_char} + 0x0020)};                 tag_name => chr ($self->{next_char} + 0x0020)};
# Line 486  sub _get_next_token ($) { Line 510  sub _get_next_token ($) {
510            redo A;            redo A;
511          } elsif (0x0061 <= $self->{next_char} and          } elsif (0x0061 <= $self->{next_char} and
512                   $self->{next_char} <= 0x007A) { # a..z                   $self->{next_char} <= 0x007A) { # a..z
513              !!!cp (20);
514            $self->{current_token} = {type => START_TAG_TOKEN,            $self->{current_token} = {type => START_TAG_TOKEN,
515                              tag_name => chr ($self->{next_char})};                              tag_name => chr ($self->{next_char})};
516            $self->{state} = TAG_NAME_STATE;            $self->{state} = TAG_NAME_STATE;
517            !!!next-input-character;            !!!next-input-character;
518            redo A;            redo A;
519          } elsif ($self->{next_char} == 0x003E) { # >          } elsif ($self->{next_char} == 0x003E) { # >
520              !!!cp (21);
521            !!!parse-error (type => 'empty start tag');            !!!parse-error (type => 'empty start tag');
522            $self->{state} = DATA_STATE;            $self->{state} = DATA_STATE;
523            !!!next-input-character;            !!!next-input-character;
# Line 500  sub _get_next_token ($) { Line 526  sub _get_next_token ($) {
526    
527            redo A;            redo A;
528          } elsif ($self->{next_char} == 0x003F) { # ?          } elsif ($self->{next_char} == 0x003F) { # ?
529              !!!cp (22);
530            !!!parse-error (type => 'pio');            !!!parse-error (type => 'pio');
531            $self->{state} = BOGUS_COMMENT_STATE;            $self->{state} = BOGUS_COMMENT_STATE;
532            ## $self->{next_char} is intentionally left as is            ## $self->{next_char} is intentionally left as is
533            redo A;            redo A;
534          } else {          } else {
535              !!!cp (23);
536            !!!parse-error (type => 'bare stago');            !!!parse-error (type => 'bare stago');
537            $self->{state} = DATA_STATE;            $self->{state} = DATA_STATE;
538            ## reconsume            ## reconsume
# Line 526  sub _get_next_token ($) { Line 554  sub _get_next_token ($) {
554              my $c = ord substr ($self->{last_emitted_start_tag_name}, $i, 1);              my $c = ord substr ($self->{last_emitted_start_tag_name}, $i, 1);
555              my $C = 0x0061 <= $c && $c <= 0x007A ? $c - 0x0020 : $c;              my $C = 0x0061 <= $c && $c <= 0x007A ? $c - 0x0020 : $c;
556              if ($self->{next_char} == $c or $self->{next_char} == $C) {              if ($self->{next_char} == $c or $self->{next_char} == $C) {
557                  !!!cp (24);
558                !!!next-input-character;                !!!next-input-character;
559                next TAGNAME;                next TAGNAME;
560              } else {              } else {
561                  !!!cp (25);
562                $self->{next_char} = shift @next_char; # reconsume                $self->{next_char} = shift @next_char; # reconsume
563                !!!back-next-input-character (@next_char);                !!!back-next-input-character (@next_char);
564                $self->{state} = DATA_STATE;                $self->{state} = DATA_STATE;
# Line 548  sub _get_next_token ($) { Line 578  sub _get_next_token ($) {
578                    $self->{next_char} == 0x003E or # >                    $self->{next_char} == 0x003E or # >
579                    $self->{next_char} == 0x002F or # /                    $self->{next_char} == 0x002F or # /
580                    $self->{next_char} == -1) {                    $self->{next_char} == -1) {
581                !!!cp (26);
582              $self->{next_char} = shift @next_char; # reconsume              $self->{next_char} = shift @next_char; # reconsume
583              !!!back-next-input-character (@next_char);              !!!back-next-input-character (@next_char);
584              $self->{state} = DATA_STATE;              $self->{state} = DATA_STATE;
585              !!!emit ({type => CHARACTER_TOKEN, data => '</'});              !!!emit ({type => CHARACTER_TOKEN, data => '</'});
586              redo A;              redo A;
587            } else {            } else {
588                !!!cp (27);
589              $self->{next_char} = shift @next_char;              $self->{next_char} = shift @next_char;
590              !!!back-next-input-character (@next_char);              !!!back-next-input-character (@next_char);
591              # and consume...              # and consume...
592            }            }
593          } else {          } else {
594            ## No start tag token has ever been emitted            ## No start tag token has ever been emitted
595              !!!cp (28);
596            # next-input-character is already done            # next-input-character is already done
597            $self->{state} = DATA_STATE;            $self->{state} = DATA_STATE;
598            !!!emit ({type => CHARACTER_TOKEN, data => '</'});            !!!emit ({type => CHARACTER_TOKEN, data => '</'});
# Line 569  sub _get_next_token ($) { Line 602  sub _get_next_token ($) {
602                
603        if (0x0041 <= $self->{next_char} and        if (0x0041 <= $self->{next_char} and
604            $self->{next_char} <= 0x005A) { # A..Z            $self->{next_char} <= 0x005A) { # A..Z
605            !!!cp (29);
606          $self->{current_token} = {type => END_TAG_TOKEN,          $self->{current_token} = {type => END_TAG_TOKEN,
607                            tag_name => chr ($self->{next_char} + 0x0020)};                            tag_name => chr ($self->{next_char} + 0x0020)};
608          $self->{state} = TAG_NAME_STATE;          $self->{state} = TAG_NAME_STATE;
# Line 576  sub _get_next_token ($) { Line 610  sub _get_next_token ($) {
610          redo A;          redo A;
611        } elsif (0x0061 <= $self->{next_char} and        } elsif (0x0061 <= $self->{next_char} and
612                 $self->{next_char} <= 0x007A) { # a..z                 $self->{next_char} <= 0x007A) { # a..z
613            !!!cp (30);
614          $self->{current_token} = {type => END_TAG_TOKEN,          $self->{current_token} = {type => END_TAG_TOKEN,
615                            tag_name => chr ($self->{next_char})};                            tag_name => chr ($self->{next_char})};
616          $self->{state} = TAG_NAME_STATE;          $self->{state} = TAG_NAME_STATE;
617          !!!next-input-character;          !!!next-input-character;
618          redo A;          redo A;
619        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
620            !!!cp (31);
621          !!!parse-error (type => 'empty end tag');          !!!parse-error (type => 'empty end tag');
622          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
623          !!!next-input-character;          !!!next-input-character;
624          redo A;          redo A;
625        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
626            !!!cp (32);
627          !!!parse-error (type => 'bare etago');          !!!parse-error (type => 'bare etago');
628          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
629          # reconsume          # reconsume
# Line 595  sub _get_next_token ($) { Line 632  sub _get_next_token ($) {
632    
633          redo A;          redo A;
634        } else {        } else {
635            !!!cp (33);
636          !!!parse-error (type => 'bogus end tag');          !!!parse-error (type => 'bogus end tag');
637          $self->{state} = BOGUS_COMMENT_STATE;          $self->{state} = BOGUS_COMMENT_STATE;
638          ## $self->{next_char} is intentionally left as is          ## $self->{next_char} is intentionally left as is
# Line 606  sub _get_next_token ($) { Line 644  sub _get_next_token ($) {
644            $self->{next_char} == 0x000B or # VT            $self->{next_char} == 0x000B or # VT
645            $self->{next_char} == 0x000C or # FF            $self->{next_char} == 0x000C or # FF
646            $self->{next_char} == 0x0020) { # SP            $self->{next_char} == 0x0020) { # SP
647            !!!cp (34);
648          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
649          !!!next-input-character;          !!!next-input-character;
650          redo A;          redo A;
651        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
652          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
653              !!!cp (35);
654            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
655                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
656            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
657          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
658            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
659            if ($self->{current_token}->{attributes}) {            #if ($self->{current_token}->{attributes}) {
660              !!!parse-error (type => 'end tag attribute');            #  ## NOTE: This should never be reached.
661            }            #  !!! cp (36);
662              #  !!! parse-error (type => 'end tag attribute');
663              #} else {
664                !!!cp (37);
665              #}
666          } else {          } else {
667            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
668          }          }
# Line 630  sub _get_next_token ($) { Line 674  sub _get_next_token ($) {
674          redo A;          redo A;
675        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{next_char} and
676                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{next_char} <= 0x005A) { # A..Z
677            !!!cp (38);
678          $self->{current_token}->{tag_name} .= chr ($self->{next_char} + 0x0020);          $self->{current_token}->{tag_name} .= chr ($self->{next_char} + 0x0020);
679            # start tag or end tag            # start tag or end tag
680          ## Stay in this state          ## Stay in this state
# Line 638  sub _get_next_token ($) { Line 683  sub _get_next_token ($) {
683        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
684          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
685          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
686              !!!cp (39);
687            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
688                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
689            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
690          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
691            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
692            if ($self->{current_token}->{attributes}) {            #if ($self->{current_token}->{attributes}) {
693              !!!parse-error (type => 'end tag attribute');            #  ## NOTE: This state should never be reached.
694            }            #  !!! cp (40);
695              #  !!! parse-error (type => 'end tag attribute');
696              #} else {
697                !!!cp (41);
698              #}
699          } else {          } else {
700            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
701          }          }
# Line 661  sub _get_next_token ($) { Line 711  sub _get_next_token ($) {
711              $self->{current_token}->{type} == START_TAG_TOKEN and              $self->{current_token}->{type} == START_TAG_TOKEN and
712              $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {              $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {
713            # permitted slash            # permitted slash
714              !!!cp (42);
715            #            #
716          } else {          } else {
717              !!!cp (43);
718            !!!parse-error (type => 'nestc');            !!!parse-error (type => 'nestc');
719          }          }
720          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
721          # next-input-character is already done          # next-input-character is already done
722          redo A;          redo A;
723        } else {        } else {
724            !!!cp (44);
725          $self->{current_token}->{tag_name} .= chr $self->{next_char};          $self->{current_token}->{tag_name} .= chr $self->{next_char};
726            # start tag or end tag            # start tag or end tag
727          ## Stay in the state          ## Stay in the state
# Line 681  sub _get_next_token ($) { Line 734  sub _get_next_token ($) {
734            $self->{next_char} == 0x000B or # VT            $self->{next_char} == 0x000B or # VT
735            $self->{next_char} == 0x000C or # FF            $self->{next_char} == 0x000C or # FF
736            $self->{next_char} == 0x0020) { # SP            $self->{next_char} == 0x0020) { # SP
737            !!!cp (45);
738          ## Stay in the state          ## Stay in the state
739          !!!next-input-character;          !!!next-input-character;
740          redo A;          redo A;
741        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
742          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
743              !!!cp (46);
744            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
745                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
746            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
747          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
748            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
749            if ($self->{current_token}->{attributes}) {            if ($self->{current_token}->{attributes}) {
750                !!!cp (47);
751              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
752              } else {
753                !!!cp (48);
754            }            }
755          } else {          } else {
756            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
# Line 705  sub _get_next_token ($) { Line 763  sub _get_next_token ($) {
763          redo A;          redo A;
764        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{next_char} and
765                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{next_char} <= 0x005A) { # A..Z
766            !!!cp (49);
767          $self->{current_attribute} = {name => chr ($self->{next_char} + 0x0020),          $self->{current_attribute} = {name => chr ($self->{next_char} + 0x0020),
768                                value => ''};                                value => ''};
769          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
# Line 716  sub _get_next_token ($) { Line 775  sub _get_next_token ($) {
775              $self->{current_token}->{type} == START_TAG_TOKEN and              $self->{current_token}->{type} == START_TAG_TOKEN and
776              $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {              $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {
777            # permitted slash            # permitted slash
778              !!!cp (50);
779            #            #
780          } else {          } else {
781              !!!cp (51);
782            !!!parse-error (type => 'nestc');            !!!parse-error (type => 'nestc');
783          }          }
784          ## Stay in the state          ## Stay in the state
# Line 726  sub _get_next_token ($) { Line 787  sub _get_next_token ($) {
787        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
788          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
789          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
790              !!!cp (52);
791            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
792                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
793            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
794          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
795            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
796            if ($self->{current_token}->{attributes}) {            if ($self->{current_token}->{attributes}) {
797                !!!cp (53);
798              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
799              } else {
800                !!!cp (54);
801            }            }
802          } else {          } else {
803            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
# Line 749  sub _get_next_token ($) { Line 814  sub _get_next_token ($) {
814               0x0027 => 1, # '               0x0027 => 1, # '
815               0x003D => 1, # =               0x003D => 1, # =
816              }->{$self->{next_char}}) {              }->{$self->{next_char}}) {
817              !!!cp (55);
818            !!!parse-error (type => 'bad attribute name');            !!!parse-error (type => 'bad attribute name');
819            } else {
820              !!!cp (56);
821          }          }
822          $self->{current_attribute} = {name => chr ($self->{next_char}),          $self->{current_attribute} = {name => chr ($self->{next_char}),
823                                value => ''};                                value => ''};
# Line 761  sub _get_next_token ($) { Line 829  sub _get_next_token ($) {
829        my $before_leave = sub {        my $before_leave = sub {
830          if (exists $self->{current_token}->{attributes} # start tag or end tag          if (exists $self->{current_token}->{attributes} # start tag or end tag
831              ->{$self->{current_attribute}->{name}}) { # MUST              ->{$self->{current_attribute}->{name}}) { # MUST
832              !!!cp (57);
833            !!!parse-error (type => 'duplicate attribute:'.$self->{current_attribute}->{name});            !!!parse-error (type => 'duplicate attribute:'.$self->{current_attribute}->{name});
834            ## Discard $self->{current_attribute} # MUST            ## Discard $self->{current_attribute} # MUST
835          } else {          } else {
836              !!!cp (58);
837            $self->{current_token}->{attributes}->{$self->{current_attribute}->{name}}            $self->{current_token}->{attributes}->{$self->{current_attribute}->{name}}
838              = $self->{current_attribute};              = $self->{current_attribute};
839          }          }
# Line 774  sub _get_next_token ($) { Line 844  sub _get_next_token ($) {
844            $self->{next_char} == 0x000B or # VT            $self->{next_char} == 0x000B or # VT
845            $self->{next_char} == 0x000C or # FF            $self->{next_char} == 0x000C or # FF
846            $self->{next_char} == 0x0020) { # SP            $self->{next_char} == 0x0020) { # SP
847            !!!cp (59);
848          $before_leave->();          $before_leave->();
849          $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;          $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;
850          !!!next-input-character;          !!!next-input-character;
851          redo A;          redo A;
852        } elsif ($self->{next_char} == 0x003D) { # =        } elsif ($self->{next_char} == 0x003D) { # =
853            !!!cp (60);
854          $before_leave->();          $before_leave->();
855          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
856          !!!next-input-character;          !!!next-input-character;
# Line 786  sub _get_next_token ($) { Line 858  sub _get_next_token ($) {
858        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
859          $before_leave->();          $before_leave->();
860          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
861              !!!cp (61);
862            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
863                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
864            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
865          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
866              !!!cp (62);
867            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
868            if ($self->{current_token}->{attributes}) {            if ($self->{current_token}->{attributes}) {
869              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
# Line 805  sub _get_next_token ($) { Line 879  sub _get_next_token ($) {
879          redo A;          redo A;
880        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{next_char} and
881                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{next_char} <= 0x005A) { # A..Z
882            !!!cp (63);
883          $self->{current_attribute}->{name} .= chr ($self->{next_char} + 0x0020);          $self->{current_attribute}->{name} .= chr ($self->{next_char} + 0x0020);
884          ## Stay in the state          ## Stay in the state
885          !!!next-input-character;          !!!next-input-character;
# Line 816  sub _get_next_token ($) { Line 891  sub _get_next_token ($) {
891              $self->{current_token}->{type} == START_TAG_TOKEN and              $self->{current_token}->{type} == START_TAG_TOKEN and
892              $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {              $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {
893            # permitted slash            # permitted slash
894              !!!cp (64);
895            #            #
896          } else {          } else {
897              !!!cp (65);
898            !!!parse-error (type => 'nestc');            !!!parse-error (type => 'nestc');
899          }          }
900          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
# Line 827  sub _get_next_token ($) { Line 904  sub _get_next_token ($) {
904          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
905          $before_leave->();          $before_leave->();
906          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
907              !!!cp (66);
908            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
909                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
910            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
911          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
912            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
913            if ($self->{current_token}->{attributes}) {            if ($self->{current_token}->{attributes}) {
914                !!!cp (67);
915              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
916              } else {
917                ## NOTE: This state should never be reached.
918                !!!cp (68);
919            }            }
920          } else {          } else {
921            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
# Line 847  sub _get_next_token ($) { Line 929  sub _get_next_token ($) {
929        } else {        } else {
930          if ($self->{next_char} == 0x0022 or # "          if ($self->{next_char} == 0x0022 or # "
931              $self->{next_char} == 0x0027) { # '              $self->{next_char} == 0x0027) { # '
932              !!!cp (69);
933            !!!parse-error (type => 'bad attribute name');            !!!parse-error (type => 'bad attribute name');
934            } else {
935              !!!cp (70);
936          }          }
937          $self->{current_attribute}->{name} .= chr ($self->{next_char});          $self->{current_attribute}->{name} .= chr ($self->{next_char});
938          ## Stay in the state          ## Stay in the state
# Line 860  sub _get_next_token ($) { Line 945  sub _get_next_token ($) {
945            $self->{next_char} == 0x000B or # VT            $self->{next_char} == 0x000B or # VT
946            $self->{next_char} == 0x000C or # FF            $self->{next_char} == 0x000C or # FF
947            $self->{next_char} == 0x0020) { # SP            $self->{next_char} == 0x0020) { # SP
948            !!!cp (71);
949          ## Stay in the state          ## Stay in the state
950          !!!next-input-character;          !!!next-input-character;
951          redo A;          redo A;
952        } elsif ($self->{next_char} == 0x003D) { # =        } elsif ($self->{next_char} == 0x003D) { # =
953            !!!cp (72);
954          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
955          !!!next-input-character;          !!!next-input-character;
956          redo A;          redo A;
957        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
958          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
959              !!!cp (73);
960            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
961                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
962            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
963          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
964            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
965            if ($self->{current_token}->{attributes}) {            if ($self->{current_token}->{attributes}) {
966                !!!cp (74);
967              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
968              } else {
969                ## NOTE: This state should never be reached.
970                !!!cp (75);
971            }            }
972          } else {          } else {
973            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
# Line 888  sub _get_next_token ($) { Line 980  sub _get_next_token ($) {
980          redo A;          redo A;
981        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{next_char} and
982                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{next_char} <= 0x005A) { # A..Z
983            !!!cp (76);
984          $self->{current_attribute} = {name => chr ($self->{next_char} + 0x0020),          $self->{current_attribute} = {name => chr ($self->{next_char} + 0x0020),
985                                value => ''};                                value => ''};
986          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
# Line 899  sub _get_next_token ($) { Line 992  sub _get_next_token ($) {
992              $self->{current_token}->{type} == START_TAG_TOKEN and              $self->{current_token}->{type} == START_TAG_TOKEN and
993              $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {              $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {
994            # permitted slash            # permitted slash
995              !!!cp (77);
996            #            #
997          } else {          } else {
998              !!!cp (78);
999            !!!parse-error (type => 'nestc');            !!!parse-error (type => 'nestc');
1000            ## TODO: Different error type for <aa / bb> than <aa/>            ## TODO: Different error type for <aa / bb> than <aa/>
1001          }          }
# Line 910  sub _get_next_token ($) { Line 1005  sub _get_next_token ($) {
1005        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
1006          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1007          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1008              !!!cp (79);
1009            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
1010                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
1011            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1012          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1013            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1014            if ($self->{current_token}->{attributes}) {            if ($self->{current_token}->{attributes}) {
1015                !!!cp (80);
1016              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1017              } else {
1018                ## NOTE: This state should never be reached.
1019                !!!cp (81);
1020            }            }
1021          } else {          } else {
1022            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
# Line 928  sub _get_next_token ($) { Line 1028  sub _get_next_token ($) {
1028    
1029          redo A;          redo A;
1030        } else {        } else {
1031            !!!cp (82);
1032          $self->{current_attribute} = {name => chr ($self->{next_char}),          $self->{current_attribute} = {name => chr ($self->{next_char}),
1033                                value => ''};                                value => ''};
1034          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
# Line 940  sub _get_next_token ($) { Line 1041  sub _get_next_token ($) {
1041            $self->{next_char} == 0x000B or # VT            $self->{next_char} == 0x000B or # VT
1042            $self->{next_char} == 0x000C or # FF            $self->{next_char} == 0x000C or # FF
1043            $self->{next_char} == 0x0020) { # SP                  $self->{next_char} == 0x0020) { # SP      
1044            !!!cp (83);
1045          ## Stay in the state          ## Stay in the state
1046          !!!next-input-character;          !!!next-input-character;
1047          redo A;          redo A;
1048        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{next_char} == 0x0022) { # "
1049            !!!cp (84);
1050          $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;
1051          !!!next-input-character;          !!!next-input-character;
1052          redo A;          redo A;
1053        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{next_char} == 0x0026) { # &
1054            !!!cp (85);
1055          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1056          ## reconsume          ## reconsume
1057          redo A;          redo A;
1058        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{next_char} == 0x0027) { # '
1059            !!!cp (86);
1060          $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;
1061          !!!next-input-character;          !!!next-input-character;
1062          redo A;          redo A;
1063        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
1064          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1065              !!!cp (87);
1066            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
1067                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
1068            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1069          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1070            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1071            if ($self->{current_token}->{attributes}) {            if ($self->{current_token}->{attributes}) {
1072                !!!cp (88);
1073              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1074              } else {
1075                ## NOTE: This state should never be reached.
1076                !!!cp (89);
1077            }            }
1078          } else {          } else {
1079            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
# Line 977  sub _get_next_token ($) { Line 1087  sub _get_next_token ($) {
1087        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
1088          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1089          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1090              !!!cp (90);
1091            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
1092                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
1093            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1094          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1095            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1096            if ($self->{current_token}->{attributes}) {            if ($self->{current_token}->{attributes}) {
1097                !!!cp (91);
1098              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1099              } else {
1100                ## NOTE: This state should never be reached.
1101                !!!cp (92);
1102            }            }
1103          } else {          } else {
1104            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
# Line 996  sub _get_next_token ($) { Line 1111  sub _get_next_token ($) {
1111          redo A;          redo A;
1112        } else {        } else {
1113          if ($self->{next_char} == 0x003D) { # =          if ($self->{next_char} == 0x003D) { # =
1114              !!!cp (93);
1115            !!!parse-error (type => 'bad attribute value');            !!!parse-error (type => 'bad attribute value');
1116            } else {
1117              !!!cp (94);
1118          }          }
1119          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{current_attribute}->{value} .= chr ($self->{next_char});
1120          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
# Line 1005  sub _get_next_token ($) { Line 1123  sub _get_next_token ($) {
1123        }        }
1124      } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {
1125        if ($self->{next_char} == 0x0022) { # "        if ($self->{next_char} == 0x0022) { # "
1126            !!!cp (95);
1127          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1128          !!!next-input-character;          !!!next-input-character;
1129          redo A;          redo A;
1130        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{next_char} == 0x0026) { # &
1131            !!!cp (96);
1132          $self->{last_attribute_value_state} = $self->{state};          $self->{last_attribute_value_state} = $self->{state};
1133          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;
1134          !!!next-input-character;          !!!next-input-character;
# Line 1016  sub _get_next_token ($) { Line 1136  sub _get_next_token ($) {
1136        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
1137          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
1138          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1139              !!!cp (97);
1140            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
1141                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
1142            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1143          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1144            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1145            if ($self->{current_token}->{attributes}) {            if ($self->{current_token}->{attributes}) {
1146                !!!cp (98);
1147              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1148              } else {
1149                ## NOTE: This state should never be reached.
1150                !!!cp (99);
1151            }            }
1152          } else {          } else {
1153            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
# Line 1034  sub _get_next_token ($) { Line 1159  sub _get_next_token ($) {
1159    
1160          redo A;          redo A;
1161        } else {        } else {
1162            !!!cp (100);
1163          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{current_attribute}->{value} .= chr ($self->{next_char});
1164          ## Stay in the state          ## Stay in the state
1165          !!!next-input-character;          !!!next-input-character;
# Line 1041  sub _get_next_token ($) { Line 1167  sub _get_next_token ($) {
1167        }        }
1168      } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {
1169        if ($self->{next_char} == 0x0027) { # '        if ($self->{next_char} == 0x0027) { # '
1170            !!!cp (101);
1171          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1172          !!!next-input-character;          !!!next-input-character;
1173          redo A;          redo A;
1174        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{next_char} == 0x0026) { # &
1175            !!!cp (102);
1176          $self->{last_attribute_value_state} = $self->{state};          $self->{last_attribute_value_state} = $self->{state};
1177          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;
1178          !!!next-input-character;          !!!next-input-character;
# Line 1052  sub _get_next_token ($) { Line 1180  sub _get_next_token ($) {
1180        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
1181          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
1182          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1183              !!!cp (103);
1184            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
1185                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
1186            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1187          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1188            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1189            if ($self->{current_token}->{attributes}) {            if ($self->{current_token}->{attributes}) {
1190                !!!cp (104);
1191              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1192              } else {
1193                ## NOTE: This state should never be reached.
1194                !!!cp (105);
1195            }            }
1196          } else {          } else {
1197            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
# Line 1070  sub _get_next_token ($) { Line 1203  sub _get_next_token ($) {
1203    
1204          redo A;          redo A;
1205        } else {        } else {
1206            !!!cp (106);
1207          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{current_attribute}->{value} .= chr ($self->{next_char});
1208          ## Stay in the state          ## Stay in the state
1209          !!!next-input-character;          !!!next-input-character;
# Line 1081  sub _get_next_token ($) { Line 1215  sub _get_next_token ($) {
1215            $self->{next_char} == 0x000B or # HT            $self->{next_char} == 0x000B or # HT
1216            $self->{next_char} == 0x000C or # FF            $self->{next_char} == 0x000C or # FF
1217            $self->{next_char} == 0x0020) { # SP            $self->{next_char} == 0x0020) { # SP
1218            !!!cp (107);
1219          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1220          !!!next-input-character;          !!!next-input-character;
1221          redo A;          redo A;
1222        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{next_char} == 0x0026) { # &
1223            !!!cp (108);
1224          $self->{last_attribute_value_state} = $self->{state};          $self->{last_attribute_value_state} = $self->{state};
1225          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;
1226          !!!next-input-character;          !!!next-input-character;
1227          redo A;          redo A;
1228        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
1229          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1230              !!!cp (109);
1231            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
1232                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
1233            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1234          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1235            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1236            if ($self->{current_token}->{attributes}) {            if ($self->{current_token}->{attributes}) {
1237                !!!cp (110);
1238              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1239              } else {
1240                ## NOTE: This state should never be reached.
1241                !!!cp (111);
1242            }            }
1243          } else {          } else {
1244            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
# Line 1111  sub _get_next_token ($) { Line 1252  sub _get_next_token ($) {
1252        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
1253          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1254          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1255              !!!cp (112);
1256            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
1257                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
1258            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1259          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1260            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1261            if ($self->{current_token}->{attributes}) {            if ($self->{current_token}->{attributes}) {
1262                !!!cp (113);
1263              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1264              } else {
1265                ## NOTE: This state should never be reached.
1266                !!!cp (114);
1267            }            }
1268          } else {          } else {
1269            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
# Line 1134  sub _get_next_token ($) { Line 1280  sub _get_next_token ($) {
1280               0x0027 => 1, # '               0x0027 => 1, # '
1281               0x003D => 1, # =               0x003D => 1, # =
1282              }->{$self->{next_char}}) {              }->{$self->{next_char}}) {
1283              !!!cp (115);
1284            !!!parse-error (type => 'bad attribute value');            !!!parse-error (type => 'bad attribute value');
1285            } else {
1286              !!!cp (116);
1287          }          }
1288          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{current_attribute}->{value} .= chr ($self->{next_char});
1289          ## Stay in the state          ## Stay in the state
# Line 1151  sub _get_next_token ($) { Line 1300  sub _get_next_token ($) {
1300             -1);             -1);
1301    
1302        unless (defined $token) {        unless (defined $token) {
1303            !!!cp (117);
1304          $self->{current_attribute}->{value} .= '&';          $self->{current_attribute}->{value} .= '&';
1305        } else {        } else {
1306            !!!cp (118);
1307          $self->{current_attribute}->{value} .= $token->{data};          $self->{current_attribute}->{value} .= $token->{data};
1308          $self->{current_attribute}->{has_reference} = $token->{has_reference};          $self->{current_attribute}->{has_reference} = $token->{has_reference};
1309          ## ISSUE: spec says "append the returned character token to the current attribute's value"          ## ISSUE: spec says "append the returned character token to the current attribute's value"
# Line 1167  sub _get_next_token ($) { Line 1318  sub _get_next_token ($) {
1318            $self->{next_char} == 0x000B or # VT            $self->{next_char} == 0x000B or # VT
1319            $self->{next_char} == 0x000C or # FF            $self->{next_char} == 0x000C or # FF
1320            $self->{next_char} == 0x0020) { # SP            $self->{next_char} == 0x0020) { # SP
1321            !!!cp (118);
1322          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1323          !!!next-input-character;          !!!next-input-character;
1324          redo A;          redo A;
1325        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
1326          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1327              !!!cp (119);
1328            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
1329                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
1330            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1331          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1332            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1333            if ($self->{current_token}->{attributes}) {            if ($self->{current_token}->{attributes}) {
1334                !!!cp (120);
1335              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1336              } else {
1337                ## NOTE: This state should never be reached.
1338                !!!cp (121);
1339            }            }
1340          } else {          } else {
1341            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
# Line 1195  sub _get_next_token ($) { Line 1352  sub _get_next_token ($) {
1352              $self->{current_token}->{type} == START_TAG_TOKEN and              $self->{current_token}->{type} == START_TAG_TOKEN and
1353              $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {              $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {
1354            # permitted slash            # permitted slash
1355              !!!cp (122);
1356            #            #
1357          } else {          } else {
1358              !!!cp (123);
1359            !!!parse-error (type => 'nestc');            !!!parse-error (type => 'nestc');
1360          }          }
1361          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1362          # next-input-character is already done          # next-input-character is already done
1363          redo A;          redo A;
1364        } else {        } else {
1365            !!!cp (124);
1366          !!!parse-error (type => 'no space between attributes');          !!!parse-error (type => 'no space between attributes');
1367          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1368          ## reconsume          ## reconsume
# Line 1215  sub _get_next_token ($) { Line 1375  sub _get_next_token ($) {
1375    
1376        BC: {        BC: {
1377          if ($self->{next_char} == 0x003E) { # >          if ($self->{next_char} == 0x003E) { # >
1378              !!!cp (124);
1379            $self->{state} = DATA_STATE;            $self->{state} = DATA_STATE;
1380            !!!next-input-character;            !!!next-input-character;
1381    
# Line 1222  sub _get_next_token ($) { Line 1383  sub _get_next_token ($) {
1383    
1384            redo A;            redo A;
1385          } elsif ($self->{next_char} == -1) {          } elsif ($self->{next_char} == -1) {
1386              !!!cp (125);
1387            $self->{state} = DATA_STATE;            $self->{state} = DATA_STATE;
1388            ## reconsume            ## reconsume
1389    
# Line 1229  sub _get_next_token ($) { Line 1391  sub _get_next_token ($) {
1391    
1392            redo A;            redo A;
1393          } else {          } else {
1394              !!!cp (126);
1395            $token->{data} .= chr ($self->{next_char});            $token->{data} .= chr ($self->{next_char});
1396            !!!next-input-character;            !!!next-input-character;
1397            redo BC;            redo BC;
1398          }          }
1399        } # BC        } # BC
1400    
1401          die "$0: _get_next_token: unexpected case [BC]";
1402      } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {      } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {
1403        ## (only happen if PCDATA state)        ## (only happen if PCDATA state)
1404    
# Line 1244  sub _get_next_token ($) { Line 1409  sub _get_next_token ($) {
1409          !!!next-input-character;          !!!next-input-character;
1410          push @next_char, $self->{next_char};          push @next_char, $self->{next_char};
1411          if ($self->{next_char} == 0x002D) { # -          if ($self->{next_char} == 0x002D) { # -
1412              !!!cp (127);
1413            $self->{current_token} = {type => COMMENT_TOKEN, data => ''};            $self->{current_token} = {type => COMMENT_TOKEN, data => ''};
1414            $self->{state} = COMMENT_START_STATE;            $self->{state} = COMMENT_START_STATE;
1415            !!!next-input-character;            !!!next-input-character;
1416            redo A;            redo A;
1417            } else {
1418              !!!cp (128);
1419          }          }
1420        } elsif ($self->{next_char} == 0x0044 or # D        } elsif ($self->{next_char} == 0x0044 or # D
1421                 $self->{next_char} == 0x0064) { # d                 $self->{next_char} == 0x0064) { # d
# Line 1275  sub _get_next_token ($) { Line 1443  sub _get_next_token ($) {
1443                    push @next_char, $self->{next_char};                    push @next_char, $self->{next_char};
1444                    if ($self->{next_char} == 0x0045 or # E                    if ($self->{next_char} == 0x0045 or # E
1445                        $self->{next_char} == 0x0065) { # e                        $self->{next_char} == 0x0065) { # e
1446                      ## ISSUE: What a stupid code this is!                      !!!cp (129);
1447                        ## TODO: What a stupid code this is!
1448                      $self->{state} = DOCTYPE_STATE;                      $self->{state} = DOCTYPE_STATE;
1449                      !!!next-input-character;                      !!!next-input-character;
1450                      redo A;                      redo A;
1451                      } else {
1452                        !!!cp (130);
1453                    }                    }
1454                    } else {
1455                      !!!cp (131);
1456                  }                  }
1457                  } else {
1458                    !!!cp (132);
1459                }                }
1460                } else {
1461                  !!!cp (133);
1462              }              }
1463              } else {
1464                !!!cp (134);
1465            }            }
1466            } else {
1467              !!!cp (135);
1468          }          }
1469          } else {
1470            !!!cp (136);
1471        }        }
1472    
1473        !!!parse-error (type => 'bogus comment');        !!!parse-error (type => 'bogus comment');
# Line 1297  sub _get_next_token ($) { Line 1480  sub _get_next_token ($) {
1480        ## ISSUE: spec is somewhat unclear on "is the first character that will be in the comment"; what is "that will be in the comment" is what the algorithm defines, isn't it?        ## ISSUE: spec is somewhat unclear on "is the first character that will be in the comment"; what is "that will be in the comment" is what the algorithm defines, isn't it?
1481      } elsif ($self->{state} == COMMENT_START_STATE) {      } elsif ($self->{state} == COMMENT_START_STATE) {
1482        if ($self->{next_char} == 0x002D) { # -        if ($self->{next_char} == 0x002D) { # -
1483            !!!cp (137);
1484          $self->{state} = COMMENT_START_DASH_STATE;          $self->{state} = COMMENT_START_DASH_STATE;
1485          !!!next-input-character;          !!!next-input-character;
1486          redo A;          redo A;
1487        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
1488            !!!cp (138);
1489          !!!parse-error (type => 'bogus comment');          !!!parse-error (type => 'bogus comment');
1490          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1491          !!!next-input-character;          !!!next-input-character;
# Line 1309  sub _get_next_token ($) { Line 1494  sub _get_next_token ($) {
1494    
1495          redo A;          redo A;
1496        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
1497            !!!cp (139);
1498          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
1499          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1500          ## reconsume          ## reconsume
# Line 1317  sub _get_next_token ($) { Line 1503  sub _get_next_token ($) {
1503    
1504          redo A;          redo A;
1505        } else {        } else {
1506            !!!cp (140);
1507          $self->{current_token}->{data} # comment          $self->{current_token}->{data} # comment
1508              .= chr ($self->{next_char});              .= chr ($self->{next_char});
1509          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
# Line 1325  sub _get_next_token ($) { Line 1512  sub _get_next_token ($) {
1512        }        }
1513      } elsif ($self->{state} == COMMENT_START_DASH_STATE) {      } elsif ($self->{state} == COMMENT_START_DASH_STATE) {
1514        if ($self->{next_char} == 0x002D) { # -        if ($self->{next_char} == 0x002D) { # -
1515            !!!cp (141);
1516          $self->{state} = COMMENT_END_STATE;          $self->{state} = COMMENT_END_STATE;
1517          !!!next-input-character;          !!!next-input-character;
1518          redo A;          redo A;
1519        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
1520            !!!cp (142);
1521          !!!parse-error (type => 'bogus comment');          !!!parse-error (type => 'bogus comment');
1522          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1523          !!!next-input-character;          !!!next-input-character;
# Line 1337  sub _get_next_token ($) { Line 1526  sub _get_next_token ($) {
1526    
1527          redo A;          redo A;
1528        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
1529            !!!cp (143);
1530          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
1531          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1532          ## reconsume          ## reconsume
# Line 1345  sub _get_next_token ($) { Line 1535  sub _get_next_token ($) {
1535    
1536          redo A;          redo A;
1537        } else {        } else {
1538            !!!cp (144);
1539          $self->{current_token}->{data} # comment          $self->{current_token}->{data} # comment
1540              .= '-' . chr ($self->{next_char});              .= '-' . chr ($self->{next_char});
1541          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
# Line 1353  sub _get_next_token ($) { Line 1544  sub _get_next_token ($) {
1544        }        }
1545      } elsif ($self->{state} == COMMENT_STATE) {      } elsif ($self->{state} == COMMENT_STATE) {
1546        if ($self->{next_char} == 0x002D) { # -        if ($self->{next_char} == 0x002D) { # -
1547            !!!cp (145);
1548          $self->{state} = COMMENT_END_DASH_STATE;          $self->{state} = COMMENT_END_DASH_STATE;
1549          !!!next-input-character;          !!!next-input-character;
1550          redo A;          redo A;
1551        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
1552            !!!cp (146);
1553          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
1554          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1555          ## reconsume          ## reconsume
# Line 1365  sub _get_next_token ($) { Line 1558  sub _get_next_token ($) {
1558    
1559          redo A;          redo A;
1560        } else {        } else {
1561            !!!cp (147);
1562          $self->{current_token}->{data} .= chr ($self->{next_char}); # comment          $self->{current_token}->{data} .= chr ($self->{next_char}); # comment
1563          ## Stay in the state          ## Stay in the state
1564          !!!next-input-character;          !!!next-input-character;
# Line 1372  sub _get_next_token ($) { Line 1566  sub _get_next_token ($) {
1566        }        }
1567      } elsif ($self->{state} == COMMENT_END_DASH_STATE) {      } elsif ($self->{state} == COMMENT_END_DASH_STATE) {
1568        if ($self->{next_char} == 0x002D) { # -        if ($self->{next_char} == 0x002D) { # -
1569            !!!cp (148);
1570          $self->{state} = COMMENT_END_STATE;          $self->{state} = COMMENT_END_STATE;
1571          !!!next-input-character;          !!!next-input-character;
1572          redo A;          redo A;
1573        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
1574            !!!cp (149);
1575          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
1576          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1577          ## reconsume          ## reconsume
# Line 1384  sub _get_next_token ($) { Line 1580  sub _get_next_token ($) {
1580    
1581          redo A;          redo A;
1582        } else {        } else {
1583            !!!cp (150);
1584          $self->{current_token}->{data} .= '-' . chr ($self->{next_char}); # comment          $self->{current_token}->{data} .= '-' . chr ($self->{next_char}); # comment
1585          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
1586          !!!next-input-character;          !!!next-input-character;
# Line 1391  sub _get_next_token ($) { Line 1588  sub _get_next_token ($) {
1588        }        }
1589      } elsif ($self->{state} == COMMENT_END_STATE) {      } elsif ($self->{state} == COMMENT_END_STATE) {
1590        if ($self->{next_char} == 0x003E) { # >        if ($self->{next_char} == 0x003E) { # >
1591            !!!cp (151);
1592          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1593          !!!next-input-character;          !!!next-input-character;
1594    
# Line 1398  sub _get_next_token ($) { Line 1596  sub _get_next_token ($) {
1596    
1597          redo A;          redo A;
1598        } elsif ($self->{next_char} == 0x002D) { # -        } elsif ($self->{next_char} == 0x002D) { # -
1599            !!!cp (152);
1600          !!!parse-error (type => 'dash in comment');          !!!parse-error (type => 'dash in comment');
1601          $self->{current_token}->{data} .= '-'; # comment          $self->{current_token}->{data} .= '-'; # comment
1602          ## Stay in the state          ## Stay in the state
1603          !!!next-input-character;          !!!next-input-character;
1604          redo A;          redo A;
1605        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
1606            !!!cp (153);
1607          !!!parse-error (type => 'unclosed comment');          !!!parse-error (type => 'unclosed comment');
1608          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1609          ## reconsume          ## reconsume
# Line 1412  sub _get_next_token ($) { Line 1612  sub _get_next_token ($) {
1612    
1613          redo A;          redo A;
1614        } else {        } else {
1615            !!!cp (154);
1616          !!!parse-error (type => 'dash in comment');          !!!parse-error (type => 'dash in comment');
1617          $self->{current_token}->{data} .= '--' . chr ($self->{next_char}); # comment          $self->{current_token}->{data} .= '--' . chr ($self->{next_char}); # comment
1618          $self->{state} = COMMENT_STATE;          $self->{state} = COMMENT_STATE;
# Line 1424  sub _get_next_token ($) { Line 1625  sub _get_next_token ($) {
1625            $self->{next_char} == 0x000B or # VT            $self->{next_char} == 0x000B or # VT
1626            $self->{next_char} == 0x000C or # FF            $self->{next_char} == 0x000C or # FF
1627            $self->{next_char} == 0x0020) { # SP            $self->{next_char} == 0x0020) { # SP
1628            !!!cp (155);
1629          $self->{state} = BEFORE_DOCTYPE_NAME_STATE;          $self->{state} = BEFORE_DOCTYPE_NAME_STATE;
1630          !!!next-input-character;          !!!next-input-character;
1631          redo A;          redo A;
1632        } else {        } else {
1633            !!!cp (156);
1634          !!!parse-error (type => 'no space before DOCTYPE name');          !!!parse-error (type => 'no space before DOCTYPE name');
1635          $self->{state} = BEFORE_DOCTYPE_NAME_STATE;          $self->{state} = BEFORE_DOCTYPE_NAME_STATE;
1636          ## reconsume          ## reconsume
# Line 1439  sub _get_next_token ($) { Line 1642  sub _get_next_token ($) {
1642            $self->{next_char} == 0x000B or # VT            $self->{next_char} == 0x000B or # VT
1643            $self->{next_char} == 0x000C or # FF            $self->{next_char} == 0x000C or # FF
1644            $self->{next_char} == 0x0020) { # SP            $self->{next_char} == 0x0020) { # SP
1645            !!!cp (157);
1646          ## Stay in the state          ## Stay in the state
1647          !!!next-input-character;          !!!next-input-character;
1648          redo A;          redo A;
1649        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
1650            !!!cp (158);
1651          !!!parse-error (type => 'no DOCTYPE name');          !!!parse-error (type => 'no DOCTYPE name');
1652          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1653          !!!next-input-character;          !!!next-input-character;
# Line 1450  sub _get_next_token ($) { Line 1655  sub _get_next_token ($) {
1655          !!!emit ({type => DOCTYPE_TOKEN, quirks => 1});          !!!emit ({type => DOCTYPE_TOKEN, quirks => 1});
1656    
1657          redo A;          redo A;
1658        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
1659            !!!cp (159);
1660          !!!parse-error (type => 'no DOCTYPE name');          !!!parse-error (type => 'no DOCTYPE name');
1661          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1662          ## reconsume          ## reconsume
# Line 1459  sub _get_next_token ($) { Line 1665  sub _get_next_token ($) {
1665    
1666          redo A;          redo A;
1667        } else {        } else {
1668            !!!cp (160);
1669          $self->{current_token}          $self->{current_token}
1670              = {type => DOCTYPE_TOKEN,              = {type => DOCTYPE_TOKEN,
1671                 name => chr ($self->{next_char}),                 name => chr ($self->{next_char}),
# Line 1476  sub _get_next_token ($) { Line 1683  sub _get_next_token ($) {
1683            $self->{next_char} == 0x000B or # VT            $self->{next_char} == 0x000B or # VT
1684            $self->{next_char} == 0x000C or # FF            $self->{next_char} == 0x000C or # FF
1685            $self->{next_char} == 0x0020) { # SP            $self->{next_char} == 0x0020) { # SP
1686            !!!cp (161);
1687          $self->{state} = AFTER_DOCTYPE_NAME_STATE;          $self->{state} = AFTER_DOCTYPE_NAME_STATE;
1688          !!!next-input-character;          !!!next-input-character;
1689          redo A;          redo A;
1690        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
1691            !!!cp (162);
1692          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1693          !!!next-input-character;          !!!next-input-character;
1694    
# Line 1487  sub _get_next_token ($) { Line 1696  sub _get_next_token ($) {
1696    
1697          redo A;          redo A;
1698        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
1699            !!!cp (163);
1700          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
1701          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1702          ## reconsume          ## reconsume
# Line 1496  sub _get_next_token ($) { Line 1706  sub _get_next_token ($) {
1706    
1707          redo A;          redo A;
1708        } else {        } else {
1709            !!!cp (164);
1710          $self->{current_token}->{name}          $self->{current_token}->{name}
1711            .= chr ($self->{next_char}); # DOCTYPE            .= chr ($self->{next_char}); # DOCTYPE
1712          ## Stay in the state          ## Stay in the state
# Line 1508  sub _get_next_token ($) { Line 1719  sub _get_next_token ($) {
1719            $self->{next_char} == 0x000B or # VT            $self->{next_char} == 0x000B or # VT
1720            $self->{next_char} == 0x000C or # FF            $self->{next_char} == 0x000C or # FF
1721            $self->{next_char} == 0x0020) { # SP            $self->{next_char} == 0x0020) { # SP
1722            !!!cp (165);
1723          ## Stay in the state          ## Stay in the state
1724          !!!next-input-character;          !!!next-input-character;
1725          redo A;          redo A;
1726        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
1727            !!!cp (166);
1728          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1729          !!!next-input-character;          !!!next-input-character;
1730    
# Line 1519  sub _get_next_token ($) { Line 1732  sub _get_next_token ($) {
1732    
1733          redo A;          redo A;
1734        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
1735            !!!cp (167);
1736          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
1737          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1738          ## reconsume          ## reconsume
# Line 1544  sub _get_next_token ($) { Line 1758  sub _get_next_token ($) {
1758                  !!!next-input-character;                  !!!next-input-character;
1759                  if ($self->{next_char} == 0x0043 or # C                  if ($self->{next_char} == 0x0043 or # C
1760                      $self->{next_char} == 0x0063) { # c                      $self->{next_char} == 0x0063) { # c
1761                      !!!cp (168);
1762                    $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;                    $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
1763                    !!!next-input-character;                    !!!next-input-character;
1764                    redo A;                    redo A;
1765                    } else {
1766                      !!!cp (169);
1767                  }                  }
1768                  } else {
1769                    !!!cp (170);
1770                }                }
1771                } else {
1772                  !!!cp (171);
1773              }              }
1774              } else {
1775                !!!cp (172);
1776            }            }
1777            } else {
1778              !!!cp (173);
1779          }          }
1780    
1781          #          #
# Line 1571  sub _get_next_token ($) { Line 1796  sub _get_next_token ($) {
1796                  !!!next-input-character;                  !!!next-input-character;
1797                  if ($self->{next_char} == 0x004D or # M                  if ($self->{next_char} == 0x004D or # M
1798                      $self->{next_char} == 0x006D) { # m                      $self->{next_char} == 0x006D) { # m
1799                      !!!cp (174);
1800                    $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;                    $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
1801                    !!!next-input-character;                    !!!next-input-character;
1802                    redo A;                    redo A;
1803                    } else {
1804                      !!!cp (175);
1805                  }                  }
1806                  } else {
1807                    !!!cp (176);
1808                }                }
1809                } else {
1810                  !!!cp (177);
1811              }              }
1812              } else {
1813                !!!cp (178);
1814            }            }
1815            } else {
1816              !!!cp (179);
1817          }          }
1818    
1819          #          #
1820        } else {        } else {
1821            !!!cp (180);
1822          !!!next-input-character;          !!!next-input-character;
1823          #          #
1824        }        }
# Line 1597  sub _get_next_token ($) { Line 1834  sub _get_next_token ($) {
1834              0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,              0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
1835              #0x000D => 1, # HT, LF, VT, FF, SP, CR              #0x000D => 1, # HT, LF, VT, FF, SP, CR
1836            }->{$self->{next_char}}) {            }->{$self->{next_char}}) {
1837            !!!cp (181);
1838          ## Stay in the state          ## Stay in the state
1839          !!!next-input-character;          !!!next-input-character;
1840          redo A;          redo A;
1841        } elsif ($self->{next_char} eq 0x0022) { # "        } elsif ($self->{next_char} eq 0x0022) { # "
1842            !!!cp (182);
1843          $self->{current_token}->{public_identifier} = ''; # DOCTYPE          $self->{current_token}->{public_identifier} = ''; # DOCTYPE
1844          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE;
1845          !!!next-input-character;          !!!next-input-character;
1846          redo A;          redo A;
1847        } elsif ($self->{next_char} eq 0x0027) { # '        } elsif ($self->{next_char} eq 0x0027) { # '
1848            !!!cp (183);
1849          $self->{current_token}->{public_identifier} = ''; # DOCTYPE          $self->{current_token}->{public_identifier} = ''; # DOCTYPE
1850          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE;
1851          !!!next-input-character;          !!!next-input-character;
1852          redo A;          redo A;
1853        } elsif ($self->{next_char} eq 0x003E) { # >        } elsif ($self->{next_char} eq 0x003E) { # >
1854            !!!cp (184);
1855          !!!parse-error (type => 'no PUBLIC literal');          !!!parse-error (type => 'no PUBLIC literal');
1856    
1857          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
# Line 1621  sub _get_next_token ($) { Line 1862  sub _get_next_token ($) {
1862    
1863          redo A;          redo A;
1864        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
1865            !!!cp (185);
1866          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
1867    
1868          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
# Line 1631  sub _get_next_token ($) { Line 1873  sub _get_next_token ($) {
1873    
1874          redo A;          redo A;
1875        } else {        } else {
1876            !!!cp (186);
1877          !!!parse-error (type => 'string after PUBLIC');          !!!parse-error (type => 'string after PUBLIC');
1878          $self->{current_token}->{quirks} = 1;          $self->{current_token}->{quirks} = 1;
1879    
# Line 1640  sub _get_next_token ($) { Line 1883  sub _get_next_token ($) {
1883        }        }
1884      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_DOUBLE_QUOTED_STATE) {
1885        if ($self->{next_char} == 0x0022) { # "        if ($self->{next_char} == 0x0022) { # "
1886            !!!cp (187);
1887          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
1888          !!!next-input-character;          !!!next-input-character;
1889          redo A;          redo A;
1890        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
1891            !!!cp (188);
1892          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
1893    
1894          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
# Line 1654  sub _get_next_token ($) { Line 1899  sub _get_next_token ($) {
1899    
1900          redo A;          redo A;
1901        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
1902            !!!cp (189);
1903          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
1904    
1905          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
# Line 1664  sub _get_next_token ($) { Line 1910  sub _get_next_token ($) {
1910    
1911          redo A;          redo A;
1912        } else {        } else {
1913            !!!cp (190);
1914          $self->{current_token}->{public_identifier} # DOCTYPE          $self->{current_token}->{public_identifier} # DOCTYPE
1915              .= chr $self->{next_char};              .= chr $self->{next_char};
1916          ## Stay in the state          ## Stay in the state
# Line 1672  sub _get_next_token ($) { Line 1919  sub _get_next_token ($) {
1919        }        }
1920      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_PUBLIC_IDENTIFIER_SINGLE_QUOTED_STATE) {
1921        if ($self->{next_char} == 0x0027) { # '        if ($self->{next_char} == 0x0027) { # '
1922            !!!cp (191);
1923          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
1924          !!!next-input-character;          !!!next-input-character;
1925          redo A;          redo A;
1926        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
1927            !!!cp (192);
1928          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
1929    
1930          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
# Line 1686  sub _get_next_token ($) { Line 1935  sub _get_next_token ($) {
1935    
1936          redo A;          redo A;
1937        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
1938            !!!cp (193);
1939          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
1940    
1941          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
# Line 1696  sub _get_next_token ($) { Line 1946  sub _get_next_token ($) {
1946    
1947          redo A;          redo A;
1948        } else {        } else {
1949            !!!cp (194);
1950          $self->{current_token}->{public_identifier} # DOCTYPE          $self->{current_token}->{public_identifier} # DOCTYPE
1951              .= chr $self->{next_char};              .= chr $self->{next_char};
1952          ## Stay in the state          ## Stay in the state
# Line 1707  sub _get_next_token ($) { Line 1958  sub _get_next_token ($) {
1958              0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,              0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
1959              #0x000D => 1, # HT, LF, VT, FF, SP, CR              #0x000D => 1, # HT, LF, VT, FF, SP, CR
1960            }->{$self->{next_char}}) {            }->{$self->{next_char}}) {
1961            !!!cp (195);
1962          ## Stay in the state          ## Stay in the state
1963          !!!next-input-character;          !!!next-input-character;
1964          redo A;          redo A;
1965        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{next_char} == 0x0022) { # "
1966            !!!cp (196);
1967          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{current_token}->{system_identifier} = ''; # DOCTYPE
1968          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
1969          !!!next-input-character;          !!!next-input-character;
1970          redo A;          redo A;
1971        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{next_char} == 0x0027) { # '
1972            !!!cp (197);
1973          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{current_token}->{system_identifier} = ''; # DOCTYPE
1974          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
1975          !!!next-input-character;          !!!next-input-character;
1976          redo A;          redo A;
1977        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
1978            !!!cp (198);
1979          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
1980          !!!next-input-character;          !!!next-input-character;
1981    
# Line 1728  sub _get_next_token ($) { Line 1983  sub _get_next_token ($) {
1983    
1984          redo A;          redo A;
1985        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
1986            !!!cp (199);
1987          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
1988    
1989          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
# Line 1738  sub _get_next_token ($) { Line 1994  sub _get_next_token ($) {
1994    
1995          redo A;          redo A;
1996        } else {        } else {
1997            !!!cp (200);
1998          !!!parse-error (type => 'string after PUBLIC literal');          !!!parse-error (type => 'string after PUBLIC literal');
1999          $self->{current_token}->{quirks} = 1;          $self->{current_token}->{quirks} = 1;
2000    
# Line 1750  sub _get_next_token ($) { Line 2007  sub _get_next_token ($) {
2007              0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,              0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
2008              #0x000D => 1, # HT, LF, VT, FF, SP, CR              #0x000D => 1, # HT, LF, VT, FF, SP, CR
2009            }->{$self->{next_char}}) {            }->{$self->{next_char}}) {
2010            !!!cp (201);
2011          ## Stay in the state          ## Stay in the state
2012          !!!next-input-character;          !!!next-input-character;
2013          redo A;          redo A;
2014        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{next_char} == 0x0022) { # "
2015            !!!cp (202);
2016          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{current_token}->{system_identifier} = ''; # DOCTYPE
2017          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE;
2018          !!!next-input-character;          !!!next-input-character;
2019          redo A;          redo A;
2020        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{next_char} == 0x0027) { # '
2021            !!!cp (203);
2022          $self->{current_token}->{system_identifier} = ''; # DOCTYPE          $self->{current_token}->{system_identifier} = ''; # DOCTYPE
2023          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;          $self->{state} = DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE;
2024          !!!next-input-character;          !!!next-input-character;
2025          redo A;          redo A;
2026        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
2027            !!!cp (204);
2028          !!!parse-error (type => 'no SYSTEM literal');          !!!parse-error (type => 'no SYSTEM literal');
2029          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2030          !!!next-input-character;          !!!next-input-character;
# Line 1773  sub _get_next_token ($) { Line 2034  sub _get_next_token ($) {
2034    
2035          redo A;          redo A;
2036        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
2037            !!!cp (205);
2038          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2039    
2040          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
# Line 1783  sub _get_next_token ($) { Line 2045  sub _get_next_token ($) {
2045    
2046          redo A;          redo A;
2047        } else {        } else {
2048            !!!cp (206);
2049          !!!parse-error (type => 'string after SYSTEM');          !!!parse-error (type => 'string after SYSTEM');
2050          $self->{current_token}->{quirks} = 1;          $self->{current_token}->{quirks} = 1;
2051    
# Line 1792  sub _get_next_token ($) { Line 2055  sub _get_next_token ($) {
2055        }        }
2056      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUOTED_STATE) {
2057        if ($self->{next_char} == 0x0022) { # "        if ($self->{next_char} == 0x0022) { # "
2058            !!!cp (207);
2059          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2060          !!!next-input-character;          !!!next-input-character;
2061          redo A;          redo A;
2062        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
2063            !!!cp (208);
2064          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2065    
2066          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
# Line 1806  sub _get_next_token ($) { Line 2071  sub _get_next_token ($) {
2071    
2072          redo A;          redo A;
2073        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
2074            !!!cp (209);
2075          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2076    
2077          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
# Line 1816  sub _get_next_token ($) { Line 2082  sub _get_next_token ($) {
2082    
2083          redo A;          redo A;
2084        } else {        } else {
2085            !!!cp (210);
2086          $self->{current_token}->{system_identifier} # DOCTYPE          $self->{current_token}->{system_identifier} # DOCTYPE
2087              .= chr $self->{next_char};              .= chr $self->{next_char};
2088          ## Stay in the state          ## Stay in the state
# Line 1824  sub _get_next_token ($) { Line 2091  sub _get_next_token ($) {
2091        }        }
2092      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE) {
2093        if ($self->{next_char} == 0x0027) { # '        if ($self->{next_char} == 0x0027) { # '
2094            !!!cp (211);
2095          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
2096          !!!next-input-character;          !!!next-input-character;
2097          redo A;          redo A;
2098        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
2099            !!!cp (212);
2100          !!!parse-error (type => 'unclosed PUBLIC literal');          !!!parse-error (type => 'unclosed PUBLIC literal');
2101    
2102          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
# Line 1838  sub _get_next_token ($) { Line 2107  sub _get_next_token ($) {
2107    
2108          redo A;          redo A;
2109        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
2110            !!!cp (213);
2111          !!!parse-error (type => 'unclosed SYSTEM literal');          !!!parse-error (type => 'unclosed SYSTEM literal');
2112    
2113          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
# Line 1848  sub _get_next_token ($) { Line 2118  sub _get_next_token ($) {
2118    
2119          redo A;          redo A;
2120        } else {        } else {
2121            !!!cp (214);
2122          $self->{current_token}->{system_identifier} # DOCTYPE          $self->{current_token}->{system_identifier} # DOCTYPE
2123              .= chr $self->{next_char};              .= chr $self->{next_char};
2124          ## Stay in the state          ## Stay in the state
# Line 1859  sub _get_next_token ($) { Line 2130  sub _get_next_token ($) {
2130              0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,              0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
2131              #0x000D => 1, # HT, LF, VT, FF, SP, CR              #0x000D => 1, # HT, LF, VT, FF, SP, CR
2132            }->{$self->{next_char}}) {            }->{$self->{next_char}}) {
2133            !!!cp (215);
2134          ## Stay in the state          ## Stay in the state
2135          !!!next-input-character;          !!!next-input-character;
2136          redo A;          redo A;
2137        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
2138            !!!cp (216);
2139          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2140          !!!next-input-character;          !!!next-input-character;
2141    
# Line 1870  sub _get_next_token ($) { Line 2143  sub _get_next_token ($) {
2143    
2144          redo A;          redo A;
2145        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
2146            !!!cp (217);
2147          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2148    
2149          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
# Line 1880  sub _get_next_token ($) { Line 2154  sub _get_next_token ($) {
2154    
2155          redo A;          redo A;
2156        } else {        } else {
2157            !!!cp (218);
2158          !!!parse-error (type => 'string after SYSTEM literal');          !!!parse-error (type => 'string after SYSTEM literal');
2159          #$self->{current_token}->{quirks} = 1;          #$self->{current_token}->{quirks} = 1;
2160    
# Line 1889  sub _get_next_token ($) { Line 2164  sub _get_next_token ($) {
2164        }        }
2165      } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {      } elsif ($self->{state} == BOGUS_DOCTYPE_STATE) {
2166        if ($self->{next_char} == 0x003E) { # >        if ($self->{next_char} == 0x003E) { # >
2167            !!!cp (219);
2168          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2169          !!!next-input-character;          !!!next-input-character;
2170    
# Line 1896  sub _get_next_token ($) { Line 2172  sub _get_next_token ($) {
2172    
2173          redo A;          redo A;
2174        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
2175            !!!cp (220);
2176          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
2177          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
2178          ## reconsume          ## reconsume
# Line 1904  sub _get_next_token ($) { Line 2181  sub _get_next_token ($) {
2181    
2182          redo A;          redo A;
2183        } else {        } else {
2184            !!!cp (221);
2185          ## Stay in the state          ## Stay in the state
2186          !!!next-input-character;          !!!next-input-character;
2187          redo A;          redo A;
# Line 1924  sub _tokenize_attempt_to_consume_an_enti Line 2202  sub _tokenize_attempt_to_consume_an_enti
2202         0x0020 => 1, 0x003C => 1, 0x0026 => 1, -1 => 1, # SP, <, & # 0x000D # CR         0x0020 => 1, 0x003C => 1, 0x0026 => 1, -1 => 1, # SP, <, & # 0x000D # CR
2203         $additional => 1,         $additional => 1,
2204        }->{$self->{next_char}}) {        }->{$self->{next_char}}) {
2205        !!!cp (1001);
2206      ## Don't consume      ## Don't consume
2207      ## No error      ## No error
2208      return undef;      return undef;
# Line 1937  sub _tokenize_attempt_to_consume_an_enti Line 2216  sub _tokenize_attempt_to_consume_an_enti
2216          !!!next-input-character;          !!!next-input-character;
2217          if (0x0030 <= $self->{next_char} and          if (0x0030 <= $self->{next_char} and
2218              $self->{next_char} <= 0x0039) { # 0..9              $self->{next_char} <= 0x0039) { # 0..9
2219              !!!cp (1002);
2220            $code ||= 0;            $code ||= 0;
2221            $code *= 0x10;            $code *= 0x10;
2222            $code += $self->{next_char} - 0x0030;            $code += $self->{next_char} - 0x0030;
2223            redo X;            redo X;
2224          } elsif (0x0061 <= $self->{next_char} and          } elsif (0x0061 <= $self->{next_char} and
2225                   $self->{next_char} <= 0x0066) { # a..f                   $self->{next_char} <= 0x0066) { # a..f
2226              !!!cp (1003);
2227            $code ||= 0;            $code ||= 0;
2228            $code *= 0x10;            $code *= 0x10;
2229            $code += $self->{next_char} - 0x0060 + 9;            $code += $self->{next_char} - 0x0060 + 9;
2230            redo X;            redo X;
2231          } elsif (0x0041 <= $self->{next_char} and          } elsif (0x0041 <= $self->{next_char} and
2232                   $self->{next_char} <= 0x0046) { # A..F                   $self->{next_char} <= 0x0046) { # A..F
2233              !!!cp (1004);
2234            $code ||= 0;            $code ||= 0;
2235            $code *= 0x10;            $code *= 0x10;
2236            $code += $self->{next_char} - 0x0040 + 9;            $code += $self->{next_char} - 0x0040 + 9;
2237            redo X;            redo X;
2238          } elsif (not defined $code) { # no hexadecimal digit          } elsif (not defined $code) { # no hexadecimal digit
2239              !!!cp (1005);
2240            !!!parse-error (type => 'bare hcro');            !!!parse-error (type => 'bare hcro');
2241            !!!back-next-input-character ($x_char, $self->{next_char});            !!!back-next-input-character ($x_char, $self->{next_char});
2242            $self->{next_char} = 0x0023; # #            $self->{next_char} = 0x0023; # #
2243            return undef;            return undef;
2244          } elsif ($self->{next_char} == 0x003B) { # ;          } elsif ($self->{next_char} == 0x003B) { # ;
2245              !!!cp (1006);
2246            !!!next-input-character;            !!!next-input-character;
2247          } else {          } else {
2248              !!!cp (1007);
2249            !!!parse-error (type => 'no refc');            !!!parse-error (type => 'no refc');
2250          }          }
2251    
2252          if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {          if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {
2253              !!!cp (1008);
2254            !!!parse-error (type => sprintf 'invalid character reference:U+%04X', $code);            !!!parse-error (type => sprintf 'invalid character reference:U+%04X', $code);
2255            $code = 0xFFFD;            $code = 0xFFFD;
2256          } elsif ($code > 0x10FFFF) {          } elsif ($code > 0x10FFFF) {
2257              !!!cp (1009);
2258            !!!parse-error (type => sprintf 'invalid character reference:U-%08X', $code);            !!!parse-error (type => sprintf 'invalid character reference:U-%08X', $code);
2259            $code = 0xFFFD;            $code = 0xFFFD;
2260          } elsif ($code == 0x000D) {          } elsif ($code == 0x000D) {
2261              !!!cp (1010);
2262            !!!parse-error (type => 'CR character reference');            !!!parse-error (type => 'CR character reference');
2263            $code = 0x000A;            $code = 0x000A;
2264          } elsif (0x80 <= $code and $code <= 0x9F) {          } elsif (0x80 <= $code and $code <= 0x9F) {
2265              !!!cp (1011);
2266            !!!parse-error (type => sprintf 'C1 character reference:U+%04X', $code);            !!!parse-error (type => sprintf 'C1 character reference:U+%04X', $code);
2267            $code = $c1_entity_char->{$code};            $code = $c1_entity_char->{$code};
2268          }          }
# Line 1988  sub _tokenize_attempt_to_consume_an_enti Line 2277  sub _tokenize_attempt_to_consume_an_enti
2277                
2278        while (0x0030 <= $self->{next_char} and        while (0x0030 <= $self->{next_char} and
2279                  $self->{next_char} <= 0x0039) { # 0..9                  $self->{next_char} <= 0x0039) { # 0..9
2280            !!!cp (1012);
2281          $code *= 10;          $code *= 10;
2282          $code += $self->{next_char} - 0x0030;          $code += $self->{next_char} - 0x0030;
2283                    
# Line 1995  sub _tokenize_attempt_to_consume_an_enti Line 2285  sub _tokenize_attempt_to_consume_an_enti
2285        }        }
2286    
2287        if ($self->{next_char} == 0x003B) { # ;        if ($self->{next_char} == 0x003B) { # ;
2288            !!!cp (1013);
2289          !!!next-input-character;          !!!next-input-character;
2290        } else {        } else {
2291            !!!cp (1014);
2292          !!!parse-error (type => 'no refc');          !!!parse-error (type => 'no refc');
2293        }        }
2294    
2295        if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {        if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {
2296            !!!cp (1015);
2297          !!!parse-error (type => sprintf 'invalid character reference:U+%04X', $code);          !!!parse-error (type => sprintf 'invalid character reference:U+%04X', $code);
2298          $code = 0xFFFD;          $code = 0xFFFD;
2299        } elsif ($code > 0x10FFFF) {        } elsif ($code > 0x10FFFF) {
2300            !!!cp (1016);
2301          !!!parse-error (type => sprintf 'invalid character reference:U-%08X', $code);          !!!parse-error (type => sprintf 'invalid character reference:U-%08X', $code);
2302          $code = 0xFFFD;          $code = 0xFFFD;
2303        } elsif ($code == 0x000D) {        } elsif ($code == 0x000D) {
2304            !!!cp (1017);
2305          !!!parse-error (type => 'CR character reference');          !!!parse-error (type => 'CR character reference');
2306          $code = 0x000A;          $code = 0x000A;
2307        } elsif (0x80 <= $code and $code <= 0x9F) {        } elsif (0x80 <= $code and $code <= 0x9F) {
2308            !!!cp (1018);
2309          !!!parse-error (type => sprintf 'C1 character reference:U+%04X', $code);          !!!parse-error (type => sprintf 'C1 character reference:U+%04X', $code);
2310          $code = $c1_entity_char->{$code};          $code = $c1_entity_char->{$code};
2311        }        }
2312                
2313        return {type => CHARACTER_TOKEN, data => chr $code, has_reference => 1};        return {type => CHARACTER_TOKEN, data => chr $code, has_reference => 1};
2314      } else {      } else {
2315          !!!cp (1019);
2316        !!!parse-error (type => 'bare nero');        !!!parse-error (type => 'bare nero');
2317        !!!back-next-input-character ($self->{next_char});        !!!back-next-input-character ($self->{next_char});
2318        $self->{next_char} = 0x0023; # #        $self->{next_char} = 0x0023; # #
# Line 2045  sub _tokenize_attempt_to_consume_an_enti Line 2342  sub _tokenize_attempt_to_consume_an_enti
2342        $entity_name .= chr $self->{next_char};        $entity_name .= chr $self->{next_char};
2343        if (defined $EntityChar->{$entity_name}) {        if (defined $EntityChar->{$entity_name}) {
2344          if ($self->{next_char} == 0x003B) { # ;          if ($self->{next_char} == 0x003B) { # ;
2345              !!!cp (1020);
2346            $value = $EntityChar->{$entity_name};            $value = $EntityChar->{$entity_name};
2347            $match = 1;            $match = 1;
2348            !!!next-input-character;            !!!next-input-character;
2349            last;            last;
2350          } else {          } else {
2351              !!!cp (1021);
2352            $value = $EntityChar->{$entity_name};            $value = $EntityChar->{$entity_name};
2353            $match = -1;            $match = -1;
2354            !!!next-input-character;            !!!next-input-character;
2355          }          }
2356        } else {        } else {
2357            !!!cp (1022);
2358          $value .= chr $self->{next_char};          $value .= chr $self->{next_char};
2359          $match *= 2;          $match *= 2;
2360          !!!next-input-character;          !!!next-input-character;
# Line 2062  sub _tokenize_attempt_to_consume_an_enti Line 2362  sub _tokenize_attempt_to_consume_an_enti
2362      }      }
2363            
2364      if ($match > 0) {      if ($match > 0) {
2365          !!!cp (1023);
2366        return {type => CHARACTER_TOKEN, data => $value, has_reference => 1};        return {type => CHARACTER_TOKEN, data => $value, has_reference => 1};
2367      } elsif ($match < 0) {      } elsif ($match < 0) {
2368        !!!parse-error (type => 'no refc');        !!!parse-error (type => 'no refc');
2369        if ($in_attr and $match < -1) {        if ($in_attr and $match < -1) {
2370            !!!cp (1024);
2371          return {type => CHARACTER_TOKEN, data => '&'.$entity_name};          return {type => CHARACTER_TOKEN, data => '&'.$entity_name};
2372        } else {        } else {
2373            !!!cp (1025);
2374          return {type => CHARACTER_TOKEN, data => $value, has_reference => 1};          return {type => CHARACTER_TOKEN, data => $value, has_reference => 1};
2375        }        }
2376      } else {      } else {
2377          !!!cp (1026);
2378        !!!parse-error (type => 'bare ero');        !!!parse-error (type => 'bare ero');
2379        ## NOTE: "No characters are consumed" in the spec.        ## NOTE: "No characters are consumed" in the spec.
2380        return {type => CHARACTER_TOKEN, data => '&'.$value};        return {type => CHARACTER_TOKEN, data => '&'.$value};
2381      }      }
2382    } else {    } else {
2383        !!!cp (1027);
2384      ## no characters are consumed      ## no characters are consumed
2385      !!!parse-error (type => 'bare ero');      !!!parse-error (type => 'bare ero');
2386      return undef;      return undef;

Legend:
Removed from v.1.76  
changed lines
  Added in v.1.78

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24