/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.76 by wakaba, Mon Mar 3 09:17:10 2008 UTC revision 1.98 by wakaba, Sun Mar 9 03:23:43 2008 UTC
# Line 304  sub ROW_IMS ()        { 0b10000000 } Line 304  sub ROW_IMS ()        { 0b10000000 }
304  sub BODY_AFTER_IMS () { 0b100000000 }  sub BODY_AFTER_IMS () { 0b100000000 }
305  sub FRAME_IMS ()      { 0b1000000000 }  sub FRAME_IMS ()      { 0b1000000000 }
306    
307    ## NOTE: "initial" and "before html" insertion modes have no constants.
308    
309    ## NOTE: "after after body" insertion mode.
310  sub AFTER_HTML_BODY_IM () { AFTER_HTML_IMS | BODY_AFTER_IMS }  sub AFTER_HTML_BODY_IM () { AFTER_HTML_IMS | BODY_AFTER_IMS }
311    
312    ## NOTE: "after after frameset" insertion mode.
313  sub AFTER_HTML_FRAMESET_IM () { AFTER_HTML_IMS | FRAME_IMS }  sub AFTER_HTML_FRAMESET_IM () { AFTER_HTML_IMS | FRAME_IMS }
314    
315  sub IN_HEAD_IM () { HEAD_IMS | 0b00 }  sub IN_HEAD_IM () { HEAD_IMS | 0b00 }
316  sub IN_HEAD_NOSCRIPT_IM () { HEAD_IMS | 0b01 }  sub IN_HEAD_NOSCRIPT_IM () { HEAD_IMS | 0b01 }
317  sub AFTER_HEAD_IM () { HEAD_IMS | 0b10 }  sub AFTER_HEAD_IM () { HEAD_IMS | 0b10 }
# Line 387  sub _get_next_token ($) { Line 393  sub _get_next_token ($) {
393        if ($self->{next_char} == 0x0026) { # &        if ($self->{next_char} == 0x0026) { # &
394          if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA          if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA
395              not $self->{escape}) {              not $self->{escape}) {
396              !!!cp (1);
397            $self->{state} = ENTITY_DATA_STATE;            $self->{state} = ENTITY_DATA_STATE;
398            !!!next-input-character;            !!!next-input-character;
399            redo A;            redo A;
400          } else {          } else {
401              !!!cp (2);
402            #            #
403          }          }
404        } elsif ($self->{next_char} == 0x002D) { # -        } elsif ($self->{next_char} == 0x002D) { # -
# Line 399  sub _get_next_token ($) { Line 407  sub _get_next_token ($) {
407              if ($self->{prev_char}->[0] == 0x002D and # -              if ($self->{prev_char}->[0] == 0x002D and # -
408                  $self->{prev_char}->[1] == 0x0021 and # !                  $self->{prev_char}->[1] == 0x0021 and # !
409                  $self->{prev_char}->[2] == 0x003C) { # <                  $self->{prev_char}->[2] == 0x003C) { # <
410                  !!!cp (3);
411                $self->{escape} = 1;                $self->{escape} = 1;
412                } else {
413                  !!!cp (4);
414              }              }
415              } else {
416                !!!cp (5);
417            }            }
418          }          }
419                    
# Line 409  sub _get_next_token ($) { Line 422  sub _get_next_token ($) {
422          if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA          if ($self->{content_model} & CM_FULL_MARKUP or # PCDATA
423              (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA              (($self->{content_model} & CM_LIMITED_MARKUP) and # CDATA | RCDATA
424               not $self->{escape})) {               not $self->{escape})) {
425              !!!cp (6);
426            $self->{state} = TAG_OPEN_STATE;            $self->{state} = TAG_OPEN_STATE;
427            !!!next-input-character;            !!!next-input-character;
428            redo A;            redo A;
429          } else {          } else {
430              !!!cp (7);
431            #            #
432          }          }
433        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
# Line 420  sub _get_next_token ($) { Line 435  sub _get_next_token ($) {
435              ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA              ($self->{content_model} & CM_LIMITED_MARKUP)) { # RCDATA | CDATA
436            if ($self->{prev_char}->[0] == 0x002D and # -            if ($self->{prev_char}->[0] == 0x002D and # -
437                $self->{prev_char}->[1] == 0x002D) { # -                $self->{prev_char}->[1] == 0x002D) { # -
438                !!!cp (8);
439              delete $self->{escape};              delete $self->{escape};
440              } else {
441                !!!cp (9);
442            }            }
443            } else {
444              !!!cp (10);
445          }          }
446                    
447          #          #
448        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
449            !!!cp (11);
450          !!!emit ({type => END_OF_FILE_TOKEN});          !!!emit ({type => END_OF_FILE_TOKEN});
451          last A; ## TODO: ok?          last A; ## TODO: ok?
452          } else {
453            !!!cp (12);
454        }        }
455        # Anything else        # Anything else
456        my $token = {type => CHARACTER_TOKEN,        my $token = {type => CHARACTER_TOKEN,
# Line 447  sub _get_next_token ($) { Line 470  sub _get_next_token ($) {
470        # next-input-character is already done        # next-input-character is already done
471    
472        unless (defined $token) {        unless (defined $token) {
473            !!!cp (13);
474          !!!emit ({type => CHARACTER_TOKEN, data => '&'});          !!!emit ({type => CHARACTER_TOKEN, data => '&'});
475        } else {        } else {
476            !!!cp (14);
477          !!!emit ($token);          !!!emit ($token);
478        }        }
479    
# Line 456  sub _get_next_token ($) { Line 481  sub _get_next_token ($) {
481      } elsif ($self->{state} == TAG_OPEN_STATE) {      } elsif ($self->{state} == TAG_OPEN_STATE) {
482        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
483          if ($self->{next_char} == 0x002F) { # /          if ($self->{next_char} == 0x002F) { # /
484              !!!cp (15);
485            !!!next-input-character;            !!!next-input-character;
486            $self->{state} = CLOSE_TAG_OPEN_STATE;            $self->{state} = CLOSE_TAG_OPEN_STATE;
487            redo A;            redo A;
488          } else {          } else {
489              !!!cp (16);
490            ## reconsume            ## reconsume
491            $self->{state} = DATA_STATE;            $self->{state} = DATA_STATE;
492    
# Line 469  sub _get_next_token ($) { Line 496  sub _get_next_token ($) {
496          }          }
497        } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA        } elsif ($self->{content_model} & CM_FULL_MARKUP) { # PCDATA
498          if ($self->{next_char} == 0x0021) { # !          if ($self->{next_char} == 0x0021) { # !
499              !!!cp (17);
500            $self->{state} = MARKUP_DECLARATION_OPEN_STATE;            $self->{state} = MARKUP_DECLARATION_OPEN_STATE;
501            !!!next-input-character;            !!!next-input-character;
502            redo A;            redo A;
503          } elsif ($self->{next_char} == 0x002F) { # /          } elsif ($self->{next_char} == 0x002F) { # /
504              !!!cp (18);
505            $self->{state} = CLOSE_TAG_OPEN_STATE;            $self->{state} = CLOSE_TAG_OPEN_STATE;
506            !!!next-input-character;            !!!next-input-character;
507            redo A;            redo A;
508          } elsif (0x0041 <= $self->{next_char} and          } elsif (0x0041 <= $self->{next_char} and
509                   $self->{next_char} <= 0x005A) { # A..Z                   $self->{next_char} <= 0x005A) { # A..Z
510              !!!cp (19);
511            $self->{current_token}            $self->{current_token}
512              = {type => START_TAG_TOKEN,              = {type => START_TAG_TOKEN,
513                 tag_name => chr ($self->{next_char} + 0x0020)};                 tag_name => chr ($self->{next_char} + 0x0020)};
# Line 486  sub _get_next_token ($) { Line 516  sub _get_next_token ($) {
516            redo A;            redo A;
517          } elsif (0x0061 <= $self->{next_char} and          } elsif (0x0061 <= $self->{next_char} and
518                   $self->{next_char} <= 0x007A) { # a..z                   $self->{next_char} <= 0x007A) { # a..z
519              !!!cp (20);
520            $self->{current_token} = {type => START_TAG_TOKEN,            $self->{current_token} = {type => START_TAG_TOKEN,
521                              tag_name => chr ($self->{next_char})};                              tag_name => chr ($self->{next_char})};
522            $self->{state} = TAG_NAME_STATE;            $self->{state} = TAG_NAME_STATE;
523            !!!next-input-character;            !!!next-input-character;
524            redo A;            redo A;
525          } elsif ($self->{next_char} == 0x003E) { # >          } elsif ($self->{next_char} == 0x003E) { # >
526              !!!cp (21);
527            !!!parse-error (type => 'empty start tag');            !!!parse-error (type => 'empty start tag');
528            $self->{state} = DATA_STATE;            $self->{state} = DATA_STATE;
529            !!!next-input-character;            !!!next-input-character;
# Line 500  sub _get_next_token ($) { Line 532  sub _get_next_token ($) {
532    
533            redo A;            redo A;
534          } elsif ($self->{next_char} == 0x003F) { # ?          } elsif ($self->{next_char} == 0x003F) { # ?
535              !!!cp (22);
536            !!!parse-error (type => 'pio');            !!!parse-error (type => 'pio');
537            $self->{state} = BOGUS_COMMENT_STATE;            $self->{state} = BOGUS_COMMENT_STATE;
538            ## $self->{next_char} is intentionally left as is            ## $self->{next_char} is intentionally left as is
539            redo A;            redo A;
540          } else {          } else {
541              !!!cp (23);
542            !!!parse-error (type => 'bare stago');            !!!parse-error (type => 'bare stago');
543            $self->{state} = DATA_STATE;            $self->{state} = DATA_STATE;
544            ## reconsume            ## reconsume
# Line 526  sub _get_next_token ($) { Line 560  sub _get_next_token ($) {
560              my $c = ord substr ($self->{last_emitted_start_tag_name}, $i, 1);              my $c = ord substr ($self->{last_emitted_start_tag_name}, $i, 1);
561              my $C = 0x0061 <= $c && $c <= 0x007A ? $c - 0x0020 : $c;              my $C = 0x0061 <= $c && $c <= 0x007A ? $c - 0x0020 : $c;
562              if ($self->{next_char} == $c or $self->{next_char} == $C) {              if ($self->{next_char} == $c or $self->{next_char} == $C) {
563                  !!!cp (24);
564                !!!next-input-character;                !!!next-input-character;
565                next TAGNAME;                next TAGNAME;
566              } else {              } else {
567                  !!!cp (25);
568                $self->{next_char} = shift @next_char; # reconsume                $self->{next_char} = shift @next_char; # reconsume
569                !!!back-next-input-character (@next_char);                !!!back-next-input-character (@next_char);
570                $self->{state} = DATA_STATE;                $self->{state} = DATA_STATE;
# Line 548  sub _get_next_token ($) { Line 584  sub _get_next_token ($) {
584                    $self->{next_char} == 0x003E or # >                    $self->{next_char} == 0x003E or # >
585                    $self->{next_char} == 0x002F or # /                    $self->{next_char} == 0x002F or # /
586                    $self->{next_char} == -1) {                    $self->{next_char} == -1) {
587                !!!cp (26);
588              $self->{next_char} = shift @next_char; # reconsume              $self->{next_char} = shift @next_char; # reconsume
589              !!!back-next-input-character (@next_char);              !!!back-next-input-character (@next_char);
590              $self->{state} = DATA_STATE;              $self->{state} = DATA_STATE;
591              !!!emit ({type => CHARACTER_TOKEN, data => '</'});              !!!emit ({type => CHARACTER_TOKEN, data => '</'});
592              redo A;              redo A;
593            } else {            } else {
594                !!!cp (27);
595              $self->{next_char} = shift @next_char;              $self->{next_char} = shift @next_char;
596              !!!back-next-input-character (@next_char);              !!!back-next-input-character (@next_char);
597              # and consume...              # and consume...
598            }            }
599          } else {          } else {
600            ## No start tag token has ever been emitted            ## No start tag token has ever been emitted
601              !!!cp (28);
602            # next-input-character is already done            # next-input-character is already done
603            $self->{state} = DATA_STATE;            $self->{state} = DATA_STATE;
604            !!!emit ({type => CHARACTER_TOKEN, data => '</'});            !!!emit ({type => CHARACTER_TOKEN, data => '</'});
# Line 569  sub _get_next_token ($) { Line 608  sub _get_next_token ($) {
608                
609        if (0x0041 <= $self->{next_char} and        if (0x0041 <= $self->{next_char} and
610            $self->{next_char} <= 0x005A) { # A..Z            $self->{next_char} <= 0x005A) { # A..Z
611            !!!cp (29);
612          $self->{current_token} = {type => END_TAG_TOKEN,          $self->{current_token} = {type => END_TAG_TOKEN,
613                            tag_name => chr ($self->{next_char} + 0x0020)};                            tag_name => chr ($self->{next_char} + 0x0020)};
614          $self->{state} = TAG_NAME_STATE;          $self->{state} = TAG_NAME_STATE;
# Line 576  sub _get_next_token ($) { Line 616  sub _get_next_token ($) {
616          redo A;          redo A;
617        } elsif (0x0061 <= $self->{next_char} and        } elsif (0x0061 <= $self->{next_char} and
618                 $self->{next_char} <= 0x007A) { # a..z                 $self->{next_char} <= 0x007A) { # a..z
619            !!!cp (30);
620          $self->{current_token} = {type => END_TAG_TOKEN,          $self->{current_token} = {type => END_TAG_TOKEN,
621                            tag_name => chr ($self->{next_char})};                            tag_name => chr ($self->{next_char})};
622          $self->{state} = TAG_NAME_STATE;          $self->{state} = TAG_NAME_STATE;
623          !!!next-input-character;          !!!next-input-character;
624          redo A;          redo A;
625        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
626            !!!cp (31);
627          !!!parse-error (type => 'empty end tag');          !!!parse-error (type => 'empty end tag');
628          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
629          !!!next-input-character;          !!!next-input-character;
630          redo A;          redo A;
631        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
632            !!!cp (32);
633          !!!parse-error (type => 'bare etago');          !!!parse-error (type => 'bare etago');
634          $self->{state} = DATA_STATE;          $self->{state} = DATA_STATE;
635          # reconsume          # reconsume
# Line 595  sub _get_next_token ($) { Line 638  sub _get_next_token ($) {
638    
639          redo A;          redo A;
640        } else {        } else {
641            !!!cp (33);
642          !!!parse-error (type => 'bogus end tag');          !!!parse-error (type => 'bogus end tag');
643          $self->{state} = BOGUS_COMMENT_STATE;          $self->{state} = BOGUS_COMMENT_STATE;
644          ## $self->{next_char} is intentionally left as is          ## $self->{next_char} is intentionally left as is
# Line 606  sub _get_next_token ($) { Line 650  sub _get_next_token ($) {
650            $self->{next_char} == 0x000B or # VT            $self->{next_char} == 0x000B or # VT
651            $self->{next_char} == 0x000C or # FF            $self->{next_char} == 0x000C or # FF
652            $self->{next_char} == 0x0020) { # SP            $self->{next_char} == 0x0020) { # SP
653            !!!cp (34);
654          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
655          !!!next-input-character;          !!!next-input-character;
656          redo A;          redo A;
657        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
658          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
659              !!!cp (35);
660            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
661                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
662            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
663          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
664            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
665            if ($self->{current_token}->{attributes}) {            #if ($self->{current_token}->{attributes}) {
666              !!!parse-error (type => 'end tag attribute');            #  ## NOTE: This should never be reached.
667            }            #  !!! cp (36);
668              #  !!! parse-error (type => 'end tag attribute');
669              #} else {
670                !!!cp (37);
671              #}
672          } else {          } else {
673            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
674          }          }
# Line 630  sub _get_next_token ($) { Line 680  sub _get_next_token ($) {
680          redo A;          redo A;
681        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{next_char} and
682                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{next_char} <= 0x005A) { # A..Z
683            !!!cp (38);
684          $self->{current_token}->{tag_name} .= chr ($self->{next_char} + 0x0020);          $self->{current_token}->{tag_name} .= chr ($self->{next_char} + 0x0020);
685            # start tag or end tag            # start tag or end tag
686          ## Stay in this state          ## Stay in this state
# Line 638  sub _get_next_token ($) { Line 689  sub _get_next_token ($) {
689        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
690          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
691          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
692              !!!cp (39);
693            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
694                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
695            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
696          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
697            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
698            if ($self->{current_token}->{attributes}) {            #if ($self->{current_token}->{attributes}) {
699              !!!parse-error (type => 'end tag attribute');            #  ## NOTE: This state should never be reached.
700            }            #  !!! cp (40);
701              #  !!! parse-error (type => 'end tag attribute');
702              #} else {
703                !!!cp (41);
704              #}
705          } else {          } else {
706            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
707          }          }
# Line 661  sub _get_next_token ($) { Line 717  sub _get_next_token ($) {
717              $self->{current_token}->{type} == START_TAG_TOKEN and              $self->{current_token}->{type} == START_TAG_TOKEN and
718              $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {              $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {
719            # permitted slash            # permitted slash
720              !!!cp (42);
721            #            #
722          } else {          } else {
723              !!!cp (43);
724            !!!parse-error (type => 'nestc');            !!!parse-error (type => 'nestc');
725          }          }
726          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
727          # next-input-character is already done          # next-input-character is already done
728          redo A;          redo A;
729        } else {        } else {
730            !!!cp (44);
731          $self->{current_token}->{tag_name} .= chr $self->{next_char};          $self->{current_token}->{tag_name} .= chr $self->{next_char};
732            # start tag or end tag            # start tag or end tag
733          ## Stay in the state          ## Stay in the state
# Line 681  sub _get_next_token ($) { Line 740  sub _get_next_token ($) {
740            $self->{next_char} == 0x000B or # VT            $self->{next_char} == 0x000B or # VT
741            $self->{next_char} == 0x000C or # FF            $self->{next_char} == 0x000C or # FF
742            $self->{next_char} == 0x0020) { # SP            $self->{next_char} == 0x0020) { # SP
743            !!!cp (45);
744          ## Stay in the state          ## Stay in the state
745          !!!next-input-character;          !!!next-input-character;
746          redo A;          redo A;
747        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
748          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
749              !!!cp (46);
750            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
751                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
752            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
753          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
754            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
755            if ($self->{current_token}->{attributes}) {            if ($self->{current_token}->{attributes}) {
756                !!!cp (47);
757              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
758              } else {
759                !!!cp (48);
760            }            }
761          } else {          } else {
762            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
# Line 705  sub _get_next_token ($) { Line 769  sub _get_next_token ($) {
769          redo A;          redo A;
770        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{next_char} and
771                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{next_char} <= 0x005A) { # A..Z
772            !!!cp (49);
773          $self->{current_attribute} = {name => chr ($self->{next_char} + 0x0020),          $self->{current_attribute} = {name => chr ($self->{next_char} + 0x0020),
774                                value => ''};                                value => ''};
775          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
# Line 716  sub _get_next_token ($) { Line 781  sub _get_next_token ($) {
781              $self->{current_token}->{type} == START_TAG_TOKEN and              $self->{current_token}->{type} == START_TAG_TOKEN and
782              $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {              $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {
783            # permitted slash            # permitted slash
784              !!!cp (50);
785            #            #
786          } else {          } else {
787              !!!cp (51);
788            !!!parse-error (type => 'nestc');            !!!parse-error (type => 'nestc');
789          }          }
790          ## Stay in the state          ## Stay in the state
# Line 726  sub _get_next_token ($) { Line 793  sub _get_next_token ($) {
793        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
794          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
795          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
796              !!!cp (52);
797            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
798                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
799            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
800          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
801            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
802            if ($self->{current_token}->{attributes}) {            if ($self->{current_token}->{attributes}) {
803                !!!cp (53);
804              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
805              } else {
806                !!!cp (54);
807            }            }
808          } else {          } else {
809            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
# Line 749  sub _get_next_token ($) { Line 820  sub _get_next_token ($) {
820               0x0027 => 1, # '               0x0027 => 1, # '
821               0x003D => 1, # =               0x003D => 1, # =
822              }->{$self->{next_char}}) {              }->{$self->{next_char}}) {
823              !!!cp (55);
824            !!!parse-error (type => 'bad attribute name');            !!!parse-error (type => 'bad attribute name');
825            } else {
826              !!!cp (56);
827          }          }
828          $self->{current_attribute} = {name => chr ($self->{next_char}),          $self->{current_attribute} = {name => chr ($self->{next_char}),
829                                value => ''};                                value => ''};
# Line 761  sub _get_next_token ($) { Line 835  sub _get_next_token ($) {
835        my $before_leave = sub {        my $before_leave = sub {
836          if (exists $self->{current_token}->{attributes} # start tag or end tag          if (exists $self->{current_token}->{attributes} # start tag or end tag
837              ->{$self->{current_attribute}->{name}}) { # MUST              ->{$self->{current_attribute}->{name}}) { # MUST
838              !!!cp (57);
839            !!!parse-error (type => 'duplicate attribute:'.$self->{current_attribute}->{name});            !!!parse-error (type => 'duplicate attribute:'.$self->{current_attribute}->{name});
840            ## Discard $self->{current_attribute} # MUST            ## Discard $self->{current_attribute} # MUST
841          } else {          } else {
842              !!!cp (58);
843            $self->{current_token}->{attributes}->{$self->{current_attribute}->{name}}            $self->{current_token}->{attributes}->{$self->{current_attribute}->{name}}
844              = $self->{current_attribute};              = $self->{current_attribute};
845          }          }
# Line 774  sub _get_next_token ($) { Line 850  sub _get_next_token ($) {
850            $self->{next_char} == 0x000B or # VT            $self->{next_char} == 0x000B or # VT
851            $self->{next_char} == 0x000C or # FF            $self->{next_char} == 0x000C or # FF
852            $self->{next_char} == 0x0020) { # SP            $self->{next_char} == 0x0020) { # SP
853            !!!cp (59);
854          $before_leave->();          $before_leave->();
855          $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;          $self->{state} = AFTER_ATTRIBUTE_NAME_STATE;
856          !!!next-input-character;          !!!next-input-character;
857          redo A;          redo A;
858        } elsif ($self->{next_char} == 0x003D) { # =        } elsif ($self->{next_char} == 0x003D) { # =
859            !!!cp (60);
860          $before_leave->();          $before_leave->();
861          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
862          !!!next-input-character;          !!!next-input-character;
# Line 786  sub _get_next_token ($) { Line 864  sub _get_next_token ($) {
864        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
865          $before_leave->();          $before_leave->();
866          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
867              !!!cp (61);
868            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
869                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
870            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
871          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
872              !!!cp (62);
873            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
874            if ($self->{current_token}->{attributes}) {            if ($self->{current_token}->{attributes}) {
875              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
# Line 805  sub _get_next_token ($) { Line 885  sub _get_next_token ($) {
885          redo A;          redo A;
886        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{next_char} and
887                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{next_char} <= 0x005A) { # A..Z
888            !!!cp (63);
889          $self->{current_attribute}->{name} .= chr ($self->{next_char} + 0x0020);          $self->{current_attribute}->{name} .= chr ($self->{next_char} + 0x0020);
890          ## Stay in the state          ## Stay in the state
891          !!!next-input-character;          !!!next-input-character;
# Line 816  sub _get_next_token ($) { Line 897  sub _get_next_token ($) {
897              $self->{current_token}->{type} == START_TAG_TOKEN and              $self->{current_token}->{type} == START_TAG_TOKEN and
898              $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {              $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {
899            # permitted slash            # permitted slash
900              !!!cp (64);
901            #            #
902          } else {          } else {
903              !!!cp (65);
904            !!!parse-error (type => 'nestc');            !!!parse-error (type => 'nestc');
905          }          }
906          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
# Line 827  sub _get_next_token ($) { Line 910  sub _get_next_token ($) {
910          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
911          $before_leave->();          $before_leave->();
912          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
913              !!!cp (66);
914            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
915                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
916            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
917          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
918            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
919            if ($self->{current_token}->{attributes}) {            if ($self->{current_token}->{attributes}) {
920                !!!cp (67);
921              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
922              } else {
923                ## NOTE: This state should never be reached.
924                !!!cp (68);
925            }            }
926          } else {          } else {
927            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
# Line 847  sub _get_next_token ($) { Line 935  sub _get_next_token ($) {
935        } else {        } else {
936          if ($self->{next_char} == 0x0022 or # "          if ($self->{next_char} == 0x0022 or # "
937              $self->{next_char} == 0x0027) { # '              $self->{next_char} == 0x0027) { # '
938              !!!cp (69);
939            !!!parse-error (type => 'bad attribute name');            !!!parse-error (type => 'bad attribute name');
940            } else {
941              !!!cp (70);
942          }          }
943          $self->{current_attribute}->{name} .= chr ($self->{next_char});          $self->{current_attribute}->{name} .= chr ($self->{next_char});
944          ## Stay in the state          ## Stay in the state
# Line 860  sub _get_next_token ($) { Line 951  sub _get_next_token ($) {
951            $self->{next_char} == 0x000B or # VT            $self->{next_char} == 0x000B or # VT
952            $self->{next_char} == 0x000C or # FF            $self->{next_char} == 0x000C or # FF
953            $self->{next_char} == 0x0020) { # SP            $self->{next_char} == 0x0020) { # SP
954            !!!cp (71);
955          ## Stay in the state          ## Stay in the state
956          !!!next-input-character;          !!!next-input-character;
957          redo A;          redo A;
958        } elsif ($self->{next_char} == 0x003D) { # =        } elsif ($self->{next_char} == 0x003D) { # =
959            !!!cp (72);
960          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;          $self->{state} = BEFORE_ATTRIBUTE_VALUE_STATE;
961          !!!next-input-character;          !!!next-input-character;
962          redo A;          redo A;
963        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
964          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
965              !!!cp (73);
966            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
967                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
968            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
969          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
970            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
971            if ($self->{current_token}->{attributes}) {            if ($self->{current_token}->{attributes}) {
972                !!!cp (74);
973              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
974              } else {
975                ## NOTE: This state should never be reached.
976                !!!cp (75);
977            }            }
978          } else {          } else {
979            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
# Line 888  sub _get_next_token ($) { Line 986  sub _get_next_token ($) {
986          redo A;          redo A;
987        } elsif (0x0041 <= $self->{next_char} and        } elsif (0x0041 <= $self->{next_char} and
988                 $self->{next_char} <= 0x005A) { # A..Z                 $self->{next_char} <= 0x005A) { # A..Z
989            !!!cp (76);
990          $self->{current_attribute} = {name => chr ($self->{next_char} + 0x0020),          $self->{current_attribute} = {name => chr ($self->{next_char} + 0x0020),
991                                value => ''};                                value => ''};
992          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
# Line 899  sub _get_next_token ($) { Line 998  sub _get_next_token ($) {
998              $self->{current_token}->{type} == START_TAG_TOKEN and              $self->{current_token}->{type} == START_TAG_TOKEN and
999              $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {              $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {
1000            # permitted slash            # permitted slash
1001              !!!cp (77);
1002            #            #
1003          } else {          } else {
1004              !!!cp (78);
1005            !!!parse-error (type => 'nestc');            !!!parse-error (type => 'nestc');
1006            ## TODO: Different error type for <aa / bb> than <aa/>            ## TODO: Different error type for <aa / bb> than <aa/>
1007          }          }
# Line 910  sub _get_next_token ($) { Line 1011  sub _get_next_token ($) {
1011        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
1012          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1013          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1014              !!!cp (79);
1015            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
1016                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
1017            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1018          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1019            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1020            if ($self->{current_token}->{attributes}) {            if ($self->{current_token}->{attributes}) {
1021                !!!cp (80);
1022              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1023              } else {
1024                ## NOTE: This state should never be reached.
1025                !!!cp (81);
1026            }            }
1027          } else {          } else {
1028            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
# Line 928  sub _get_next_token ($) { Line 1034  sub _get_next_token ($) {
1034    
1035          redo A;          redo A;
1036        } else {        } else {
1037            !!!cp (82);
1038          $self->{current_attribute} = {name => chr ($self->{next_char}),          $self->{current_attribute} = {name => chr ($self->{next_char}),
1039                                value => ''};                                value => ''};
1040          $self->{state} = ATTRIBUTE_NAME_STATE;          $self->{state} = ATTRIBUTE_NAME_STATE;
# Line 940  sub _get_next_token ($) { Line 1047  sub _get_next_token ($) {
1047            $self->{next_char} == 0x000B or # VT            $self->{next_char} == 0x000B or # VT
1048            $self->{next_char} == 0x000C or # FF            $self->{next_char} == 0x000C or # FF
1049            $self->{next_char} == 0x0020) { # SP                  $self->{next_char} == 0x0020) { # SP      
1050            !!!cp (83);
1051          ## Stay in the state          ## Stay in the state
1052          !!!next-input-character;          !!!next-input-character;
1053          redo A;          redo A;
1054        } elsif ($self->{next_char} == 0x0022) { # "        } elsif ($self->{next_char} == 0x0022) { # "
1055            !!!cp (84);
1056          $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE;
1057          !!!next-input-character;          !!!next-input-character;
1058          redo A;          redo A;
1059        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{next_char} == 0x0026) { # &
1060            !!!cp (85);
1061          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
1062          ## reconsume          ## reconsume
1063          redo A;          redo A;
1064        } elsif ($self->{next_char} == 0x0027) { # '        } elsif ($self->{next_char} == 0x0027) { # '
1065            !!!cp (86);
1066          $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE;
1067          !!!next-input-character;          !!!next-input-character;
1068          redo A;          redo A;
1069        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
1070          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1071              !!!cp (87);
1072            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
1073                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
1074            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1075          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1076            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1077            if ($self->{current_token}->{attributes}) {            if ($self->{current_token}->{attributes}) {
1078                !!!cp (88);
1079              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1080              } else {
1081                ## NOTE: This state should never be reached.
1082                !!!cp (89);
1083            }            }
1084          } else {          } else {
1085            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
# Line 977  sub _get_next_token ($) { Line 1093  sub _get_next_token ($) {
1093        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
1094          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1095          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1096              !!!cp (90);
1097            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
1098                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
1099            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1100          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1101            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1102            if ($self->{current_token}->{attributes}) {            if ($self->{current_token}->{attributes}) {
1103                !!!cp (91);
1104              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1105              } else {
1106                ## NOTE: This state should never be reached.
1107                !!!cp (92);
1108            }            }
1109          } else {          } else {
1110            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
# Line 996  sub _get_next_token ($) { Line 1117  sub _get_next_token ($) {
1117          redo A;          redo A;
1118        } else {        } else {
1119          if ($self->{next_char} == 0x003D) { # =          if ($self->{next_char} == 0x003D) { # =
1120              !!!cp (93);
1121            !!!parse-error (type => 'bad attribute value');            !!!parse-error (type => 'bad attribute value');
1122            } else {
1123              !!!cp (94);
1124          }          }
1125          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{current_attribute}->{value} .= chr ($self->{next_char});
1126          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
# Line 1005  sub _get_next_token ($) { Line 1129  sub _get_next_token ($) {
1129        }        }
1130      } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {
1131        if ($self->{next_char} == 0x0022) { # "        if ($self->{next_char} == 0x0022) { # "
1132            !!!cp (95);
1133          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1134          !!!next-input-character;          !!!next-input-character;
1135          redo A;          redo A;
1136        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{next_char} == 0x0026) { # &
1137            !!!cp (96);
1138          $self->{last_attribute_value_state} = $self->{state};          $self->{last_attribute_value_state} = $self->{state};
1139          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;
1140          !!!next-input-character;          !!!next-input-character;
# Line 1016  sub _get_next_token ($) { Line 1142  sub _get_next_token ($) {
1142        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
1143          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
1144          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1145              !!!cp (97);
1146            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
1147                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
1148            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1149          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1150            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1151            if ($self->{current_token}->{attributes}) {            if ($self->{current_token}->{attributes}) {
1152                !!!cp (98);
1153              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1154              } else {
1155                ## NOTE: This state should never be reached.
1156                !!!cp (99);
1157            }            }
1158          } else {          } else {
1159            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
# Line 1034  sub _get_next_token ($) { Line 1165  sub _get_next_token ($) {
1165    
1166          redo A;          redo A;
1167        } else {        } else {
1168            !!!cp (100);
1169          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{current_attribute}->{value} .= chr ($self->{next_char});
1170          ## Stay in the state          ## Stay in the state
1171          !!!next-input-character;          !!!next-input-character;
# Line 1041  sub _get_next_token ($) { Line 1173  sub _get_next_token ($) {
1173        }        }
1174      } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {      } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {
1175        if ($self->{next_char} == 0x0027) { # '        if ($self->{next_char} == 0x0027) { # '
1176            !!!cp (101);
1177          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;          $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
1178          !!!next-input-character;          !!!next-input-character;
1179          redo A;          redo A;
1180        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{next_char} == 0x0026) { # &
1181            !!!cp (102);
1182          $self->{last_attribute_value_state} = $self->{state};          $self->{last_attribute_value_state} = $self->{state};
1183          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;
1184          !!!next-input-character;          !!!next-input-character;
# Line 1052  sub _get_next_token ($) { Line 1186  sub _get_next_token ($) {
1186        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
1187          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
1188          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1189              !!!cp (103);
1190            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
1191                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
1192            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1193          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1194            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1195            if ($self->{current_token}->{attributes}) {            if ($self->{current_token}->{attributes}) {
1196                !!!cp (104);
1197              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1198              } else {
1199                ## NOTE: This state should never be reached.
1200                !!!cp (105);
1201            }            }
1202          } else {          } else {
1203            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
# Line 1070  sub _get_next_token ($) { Line 1209  sub _get_next_token ($) {
1209    
1210          redo A;          redo A;
1211        } else {        } else {
1212            !!!cp (106);
1213          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{current_attribute}->{value} .= chr ($self->{next_char});
1214          ## Stay in the state          ## Stay in the state
1215          !!!next-input-character;          !!!next-input-character;
# Line 1081  sub _get_next_token ($) { Line 1221  sub _get_next_token ($) {
1221            $self->{next_char} == 0x000B or # HT            $self->{next_char} == 0x000B or # HT
1222            $self->{next_char} == 0x000C or # FF            $self->{next_char} == 0x000C or # FF
1223            $self->{next_char} == 0x0020) { # SP            $self->{next_char} == 0x0020) { # SP
1224            !!!cp (107);
1225          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1226          !!!next-input-character;          !!!next-input-character;
1227          redo A;          redo A;
1228        } elsif ($self->{next_char} == 0x0026) { # &        } elsif ($self->{next_char} == 0x0026) { # &
1229            !!!cp (108);
1230          $self->{last_attribute_value_state} = $self->{state};          $self->{last_attribute_value_state} = $self->{state};
1231          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;          $self->{state} = ENTITY_IN_ATTRIBUTE_VALUE_STATE;
1232          !!!next-input-character;          !!!next-input-character;
1233          redo A;          redo A;
1234        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
1235          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1236              !!!cp (109);
1237            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
1238                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
1239            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1240          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1241            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1242            if ($self->{current_token}->{attributes}) {            if ($self->{current_token}->{attributes}) {
1243                !!!cp (110);
1244              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1245              } else {
1246                ## NOTE: This state should never be reached.
1247                !!!cp (111);
1248            }            }
1249          } else {          } else {
1250            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
# Line 1111  sub _get_next_token ($) { Line 1258  sub _get_next_token ($) {
1258        } elsif ($self->{next_char} == -1) {        } elsif ($self->{next_char} == -1) {
1259          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
1260          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1261              !!!cp (112);
1262            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
1263                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
1264            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1265          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1266            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1267            if ($self->{current_token}->{attributes}) {            if ($self->{current_token}->{attributes}) {
1268                !!!cp (113);
1269              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1270              } else {
1271                ## NOTE: This state should never be reached.
1272                !!!cp (114);
1273            }            }
1274          } else {          } else {
1275            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
# Line 1134  sub _get_next_token ($) { Line 1286  sub _get_next_token ($) {
1286               0x0027 => 1, # '               0x0027 => 1, # '
1287               0x003D => 1, # =               0x003D => 1, # =
1288              }->{$self->{next_char}}) {              }->{$self->{next_char}}) {
1289              !!!cp (115);
1290            !!!parse-error (type => 'bad attribute value');            !!!parse-error (type => 'bad attribute value');
1291            } else {
1292              !!!cp (116);
1293          }          }
1294          $self->{current_attribute}->{value} .= chr ($self->{next_char});          $self->{current_attribute}->{value} .= chr ($self->{next_char});
1295          ## Stay in the state          ## Stay in the state
# Line 1151  sub _get_next_token ($) { Line 1306  sub _get_next_token ($) {
1306             -1);             -1);
1307    
1308        unless (defined $token) {        unless (defined $token) {
1309            !!!cp (117);
1310          $self->{current_attribute}->{value} .= '&';          $self->{current_attribute}->{value} .= '&';
1311        } else {        } else {
1312            !!!cp (118);
1313          $self->{current_attribute}->{value} .= $token->{data};          $self->{current_attribute}->{value} .= $token->{data};
1314          $self->{current_attribute}->{has_reference} = $token->{has_reference};          $self->{current_attribute}->{has_reference} = $token->{has_reference};
1315          ## ISSUE: spec says "append the returned character token to the current attribute's value"          ## ISSUE: spec says "append the returned character token to the current attribute's value"
# Line 1167  sub _get_next_token ($) { Line 1324  sub _get_next_token ($) {
1324            $self->{next_char} == 0x000B or # VT            $self->{next_char} == 0x000B or # VT
1325            $self->{next_char} == 0x000C or # FF            $self->{next_char} == 0x000C or # FF
1326            $self->{next_char} == 0x0020) { # SP            $self->{next_char} == 0x0020) { # SP
1327            !!!cp (118);
1328          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1329          !!!next-input-character;          !!!next-input-character;
1330          redo A;          redo A;
1331        } elsif ($self->{next_char} == 0x003E) { # >        } elsif ($self->{next_char} == 0x003E) { # >
1332          if ($self->{current_token}->{type} == START_TAG_TOKEN) {          if ($self->{current_token}->{type} == START_TAG_TOKEN) {
1333              !!!cp (119);
1334            $self->{current_token}->{first_start_tag}            $self->{current_token}->{first_start_tag}
1335                = not defined $self->{last_emitted_start_tag_name};                = not defined $self->{last_emitted_start_tag_name};
1336            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
1337          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {          } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
1338            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST            $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
1339            if ($self->{current_token}->{attributes}) {            if ($self->{current_token}->{attributes}) {
1340                !!!cp (120);
1341              !!!parse-error (type => 'end tag attribute');              !!!parse-error (type => 'end tag attribute');
1342              } else {
1343                ## NOTE: This state should never be reached.
1344                !!!cp (121);
1345            }            }
1346          } else {          } else {
1347            die "$0: $self->{current_token}->{type}: Unknown token type";            die "$0: $self->{current_token}->{type}: Unknown token type";
# Line 1195  sub _get_next_token ($) { Line 1358  sub _get_next_token ($) {
1358              $self->{current_token}->{type} == START_TAG_TOKEN and              $self->{current_token}->{type} == START_TAG_TOKEN and
1359              $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {              $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {
1360            # permitted slash            # permitted slash
1361              !!!cp (122);
1362            #            #
1363          } else {          } else {
1364              !!!cp (123);
1365            !!!parse-error (type => 'nestc');            !!!parse-error (type => 'nestc');
1366          }          }
1367          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1368          # next-input-character is already done          # next-input-character is already done
1369          redo A;          redo A;
1370        } else {        } else {
1371            !!!cp (124);
1372          !!!parse-error (type => 'no space between attributes');          !!!parse-error (type => 'no space between attributes');
1373          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;          $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
1374          ## reconsume          ## reconsume
# Line 1215  sub _get_next_token ($) { Line 1381  sub _get_next_token ($) {
1381    
1382        BC: {        BC: {
1383          if ($self->{next_char} == 0x003E) { # >          if ($self->{next_char} == 0x003E) { # >
1384              !!!cp (124);
1385            $self->{state} = DATA_STATE;            $self->{state} = DATA_STATE;
1386            !!!next-input-character;            !!!next-input-character;
1387    
# Line 1222  sub _get_next_token ($) { Line 1389  sub _get_next_token ($) {
1389    
1390            redo A;            redo A;
1391          } elsif ($self->{next_char} == -1) {          } elsif ($self->{next_char} == -1) {
1392              !!!cp (125);
1393            $self->{state} = DATA_STATE;            $self->{state} = DATA_STATE;
1394            ## reconsume            ## reconsume
1395    
# Line 1229  sub _get_next_token ($) { Line 1397  sub _get_next_token ($) {
1397    
1398            redo A;            redo A;
1399          } else {          } else {
1400              !!!cp (126);
1401            $token->{data} .= chr ($self->{next_char});            $token->{data} .= chr ($self->{next_char});
1402            !!!next-input-character;            !!!next-input-character;
1403            redo BC;            redo BC;
1404          }          }
1405        } # BC        } # BC
1406    
1407          die "$0: _get_next_token: unexpected case [BC]";
1408      } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {      } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {
1409        ## (only happen if PCDATA state)        ## (only happen if PCDATA state)
1410    
# Line 1244  sub _get_next_token ($) { Line 1415  sub _get_next_token ($) {
1415          !!!next-input-character;          !!!next-input-character;
1416          push @next_char, $self->{next_char};          push @next_char, $self->{next_char};
1417          if ($self->{next_char} == 0x002D) { # -          if ($self->{next_char} == 0x002D) { # -
1418              !!!cp (127);
1419            $self->{current_token} = {type => COMMENT_TOKEN, data => ''};            $self->{current_token} = {type => COMMENT_TOKEN, data => ''};
1420            $self->{state} = COMMENT_START_STATE;            $self->{state} = COMMENT_START_STATE;
1421            !!!next-input-character;            !!!next-input-character;
1422            redo A;            redo A;
1423            } else {
1424              !!!cp (128);
1425          }          }
1426        } elsif ($self->{next_char} == 0x0044 or # D        } elsif ($self->{next_char} == 0x0044 or # D
1427                 $self->{next_char} == 0x0064) { # d                 $self->{next_char} == 0x0064) { # d
# Line 1275  sub _get_next_token ($) { Line 1449  sub _get_next_token ($) {
1449                    push @next_char, $self->{next_char};                    push @next_char, $self->{next_char};
1450                    if ($self->{next_char} == 0x0045 or # E                    if ($self->{next_char} == 0x0045 or # E
1451                        $self->{next_char} == 0x0065) { # e                        $self->{next_char} == 0x0065) { # e
1452                      ## ISSUE: What a stupid code this is!                      !!!cp (129);
1453                        ## TODO: What a stupid code this is!
1454                      $self->{state} = DOCTYPE_STATE;                      $self->{state} = DOCTYPE_STATE;
1455                      !!!next-input-character;                      !!!next-input-character;
1456                      redo A;                      redo A;
1457                      } else {
1458                        !!!cp (130);
1459                    }                    }
1460                    } else {
1461                      !!!cp (131);
1462                  }                  }
1463                  } else {
1464                    !!!cp (132);
1465                }                }
1466                } else {