/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.20 by wakaba, Sat Jun 23 14:25:05 2007 UTC revision 1.32 by wakaba, Sun Jul 1 06:18:57 2007 UTC
# Line 7  our $VERSION=do{my @r=(q$Revision$=~/\d+ Line 7  our $VERSION=do{my @r=(q$Revision$=~/\d+
7  ## doc.write ('');  ## doc.write ('');
8  ## alert (doc.compatMode);  ## alert (doc.compatMode);
9    
10    ## ISSUE: HTML5 revision 967 says that the encoding layer MUST NOT
11    ## strip BOM and the HTML layer MUST ignore it.  Whether we can do it
12    ## is not yet clear.
13    ## "{U+FEFF}..." in UTF-16BE/UTF-16LE is three or four characters?
14    ## "{U+FEFF}..." in GB18030?
15    
16  my $permitted_slash_tag_name = {  my $permitted_slash_tag_name = {
17    base => 1,    base => 1,
18    link => 1,    link => 1,
# Line 247  sub _get_next_token ($) { Line 253  sub _get_next_token ($) {
253      } elsif ($self->{state} eq 'entity data') {      } elsif ($self->{state} eq 'entity data') {
254        ## (cannot happen in CDATA state)        ## (cannot happen in CDATA state)
255                
256        my $token = $self->_tokenize_attempt_to_consume_an_entity;        my $token = $self->_tokenize_attempt_to_consume_an_entity (0);
257    
258        $self->{state} = 'data';        $self->{state} = 'data';
259        # next-input-character is already done        # next-input-character is already done
# Line 326  sub _get_next_token ($) { Line 332  sub _get_next_token ($) {
332      } elsif ($self->{state} eq 'close tag open') {      } elsif ($self->{state} eq 'close tag open') {
333        if ($self->{content_model_flag} eq 'RCDATA' or        if ($self->{content_model_flag} eq 'RCDATA' or
334            $self->{content_model_flag} eq 'CDATA') {            $self->{content_model_flag} eq 'CDATA') {
335          my @next_char;          if (defined $self->{last_emitted_start_tag_name}) {
336          TAGNAME: for (my $i = 0; $i < length $self->{last_emitted_start_tag_name}; $i++) {            ## NOTE: <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>
337              my @next_char;
338              TAGNAME: for (my $i = 0; $i < length $self->{last_emitted_start_tag_name}; $i++) {
339                push @next_char, $self->{next_input_character};
340                my $c = ord substr ($self->{last_emitted_start_tag_name}, $i, 1);
341                my $C = 0x0061 <= $c && $c <= 0x007A ? $c - 0x0020 : $c;
342                if ($self->{next_input_character} == $c or $self->{next_input_character} == $C) {
343                  !!!next-input-character;
344                  next TAGNAME;
345                } else {
346                  $self->{next_input_character} = shift @next_char; # reconsume
347                  !!!back-next-input-character (@next_char);
348                  $self->{state} = 'data';
349    
350                  !!!emit ({type => 'character', data => '</'});
351      
352                  redo A;
353                }
354              }
355            push @next_char, $self->{next_input_character};            push @next_char, $self->{next_input_character};
356            my $c = ord substr ($self->{last_emitted_start_tag_name}, $i, 1);        
357            my $C = 0x0061 <= $c && $c <= 0x007A ? $c - 0x0020 : $c;            unless ($self->{next_input_character} == 0x0009 or # HT
358            if ($self->{next_input_character} == $c or $self->{next_input_character} == $C) {                    $self->{next_input_character} == 0x000A or # LF
359              !!!next-input-character;                    $self->{next_input_character} == 0x000B or # VT
360              next TAGNAME;                    $self->{next_input_character} == 0x000C or # FF
361            } else {                    $self->{next_input_character} == 0x0020 or # SP
362                      $self->{next_input_character} == 0x003E or # >
363                      $self->{next_input_character} == 0x002F or # /
364                      $self->{next_input_character} == -1) {
365              $self->{next_input_character} = shift @next_char; # reconsume              $self->{next_input_character} = shift @next_char; # reconsume
366              !!!back-next-input-character (@next_char);              !!!back-next-input-character (@next_char);
367              $self->{state} = 'data';              $self->{state} = 'data';
   
368              !!!emit ({type => 'character', data => '</'});              !!!emit ({type => 'character', data => '</'});
   
369              redo A;              redo A;
370              } else {
371                $self->{next_input_character} = shift @next_char;
372                !!!back-next-input-character (@next_char);
373                # and consume...
374            }            }
375          }          } else {
376          push @next_char, $self->{next_input_character};            ## No start tag token has ever been emitted
377                  # next-input-character is already done
         unless ($self->{next_input_character} == 0x0009 or # HT  
                 $self->{next_input_character} == 0x000A or # LF  
                 $self->{next_input_character} == 0x000B or # VT  
                 $self->{next_input_character} == 0x000C or # FF  
                 $self->{next_input_character} == 0x0020 or # SP  
                 $self->{next_input_character} == 0x003E or # >  
                 $self->{next_input_character} == 0x002F or # /  
                 $self->{next_input_character} == -1) {  
           $self->{next_input_character} = shift @next_char; # reconsume  
           !!!back-next-input-character (@next_char);  
378            $self->{state} = 'data';            $self->{state} = 'data';
   
379            !!!emit ({type => 'character', data => '</'});            !!!emit ({type => 'character', data => '</'});
   
380            redo A;            redo A;
         } else {  
           $self->{next_input_character} = shift @next_char;  
           !!!back-next-input-character (@next_char);  
           # and consume...  
381          }          }
382        }        }
383                
# Line 412  sub _get_next_token ($) { Line 425  sub _get_next_token ($) {
425          redo A;          redo A;
426        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
427          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
428              $self->{current_token}->{first_start_tag}
429                  = not defined $self->{last_emitted_start_tag_name};
430            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
431          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
432            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 425  sub _get_next_token ($) { Line 440  sub _get_next_token ($) {
440          !!!next-input-character;          !!!next-input-character;
441    
442          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
443    
444          redo A;          redo A;
445        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{next_input_character} and
# Line 438  sub _get_next_token ($) { Line 452  sub _get_next_token ($) {
452        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
453          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
454          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
455              $self->{current_token}->{first_start_tag}
456                  = not defined $self->{last_emitted_start_tag_name};
457            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
458          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
459            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 451  sub _get_next_token ($) { Line 467  sub _get_next_token ($) {
467          # reconsume          # reconsume
468    
469          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
470    
471          redo A;          redo A;
472        } elsif ($self->{next_input_character} == 0x002F) { # /        } elsif ($self->{next_input_character} == 0x002F) { # /
# Line 485  sub _get_next_token ($) { Line 500  sub _get_next_token ($) {
500          redo A;          redo A;
501        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
502          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
503              $self->{current_token}->{first_start_tag}
504                  = not defined $self->{last_emitted_start_tag_name};
505            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
506          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
507            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 498  sub _get_next_token ($) { Line 515  sub _get_next_token ($) {
515          !!!next-input-character;          !!!next-input-character;
516    
517          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
518    
519          redo A;          redo A;
520        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{next_input_character} and
# Line 524  sub _get_next_token ($) { Line 540  sub _get_next_token ($) {
540        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
541          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
542          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
543              $self->{current_token}->{first_start_tag}
544                  = not defined $self->{last_emitted_start_tag_name};
545            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
546          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
547            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 537  sub _get_next_token ($) { Line 555  sub _get_next_token ($) {
555          # reconsume          # reconsume
556    
557          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
558    
559          redo A;          redo A;
560        } else {        } else {
# Line 576  sub _get_next_token ($) { Line 593  sub _get_next_token ($) {
593        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
594          $before_leave->();          $before_leave->();
595          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
596              $self->{current_token}->{first_start_tag}
597                  = not defined $self->{last_emitted_start_tag_name};
598            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
599          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
600            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 589  sub _get_next_token ($) { Line 608  sub _get_next_token ($) {
608          !!!next-input-character;          !!!next-input-character;
609    
610          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
611    
612          redo A;          redo A;
613        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{next_input_character} and
# Line 616  sub _get_next_token ($) { Line 634  sub _get_next_token ($) {
634          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
635          $before_leave->();          $before_leave->();
636          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
637              $self->{current_token}->{first_start_tag}
638                  = not defined $self->{last_emitted_start_tag_name};
639            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
640          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
641            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 629  sub _get_next_token ($) { Line 649  sub _get_next_token ($) {
649          # reconsume          # reconsume
650    
651          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
652    
653          redo A;          redo A;
654        } else {        } else {
# Line 653  sub _get_next_token ($) { Line 672  sub _get_next_token ($) {
672          redo A;          redo A;
673        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
674          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
675              $self->{current_token}->{first_start_tag}
676                  = not defined $self->{last_emitted_start_tag_name};
677            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
678          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
679            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 666  sub _get_next_token ($) { Line 687  sub _get_next_token ($) {
687          !!!next-input-character;          !!!next-input-character;
688    
689          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
690    
691          redo A;          redo A;
692        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{next_input_character} and
# Line 692  sub _get_next_token ($) { Line 712  sub _get_next_token ($) {
712        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
713          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
714          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
715              $self->{current_token}->{first_start_tag}
716                  = not defined $self->{last_emitted_start_tag_name};
717            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
718          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
719            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 705  sub _get_next_token ($) { Line 727  sub _get_next_token ($) {
727          # reconsume          # reconsume
728    
729          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
730    
731          redo A;          redo A;
732        } else {        } else {
# Line 738  sub _get_next_token ($) { Line 759  sub _get_next_token ($) {
759          redo A;          redo A;
760        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
761          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
762              $self->{current_token}->{first_start_tag}
763                  = not defined $self->{last_emitted_start_tag_name};
764            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
765          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
766            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 751  sub _get_next_token ($) { Line 774  sub _get_next_token ($) {
774          !!!next-input-character;          !!!next-input-character;
775    
776          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
777    
778          redo A;          redo A;
779        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
780          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
781          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
782              $self->{current_token}->{first_start_tag}
783                  = not defined $self->{last_emitted_start_tag_name};
784            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
785          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
786            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 770  sub _get_next_token ($) { Line 794  sub _get_next_token ($) {
794          ## reconsume          ## reconsume
795    
796          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
797    
798          redo A;          redo A;
799        } else {        } else {
# Line 792  sub _get_next_token ($) { Line 815  sub _get_next_token ($) {
815        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
816          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
817          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
818              $self->{current_token}->{first_start_tag}
819                  = not defined $self->{last_emitted_start_tag_name};
820            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
821          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
822            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 805  sub _get_next_token ($) { Line 830  sub _get_next_token ($) {
830          ## reconsume          ## reconsume
831    
832          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
833    
834          redo A;          redo A;
835        } else {        } else {
# Line 827  sub _get_next_token ($) { Line 851  sub _get_next_token ($) {
851        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
852          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
853          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
854              $self->{current_token}->{first_start_tag}
855                  = not defined $self->{last_emitted_start_tag_name};
856            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
857          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
858            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 840  sub _get_next_token ($) { Line 866  sub _get_next_token ($) {
866          ## reconsume          ## reconsume
867    
868          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
869    
870          redo A;          redo A;
871        } else {        } else {
# Line 865  sub _get_next_token ($) { Line 890  sub _get_next_token ($) {
890          redo A;          redo A;
891        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
892          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
893              $self->{current_token}->{first_start_tag}
894                  = not defined $self->{last_emitted_start_tag_name};
895            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
896          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
897            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 878  sub _get_next_token ($) { Line 905  sub _get_next_token ($) {
905          !!!next-input-character;          !!!next-input-character;
906    
907          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
908    
909          redo A;          redo A;
910        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
911          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
912          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
913              $self->{current_token}->{first_start_tag}
914                  = not defined $self->{last_emitted_start_tag_name};
915            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
916          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
917            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 897  sub _get_next_token ($) { Line 925  sub _get_next_token ($) {
925          ## reconsume          ## reconsume
926    
927          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
928    
929          redo A;          redo A;
930        } else {        } else {
# Line 907  sub _get_next_token ($) { Line 934  sub _get_next_token ($) {
934          redo A;          redo A;
935        }        }
936      } elsif ($self->{state} eq 'entity in attribute value') {      } elsif ($self->{state} eq 'entity in attribute value') {
937        my $token = $self->_tokenize_attempt_to_consume_an_entity;        my $token = $self->_tokenize_attempt_to_consume_an_entity (1);
938    
939        unless (defined $token) {        unless (defined $token) {
940          $self->{current_attribute}->{value} .= '&';          $self->{current_attribute}->{value} .= '&';
# Line 956  sub _get_next_token ($) { Line 983  sub _get_next_token ($) {
983          push @next_char, $self->{next_input_character};          push @next_char, $self->{next_input_character};
984          if ($self->{next_input_character} == 0x002D) { # -          if ($self->{next_input_character} == 0x002D) { # -
985            $self->{current_token} = {type => 'comment', data => ''};            $self->{current_token} = {type => 'comment', data => ''};
986            $self->{state} = 'comment';            $self->{state} = 'comment start';
987            !!!next-input-character;            !!!next-input-character;
988            redo A;            redo A;
989          }          }
# Line 998  sub _get_next_token ($) { Line 1025  sub _get_next_token ($) {
1025          }          }
1026        }        }
1027    
1028        !!!parse-error (type => 'bogus comment open');        !!!parse-error (type => 'bogus comment');
1029        $self->{next_input_character} = shift @next_char;        $self->{next_input_character} = shift @next_char;
1030        !!!back-next-input-character (@next_char);        !!!back-next-input-character (@next_char);
1031        $self->{state} = 'bogus comment';        $self->{state} = 'bogus comment';
# Line 1006  sub _get_next_token ($) { Line 1033  sub _get_next_token ($) {
1033                
1034        ## ISSUE: typos in spec: chacacters, is is a parse error        ## ISSUE: typos in spec: chacacters, is is a parse error
1035        ## ISSUE: spec is somewhat unclear on "is the first character that will be in the comment"; what is "that will be in the comment" is what the algorithm defines, isn't it?        ## ISSUE: spec is somewhat unclear on "is the first character that will be in the comment"; what is "that will be in the comment" is what the algorithm defines, isn't it?
1036        } elsif ($self->{state} eq 'comment start') {
1037          if ($self->{next_input_character} == 0x002D) { # -
1038            $self->{state} = 'comment start dash';
1039            !!!next-input-character;
1040            redo A;
1041          } elsif ($self->{next_input_character} == 0x003E) { # >
1042            !!!parse-error (type => 'bogus comment');
1043            $self->{state} = 'data';
1044            !!!next-input-character;
1045    
1046            !!!emit ($self->{current_token}); # comment
1047    
1048            redo A;
1049          } elsif ($self->{next_input_character} == -1) {
1050            !!!parse-error (type => 'unclosed comment');
1051            $self->{state} = 'data';
1052            ## reconsume
1053    
1054            !!!emit ($self->{current_token}); # comment
1055    
1056            redo A;
1057          } else {
1058            $self->{current_token}->{data} # comment
1059                .= chr ($self->{next_input_character});
1060            $self->{state} = 'comment';
1061            !!!next-input-character;
1062            redo A;
1063          }
1064        } elsif ($self->{state} eq 'comment start dash') {
1065          if ($self->{next_input_character} == 0x002D) { # -
1066            $self->{state} = 'comment end';
1067            !!!next-input-character;
1068            redo A;
1069          } elsif ($self->{next_input_character} == 0x003E) { # >
1070            !!!parse-error (type => 'bogus comment');
1071            $self->{state} = 'data';
1072            !!!next-input-character;
1073    
1074            !!!emit ($self->{current_token}); # comment
1075    
1076            redo A;
1077          } elsif ($self->{next_input_character} == -1) {
1078            !!!parse-error (type => 'unclosed comment');
1079            $self->{state} = 'data';
1080            ## reconsume
1081    
1082            !!!emit ($self->{current_token}); # comment
1083    
1084            redo A;
1085          } else {
1086            $self->{current_token}->{data} # comment
1087                .= chr ($self->{next_input_character});
1088            $self->{state} = 'comment';
1089            !!!next-input-character;
1090            redo A;
1091          }
1092      } elsif ($self->{state} eq 'comment') {      } elsif ($self->{state} eq 'comment') {
1093        if ($self->{next_input_character} == 0x002D) { # -        if ($self->{next_input_character} == 0x002D) { # -
1094          $self->{state} = 'comment dash';          $self->{state} = 'comment end dash';
1095          !!!next-input-character;          !!!next-input-character;
1096          redo A;          redo A;
1097        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
# Line 1017  sub _get_next_token ($) { Line 1100  sub _get_next_token ($) {
1100          ## reconsume          ## reconsume
1101    
1102          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{current_token}); # comment
         undef $self->{current_token};  
1103    
1104          redo A;          redo A;
1105        } else {        } else {
# Line 1026  sub _get_next_token ($) { Line 1108  sub _get_next_token ($) {
1108          !!!next-input-character;          !!!next-input-character;
1109          redo A;          redo A;
1110        }        }
1111      } elsif ($self->{state} eq 'comment dash') {      } elsif ($self->{state} eq 'comment end dash') {
1112        if ($self->{next_input_character} == 0x002D) { # -        if ($self->{next_input_character} == 0x002D) { # -
1113          $self->{state} = 'comment end';          $self->{state} = 'comment end';
1114          !!!next-input-character;          !!!next-input-character;
# Line 1037  sub _get_next_token ($) { Line 1119  sub _get_next_token ($) {
1119          ## reconsume          ## reconsume
1120    
1121          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{current_token}); # comment
         undef $self->{current_token};  
1122    
1123          redo A;          redo A;
1124        } else {        } else {
# Line 1052  sub _get_next_token ($) { Line 1133  sub _get_next_token ($) {
1133          !!!next-input-character;          !!!next-input-character;
1134    
1135          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{current_token}); # comment
         undef $self->{current_token};  
1136    
1137          redo A;          redo A;
1138        } elsif ($self->{next_input_character} == 0x002D) { # -        } elsif ($self->{next_input_character} == 0x002D) { # -
# Line 1067  sub _get_next_token ($) { Line 1147  sub _get_next_token ($) {
1147          ## reconsume          ## reconsume
1148    
1149          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{current_token}); # comment
         undef $self->{current_token};  
1150    
1151          redo A;          redo A;
1152        } else {        } else {
# Line 1142  sub _get_next_token ($) { Line 1221  sub _get_next_token ($) {
1221          !!!next-input-character;          !!!next-input-character;
1222    
1223          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1224    
1225          redo A;          redo A;
1226        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
# Line 1152  sub _get_next_token ($) { Line 1230  sub _get_next_token ($) {
1230    
1231          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1232          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1233    
1234          redo A;          redo A;
1235        } else {        } else {
# Line 1176  sub _get_next_token ($) { Line 1253  sub _get_next_token ($) {
1253          !!!next-input-character;          !!!next-input-character;
1254    
1255          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1256    
1257          redo A;          redo A;
1258        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
# Line 1186  sub _get_next_token ($) { Line 1262  sub _get_next_token ($) {
1262    
1263          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1264          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1265    
1266          redo A;          redo A;
1267        } elsif ($self->{next_input_character} == 0x0050 or # P        } elsif ($self->{next_input_character} == 0x0050 or # P
# Line 1278  sub _get_next_token ($) { Line 1353  sub _get_next_token ($) {
1353    
1354          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1355          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1356    
1357          redo A;          redo A;
1358        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
# Line 1289  sub _get_next_token ($) { Line 1363  sub _get_next_token ($) {
1363    
1364          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1365          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1366    
1367          redo A;          redo A;
1368        } else {        } else {
# Line 1311  sub _get_next_token ($) { Line 1384  sub _get_next_token ($) {
1384    
1385          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1386          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1387    
1388          redo A;          redo A;
1389        } else {        } else {
# Line 1334  sub _get_next_token ($) { Line 1406  sub _get_next_token ($) {
1406    
1407          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1408          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1409    
1410          redo A;          redo A;
1411        } else {        } else {
# Line 1367  sub _get_next_token ($) { Line 1438  sub _get_next_token ($) {
1438          !!!next-input-character;          !!!next-input-character;
1439    
1440          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1441    
1442          redo A;          redo A;
1443        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
1444          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
1445    
1446          $self->{state} = 'data';          $self->{state} = 'data';
1447          ## recomsume          ## reconsume
1448    
1449          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1450          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1451    
1452          redo A;          redo A;
1453        } else {        } else {
# Line 1412  sub _get_next_token ($) { Line 1481  sub _get_next_token ($) {
1481    
1482          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1483          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1484    
1485          redo A;          redo A;
1486        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
1487          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
1488    
1489          $self->{state} = 'data';          $self->{state} = 'data';
1490          ## recomsume          ## reconsume
1491    
1492          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1493          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1494    
1495          redo A;          redo A;
1496        } else {        } else {
1497          !!!parse-error (type => 'string after PUBLIC literal');          !!!parse-error (type => 'string after SYSTEM');
1498          $self->{state} = 'bogus DOCTYPE';          $self->{state} = 'bogus DOCTYPE';
1499          !!!next-input-character;          !!!next-input-character;
1500          redo A;          redo A;
# Line 1445  sub _get_next_token ($) { Line 1512  sub _get_next_token ($) {
1512    
1513          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1514          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1515    
1516          redo A;          redo A;
1517        } else {        } else {
# Line 1468  sub _get_next_token ($) { Line 1534  sub _get_next_token ($) {
1534    
1535          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1536          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1537    
1538          redo A;          redo A;
1539        } else {        } else {
# Line 1491  sub _get_next_token ($) { Line 1556  sub _get_next_token ($) {
1556          !!!next-input-character;          !!!next-input-character;
1557    
1558          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1559    
1560          redo A;          redo A;
1561        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
1562          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
1563    
1564          $self->{state} = 'data';          $self->{state} = 'data';
1565          ## recomsume          ## reconsume
1566    
1567          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1568          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1569    
1570          redo A;          redo A;
1571        } else {        } else {
# Line 1518  sub _get_next_token ($) { Line 1581  sub _get_next_token ($) {
1581    
1582          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1583          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1584    
1585          redo A;          redo A;
1586        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
# Line 1528  sub _get_next_token ($) { Line 1590  sub _get_next_token ($) {
1590    
1591          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1592          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1593    
1594          redo A;          redo A;
1595        } else {        } else {
# Line 1544  sub _get_next_token ($) { Line 1605  sub _get_next_token ($) {
1605    die "$0: _get_next_token: unexpected case";    die "$0: _get_next_token: unexpected case";
1606  } # _get_next_token  } # _get_next_token
1607    
1608  sub _tokenize_attempt_to_consume_an_entity ($) {  sub _tokenize_attempt_to_consume_an_entity ($$) {
1609    my $self = shift;    my ($self, $in_attr) = @_;
1610    
1611    if ({    if ({
1612         0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, # HT, LF, VT, FF,         0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, # HT, LF, VT, FF,
# Line 1558  sub _tokenize_attempt_to_consume_an_enti Line 1619  sub _tokenize_attempt_to_consume_an_enti
1619      !!!next-input-character;      !!!next-input-character;
1620      if ($self->{next_input_character} == 0x0078 or # x      if ($self->{next_input_character} == 0x0078 or # x
1621          $self->{next_input_character} == 0x0058) { # X          $self->{next_input_character} == 0x0058) { # X
1622        my $num;        my $code;
1623        X: {        X: {
1624          my $x_char = $self->{next_input_character};          my $x_char = $self->{next_input_character};
1625          !!!next-input-character;          !!!next-input-character;
1626          if (0x0030 <= $self->{next_input_character} and          if (0x0030 <= $self->{next_input_character} and
1627              $self->{next_input_character} <= 0x0039) { # 0..9              $self->{next_input_character} <= 0x0039) { # 0..9
1628            $num ||= 0;            $code ||= 0;
1629            $num *= 0x10;            $code *= 0x10;
1630            $num += $self->{next_input_character} - 0x0030;            $code += $self->{next_input_character} - 0x0030;
1631            redo X;            redo X;
1632          } elsif (0x0061 <= $self->{next_input_character} and          } elsif (0x0061 <= $self->{next_input_character} and
1633                   $self->{next_input_character} <= 0x0066) { # a..f                   $self->{next_input_character} <= 0x0066) { # a..f
1634            ## ISSUE: the spec says U+0078, which is apparently incorrect            $code ||= 0;
1635            $num ||= 0;            $code *= 0x10;
1636            $num *= 0x10;            $code += $self->{next_input_character} - 0x0060 + 9;
           $num += $self->{next_input_character} - 0x0060 + 9;  
1637            redo X;            redo X;
1638          } elsif (0x0041 <= $self->{next_input_character} and          } elsif (0x0041 <= $self->{next_input_character} and
1639                   $self->{next_input_character} <= 0x0046) { # A..F                   $self->{next_input_character} <= 0x0046) { # A..F
1640            ## ISSUE: the spec says U+0058, which is apparently incorrect            $code ||= 0;
1641            $num ||= 0;            $code *= 0x10;
1642            $num *= 0x10;            $code += $self->{next_input_character} - 0x0040 + 9;
           $num += $self->{next_input_character} - 0x0040 + 9;  
1643            redo X;            redo X;
1644          } elsif (not defined $num) { # no hexadecimal digit          } elsif (not defined $code) { # no hexadecimal digit
1645            !!!parse-error (type => 'bare hcro');            !!!parse-error (type => 'bare hcro');
1646            $self->{next_input_character} = 0x0023; # #            $self->{next_input_character} = 0x0023; # #
1647            !!!back-next-input-character ($x_char);            !!!back-next-input-character ($x_char);
# Line 1593  sub _tokenize_attempt_to_consume_an_enti Line 1652  sub _tokenize_attempt_to_consume_an_enti
1652            !!!parse-error (type => 'no refc');            !!!parse-error (type => 'no refc');
1653          }          }
1654    
1655          ## TODO: check the definition for |a valid Unicode character|.          if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {
1656          ## <http://lists.whatwg.org/pipermail/whatwg-whatwg.org/2006-December/thread.html#8189>            !!!parse-error (type => sprintf 'invalid character reference:U+%04X', $code);
1657          if ($num > 1114111 or $num == 0) {            $code = 0xFFFD;
1658            $num = 0xFFFD; # REPLACEMENT CHARACTER          } elsif ($code > 0x10FFFF) {
1659            ## ISSUE: Why this is not an error?            !!!parse-error (type => sprintf 'invalid character reference:U-%08X', $code);
1660          } elsif (0x80 <= $num and $num <= 0x9F) {            $code = 0xFFFD;
1661            !!!parse-error (type => sprintf 'c1 entity:U+%04X', $num);          } elsif ($code == 0x000D) {
1662            $num = $c1_entity_char->{$num};            !!!parse-error (type => 'CR character reference');
1663              $code = 0x000A;
1664            } elsif (0x80 <= $code and $code <= 0x9F) {
1665              !!!parse-error (type => sprintf 'C1 character reference:U+%04X', $code);
1666              $code = $c1_entity_char->{$code};
1667          }          }
1668    
1669          return {type => 'character', data => chr $num};          return {type => 'character', data => chr $code};
1670        } # X        } # X
1671      } elsif (0x0030 <= $self->{next_input_character} and      } elsif (0x0030 <= $self->{next_input_character} and
1672               $self->{next_input_character} <= 0x0039) { # 0..9               $self->{next_input_character} <= 0x0039) { # 0..9
# Line 1624  sub _tokenize_attempt_to_consume_an_enti Line 1687  sub _tokenize_attempt_to_consume_an_enti
1687          !!!parse-error (type => 'no refc');          !!!parse-error (type => 'no refc');
1688        }        }
1689    
1690        ## TODO: check the definition for |a valid Unicode character|.        if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {
1691        if ($code > 1114111 or $code == 0) {          !!!parse-error (type => sprintf 'invalid character reference:U+%04X', $code);
1692          $code = 0xFFFD; # REPLACEMENT CHARACTER          $code = 0xFFFD;
1693          ## ISSUE: Why this is not an error?        } elsif ($code > 0x10FFFF) {
1694            !!!parse-error (type => sprintf 'invalid character reference:U-%08X', $code);
1695            $code = 0xFFFD;
1696          } elsif ($code == 0x000D) {
1697            !!!parse-error (type => 'CR character reference');
1698            $code = 0x000A;
1699        } elsif (0x80 <= $code and $code <= 0x9F) {        } elsif (0x80 <= $code and $code <= 0x9F) {
1700          !!!parse-error (type => sprintf 'c1 entity:U+%04X', $code);          !!!parse-error (type => sprintf 'C1 character reference:U+%04X', $code);
1701          $code = $c1_entity_char->{$code};          $code = $c1_entity_char->{$code};
1702        }        }
1703                
# Line 1663  sub _tokenize_attempt_to_consume_an_enti Line 1731  sub _tokenize_attempt_to_consume_an_enti
1731              $self->{next_input_character} == 0x003B)) { # ;              $self->{next_input_character} == 0x003B)) { # ;
1732        $entity_name .= chr $self->{next_input_character};        $entity_name .= chr $self->{next_input_character};
1733        if (defined $EntityChar->{$entity_name}) {        if (defined $EntityChar->{$entity_name}) {
         $value = $EntityChar->{$entity_name};  
1734          if ($self->{next_input_character} == 0x003B) { # ;          if ($self->{next_input_character} == 0x003B) { # ;
1735              $value = $EntityChar->{$entity_name};
1736            $match = 1;            $match = 1;
1737            !!!next-input-character;            !!!next-input-character;
1738            last;            last;
1739          } else {          } elsif (not $in_attr) {
1740              $value = $EntityChar->{$entity_name};
1741            $match = -1;            $match = -1;
1742            } else {
1743              $value .= chr $self->{next_input_character};
1744          }          }
1745        } else {        } else {
1746          $value .= chr $self->{next_input_character};          $value .= chr $self->{next_input_character};
# Line 1680  sub _tokenize_attempt_to_consume_an_enti Line 1751  sub _tokenize_attempt_to_consume_an_enti
1751      if ($match > 0) {      if ($match > 0) {
1752        return {type => 'character', data => $value};        return {type => 'character', data => $value};
1753      } elsif ($match < 0) {      } elsif ($match < 0) {
1754        !!!parse-error (type => 'refc');        !!!parse-error (type => 'no refc');
1755        return {type => 'character', data => $value};        return {type => 'character', data => $value};
1756      } else {      } else {
1757        !!!parse-error (type => 'bare ero');        !!!parse-error (type => 'bare ero');
1758        ## NOTE: No characters are consumed in the spec.        ## NOTE: No characters are consumed in the spec.
1759        !!!back-token ({type => 'character', data => $value});        return {type => 'character', data => '&'.$value};
       return undef;  
1760      }      }
1761    } else {    } else {
1762      ## no characters are consumed      ## no characters are consumed
# Line 1881  sub _tree_construction_initial ($) { Line 1951  sub _tree_construction_initial ($) {
1951      } elsif ($token->{type} eq 'character') {      } elsif ($token->{type} eq 'character') {
1952        if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D        if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D
1953          ## Ignore the token          ## Ignore the token
1954    
1955          unless (length $token->{data}) {          unless (length $token->{data}) {
1956            ## Stay in the phase            ## Stay in the phase
1957            !!!next-token;            !!!next-token;
# Line 1923  sub _tree_construction_root_element ($) Line 1994  sub _tree_construction_root_element ($)
1994          !!!next-token;          !!!next-token;
1995          redo B;          redo B;
1996        } elsif ($token->{type} eq 'character') {        } elsif ($token->{type} eq 'character') {
1997          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D
1998            $self->{document}->manakai_append_text ($1);            ## Ignore the token.
1999            ## ISSUE: DOM3 Core does not allow Document > Text  
2000            unless (length $token->{data}) {            unless (length $token->{data}) {
2001              ## Stay in the phase              ## Stay in the phase
2002              !!!next-token;              !!!next-token;
# Line 1965  sub _reset_insertion_mode ($) { Line 2036  sub _reset_insertion_mode ($) {
2036            
2037      ## Step 3      ## Step 3
2038      S3: {      S3: {
2039        $last = 1 if $self->{open_elements}->[0]->[0] eq $node->[0];        ## ISSUE: Oops! "If node is the first node in the stack of open
2040        if (defined $self->{inner_html_node}) {        ## elements, then set last to true. If the context element of the
2041          if ($self->{inner_html_node}->[1] eq 'td' or        ## HTML fragment parsing algorithm is neither a td element nor a
2042              $self->{inner_html_node}->[1] eq 'th') {        ## th element, then set node to the context element. (fragment case)":
2043            #        ## The second "if" is in the scope of the first "if"!?
2044          } else {        if ($self->{open_elements}->[0]->[0] eq $node->[0]) {
2045            $node = $self->{inner_html_node};          $last = 1;
2046            if (defined $self->{inner_html_node}) {
2047              if ($self->{inner_html_node}->[1] eq 'td' or
2048                  $self->{inner_html_node}->[1] eq 'th') {
2049                #
2050              } else {
2051                $node = $self->{inner_html_node};
2052              }
2053          }          }
2054        }        }
2055            
# Line 2102  sub _tree_construction_main ($) { Line 2180  sub _tree_construction_main ($) {
2180      }      }
2181    }; # $clear_up_to_marker    }; # $clear_up_to_marker
2182    
2183    my $style_start_tag = sub {    my $parse_rcdata = sub ($$) {
2184      my $style_el; !!!create-element ($style_el, 'style', $token->{attributes});      my ($content_model_flag, $insert) = @_;
2185      ## $self->{insertion_mode} eq 'in head' and ... (always true)  
2186      (($self->{insertion_mode} eq 'in head' and defined $self->{head_element})      ## Step 1
2187       ? $self->{head_element} : $self->{open_elements}->[-1]->[0])      my $start_tag_name = $token->{tag_name};
2188        ->append_child ($style_el);      my $el;
2189      $self->{content_model_flag} = 'CDATA';      !!!create-element ($el, $start_tag_name, $token->{attributes});
2190    
2191        ## Step 2
2192        $insert->($el); # /context node/->append_child ($el)
2193    
2194        ## Step 3
2195        $self->{content_model_flag} = $content_model_flag; # CDATA or RCDATA
2196      delete $self->{escape}; # MUST      delete $self->{escape}; # MUST
2197                  
2198        ## Step 4
2199      my $text = '';      my $text = '';
2200      !!!next-token;      !!!next-token;
2201      while ($token->{type} eq 'character') {      while ($token->{type} eq 'character') { # or until stop tokenizing
2202        $text .= $token->{data};        $text .= $token->{data};
2203        !!!next-token;        !!!next-token;
2204      } # stop if non-character token or tokenizer stops tokenising      }
2205    
2206        ## Step 5
2207      if (length $text) {      if (length $text) {
2208        $style_el->manakai_append_text ($text);        my $text = $self->{document}->create_text_node ($text);
2209          $el->append_child ($text);
2210      }      }
2211        
2212        ## Step 6
2213      $self->{content_model_flag} = 'PCDATA';      $self->{content_model_flag} = 'PCDATA';
2214                  
2215      if ($token->{type} eq 'end tag' and $token->{tag_name} eq 'style') {      ## Step 7
2216        if ($token->{type} eq 'end tag' and $token->{tag_name} eq $start_tag_name) {
2217        ## Ignore the token        ## Ignore the token
2218      } else {      } else {
2219        !!!parse-error (type => 'in CDATA:#'.$token->{type});        !!!parse-error (type => 'in '.$content_model_flag.':#'.$token->{type});
       ## ISSUE: And ignore?  
2220      }      }
2221      !!!next-token;      !!!next-token;
2222    }; # $style_start_tag    }; # $parse_rcdata
2223    
2224    my $script_start_tag = sub {    my $script_start_tag = sub ($) {
2225        my $insert = $_[0];
2226      my $script_el;      my $script_el;
2227      !!!create-element ($script_el, 'script', $token->{attributes});      !!!create-element ($script_el, 'script', $token->{attributes});
2228      ## TODO: mark as "parser-inserted"      ## TODO: mark as "parser-inserted"
# Line 2166  sub _tree_construction_main ($) { Line 2256  sub _tree_construction_main ($) {
2256      } else {      } else {
2257        ## TODO: $old_insertion_point = current insertion point        ## TODO: $old_insertion_point = current insertion point
2258        ## TODO: insertion point = just before the next input character        ## TODO: insertion point = just before the next input character
2259          
2260        (($self->{insertion_mode} eq 'in head' and defined $self->{head_element})        $insert->($script_el);
        ? $self->{head_element} : $self->{open_elements}->[-1]->[0])->append_child ($script_el);  
2261                
2262        ## TODO: insertion point = $old_insertion_point (might be "undefined")        ## TODO: insertion point = $old_insertion_point (might be "undefined")
2263                
# Line 2362  sub _tree_construction_main ($) { Line 2451  sub _tree_construction_main ($) {
2451    }; # $formatting_end_tag    }; # $formatting_end_tag
2452    
2453    my $insert_to_current = sub {    my $insert_to_current = sub {
2454      $self->{open_elements}->[-1]->[0]->append_child (shift);      $self->{open_elements}->[-1]->[0]->append_child ($_[0]);
2455    }; # $insert_to_current    }; # $insert_to_current
2456    
2457    my $insert_to_foster = sub {    my $insert_to_foster = sub {
# Line 2400  sub _tree_construction_main ($) { Line 2489  sub _tree_construction_main ($) {
2489      my $insert = shift;      my $insert = shift;
2490      if ($token->{type} eq 'start tag') {      if ($token->{type} eq 'start tag') {
2491        if ($token->{tag_name} eq 'script') {        if ($token->{tag_name} eq 'script') {
2492          $script_start_tag->();          ## NOTE: This is an "as if in head" code clone
2493            $script_start_tag->($insert);
2494          return;          return;
2495        } elsif ($token->{tag_name} eq 'style') {        } elsif ($token->{tag_name} eq 'style') {
2496          $style_start_tag->();          ## NOTE: This is an "as if in head" code clone
2497            $parse_rcdata->('CDATA', $insert);
2498          return;          return;
2499        } elsif ({        } elsif ({
2500                  base => 1, link => 1, meta => 1,                  base => 1, link => 1, meta => 1,
2501                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
2502          !!!parse-error (type => 'in body:'.$token->{tag_name});          ## NOTE: This is an "as if in head" code clone, only "-t" differs
2503          ## NOTE: This is an "as if in head" code clone          !!!insert-element-t ($token->{tag_name}, $token->{attributes});
2504          my $el;          pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
         !!!create-element ($el, $token->{tag_name}, $token->{attributes});  
         if (defined $self->{head_element}) {  
           $self->{head_element}->append_child ($el);  
         } else {  
           $insert->($el);  
         }  
           
2505          !!!next-token;          !!!next-token;
2506            ## TODO: Extracting |charset| from |meta|.
2507          return;          return;
2508        } elsif ($token->{tag_name} eq 'title') {        } elsif ($token->{tag_name} eq 'title') {
2509          !!!parse-error (type => 'in body:title');          !!!parse-error (type => 'in body:title');
2510          ## NOTE: There is an "as if in head" code clone          ## NOTE: This is an "as if in head" code clone
2511          my $title_el;          $parse_rcdata->('RCDATA', sub {
2512          !!!create-element ($title_el, 'title', $token->{attributes});            if (defined $self->{head_element}) {
2513          (defined $self->{head_element} ? $self->{head_element} : $self->{open_elements}->[-1]->[0])              $self->{head_element}->append_child ($_[0]);
2514            ->append_child ($title_el);            } else {
2515          $self->{content_model_flag} = 'RCDATA';              $insert->($_[0]);
2516          delete $self->{escape}; # MUST            }
2517                    });
         my $text = '';  
         !!!next-token;  
         while ($token->{type} eq 'character') {  
           $text .= $token->{data};  
           !!!next-token;  
         }  
         if (length $text) {  
           $title_el->manakai_append_text ($text);  
         }  
           
         $self->{content_model_flag} = 'PCDATA';  
           
         if ($token->{type} eq 'end tag' and  
             $token->{tag_name} eq 'title') {  
           ## Ignore the token  
         } else {  
           !!!parse-error (type => 'in RCDATA:#'.$token->{type});  
           ## ISSUE: And ignore?  
         }  
         !!!next-token;  
2518          return;          return;
2519        } elsif ($token->{tag_name} eq 'body') {        } elsif ($token->{tag_name} eq 'body') {
2520          !!!parse-error (type => 'in body:body');          !!!parse-error (type => 'in body:body');
# Line 2552  sub _tree_construction_main ($) { Line 2617  sub _tree_construction_main ($) {
2617              if ($i != -1) {              if ($i != -1) {
2618                !!!parse-error (type => 'end tag missing:'.                !!!parse-error (type => 'end tag missing:'.
2619                                $self->{open_elements}->[-1]->[1]);                                $self->{open_elements}->[-1]->[1]);
               ## TODO: test  
2620              }              }
2621              splice @{$self->{open_elements}}, $i;              splice @{$self->{open_elements}}, $i;
2622              last LI;              last LI;
# Line 2600  sub _tree_construction_main ($) { Line 2664  sub _tree_construction_main ($) {
2664              if ($i != -1) {              if ($i != -1) {
2665                !!!parse-error (type => 'end tag missing:'.                !!!parse-error (type => 'end tag missing:'.
2666                                $self->{open_elements}->[-1]->[1]);                                $self->{open_elements}->[-1]->[1]);
               ## TODO: test  
2667              }              }
2668              splice @{$self->{open_elements}}, $i;              splice @{$self->{open_elements}}, $i;
2669              last LI;              last LI;
# Line 2663  sub _tree_construction_main ($) { Line 2726  sub _tree_construction_main ($) {
2726            }            }
2727          } # INSCOPE          } # INSCOPE
2728                        
2729            ## NOTE: See <http://html5.org/tools/web-apps-tracker?from=925&to=926>
2730          ## has an element in scope          ## has an element in scope
2731          my $i;          #my $i;
2732          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          #INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
2733            my $node = $self->{open_elements}->[$_];          #  my $node = $self->{open_elements}->[$_];
2734            if ({          #  if ({
2735                 h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,          #       h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
2736                }->{$node->[1]}) {          #      }->{$node->[1]}) {
2737              $i = $_;          #    $i = $_;
2738              last INSCOPE;          #    last INSCOPE;
2739            } elsif ({          #  } elsif ({
2740                      table => 1, caption => 1, td => 1, th => 1,          #            table => 1, caption => 1, td => 1, th => 1,
2741                      button => 1, marquee => 1, object => 1, html => 1,          #            button => 1, marquee => 1, object => 1, html => 1,
2742                     }->{$node->[1]}) {          #           }->{$node->[1]}) {
2743              last INSCOPE;          #    last INSCOPE;
2744            }          #  }
2745          } # INSCOPE          #} # INSCOPE
2746                      #  
2747          if (defined $i) {          #if (defined $i) {
2748            !!!parse-error (type => 'in hn:hn');          #  !!! parse-error (type => 'in hn:hn');
2749            splice @{$self->{open_elements}}, $i;          #  splice @{$self->{open_elements}}, $i;
2750          }          #}
2751                        
2752          !!!insert-element-t ($token->{tag_name}, $token->{attributes});          !!!insert-element-t ($token->{tag_name}, $token->{attributes});
2753                        
# Line 2743  sub _tree_construction_main ($) { Line 2807  sub _tree_construction_main ($) {
2807          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
2808            my $node = $self->{open_elements}->[$_];            my $node = $self->{open_elements}->[$_];
2809            if ($node->[1] eq 'nobr') {            if ($node->[1] eq 'nobr') {
2810                !!!parse-error (type => 'not closed:nobr');
2811              !!!back-token;              !!!back-token;
2812              $token = {type => 'end tag', tag_name => 'nobr'};              $token = {type => 'end tag', tag_name => 'nobr'};
2813              return;              return;
# Line 2794  sub _tree_construction_main ($) { Line 2859  sub _tree_construction_main ($) {
2859          return;          return;
2860        } elsif ($token->{tag_name} eq 'xmp') {        } elsif ($token->{tag_name} eq 'xmp') {
2861          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
2862                    $parse_rcdata->('CDATA', $insert);
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
           
         $self->{content_model_flag} = 'CDATA';  
         delete $self->{escape}; # MUST  
           
         !!!next-token;  
2863          return;          return;
2864        } elsif ($token->{tag_name} eq 'table') {        } elsif ($token->{tag_name} eq 'table') {
2865          ## has a p element in scope          ## has a p element in scope
# Line 2832  sub _tree_construction_main ($) { Line 2891  sub _tree_construction_main ($) {
2891            !!!parse-error (type => 'image');            !!!parse-error (type => 'image');
2892            $token->{tag_name} = 'img';            $token->{tag_name} = 'img';
2893          }          }
2894            
2895            ## NOTE: There is an "as if <br>" code clone.
2896          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
2897                    
2898          !!!insert-element-t ($token->{tag_name}, $token->{attributes});          !!!insert-element-t ($token->{tag_name}, $token->{attributes});
# Line 2878  sub _tree_construction_main ($) { Line 2938  sub _tree_construction_main ($) {
2938            return;            return;
2939          } else {          } else {
2940            my $at = $token->{attributes};            my $at = $token->{attributes};
2941              my $form_attrs;
2942              $form_attrs->{action} = $at->{action} if $at->{action};
2943              my $prompt_attr = $at->{prompt};
2944            $at->{name} = {name => 'name', value => 'isindex'};            $at->{name} = {name => 'name', value => 'isindex'};
2945              delete $at->{action};
2946              delete $at->{prompt};
2947            my @tokens = (            my @tokens = (
2948                          {type => 'start tag', tag_name => 'form'},                          {type => 'start tag', tag_name => 'form',
2949                             attributes => $form_attrs},
2950                          {type => 'start tag', tag_name => 'hr'},                          {type => 'start tag', tag_name => 'hr'},
2951                          {type => 'start tag', tag_name => 'p'},                          {type => 'start tag', tag_name => 'p'},
2952                          {type => 'start tag', tag_name => 'label'},                          {type => 'start tag', tag_name => 'label'},
2953                          {type => 'character',                         );
2954                           data => 'This is a searchable index. Insert your search keywords here: '}, # SHOULD            if ($prompt_attr) {
2955                          ## TODO: make this configurable              push @tokens, {type => 'character', data => $prompt_attr->{value}};
2956              } else {
2957                push @tokens, {type => 'character',
2958                               data => 'This is a searchable index. Insert your search keywords here: '}; # SHOULD
2959                ## TODO: make this configurable
2960              }
2961              push @tokens,
2962                          {type => 'start tag', tag_name => 'input', attributes => $at},                          {type => 'start tag', tag_name => 'input', attributes => $at},
2963                          #{type => 'character', data => ''}, # SHOULD                          #{type => 'character', data => ''}, # SHOULD
2964                          {type => 'end tag', tag_name => 'label'},                          {type => 'end tag', tag_name => 'label'},
2965                          {type => 'end tag', tag_name => 'p'},                          {type => 'end tag', tag_name => 'p'},
2966                          {type => 'start tag', tag_name => 'hr'},                          {type => 'start tag', tag_name => 'hr'},
2967                          {type => 'end tag', tag_name => 'form'},                          {type => 'end tag', tag_name => 'form'};
                        );  
2968            $token = shift @tokens;            $token = shift @tokens;
2969            !!!back-token (@tokens);            !!!back-token (@tokens);
2970            return;            return;
2971          }          }
2972        } elsif ({        } elsif ($token->{tag_name} eq 'textarea') {
                 textarea => 1,  
                 iframe => 1,  
                 noembed => 1,  
                 noframes => 1,  
                 noscript => 0, ## TODO: 1 if scripting is enabled  
                }->{$token->{tag_name}}) {  
2973          my $tag_name = $token->{tag_name};          my $tag_name = $token->{tag_name};
2974          my $el;          my $el;
2975          !!!create-element ($el, $token->{tag_name}, $token->{attributes});          !!!create-element ($el, $token->{tag_name}, $token->{attributes});
2976                    
2977          if ($token->{tag_name} eq 'textarea') {          ## TODO: $self->{form_element} if defined
2978            ## TODO: $self->{form_element} if defined          $self->{content_model_flag} = 'RCDATA';
           $self->{content_model_flag} = 'RCDATA';  
         } else {  
           $self->{content_model_flag} = 'CDATA';  
         }  
2979          delete $self->{escape}; # MUST          delete $self->{escape}; # MUST
2980                    
2981          $insert->($el);          $insert->($el);
2982                    
2983          my $text = '';          my $text = '';
2984          if ($token->{tag_name} eq 'textarea') {          !!!next-token;
2985            !!!next-token;          if ($token->{type} eq 'character') {
2986            if ($token->{type} eq 'character') {            $token->{data} =~ s/^\x0A//;
2987              $token->{data} =~ s/^\x0A//;            unless (length $token->{data}) {
2988              unless (length $token->{data}) {              !!!next-token;
               !!!next-token;  
             }  
2989            }            }
         } else {  
           !!!next-token;  
2990          }          }
2991          while ($token->{type} eq 'character') {          while ($token->{type} eq 'character') {
2992            $text .= $token->{data};            $text .= $token->{data};
# Line 2945  sub _tree_construction_main ($) { Line 3002  sub _tree_construction_main ($) {
3002              $token->{tag_name} eq $tag_name) {              $token->{tag_name} eq $tag_name) {
3003            ## Ignore the token            ## Ignore the token
3004          } else {          } else {
3005            if ($token->{tag_name} eq 'textarea') {            !!!parse-error (type => 'in RCDATA:#'.$token->{type});
             !!!parse-error (type => 'in RCDATA:#'.$token->{type});  
           } else {  
             !!!parse-error (type => 'in CDATA:#'.$token->{type});  
           }  
           ## ISSUE: And ignore?  
3006          }          }
3007          !!!next-token;          !!!next-token;
3008          return;          return;
3009          } elsif ({
3010                    iframe => 1,
3011                    noembed => 1,
3012                    noframes => 1,
3013                    noscript => 0, ## TODO: 1 if scripting is enabled
3014                   }->{$token->{tag_name}}) {
3015            $parse_rcdata->('CDATA', $insert);
3016            return;
3017        } elsif ($token->{tag_name} eq 'select') {        } elsif ($token->{tag_name} eq 'select') {
3018          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
3019                    
# Line 2990  sub _tree_construction_main ($) { Line 3050  sub _tree_construction_main ($) {
3050              unless ({              unless ({
3051                         dd => 1, dt => 1, li => 1, p => 1, td => 1,                         dd => 1, dt => 1, li => 1, p => 1, td => 1,
3052                         th => 1, tr => 1, body => 1, html => 1,                         th => 1, tr => 1, body => 1, html => 1,
3053                         tbody => 1, tfoot => 1, thead => 1,
3054                      }->{$_->[1]}) {                      }->{$_->[1]}) {
3055                !!!parse-error (type => 'not closed:'.$_->[1]);                !!!parse-error (type => 'not closed:'.$_->[1]);
3056              }              }
# Line 3039  sub _tree_construction_main ($) { Line 3100  sub _tree_construction_main ($) {
3100                   li => ($token->{tag_name} ne 'li'),                   li => ($token->{tag_name} ne 'li'),
3101                   p => ($token->{tag_name} ne 'p'),                   p => ($token->{tag_name} ne 'p'),
3102                   td => 1, th => 1, tr => 1,                   td => 1, th => 1, tr => 1,
3103                     tbody => 1, tfoot=> 1, thead => 1,
3104                  }->{$self->{open_elements}->[-1]->[1]}) {                  }->{$self->{open_elements}->[-1]->[1]}) {
3105                !!!back-token;                !!!back-token;
3106                $token = {type => 'end tag',                $token = {type => 'end tag',
# Line 3056  sub _tree_construction_main ($) { Line 3118  sub _tree_construction_main ($) {
3118          } # INSCOPE          } # INSCOPE
3119                    
3120          if ($self->{open_elements}->[-1]->[1] ne $token->{tag_name}) {          if ($self->{open_elements}->[-1]->[1] ne $token->{tag_name}) {
3121            !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);            if (defined $i) {
3122                !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);
3123              } else {
3124                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3125              }
3126          }          }
3127                    
3128          splice @{$self->{open_elements}}, $i if defined $i;          if (defined $i) {
3129              splice @{$self->{open_elements}}, $i;
3130            } elsif ($token->{tag_name} eq 'p') {
3131              ## As if <p>, then reprocess the current token
3132              my $el;
3133              !!!create-element ($el, 'p');
3134              $insert->($el);
3135            }
3136          $clear_up_to_marker->()          $clear_up_to_marker->()
3137            if {            if {
3138              button => 1, marquee => 1, object => 1,              button => 1, marquee => 1, object => 1,
# Line 3075  sub _tree_construction_main ($) { Line 3148  sub _tree_construction_main ($) {
3148              if ({              if ({
3149                   dd => 1, dt => 1, li => 1, p => 1,                   dd => 1, dt => 1, li => 1, p => 1,
3150                   td => 1, th => 1, tr => 1,                   td => 1, th => 1, tr => 1,
3151                     tbody => 1, tfoot=> 1, thead => 1,
3152                  }->{$self->{open_elements}->[-1]->[1]}) {                  }->{$self->{open_elements}->[-1]->[1]}) {
3153                !!!back-token;                !!!back-token;
3154                $token = {type => 'end tag',                $token = {type => 'end tag',
# Line 3113  sub _tree_construction_main ($) { Line 3187  sub _tree_construction_main ($) {
3187              if ({              if ({
3188                   dd => 1, dt => 1, li => 1, p => 1,                   dd => 1, dt => 1, li => 1, p => 1,
3189                   td => 1, th => 1, tr => 1,                   td => 1, th => 1, tr => 1,
3190                     tbody => 1, tfoot=> 1, thead => 1,
3191                  }->{$self->{open_elements}->[-1]->[1]}) {                  }->{$self->{open_elements}->[-1]->[1]}) {
3192                !!!back-token;                !!!back-token;
3193                $token = {type => 'end tag',                $token = {type => 'end tag',
# Line 3143  sub _tree_construction_main ($) { Line 3218  sub _tree_construction_main ($) {
3218                  strong => 1, tt => 1, u => 1,                  strong => 1, tt => 1, u => 1,
3219                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
3220          $formatting_end_tag->($token->{tag_name});          $formatting_end_tag->($token->{tag_name});
3221  ## TODO: <http://html5.org/tools/web-apps-tracker?from=883&to=884>          return;
3222          } elsif ($token->{tag_name} eq 'br') {
3223            !!!parse-error (type => 'unmatched end tag:br');
3224    
3225            ## As if <br>
3226            $reconstruct_active_formatting_elements->($insert_to_current);
3227            
3228            my $el;
3229            !!!create-element ($el, 'br');
3230            $insert->($el);
3231            
3232            ## Ignore the token.
3233            !!!next-token;
3234          return;          return;
3235        } elsif ({        } elsif ({
3236                  caption => 1, col => 1, colgroup => 1, frame => 1,                  caption => 1, col => 1, colgroup => 1, frame => 1,
3237                  frameset => 1, head => 1, option => 1, optgroup => 1,                  frameset => 1, head => 1, option => 1, optgroup => 1,
3238                  tbody => 1, td => 1, tfoot => 1, th => 1,                  tbody => 1, td => 1, tfoot => 1, th => 1,
3239                  thead => 1, tr => 1,                  thead => 1, tr => 1,
3240                  area => 1, basefont => 1, bgsound => 1, br => 1,                  area => 1, basefont => 1, bgsound => 1,
3241                  embed => 1, hr => 1, iframe => 1, image => 1,                  embed => 1, hr => 1, iframe => 1, image => 1,
3242                  img => 1, input => 1, isindex => 1, noembed => 1,                  img => 1, input => 1, isindex => 1, noembed => 1,
3243                  noframes => 1, param => 1, select => 1, spacer => 1,                  noframes => 1, param => 1, select => 1, spacer => 1,
# Line 3177  sub _tree_construction_main ($) { Line 3264  sub _tree_construction_main ($) {
3264              if ({              if ({
3265                   dd => 1, dt => 1, li => 1, p => 1,                   dd => 1, dt => 1, li => 1, p => 1,
3266                   td => 1, th => 1, tr => 1,                   td => 1, th => 1, tr => 1,
3267                     tbody => 1, tfoot=> 1, thead => 1,
3268                  }->{$self->{open_elements}->[-1]->[1]}) {                  }->{$self->{open_elements}->[-1]->[1]}) {
3269                !!!back-token;                !!!back-token;
3270                $token = {type => 'end tag',                $token = {type => 'end tag',
# Line 3200  sub _tree_construction_main ($) { Line 3288  sub _tree_construction_main ($) {
3288                  #not $phrasing_category->{$node->[1]} and                  #not $phrasing_category->{$node->[1]} and
3289                  ($special_category->{$node->[1]} or                  ($special_category->{$node->[1]} or
3290                   $scoping_category->{$node->[1]})) {                   $scoping_category->{$node->[1]})) {
3291                !!!parse-error (type => 'not closed:'.$node->[1]);                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3292                ## Ignore the token                ## Ignore the token
3293                !!!next-token;                !!!next-token;
3294                last S2;                last S2;
# Line 3229  sub _tree_construction_main ($) { Line 3317  sub _tree_construction_main ($) {
3317          redo B;          redo B;
3318        } elsif ($token->{type} eq 'start tag' and        } elsif ($token->{type} eq 'start tag' and
3319                 $token->{tag_name} eq 'html') {                 $token->{tag_name} eq 'html') {
3320          ## TODO: unless it is the first start tag token, parse-error  ## ISSUE: "aa<html>" is not a parse error.
3321    ## ISSUE: "<html>" in fragment is not a parse error.
3322            unless ($token->{first_start_tag}) {
3323              !!!parse-error (type => 'not first start tag');
3324            }
3325          my $top_el = $self->{open_elements}->[0]->[0];          my $top_el = $self->{open_elements}->[0]->[0];
3326          for my $attr_name (keys %{$token->{attributes}}) {          for my $attr_name (keys %{$token->{attributes}}) {
3327            unless ($top_el->has_attribute_ns (undef, $attr_name)) {            unless ($top_el->has_attribute_ns (undef, $attr_name)) {
# Line 3244  sub _tree_construction_main ($) { Line 3336  sub _tree_construction_main ($) {
3336          ## Generate implied end tags          ## Generate implied end tags
3337          if ({          if ({
3338               dd => 1, dt => 1, li => 1, p => 1, td => 1, th => 1, tr => 1,               dd => 1, dt => 1, li => 1, p => 1, td => 1, th => 1, tr => 1,
3339                 tbody => 1, tfoot=> 1, thead => 1,
3340              }->{$self->{open_elements}->[-1]->[1]}) {              }->{$self->{open_elements}->[-1]->[1]}) {
3341            !!!back-token;            !!!back-token;
3342            $token = {type => 'end tag', tag_name => $self->{open_elements}->[-1]->[1]};            $token = {type => 'end tag', tag_name => $self->{open_elements}->[-1]->[1]};
# Line 3303  sub _tree_construction_main ($) { Line 3396  sub _tree_construction_main ($) {
3396              }              }
3397              redo B;              redo B;
3398            } elsif ($token->{type} eq 'end tag') {            } elsif ($token->{type} eq 'end tag') {
3399              if ($token->{tag_name} eq 'html') {              if ({
3400                     head => 1, body => 1, html => 1,
3401                     p => 1, br => 1,
3402                    }->{$token->{tag_name}}) {
3403                ## As if <head>                ## As if <head>
3404                !!!create-element ($self->{head_element}, 'head');                !!!create-element ($self->{head_element}, 'head');
3405                $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});                $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
# Line 3313  sub _tree_construction_main ($) { Line 3409  sub _tree_construction_main ($) {
3409                redo B;                redo B;
3410              } else {              } else {
3411                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3412                ## Ignore the token                ## Ignore the token ## ISSUE: An issue in the spec.
3413                !!!next-token;                !!!next-token;
3414                redo B;                redo B;
3415              }              }
3416            } else {            } else {
3417              die "$0: $token->{type}: Unknown type";              die "$0: $token->{type}: Unknown type";
3418            }            }
3419          } elsif ($self->{insertion_mode} eq 'in head') {          } elsif ($self->{insertion_mode} eq 'in head' or
3420                     $self->{insertion_mode} eq 'in head noscript' or
3421                     $self->{insertion_mode} eq 'after head') {
3422            if ($token->{type} eq 'character') {            if ($token->{type} eq 'character') {
3423              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {
3424                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
# Line 3337  sub _tree_construction_main ($) { Line 3435  sub _tree_construction_main ($) {
3435              !!!next-token;              !!!next-token;
3436              redo B;              redo B;
3437            } elsif ($token->{type} eq 'start tag') {            } elsif ($token->{type} eq 'start tag') {
3438              if ($token->{tag_name} eq 'title') {              if ({base => ($self->{insertion_mode} eq 'in head' or
3439                ## NOTE: There is an "as if in head" code clone                            $self->{insertion_mode} eq 'after head'),
3440                my $title_el;                   link => 1, meta => 1}->{$token->{tag_name}}) {
3441                !!!create-element ($title_el, 'title', $token->{attributes});                ## NOTE: There is a "as if in head" code clone.
3442                (defined $self->{head_element} ? $self->{head_element} : $self->{open_elements}->[-1]->[0])                if ($self->{insertion_mode} eq 'after head') {
3443                  ->append_child ($title_el);                  !!!parse-error (type => 'after head:'.$token->{tag_name});
3444                $self->{content_model_flag} = 'RCDATA';                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
3445                delete $self->{escape}; # MUST                }
3446                  !!!insert-element ($token->{tag_name}, $token->{attributes});
3447                my $text = '';                pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
3448                  ## TODO: Extracting |charset| from |meta|.
3449                  pop @{$self->{open_elements}}
3450                      if $self->{insertion_mode} eq 'after head';
3451                !!!next-token;                !!!next-token;
3452                while ($token->{type} eq 'character') {                redo B;
3453                  $text .= $token->{data};              } elsif ($token->{tag_name} eq 'title' and
3454                         $self->{insertion_mode} eq 'in head') {
3455                  ## NOTE: There is a "as if in head" code clone.
3456                  if ($self->{insertion_mode} eq 'after head') {
3457                    !!!parse-error (type => 'after head:'.$token->{tag_name});
3458                    push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
3459                  }
3460                  my $parent = defined $self->{head_element} ? $self->{head_element}
3461                      : $self->{open_elements}->[-1]->[0];
3462                  $parse_rcdata->('RCDATA', sub { $parent->append_child ($_[0]) });
3463                  pop @{$self->{open_elements}}
3464                      if $self->{insertion_mode} eq 'after head';
3465                  redo B;
3466                } elsif ($token->{tag_name} eq 'style') {
3467                  ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and
3468                  ## insertion mode 'in head')
3469                  ## NOTE: There is a "as if in head" code clone.
3470                  if ($self->{insertion_mode} eq 'after head') {
3471                    !!!parse-error (type => 'after head:'.$token->{tag_name});
3472                    push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
3473                  }
3474                  $parse_rcdata->('CDATA', $insert_to_current);
3475                  pop @{$self->{open_elements}}
3476                      if $self->{insertion_mode} eq 'after head';
3477                  redo B;
3478                } elsif ($token->{tag_name} eq 'noscript') {
3479                  if ($self->{insertion_mode} eq 'in head') {
3480                    ## NOTE: and scripting is disalbed
3481                    !!!insert-element ($token->{tag_name}, $token->{attributes});
3482                    $self->{insertion_mode} = 'in head noscript';
3483                  !!!next-token;                  !!!next-token;
3484                }                  redo B;
3485                if (length $text) {                } elsif ($self->{insertion_mode} eq 'in head noscript') {
3486                  $title_el->manakai_append_text ($text);                  !!!parse-error (type => 'in noscript:noscript');
               }  
                 
               $self->{content_model_flag} = 'PCDATA';  
                 
               if ($token->{type} eq 'end tag' and  
                   $token->{tag_name} eq 'title') {  
3487                  ## Ignore the token                  ## Ignore the token
3488                    redo B;
3489                } else {                } else {
3490                  !!!parse-error (type => 'in RCDATA:#'.$token->{type});                  #
                 ## ISSUE: And ignore?  
3491                }                }
3492                } elsif ($token->{tag_name} eq 'head' and
3493                         $self->{insertion_mode} ne 'after head') {
3494                  !!!parse-error (type => 'in head:head'); # or in head noscript
3495                  ## Ignore the token
3496                !!!next-token;                !!!next-token;
3497                redo B;                redo B;
3498              } elsif ($token->{tag_name} eq 'style') {              } elsif ($self->{insertion_mode} ne 'in head noscript' and
3499                $style_start_tag->();                       $token->{tag_name} eq 'script') {
3500                redo B;                if ($self->{insertion_mode} eq 'after head') {
3501              } elsif ($token->{tag_name} eq 'script') {                  !!!parse-error (type => 'after head:'.$token->{tag_name});
3502                $script_start_tag->();                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
3503                  }
3504                  ## NOTE: There is a "as if in head" code clone.
3505                  $script_start_tag->($insert_to_current);
3506                  pop @{$self->{open_elements}}
3507                      if $self->{insertion_mode} eq 'after head';
3508                redo B;                redo B;
3509              } elsif ({base => 1, link => 1, meta => 1}->{$token->{tag_name}}) {              } elsif ($self->{insertion_mode} eq 'after head' and
3510                ## NOTE: There are "as if in head" code clones                       $token->{tag_name} eq 'body') {
3511                my $el;                !!!insert-element ('body', $token->{attributes});
3512                !!!create-element ($el, $token->{tag_name}, $token->{attributes});                $self->{insertion_mode} = 'in body';
               (defined $self->{head_element} ? $self->{head_element} : $self->{open_elements}->[-1]->[0])  
                 ->append_child ($el);  
   
3513                !!!next-token;                !!!next-token;
3514                redo B;                redo B;
3515              } elsif ($token->{tag_name} eq 'head') {              } elsif ($self->{insertion_mode} eq 'after head' and
3516                !!!parse-error (type => 'in head:head');                       $token->{tag_name} eq 'frameset') {
3517                ## Ignore the token                !!!insert-element ('frameset', $token->{attributes});
3518                  $self->{insertion_mode} = 'in frameset';
3519                !!!next-token;                !!!next-token;
3520                redo B;                redo B;
3521              } else {              } else {
3522                #                #
3523              }              }
3524            } elsif ($token->{type} eq 'end tag') {            } elsif ($token->{type} eq 'end tag') {
3525              if ($token->{tag_name} eq 'head') {              if ($self->{insertion_mode} eq 'in head' and
3526                if ($self->{open_elements}->[-1]->[1] eq 'head') {                  $token->{tag_name} eq 'head') {
3527                  pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
               } else {  
                 !!!parse-error (type => 'unmatched end tag:head');  
               }  
3528                $self->{insertion_mode} = 'after head';                $self->{insertion_mode} = 'after head';
3529                !!!next-token;                !!!next-token;
3530                redo B;                redo B;
3531              } elsif ($token->{tag_name} eq 'html') {              } elsif ($self->{insertion_mode} eq 'in head noscript' and
3532                    $token->{tag_name} eq 'noscript') {
3533                  pop @{$self->{open_elements}};
3534                  $self->{insertion_mode} = 'in head';
3535                  !!!next-token;
3536                  redo B;
3537                } elsif ($self->{insertion_mode} eq 'in head' and
3538                         {
3539                          body => 1, html => 1,
3540                          p => 1, br => 1,
3541                         }->{$token->{tag_name}}) {
3542                #                #
3543              } else {              } elsif ($self->{insertion_mode} eq 'in head noscript' and
3544                         {
3545                          p => 1, br => 1,
3546                         }->{$token->{tag_name}}) {
3547                  #
3548                } elsif ($self->{insertion_mode} ne 'after head') {
3549                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3550                ## Ignore the token                ## Ignore the token
3551                !!!next-token;                !!!next-token;
3552                redo B;                redo B;
3553                } else {
3554                  #
3555              }              }
3556            } else {            } else {
3557              #              #
3558            }            }
3559    
3560            if ($self->{open_elements}->[-1]->[1] eq 'head') {            ## As if </head> or </noscript> or <body>
3561              ## As if </head>            if ($self->{insertion_mode} eq 'in head') {
3562                pop @{$self->{open_elements}};
3563                $self->{insertion_mode} = 'after head';
3564              } elsif ($self->{insertion_mode} eq 'in head noscript') {
3565              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
3566                !!!parse-error (type => 'in noscript:'.(defined $token->{tag_name} ? ($token->{type} eq 'end tag' ? '/' : '') . $token->{tag_name} : '#' . $token->{type}));
3567                $self->{insertion_mode} = 'in head';
3568              } else { # 'after head'
3569                !!!insert-element ('body');
3570                $self->{insertion_mode} = 'in body';
3571            }            }
           $self->{insertion_mode} = 'after head';  
3572            ## reprocess            ## reprocess
3573            redo B;            redo B;
3574    
3575            ## ISSUE: An issue in the spec.            ## ISSUE: An issue in the spec.
         } elsif ($self->{insertion_mode} eq 'after head') {  
           if ($token->{type} eq 'character') {  
             if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {  
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);  
               unless (length $token->{data}) {  
                 !!!next-token;  
                 redo B;  
               }  
             }  
               
             #  
           } elsif ($token->{type} eq 'comment') {  
             my $comment = $self->{document}->create_comment ($token->{data});  
             $self->{open_elements}->[-1]->[0]->append_child ($comment);  
             !!!next-token;  
             redo B;  
           } elsif ($token->{type} eq 'start tag') {  
             if ($token->{tag_name} eq 'body') {  
               !!!insert-element ('body', $token->{attributes});  
               $self->{insertion_mode} = 'in body';  
               !!!next-token;  
               redo B;  
             } elsif ($token->{tag_name} eq 'frameset') {  
               !!!insert-element ('frameset', $token->{attributes});  
               $self->{insertion_mode} = 'in frameset';  
               !!!next-token;  
               redo B;  
             } elsif ({  
                       base => 1, link => 1, meta => 1,  
                       script => 1, style => 1, title => 1,  
                      }->{$token->{tag_name}}) {  
               !!!parse-error (type => 'after head:'.$token->{tag_name});  
               $self->{insertion_mode} = 'in head';  
               ## reprocess  
               redo B;  
             } else {  
               #  
             }  
           } else {  
             #  
           }  
             
           ## As if <body>  
           !!!insert-element ('body');  
           $self->{insertion_mode} = 'in body';  
           ## reprocess  
           redo B;  
3576          } elsif ($self->{insertion_mode} eq 'in body') {          } elsif ($self->{insertion_mode} eq 'in body') {
3577            if ($token->{type} eq 'character') {            if ($token->{type} eq 'character') {
3578              ## NOTE: There is a code clone of "character in body".              ## NOTE: There is a code clone of "character in body".
# Line 3622  sub _tree_construction_main ($) { Line 3727  sub _tree_construction_main ($) {
3727                if ({                if ({
3728                     dd => 1, dt => 1, li => 1, p => 1,                     dd => 1, dt => 1, li => 1, p => 1,
3729                     td => 1, th => 1, tr => 1,                     td => 1, th => 1, tr => 1,
3730                       tbody => 1, tfoot=> 1, thead => 1,
3731                    }->{$self->{open_elements}->[-1]->[1]}) {                    }->{$self->{open_elements}->[-1]->[1]}) {
3732                  !!!back-token; # <table>                  !!!back-token; # <table>
3733                  $token = {type => 'end tag', tag_name => 'table'};                  $token = {type => 'end tag', tag_name => 'table'};
# Line 3670  sub _tree_construction_main ($) { Line 3776  sub _tree_construction_main ($) {
3776                if ({                if ({
3777                     dd => 1, dt => 1, li => 1, p => 1,                     dd => 1, dt => 1, li => 1, p => 1,
3778                     td => 1, th => 1, tr => 1,                     td => 1, th => 1, tr => 1,
3779                       tbody => 1, tfoot=> 1, thead => 1,
3780                    }->{$self->{open_elements}->[-1]->[1]}) {                    }->{$self->{open_elements}->[-1]->[1]}) {
3781                  !!!back-token;                  !!!back-token;
3782                  $token = {type => 'end tag',                  $token = {type => 'end tag',
# Line 3753  sub _tree_construction_main ($) { Line 3860  sub _tree_construction_main ($) {
3860                if ({                if ({
3861                     dd => 1, dt => 1, li => 1, p => 1,                     dd => 1, dt => 1, li => 1, p => 1,
3862                     td => 1, th => 1, tr => 1,                     td => 1, th => 1, tr => 1,
3863                       tbody => 1, tfoot=> 1, thead => 1,
3864                    }->{$self->{open_elements}->[-1]->[1]}) {                    }->{$self->{open_elements}->[-1]->[1]}) {
3865                  !!!back-token; # <?>                  !!!back-token; # <?>
3866                  $token = {type => 'end tag', tag_name => 'caption'};                  $token = {type => 'end tag', tag_name => 'caption'};
# Line 3803  sub _tree_construction_main ($) { Line 3911  sub _tree_construction_main ($) {
3911                if ({                if ({
3912                     dd => 1, dt => 1, li => 1, p => 1,                     dd => 1, dt => 1, li => 1, p => 1,
3913                     td => 1, th => 1, tr => 1,                     td => 1, th => 1, tr => 1,
3914                       tbody => 1, tfoot=> 1, thead => 1,
3915                    }->{$self->{open_elements}->[-1]->[1]}) {                    }->{$self->{open_elements}->[-1]->[1]}) {
3916                  !!!back-token;                  !!!back-token;
3917                  $token = {type => 'end tag',                  $token = {type => 'end tag',
# Line 3850  sub _tree_construction_main ($) { Line 3959  sub _tree_construction_main ($) {
3959                if ({                if ({
3960                     dd => 1, dt => 1, li => 1, p => 1,                     dd => 1, dt => 1, li => 1, p => 1,
3961                     td => 1, th => 1, tr => 1,                     td => 1, th => 1, tr => 1,
3962                       tbody => 1, tfoot=> 1, thead => 1,
3963                    }->{$self->{open_elements}->[-1]->[1]}) {                    }->{$self->{open_elements}->[-1]->[1]}) {
3964                  !!!back-token; # </table>                  !!!back-token; # </table>
3965                  $token = {type => 'end tag', tag_name => 'caption'};                  $token = {type => 'end tag', tag_name => 'caption'};
# Line 4115  sub _tree_construction_main ($) { Line 4225  sub _tree_construction_main ($) {
4225                if ({                if ({
4226                     dd => 1, dt => 1, li => 1, p => 1,                     dd => 1, dt => 1, li => 1, p => 1,
4227                     td => 1, th => 1, tr => 1,                     td => 1, th => 1, tr => 1,
4228                       tbody => 1, tfoot=> 1, thead => 1,
4229                    }->{$self->{open_elements}->[-1]->[1]}) {                    }->{$self->{open_elements}->[-1]->[1]}) {
4230                  !!!back-token; # <table>                  !!!back-token; # <table>
4231                  $token = {type => 'end tag', tag_name => 'table'};                  $token = {type => 'end tag', tag_name => 'table'};
# Line 4383  sub _tree_construction_main ($) { Line 4494  sub _tree_construction_main ($) {
4494                if ({                if ({
4495                     dd => 1, dt => 1, li => 1, p => 1,                     dd => 1, dt => 1, li => 1, p => 1,
4496                     td => 1, th => 1, tr => 1,                     td => 1, th => 1, tr => 1,
4497                       tbody => 1, tfoot=> 1, thead => 1,
4498                    }->{$self->{open_elements}->[-1]->[1]}) {                    }->{$self->{open_elements}->[-1]->[1]}) {
4499                  !!!back-token; # <table>                  !!!back-token; # <table>
4500                  $token = {type => 'end tag', tag_name => 'table'};                  $token = {type => 'end tag', tag_name => 'table'};
# Line 4624  sub _tree_construction_main ($) { Line 4736  sub _tree_construction_main ($) {
4736                     td => ($token->{tag_name} eq 'th'),                     td => ($token->{tag_name} eq 'th'),
4737                     th => ($token->{tag_name} eq 'td'),                     th => ($token->{tag_name} eq 'td'),
4738                     tr => 1,                     tr => 1,
4739                       tbody => 1, tfoot=> 1, thead => 1,
4740                    }->{$self->{open_elements}->[-1]->[1]}) {                    }->{$self->{open_elements}->[-1]->[1]}) {
4741                  !!!back-token;                  !!!back-token;
4742                  $token = {type => 'end tag',                  $token = {type => 'end tag',
# Line 4976  sub _tree_construction_main ($) { Line 5089  sub _tree_construction_main ($) {
5089            }            }
5090                        
5091            if (defined $token->{tag_name}) {            if (defined $token->{tag_name}) {
5092              !!!parse-error (type => 'in frameset:'.$token->{tag_name});              !!!parse-error (type => 'in frameset:'.($token->{type} eq 'end tag' ? '/' : '').$token->{tag_name});
5093            } else {            } else {
5094              !!!parse-error (type => 'in frameset:#'.$token->{type});              !!!parse-error (type => 'in frameset:#'.$token->{type});
5095            }            }
# Line 5020  sub _tree_construction_main ($) { Line 5133  sub _tree_construction_main ($) {
5133            }            }
5134                        
5135            if (defined $token->{tag_name}) {            if (defined $token->{tag_name}) {
5136              !!!parse-error (type => 'after frameset:'.$token->{tag_name});              !!!parse-error (type => 'after frameset:'.($token->{tag_name} eq 'end tag' ? '/' : '').$token->{tag_name});
5137            } else {            } else {
5138              !!!parse-error (type => 'after frameset:#'.$token->{type});              !!!parse-error (type => 'after frameset:#'.$token->{type});
5139            }            }
# Line 5070  sub _tree_construction_main ($) { Line 5183  sub _tree_construction_main ($) {
5183          redo B;          redo B;
5184        } elsif ($token->{type} eq 'start tag' or        } elsif ($token->{type} eq 'start tag' or
5185                 $token->{type} eq 'end tag') {                 $token->{type} eq 'end tag') {
5186          !!!parse-error (type => 'after html:'.$token->{tag_name});          !!!parse-error (type => 'after html:'.($token->{type} eq 'end tag' ? '/' : '').$token->{tag_name});
5187          $phase = 'main';          $phase = 'main';
5188          ## reprocess          ## reprocess
5189          redo B;          redo B;
# Line 5279  sub get_inner_html ($$$) { Line 5392  sub get_inner_html ($$$) {
5392            
5393      my $nt = $child->node_type;      my $nt = $child->node_type;
5394      if ($nt == 1) { # Element      if ($nt == 1) { # Element
5395        my $tag_name = lc $child->tag_name; ## ISSUE: Definition of "lowercase"        my $tag_name = $child->tag_name; ## TODO: manakai_tag_name
5396        $s .= '<' . $tag_name;        $s .= '<' . $tag_name;
5397          ## NOTE: Non-HTML case:
5398        ## ISSUE: Non-html elements        ## <http://permalink.gmane.org/gmane.org.w3c.whatwg.discuss/11191>
5399    
5400        my @attrs = @{$child->attributes}; # sort order MUST be stable        my @attrs = @{$child->attributes}; # sort order MUST be stable
5401        for my $attr (@attrs) { # order is implementation dependent        for my $attr (@attrs) { # order is implementation dependent
5402          my $attr_name = lc $attr->name; ## ISSUE: Definition of "lowercase"          my $attr_name = $attr->name; ## TODO: manakai_name
5403          $s .= ' ' . $attr_name . '="';          $s .= ' ' . $attr_name . '="';
5404          my $attr_value = $attr->value;          my $attr_value = $attr->value;
5405          ## escape          ## escape
# Line 5305  sub get_inner_html ($$$) { Line 5418  sub get_inner_html ($$$) {
5418          spacer => 1, wbr => 1,          spacer => 1, wbr => 1,
5419        }->{$tag_name};        }->{$tag_name};
5420    
5421          $s .= "\x0A" if $tag_name eq 'pre' or $tag_name eq 'textarea';
5422    
5423        if (not $in_cdata and {        if (not $in_cdata and {
5424          style => 1, script => 1, xmp => 1, iframe => 1,          style => 1, script => 1, xmp => 1, iframe => 1,
5425          noembed => 1, noframes => 1, noscript => 1,          noembed => 1, noframes => 1, noscript => 1,
5426            plaintext => 1,
5427        }->{$tag_name}) {        }->{$tag_name}) {
5428          unshift @node, 'cdata-out';          unshift @node, 'cdata-out';
5429          $in_cdata = 1;          $in_cdata = 1;

Legend:
Removed from v.1.20  
changed lines
  Added in v.1.32

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24