/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.18 by wakaba, Sat Jun 23 12:21:01 2007 UTC revision 1.27 by wakaba, Sun Jun 24 14:24:21 2007 UTC
# Line 247  sub _get_next_token ($) { Line 247  sub _get_next_token ($) {
247      } elsif ($self->{state} eq 'entity data') {      } elsif ($self->{state} eq 'entity data') {
248        ## (cannot happen in CDATA state)        ## (cannot happen in CDATA state)
249                
250        my $token = $self->_tokenize_attempt_to_consume_an_entity;        my $token = $self->_tokenize_attempt_to_consume_an_entity (0);
251    
252        $self->{state} = 'data';        $self->{state} = 'data';
253        # next-input-character is already done        # next-input-character is already done
# Line 326  sub _get_next_token ($) { Line 326  sub _get_next_token ($) {
326      } elsif ($self->{state} eq 'close tag open') {      } elsif ($self->{state} eq 'close tag open') {
327        if ($self->{content_model_flag} eq 'RCDATA' or        if ($self->{content_model_flag} eq 'RCDATA' or
328            $self->{content_model_flag} eq 'CDATA') {            $self->{content_model_flag} eq 'CDATA') {
329          my @next_char;          if (defined $self->{last_emitted_start_tag_name}) {
330          TAGNAME: for (my $i = 0; $i < length $self->{last_emitted_start_tag_name}; $i++) {            my @next_char;
331              TAGNAME: for (my $i = 0; $i < length $self->{last_emitted_start_tag_name}; $i++) {
332                push @next_char, $self->{next_input_character};
333                my $c = ord substr ($self->{last_emitted_start_tag_name}, $i, 1);
334                my $C = 0x0061 <= $c && $c <= 0x007A ? $c - 0x0020 : $c;
335                if ($self->{next_input_character} == $c or $self->{next_input_character} == $C) {
336                  !!!next-input-character;
337                  next TAGNAME;
338                } else {
339                  $self->{next_input_character} = shift @next_char; # reconsume
340                  !!!back-next-input-character (@next_char);
341                  $self->{state} = 'data';
342    
343                  !!!emit ({type => 'character', data => '</'});
344      
345                  redo A;
346                }
347              }
348            push @next_char, $self->{next_input_character};            push @next_char, $self->{next_input_character};
349            my $c = ord substr ($self->{last_emitted_start_tag_name}, $i, 1);        
350            my $C = 0x0061 <= $c && $c <= 0x007A ? $c - 0x0020 : $c;            unless ($self->{next_input_character} == 0x0009 or # HT
351            if ($self->{next_input_character} == $c or $self->{next_input_character} == $C) {                    $self->{next_input_character} == 0x000A or # LF
352              !!!next-input-character;                    $self->{next_input_character} == 0x000B or # VT
353              next TAGNAME;                    $self->{next_input_character} == 0x000C or # FF
354            } else {                    $self->{next_input_character} == 0x0020 or # SP
355              !!!parse-error (type => 'unmatched end tag');                    $self->{next_input_character} == 0x003E or # >
356                      $self->{next_input_character} == 0x002F or # /
357                      $self->{next_input_character} == -1) {
358              $self->{next_input_character} = shift @next_char; # reconsume              $self->{next_input_character} = shift @next_char; # reconsume
359              !!!back-next-input-character (@next_char);              !!!back-next-input-character (@next_char);
360              $self->{state} = 'data';              $self->{state} = 'data';
   
361              !!!emit ({type => 'character', data => '</'});              !!!emit ({type => 'character', data => '</'});
   
362              redo A;              redo A;
363              } else {
364                $self->{next_input_character} = shift @next_char;
365                !!!back-next-input-character (@next_char);
366                # and consume...
367            }            }
368          }          } else {
369          push @next_char, $self->{next_input_character};            ## No start tag token has ever been emitted
370                  # next-input-character is already done
         unless ($self->{next_input_character} == 0x0009 or # HT  
                 $self->{next_input_character} == 0x000A or # LF  
                 $self->{next_input_character} == 0x000B or # VT  
                 $self->{next_input_character} == 0x000C or # FF  
                 $self->{next_input_character} == 0x0020 or # SP  
                 $self->{next_input_character} == 0x003E or # >  
                 $self->{next_input_character} == 0x002F or # /  
                 $self->{next_input_character} == -1) {  
           !!!parse-error (type => 'unmatched end tag');  
           $self->{next_input_character} = shift @next_char; # reconsume  
           !!!back-next-input-character (@next_char);  
371            $self->{state} = 'data';            $self->{state} = 'data';
   
372            !!!emit ({type => 'character', data => '</'});            !!!emit ({type => 'character', data => '</'});
   
373            redo A;            redo A;
         } else {  
           $self->{next_input_character} = shift @next_char;  
           !!!back-next-input-character (@next_char);  
           # and consume...  
374          }          }
375        }        }
376                
# Line 427  sub _get_next_token ($) { Line 431  sub _get_next_token ($) {
431          !!!next-input-character;          !!!next-input-character;
432    
433          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
434    
435          redo A;          redo A;
436        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{next_input_character} and
# Line 453  sub _get_next_token ($) { Line 456  sub _get_next_token ($) {
456          # reconsume          # reconsume
457    
458          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
459    
460          redo A;          redo A;
461        } elsif ($self->{next_input_character} == 0x002F) { # /        } elsif ($self->{next_input_character} == 0x002F) { # /
# Line 500  sub _get_next_token ($) { Line 502  sub _get_next_token ($) {
502          !!!next-input-character;          !!!next-input-character;
503    
504          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
505    
506          redo A;          redo A;
507        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{next_input_character} and
# Line 539  sub _get_next_token ($) { Line 540  sub _get_next_token ($) {
540          # reconsume          # reconsume
541    
542          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
543    
544          redo A;          redo A;
545        } else {        } else {
# Line 591  sub _get_next_token ($) { Line 591  sub _get_next_token ($) {
591          !!!next-input-character;          !!!next-input-character;
592    
593          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
594    
595          redo A;          redo A;
596        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{next_input_character} and
# Line 631  sub _get_next_token ($) { Line 630  sub _get_next_token ($) {
630          # reconsume          # reconsume
631    
632          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
633    
634          redo A;          redo A;
635        } else {        } else {
# Line 668  sub _get_next_token ($) { Line 666  sub _get_next_token ($) {
666          !!!next-input-character;          !!!next-input-character;
667    
668          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
669    
670          redo A;          redo A;
671        } elsif (0x0041 <= $self->{next_input_character} and        } elsif (0x0041 <= $self->{next_input_character} and
# Line 707  sub _get_next_token ($) { Line 704  sub _get_next_token ($) {
704          # reconsume          # reconsume
705    
706          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
707    
708          redo A;          redo A;
709        } else {        } else {
# Line 753  sub _get_next_token ($) { Line 749  sub _get_next_token ($) {
749          !!!next-input-character;          !!!next-input-character;
750    
751          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
752    
753          redo A;          redo A;
754        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
# Line 772  sub _get_next_token ($) { Line 767  sub _get_next_token ($) {
767          ## reconsume          ## reconsume
768    
769          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
770    
771          redo A;          redo A;
772        } else {        } else {
# Line 807  sub _get_next_token ($) { Line 801  sub _get_next_token ($) {
801          ## reconsume          ## reconsume
802    
803          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
804    
805          redo A;          redo A;
806        } else {        } else {
# Line 842  sub _get_next_token ($) { Line 835  sub _get_next_token ($) {
835          ## reconsume          ## reconsume
836    
837          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
838    
839          redo A;          redo A;
840        } else {        } else {
# Line 880  sub _get_next_token ($) { Line 872  sub _get_next_token ($) {
872          !!!next-input-character;          !!!next-input-character;
873    
874          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
875    
876          redo A;          redo A;
877        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
# Line 899  sub _get_next_token ($) { Line 890  sub _get_next_token ($) {
890          ## reconsume          ## reconsume
891    
892          !!!emit ($self->{current_token}); # start tag or end tag          !!!emit ($self->{current_token}); # start tag or end tag
         undef $self->{current_token};  
893    
894          redo A;          redo A;
895        } else {        } else {
# Line 909  sub _get_next_token ($) { Line 899  sub _get_next_token ($) {
899          redo A;          redo A;
900        }        }
901      } elsif ($self->{state} eq 'entity in attribute value') {      } elsif ($self->{state} eq 'entity in attribute value') {
902        my $token = $self->_tokenize_attempt_to_consume_an_entity;        my $token = $self->_tokenize_attempt_to_consume_an_entity (1);
903    
904        unless (defined $token) {        unless (defined $token) {
905          $self->{current_attribute}->{value} .= '&';          $self->{current_attribute}->{value} .= '&';
# Line 958  sub _get_next_token ($) { Line 948  sub _get_next_token ($) {
948          push @next_char, $self->{next_input_character};          push @next_char, $self->{next_input_character};
949          if ($self->{next_input_character} == 0x002D) { # -          if ($self->{next_input_character} == 0x002D) { # -
950            $self->{current_token} = {type => 'comment', data => ''};            $self->{current_token} = {type => 'comment', data => ''};
951            $self->{state} = 'comment';            $self->{state} = 'comment start';
952            !!!next-input-character;            !!!next-input-character;
953            redo A;            redo A;
954          }          }
# Line 1008  sub _get_next_token ($) { Line 998  sub _get_next_token ($) {
998                
999        ## ISSUE: typos in spec: chacacters, is is a parse error        ## ISSUE: typos in spec: chacacters, is is a parse error
1000        ## ISSUE: spec is somewhat unclear on "is the first character that will be in the comment"; what is "that will be in the comment" is what the algorithm defines, isn't it?        ## ISSUE: spec is somewhat unclear on "is the first character that will be in the comment"; what is "that will be in the comment" is what the algorithm defines, isn't it?
1001        } elsif ($self->{state} eq 'comment start') {
1002          if ($self->{next_input_character} == 0x002D) { # -
1003            $self->{state} = 'comment start dash';
1004            !!!next-input-character;
1005            redo A;
1006          } elsif ($self->{next_input_character} == 0x003E) { # >
1007            !!!parse-error (type => 'bogus comment');
1008            $self->{state} = 'data';
1009            !!!next-input-character;
1010    
1011            !!!emit ($self->{current_token}); # comment
1012    
1013            redo A;
1014          } elsif ($self->{next_input_character} == -1) {
1015            !!!parse-error (type => 'unclosed comment');
1016            $self->{state} = 'data';
1017            ## reconsume
1018    
1019            !!!emit ($self->{current_token}); # comment
1020    
1021            redo A;
1022          } else {
1023            $self->{current_token}->{data} # comment
1024                .= chr ($self->{next_input_character});
1025            $self->{state} = 'comment';
1026            !!!next-input-character;
1027            redo A;
1028          }
1029        } elsif ($self->{state} eq 'comment start dash') {
1030          if ($self->{next_input_character} == 0x002D) { # -
1031            $self->{state} = 'comment end';
1032            !!!next-input-character;
1033            redo A;
1034          } elsif ($self->{next_input_character} == 0x003E) { # >
1035            !!!parse-error (type => 'bogus comment');
1036            $self->{state} = 'data';
1037            !!!next-input-character;
1038    
1039            !!!emit ($self->{current_token}); # comment
1040    
1041            redo A;
1042          } elsif ($self->{next_input_character} == -1) {
1043            !!!parse-error (type => 'unclosed comment');
1044            $self->{state} = 'data';
1045            ## reconsume
1046    
1047            !!!emit ($self->{current_token}); # comment
1048    
1049            redo A;
1050          } else {
1051            $self->{current_token}->{data} # comment
1052                .= chr ($self->{next_input_character});
1053            $self->{state} = 'comment';
1054            !!!next-input-character;
1055            redo A;
1056          }
1057      } elsif ($self->{state} eq 'comment') {      } elsif ($self->{state} eq 'comment') {
1058        if ($self->{next_input_character} == 0x002D) { # -        if ($self->{next_input_character} == 0x002D) { # -
1059          $self->{state} = 'comment dash';          $self->{state} = 'comment end dash';
1060          !!!next-input-character;          !!!next-input-character;
1061          redo A;          redo A;
1062        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
# Line 1019  sub _get_next_token ($) { Line 1065  sub _get_next_token ($) {
1065          ## reconsume          ## reconsume
1066    
1067          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{current_token}); # comment
         undef $self->{current_token};  
1068    
1069          redo A;          redo A;
1070        } else {        } else {
# Line 1028  sub _get_next_token ($) { Line 1073  sub _get_next_token ($) {
1073          !!!next-input-character;          !!!next-input-character;
1074          redo A;          redo A;
1075        }        }
1076      } elsif ($self->{state} eq 'comment dash') {      } elsif ($self->{state} eq 'comment end dash') {
1077        if ($self->{next_input_character} == 0x002D) { # -        if ($self->{next_input_character} == 0x002D) { # -
1078          $self->{state} = 'comment end';          $self->{state} = 'comment end';
1079          !!!next-input-character;          !!!next-input-character;
# Line 1039  sub _get_next_token ($) { Line 1084  sub _get_next_token ($) {
1084          ## reconsume          ## reconsume
1085    
1086          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{current_token}); # comment
         undef $self->{current_token};  
1087    
1088          redo A;          redo A;
1089        } else {        } else {
# Line 1054  sub _get_next_token ($) { Line 1098  sub _get_next_token ($) {
1098          !!!next-input-character;          !!!next-input-character;
1099    
1100          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{current_token}); # comment
         undef $self->{current_token};  
1101    
1102          redo A;          redo A;
1103        } elsif ($self->{next_input_character} == 0x002D) { # -        } elsif ($self->{next_input_character} == 0x002D) { # -
# Line 1069  sub _get_next_token ($) { Line 1112  sub _get_next_token ($) {
1112          ## reconsume          ## reconsume
1113    
1114          !!!emit ($self->{current_token}); # comment          !!!emit ($self->{current_token}); # comment
         undef $self->{current_token};  
1115    
1116          redo A;          redo A;
1117        } else {        } else {
# Line 1144  sub _get_next_token ($) { Line 1186  sub _get_next_token ($) {
1186          !!!next-input-character;          !!!next-input-character;
1187    
1188          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1189    
1190          redo A;          redo A;
1191        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
# Line 1154  sub _get_next_token ($) { Line 1195  sub _get_next_token ($) {
1195    
1196          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1197          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1198    
1199          redo A;          redo A;
1200        } else {        } else {
# Line 1178  sub _get_next_token ($) { Line 1218  sub _get_next_token ($) {
1218          !!!next-input-character;          !!!next-input-character;
1219    
1220          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1221    
1222          redo A;          redo A;
1223        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
# Line 1188  sub _get_next_token ($) { Line 1227  sub _get_next_token ($) {
1227    
1228          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1229          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1230    
1231          redo A;          redo A;
1232        } elsif ($self->{next_input_character} == 0x0050 or # P        } elsif ($self->{next_input_character} == 0x0050 or # P
# Line 1280  sub _get_next_token ($) { Line 1318  sub _get_next_token ($) {
1318    
1319          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1320          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1321    
1322          redo A;          redo A;
1323        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
# Line 1291  sub _get_next_token ($) { Line 1328  sub _get_next_token ($) {
1328    
1329          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1330          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1331    
1332          redo A;          redo A;
1333        } else {        } else {
# Line 1313  sub _get_next_token ($) { Line 1349  sub _get_next_token ($) {
1349    
1350          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1351          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1352    
1353          redo A;          redo A;
1354        } else {        } else {
# Line 1336  sub _get_next_token ($) { Line 1371  sub _get_next_token ($) {
1371    
1372          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1373          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1374    
1375          redo A;          redo A;
1376        } else {        } else {
# Line 1369  sub _get_next_token ($) { Line 1403  sub _get_next_token ($) {
1403          !!!next-input-character;          !!!next-input-character;
1404    
1405          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1406    
1407          redo A;          redo A;
1408        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
1409          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
1410    
1411          $self->{state} = 'data';          $self->{state} = 'data';
1412          ## recomsume          ## reconsume
1413    
1414          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1415          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1416    
1417          redo A;          redo A;
1418        } else {        } else {
# Line 1414  sub _get_next_token ($) { Line 1446  sub _get_next_token ($) {
1446    
1447          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1448          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1449    
1450          redo A;          redo A;
1451        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
1452          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
1453    
1454          $self->{state} = 'data';          $self->{state} = 'data';
1455          ## recomsume          ## reconsume
1456    
1457          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1458          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1459    
1460          redo A;          redo A;
1461        } else {        } else {
# Line 1447  sub _get_next_token ($) { Line 1477  sub _get_next_token ($) {
1477    
1478          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1479          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1480    
1481          redo A;          redo A;
1482        } else {        } else {
# Line 1470  sub _get_next_token ($) { Line 1499  sub _get_next_token ($) {
1499    
1500          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1501          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1502    
1503          redo A;          redo A;
1504        } else {        } else {
# Line 1493  sub _get_next_token ($) { Line 1521  sub _get_next_token ($) {
1521          !!!next-input-character;          !!!next-input-character;
1522    
1523          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1524    
1525          redo A;          redo A;
1526        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
1527          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
1528    
1529          $self->{state} = 'data';          $self->{state} = 'data';
1530          ## recomsume          ## reconsume
1531    
1532          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1533          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1534    
1535          redo A;          redo A;
1536        } else {        } else {
# Line 1520  sub _get_next_token ($) { Line 1546  sub _get_next_token ($) {
1546    
1547          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1548          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1549    
1550          redo A;          redo A;
1551        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
# Line 1530  sub _get_next_token ($) { Line 1555  sub _get_next_token ($) {
1555    
1556          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1557          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
         undef $self->{current_token};  
1558    
1559          redo A;          redo A;
1560        } else {        } else {
# Line 1546  sub _get_next_token ($) { Line 1570  sub _get_next_token ($) {
1570    die "$0: _get_next_token: unexpected case";    die "$0: _get_next_token: unexpected case";
1571  } # _get_next_token  } # _get_next_token
1572    
1573  sub _tokenize_attempt_to_consume_an_entity ($) {  sub _tokenize_attempt_to_consume_an_entity ($$) {
1574    my $self = shift;    my ($self, $in_attr) = @_;
1575      
1576    if ($self->{next_input_character} == 0x0023) { # #    if ({
1577           0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, # HT, LF, VT, FF,
1578           0x0020 => 1, 0x003C => 1, 0x0026 => 1, -1 => 1, # SP, <, & # 0x000D # CR
1579          }->{$self->{next_input_character}}) {
1580        ## Don't consume
1581        ## No error
1582        return undef;
1583      } elsif ($self->{next_input_character} == 0x0023) { # #
1584      !!!next-input-character;      !!!next-input-character;
1585      if ($self->{next_input_character} == 0x0078 or # x      if ($self->{next_input_character} == 0x0078 or # x
1586          $self->{next_input_character} == 0x0058) { # X          $self->{next_input_character} == 0x0058) { # X
1587        my $num;        my $code;
1588        X: {        X: {
1589          my $x_char = $self->{next_input_character};          my $x_char = $self->{next_input_character};
1590          !!!next-input-character;          !!!next-input-character;
1591          if (0x0030 <= $self->{next_input_character} and          if (0x0030 <= $self->{next_input_character} and
1592              $self->{next_input_character} <= 0x0039) { # 0..9              $self->{next_input_character} <= 0x0039) { # 0..9
1593            $num ||= 0;            $code ||= 0;
1594            $num *= 0x10;            $code *= 0x10;
1595            $num += $self->{next_input_character} - 0x0030;            $code += $self->{next_input_character} - 0x0030;
1596            redo X;            redo X;
1597          } elsif (0x0061 <= $self->{next_input_character} and          } elsif (0x0061 <= $self->{next_input_character} and
1598                   $self->{next_input_character} <= 0x0066) { # a..f                   $self->{next_input_character} <= 0x0066) { # a..f
1599            ## ISSUE: the spec says U+0078, which is apparently incorrect            $code ||= 0;
1600            $num ||= 0;            $code *= 0x10;
1601            $num *= 0x10;            $code += $self->{next_input_character} - 0x0060 + 9;
           $num += $self->{next_input_character} - 0x0060 + 9;  
1602            redo X;            redo X;
1603          } elsif (0x0041 <= $self->{next_input_character} and          } elsif (0x0041 <= $self->{next_input_character} and
1604                   $self->{next_input_character} <= 0x0046) { # A..F                   $self->{next_input_character} <= 0x0046) { # A..F
1605            ## ISSUE: the spec says U+0058, which is apparently incorrect            $code ||= 0;
1606            $num ||= 0;            $code *= 0x10;
1607            $num *= 0x10;            $code += $self->{next_input_character} - 0x0040 + 9;
           $num += $self->{next_input_character} - 0x0040 + 9;  
1608            redo X;            redo X;
1609          } elsif (not defined $num) { # no hexadecimal digit          } elsif (not defined $code) { # no hexadecimal digit
1610            !!!parse-error (type => 'bare hcro');            !!!parse-error (type => 'bare hcro');
1611            $self->{next_input_character} = 0x0023; # #            $self->{next_input_character} = 0x0023; # #
1612            !!!back-next-input-character ($x_char);            !!!back-next-input-character ($x_char);
# Line 1588  sub _tokenize_attempt_to_consume_an_enti Line 1617  sub _tokenize_attempt_to_consume_an_enti
1617            !!!parse-error (type => 'no refc');            !!!parse-error (type => 'no refc');
1618          }          }
1619    
1620          ## TODO: check the definition for |a valid Unicode character|.          if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {
1621          ## <http://lists.whatwg.org/pipermail/whatwg-whatwg.org/2006-December/thread.html#8189>            !!!parse-error (type => sprintf 'invalid character reference:U+%04X', $code);
1622          if ($num > 1114111 or $num == 0) {            $code = 0xFFFD;
1623            $num = 0xFFFD; # REPLACEMENT CHARACTER          } elsif ($code > 0x10FFFF) {
1624            ## ISSUE: Why this is not an error?            !!!parse-error (type => sprintf 'invalid character reference:U-%08X', $code);
1625          } elsif (0x80 <= $num and $num <= 0x9F) {            $code = 0xFFFD;
1626            !!!parse-error (type => sprintf 'c1 entity:U+%04X', $num);          } elsif ($code == 0x000D) {
1627            $num = $c1_entity_char->{$num};            !!!parse-error (type => 'CR character reference');
1628              $code = 0x000A;
1629            } elsif (0x80 <= $code and $code <= 0x9F) {
1630              !!!parse-error (type => sprintf 'c1 entity:U+%04X', $code);
1631              $code = $c1_entity_char->{$code};
1632          }          }
1633    
1634          return {type => 'character', data => chr $num};          return {type => 'character', data => chr $code};
1635        } # X        } # X
1636      } elsif (0x0030 <= $self->{next_input_character} and      } elsif (0x0030 <= $self->{next_input_character} and
1637               $self->{next_input_character} <= 0x0039) { # 0..9               $self->{next_input_character} <= 0x0039) { # 0..9
# Line 1619  sub _tokenize_attempt_to_consume_an_enti Line 1652  sub _tokenize_attempt_to_consume_an_enti
1652          !!!parse-error (type => 'no refc');          !!!parse-error (type => 'no refc');
1653        }        }
1654    
1655        ## TODO: check the definition for |a valid Unicode character|.        if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {
1656        if ($code > 1114111 or $code == 0) {          !!!parse-error (type => sprintf 'invalid character reference:U+%04X', $code);
1657          $code = 0xFFFD; # REPLACEMENT CHARACTER          $code = 0xFFFD;
1658          ## ISSUE: Why this is not an error?        } elsif ($code > 0x10FFFF) {
1659            !!!parse-error (type => sprintf 'invalid character reference:U-%08X', $code);
1660            $code = 0xFFFD;
1661          } elsif ($code == 0x000D) {
1662            !!!parse-error (type => 'CR character reference');
1663            $code = 0x000A;
1664        } elsif (0x80 <= $code and $code <= 0x9F) {        } elsif (0x80 <= $code and $code <= 0x9F) {
1665          !!!parse-error (type => sprintf 'c1 entity:U+%04X', $code);          !!!parse-error (type => sprintf 'c1 entity:U+%04X', $code);
1666          $code = $c1_entity_char->{$code};          $code = $c1_entity_char->{$code};
# Line 1658  sub _tokenize_attempt_to_consume_an_enti Line 1696  sub _tokenize_attempt_to_consume_an_enti
1696              $self->{next_input_character} == 0x003B)) { # ;              $self->{next_input_character} == 0x003B)) { # ;
1697        $entity_name .= chr $self->{next_input_character};        $entity_name .= chr $self->{next_input_character};
1698        if (defined $EntityChar->{$entity_name}) {        if (defined $EntityChar->{$entity_name}) {
         $value = $EntityChar->{$entity_name};  
1699          if ($self->{next_input_character} == 0x003B) { # ;          if ($self->{next_input_character} == 0x003B) { # ;
1700              $value = $EntityChar->{$entity_name};
1701            $match = 1;            $match = 1;
1702            !!!next-input-character;            !!!next-input-character;
1703            last;            last;
1704          } else {          } elsif (not $in_attr) {
1705              $value = $EntityChar->{$entity_name};
1706            $match = -1;            $match = -1;
1707            } else {
1708              $value .= chr $self->{next_input_character};
1709          }          }
1710        } else {        } else {
1711          $value .= chr $self->{next_input_character};          $value .= chr $self->{next_input_character};
# Line 1680  sub _tokenize_attempt_to_consume_an_enti Line 1721  sub _tokenize_attempt_to_consume_an_enti
1721      } else {      } else {
1722        !!!parse-error (type => 'bare ero');        !!!parse-error (type => 'bare ero');
1723        ## NOTE: No characters are consumed in the spec.        ## NOTE: No characters are consumed in the spec.
1724        !!!back-token ({type => 'character', data => $value});        return {type => 'character', data => '&'.$value};
       return undef;  
1725      }      }
1726    } else {    } else {
1727      ## no characters are consumed      ## no characters are consumed
# Line 1876  sub _tree_construction_initial ($) { Line 1916  sub _tree_construction_initial ($) {
1916      } elsif ($token->{type} eq 'character') {      } elsif ($token->{type} eq 'character') {
1917        if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D        if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D
1918          ## Ignore the token          ## Ignore the token
1919    
1920          unless (length $token->{data}) {          unless (length $token->{data}) {
1921            ## Stay in the phase            ## Stay in the phase
1922            !!!next-token;            !!!next-token;
# Line 1918  sub _tree_construction_root_element ($) Line 1959  sub _tree_construction_root_element ($)
1959          !!!next-token;          !!!next-token;
1960          redo B;          redo B;
1961        } elsif ($token->{type} eq 'character') {        } elsif ($token->{type} eq 'character') {
1962          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D
1963            $self->{document}->manakai_append_text ($1);            ## Ignore the token.
1964            ## ISSUE: DOM3 Core does not allow Document > Text  
1965            unless (length $token->{data}) {            unless (length $token->{data}) {
1966              ## Stay in the phase              ## Stay in the phase
1967              !!!next-token;              !!!next-token;
# Line 2097  sub _tree_construction_main ($) { Line 2138  sub _tree_construction_main ($) {
2138      }      }
2139    }; # $clear_up_to_marker    }; # $clear_up_to_marker
2140    
2141    my $style_start_tag = sub {    my $parse_rcdata = sub ($$) {
2142      my $style_el; !!!create-element ($style_el, 'style', $token->{attributes});      my ($content_model_flag, $insert) = @_;
2143      ## $self->{insertion_mode} eq 'in head' and ... (always true)  
2144      (($self->{insertion_mode} eq 'in head' and defined $self->{head_element})      ## Step 1
2145       ? $self->{head_element} : $self->{open_elements}->[-1]->[0])      my $start_tag_name = $token->{tag_name};
2146        ->append_child ($style_el);      my $el;
2147      $self->{content_model_flag} = 'CDATA';      !!!create-element ($el, $start_tag_name, $token->{attributes});
2148    
2149        ## Step 2
2150        $insert->($el); # /context node/->append_child ($el)
2151    
2152        ## Step 3
2153        $self->{content_model_flag} = $content_model_flag; # CDATA or RCDATA
2154      delete $self->{escape}; # MUST      delete $self->{escape}; # MUST
2155                  
2156        ## Step 4
2157      my $text = '';      my $text = '';
2158      !!!next-token;      !!!next-token;
2159      while ($token->{type} eq 'character') {      while ($token->{type} eq 'character') { # or until stop tokenizing
2160        $text .= $token->{data};        $text .= $token->{data};
2161        !!!next-token;        !!!next-token;
2162      } # stop if non-character token or tokenizer stops tokenising      }
2163    
2164        ## Step 5
2165      if (length $text) {      if (length $text) {
2166        $style_el->manakai_append_text ($text);        my $text = $self->{document}->create_text_node ($text);
2167          $el->append_child ($text);
2168      }      }
2169        
2170        ## Step 6
2171      $self->{content_model_flag} = 'PCDATA';      $self->{content_model_flag} = 'PCDATA';
2172                  
2173      if ($token->{type} eq 'end tag' and $token->{tag_name} eq 'style') {      ## Step 7
2174        if ($token->{type} eq 'end tag' and $token->{tag_name} eq $start_tag_name) {
2175        ## Ignore the token        ## Ignore the token
2176      } else {      } else {
2177        !!!parse-error (type => 'in CDATA:#'.$token->{type});        !!!parse-error (type => 'in '.$content_model_flag.':#'.$token->{type});
       ## ISSUE: And ignore?  
2178      }      }
2179      !!!next-token;      !!!next-token;
2180    }; # $style_start_tag    }; # $parse_rcdata
2181    
2182    my $script_start_tag = sub {    my $script_start_tag = sub ($) {
2183        my $insert = $_[0];
2184      my $script_el;      my $script_el;
2185      !!!create-element ($script_el, 'script', $token->{attributes});      !!!create-element ($script_el, 'script', $token->{attributes});
2186      ## TODO: mark as "parser-inserted"      ## TODO: mark as "parser-inserted"
# Line 2161  sub _tree_construction_main ($) { Line 2214  sub _tree_construction_main ($) {
2214      } else {      } else {
2215        ## TODO: $old_insertion_point = current insertion point        ## TODO: $old_insertion_point = current insertion point
2216        ## TODO: insertion point = just before the next input character        ## TODO: insertion point = just before the next input character
2217          
2218        (($self->{insertion_mode} eq 'in head' and defined $self->{head_element})        $insert->($script_el);
        ? $self->{head_element} : $self->{open_elements}->[-1]->[0])->append_child ($script_el);  
2219                
2220        ## TODO: insertion point = $old_insertion_point (might be "undefined")        ## TODO: insertion point = $old_insertion_point (might be "undefined")
2221                
# Line 2357  sub _tree_construction_main ($) { Line 2409  sub _tree_construction_main ($) {
2409    }; # $formatting_end_tag    }; # $formatting_end_tag
2410    
2411    my $insert_to_current = sub {    my $insert_to_current = sub {
2412      $self->{open_elements}->[-1]->[0]->append_child (shift);      $self->{open_elements}->[-1]->[0]->append_child ($_[0]);
2413    }; # $insert_to_current    }; # $insert_to_current
2414    
2415    my $insert_to_foster = sub {    my $insert_to_foster = sub {
# Line 2395  sub _tree_construction_main ($) { Line 2447  sub _tree_construction_main ($) {
2447      my $insert = shift;      my $insert = shift;
2448      if ($token->{type} eq 'start tag') {      if ($token->{type} eq 'start tag') {
2449        if ($token->{tag_name} eq 'script') {        if ($token->{tag_name} eq 'script') {
2450          $script_start_tag->();          ## NOTE: This is an "as if in head" code clone
2451            $script_start_tag->($insert);
2452          return;          return;
2453        } elsif ($token->{tag_name} eq 'style') {        } elsif ($token->{tag_name} eq 'style') {
2454          $style_start_tag->();          ## NOTE: This is an "as if in head" code clone
2455            $parse_rcdata->('CDATA', $insert);
2456          return;          return;
2457        } elsif ({        } elsif ({
2458                  base => 1, link => 1, meta => 1,                  base => 1, link => 1, meta => 1,
2459                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
2460          !!!parse-error (type => 'in body:'.$token->{tag_name});          ## NOTE: This is an "as if in head" code clone, only "-t" differs
2461          ## NOTE: This is an "as if in head" code clone          !!!insert-element-t ($token->{tag_name}, $token->{attributes});
2462          my $el;          pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
         !!!create-element ($el, $token->{tag_name}, $token->{attributes});  
         if (defined $self->{head_element}) {  
           $self->{head_element}->append_child ($el);  
         } else {  
           $insert->($el);  
         }  
           
2463          !!!next-token;          !!!next-token;
2464            ## TODO: Extracting |charset| from |meta|.
2465          return;          return;
2466        } elsif ($token->{tag_name} eq 'title') {        } elsif ($token->{tag_name} eq 'title') {
2467          !!!parse-error (type => 'in body:title');          !!!parse-error (type => 'in body:title');
2468          ## NOTE: There is an "as if in head" code clone          ## NOTE: This is an "as if in head" code clone
2469          my $title_el;          $parse_rcdata->('RCDATA', $insert);
         !!!create-element ($title_el, 'title', $token->{attributes});  
         (defined $self->{head_element} ? $self->{head_element} : $self->{open_elements}->[-1]->[0])  
           ->append_child ($title_el);  
         $self->{content_model_flag} = 'RCDATA';  
         delete $self->{escape}; # MUST  
           
         my $text = '';  
         !!!next-token;  
         while ($token->{type} eq 'character') {  
           $text .= $token->{data};  
           !!!next-token;  
         }  
         if (length $text) {  
           $title_el->manakai_append_text ($text);  
         }  
           
         $self->{content_model_flag} = 'PCDATA';  
           
         if ($token->{type} eq 'end tag' and  
             $token->{tag_name} eq 'title') {  
           ## Ignore the token  
         } else {  
           !!!parse-error (type => 'in RCDATA:#'.$token->{type});  
           ## ISSUE: And ignore?  
         }  
         !!!next-token;  
2470          return;          return;
2471        } elsif ($token->{tag_name} eq 'body') {        } elsif ($token->{tag_name} eq 'body') {
2472          !!!parse-error (type => 'in body:body');          !!!parse-error (type => 'in body:body');
# Line 2658  sub _tree_construction_main ($) { Line 2680  sub _tree_construction_main ($) {
2680            }            }
2681          } # INSCOPE          } # INSCOPE
2682                        
2683            ## NOTE: See <http://html5.org/tools/web-apps-tracker?from=925&to=926>
2684          ## has an element in scope          ## has an element in scope
2685          my $i;          #my $i;
2686          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          #INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
2687            my $node = $self->{open_elements}->[$_];          #  my $node = $self->{open_elements}->[$_];
2688            if ({          #  if ({
2689                 h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,          #       h1 => 1, h2 => 1, h3 => 1, h4 => 1, h5 => 1, h6 => 1,
2690                }->{$node->[1]}) {          #      }->{$node->[1]}) {
2691              $i = $_;          #    $i = $_;
2692              last INSCOPE;          #    last INSCOPE;
2693            } elsif ({          #  } elsif ({
2694                      table => 1, caption => 1, td => 1, th => 1,          #            table => 1, caption => 1, td => 1, th => 1,
2695                      button => 1, marquee => 1, object => 1, html => 1,          #            button => 1, marquee => 1, object => 1, html => 1,
2696                     }->{$node->[1]}) {          #           }->{$node->[1]}) {
2697              last INSCOPE;          #    last INSCOPE;
2698            }          #  }
2699          } # INSCOPE          #} # INSCOPE
2700                      #  
2701          if (defined $i) {          #if (defined $i) {
2702            !!!parse-error (type => 'in hn:hn');          #  !!! parse-error (type => 'in hn:hn');
2703            splice @{$self->{open_elements}}, $i;          #  splice @{$self->{open_elements}}, $i;
2704          }          #}
2705                        
2706          !!!insert-element-t ($token->{tag_name}, $token->{attributes});          !!!insert-element-t ($token->{tag_name}, $token->{attributes});
2707                        
# Line 2721  sub _tree_construction_main ($) { Line 2744  sub _tree_construction_main ($) {
2744          return;          return;
2745        } elsif ({        } elsif ({
2746                  b => 1, big => 1, em => 1, font => 1, i => 1,                  b => 1, big => 1, em => 1, font => 1, i => 1,
2747                  nobr => 1, s => 1, small => 1, strile => 1,                  s => 1, small => 1, strile => 1,
2748                  strong => 1, tt => 1, u => 1,                  strong => 1, tt => 1, u => 1,
2749                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
2750          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
# Line 2731  sub _tree_construction_main ($) { Line 2754  sub _tree_construction_main ($) {
2754                    
2755          !!!next-token;          !!!next-token;
2756          return;          return;
2757          } elsif ($token->{tag_name} eq 'nobr') {
2758            $reconstruct_active_formatting_elements->($insert_to_current);
2759    
2760            ## has a |nobr| element in scope
2761            INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
2762              my $node = $self->{open_elements}->[$_];
2763              if ($node->[1] eq 'nobr') {
2764                !!!back-token;
2765                $token = {type => 'end tag', tag_name => 'nobr'};
2766                return;
2767              } elsif ({
2768                        table => 1, caption => 1, td => 1, th => 1,
2769                        button => 1, marquee => 1, object => 1, html => 1,
2770                       }->{$node->[1]}) {
2771                last INSCOPE;
2772              }
2773            } # INSCOPE
2774            
2775            !!!insert-element-t ($token->{tag_name}, $token->{attributes});
2776            push @$active_formatting_elements, $self->{open_elements}->[-1];
2777            
2778            !!!next-token;
2779            return;
2780        } elsif ($token->{tag_name} eq 'button') {        } elsif ($token->{tag_name} eq 'button') {
2781          ## has a button element in scope          ## has a button element in scope
2782          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {          INSCOPE: for (reverse 0..$#{$self->{open_elements}}) {
# Line 2766  sub _tree_construction_main ($) { Line 2812  sub _tree_construction_main ($) {
2812          return;          return;
2813        } elsif ($token->{tag_name} eq 'xmp') {        } elsif ($token->{tag_name} eq 'xmp') {
2814          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
2815                    $parse_rcdata->('CDATA', $insert);
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
           
         $self->{content_model_flag} = 'CDATA';  
         delete $self->{escape}; # MUST  
           
         !!!next-token;  
2816          return;          return;
2817        } elsif ($token->{tag_name} eq 'table') {        } elsif ($token->{tag_name} eq 'table') {
2818          ## has a p element in scope          ## has a p element in scope
# Line 2850  sub _tree_construction_main ($) { Line 2890  sub _tree_construction_main ($) {
2890            return;            return;
2891          } else {          } else {
2892            my $at = $token->{attributes};            my $at = $token->{attributes};
2893              my $form_attrs;
2894              $form_attrs->{action} = $at->{action} if $at->{action};
2895              my $prompt_attr = $at->{prompt};
2896            $at->{name} = {name => 'name', value => 'isindex'};            $at->{name} = {name => 'name', value => 'isindex'};
2897              delete $at->{action};
2898              delete $at->{prompt};
2899            my @tokens = (            my @tokens = (
2900                          {type => 'start tag', tag_name => 'form'},                          {type => 'start tag', tag_name => 'form',
2901                             attributes => $form_attrs},
2902                          {type => 'start tag', tag_name => 'hr'},                          {type => 'start tag', tag_name => 'hr'},
2903                          {type => 'start tag', tag_name => 'p'},                          {type => 'start tag', tag_name => 'p'},
2904                          {type => 'start tag', tag_name => 'label'},                          {type => 'start tag', tag_name => 'label'},
2905                          {type => 'character',                         );
2906                           data => 'This is a searchable index. Insert your search keywords here: '}, # SHOULD            if ($prompt_attr) {
2907                          ## TODO: make this configurable              push @tokens, {type => 'character', data => $prompt_attr->{value}};
2908              } else {
2909                push @tokens, {type => 'character',
2910                               data => 'This is a searchable index. Insert your search keywords here: '}; # SHOULD
2911                ## TODO: make this configurable
2912              }
2913              push @tokens,
2914                          {type => 'start tag', tag_name => 'input', attributes => $at},                          {type => 'start tag', tag_name => 'input', attributes => $at},
2915                          #{type => 'character', data => ''}, # SHOULD                          #{type => 'character', data => ''}, # SHOULD
2916                          {type => 'end tag', tag_name => 'label'},                          {type => 'end tag', tag_name => 'label'},
2917                          {type => 'end tag', tag_name => 'p'},                          {type => 'end tag', tag_name => 'p'},
2918                          {type => 'start tag', tag_name => 'hr'},                          {type => 'start tag', tag_name => 'hr'},
2919                          {type => 'end tag', tag_name => 'form'},                          {type => 'end tag', tag_name => 'form'};
                        );  
2920            $token = shift @tokens;            $token = shift @tokens;
2921            !!!back-token (@tokens);            !!!back-token (@tokens);
2922            return;            return;
2923          }          }
2924        } elsif ({        } elsif ($token->{tag_name} eq 'textarea') {
                 textarea => 1,  
                 iframe => 1,  
                 noembed => 1,  
                 noframes => 1,  
                 noscript => 0, ## TODO: 1 if scripting is enabled  
                }->{$token->{tag_name}}) {  
2925          my $tag_name = $token->{tag_name};          my $tag_name = $token->{tag_name};
2926          my $el;          my $el;
2927          !!!create-element ($el, $token->{tag_name}, $token->{attributes});          !!!create-element ($el, $token->{tag_name}, $token->{attributes});
2928                    
2929          if ($token->{tag_name} eq 'textarea') {          ## TODO: $self->{form_element} if defined
2930            ## TODO: $self->{form_element} if defined          $self->{content_model_flag} = 'RCDATA';
           $self->{content_model_flag} = 'RCDATA';  
         } else {  
           $self->{content_model_flag} = 'CDATA';  
         }  
2931          delete $self->{escape}; # MUST          delete $self->{escape}; # MUST
2932                    
2933          $insert->($el);          $insert->($el);
2934                    
2935          my $text = '';          my $text = '';
2936          if ($token->{tag_name} eq 'textarea') {          !!!next-token;
2937            !!!next-token;          if ($token->{type} eq 'character') {
2938            if ($token->{type} eq 'character') {            $token->{data} =~ s/^\x0A//;
2939              $token->{data} =~ s/^\x0A//;            unless (length $token->{data}) {
2940              unless (length $token->{data}) {              !!!next-token;
               !!!next-token;  
             }  
2941            }            }
         } else {  
           !!!next-token;  
2942          }          }
2943          while ($token->{type} eq 'character') {          while ($token->{type} eq 'character') {
2944            $text .= $token->{data};            $text .= $token->{data};
# Line 2917  sub _tree_construction_main ($) { Line 2954  sub _tree_construction_main ($) {
2954              $token->{tag_name} eq $tag_name) {              $token->{tag_name} eq $tag_name) {
2955            ## Ignore the token            ## Ignore the token
2956          } else {          } else {
2957            if ($token->{tag_name} eq 'textarea') {            !!!parse-error (type => 'in RCDATA:#'.$token->{type});
             !!!parse-error (type => 'in RCDATA:#'.$token->{type});  
           } else {  
             !!!parse-error (type => 'in CDATA:#'.$token->{type});  
           }  
           ## ISSUE: And ignore?  
2958          }          }
2959          !!!next-token;          !!!next-token;
2960          return;          return;
2961          } elsif ({
2962                    iframe => 1,
2963                    noembed => 1,
2964                    noframes => 1,
2965                    noscript => 0, ## TODO: 1 if scripting is enabled
2966                   }->{$token->{tag_name}}) {
2967            $parse_rcdata->('CDATA', $insert);
2968            return;
2969        } elsif ($token->{tag_name} eq 'select') {        } elsif ($token->{tag_name} eq 'select') {
2970          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
2971                    
# Line 2956  sub _tree_construction_main ($) { Line 2996  sub _tree_construction_main ($) {
2996        }        }
2997      } elsif ($token->{type} eq 'end tag') {      } elsif ($token->{type} eq 'end tag') {
2998        if ($token->{tag_name} eq 'body') {        if ($token->{tag_name} eq 'body') {
2999          if (@{$self->{open_elements}} > 1 and $self->{open_elements}->[1]->[1] eq 'body') {          if (@{$self->{open_elements}} > 1 and
3000            ## ISSUE: There is an issue in the spec.              $self->{open_elements}->[1]->[1] eq 'body') {
3001            if ($self->{open_elements}->[-1]->[1] ne 'body') {            for (@{$self->{open_elements}}) {
3002              !!!parse-error (type => 'not closed:'.$self->{open_elements}->[-1]->[1]);              unless ({
3003                           dd => 1, dt => 1, li => 1, p => 1, td => 1,
3004                           th => 1, tr => 1, body => 1, html => 1,
3005                        }->{$_->[1]}) {
3006                  !!!parse-error (type => 'not closed:'.$_->[1]);
3007                }
3008            }            }
3009    
3010            $self->{insertion_mode} = 'after body';            $self->{insertion_mode} = 'after body';
3011            !!!next-token;            !!!next-token;
3012            return;            return;
# Line 3166  sub _tree_construction_main ($) { Line 3212  sub _tree_construction_main ($) {
3212                  #not $phrasing_category->{$node->[1]} and                  #not $phrasing_category->{$node->[1]} and
3213                  ($special_category->{$node->[1]} or                  ($special_category->{$node->[1]} or
3214                   $scoping_category->{$node->[1]})) {                   $scoping_category->{$node->[1]})) {
3215                !!!parse-error (type => 'not closed:'.$node->[1]);                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3216                ## Ignore the token                ## Ignore the token
3217                !!!next-token;                !!!next-token;
3218                last S2;                last S2;
# Line 3269  sub _tree_construction_main ($) { Line 3315  sub _tree_construction_main ($) {
3315              }              }
3316              redo B;              redo B;
3317            } elsif ($token->{type} eq 'end tag') {            } elsif ($token->{type} eq 'end tag') {
3318              if ($token->{tag_name} eq 'html') {              if ({head => 1, body => 1, html => 1}->{$token->{tag_name}}) {
3319                ## As if <head>                ## As if <head>
3320                !!!create-element ($self->{head_element}, 'head');                !!!create-element ($self->{head_element}, 'head');
3321                $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});                $self->{open_elements}->[-1]->[0]->append_child ($self->{head_element});
# Line 3279  sub _tree_construction_main ($) { Line 3325  sub _tree_construction_main ($) {
3325                redo B;                redo B;
3326              } else {              } else {
3327                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3328                ## Ignore the token                ## Ignore the token ## ISSUE: An issue in the spec.
3329                !!!next-token;                !!!next-token;
3330                redo B;                redo B;
3331              }              }
3332            } else {            } else {
3333              die "$0: $token->{type}: Unknown type";              die "$0: $token->{type}: Unknown type";
3334            }            }
3335          } elsif ($self->{insertion_mode} eq 'in head') {          } elsif ($self->{insertion_mode} eq 'in head' or
3336                     $self->{insertion_mode} eq 'in head noscript' or
3337                     $self->{insertion_mode} eq 'after head') {
3338            if ($token->{type} eq 'character') {            if ($token->{type} eq 'character') {
3339              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {
3340                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
# Line 3303  sub _tree_construction_main ($) { Line 3351  sub _tree_construction_main ($) {
3351              !!!next-token;              !!!next-token;
3352              redo B;              redo B;
3353            } elsif ($token->{type} eq 'start tag') {            } elsif ($token->{type} eq 'start tag') {
3354              if ($token->{tag_name} eq 'title') {              if ({base => ($self->{insertion_mode} eq 'in head' or
3355                ## NOTE: There is an "as if in head" code clone                            $self->{insertion_mode} eq 'after head'),
3356                my $title_el;                   link => 1, meta => 1}->{$token->{tag_name}}) {
3357                !!!create-element ($title_el, 'title', $token->{attributes});                ## NOTE: There is a "as if in head" code clone.
3358                (defined $self->{head_element} ? $self->{head_element} : $self->{open_elements}->[-1]->[0])                if ($self->{insertion_mode} eq 'after head') {
3359                  ->append_child ($title_el);                  !!!parse-error (type => 'after head:'.$token->{tag_name});
3360                $self->{content_model_flag} = 'RCDATA';                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
3361                delete $self->{escape}; # MUST                }
3362                  !!!insert-element ($token->{tag_name}, $token->{attributes});
3363                my $text = '';                pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
3364                  ## TODO: Extracting |charset| from |meta|.
3365                  pop @{$self->{open_elements}}
3366                      if $self->{insertion_mode} eq 'after head';
3367                !!!next-token;                !!!next-token;
3368                while ($token->{type} eq 'character') {                redo B;
3369                  $text .= $token->{data};              } elsif ($token->{tag_name} eq 'title' and
3370                         $self->{insertion_mode} eq 'in head') {
3371                  ## NOTE: There is a "as if in head" code clone.
3372                  if ($self->{insertion_mode} eq 'after head') {
3373                    !!!parse-error (type => 'after head:'.$token->{tag_name});
3374                    push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
3375                  }
3376                  $parse_rcdata->('RCDATA', $insert_to_current);
3377                  pop @{$self->{open_elements}}
3378                      if $self->{insertion_mode} eq 'after head';
3379                  redo B;
3380                } elsif ($token->{tag_name} eq 'style') {
3381                  ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and
3382                  ## insertion mode 'in head')
3383                  ## NOTE: There is a "as if in head" code clone.
3384                  if ($self->{insertion_mode} eq 'after head') {
3385                    !!!parse-error (type => 'after head:'.$token->{tag_name});
3386                    push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
3387                  }
3388                  $parse_rcdata->('CDATA', $insert_to_current);
3389                  pop @{$self->{open_elements}}
3390                      if $self->{insertion_mode} eq 'after head';
3391                  redo B;
3392                } elsif ($token->{tag_name} eq 'noscript') {
3393                  if ($self->{insertion_mode} eq 'in head') {
3394                    ## NOTE: and scripting is disalbed
3395                    !!!insert-element ($token->{tag_name}, $token->{attributes});
3396                    $self->{insertion_mode} = 'in head noscript';
3397                  !!!next-token;                  !!!next-token;
3398                }                  redo B;
3399                if (length $text) {                } elsif ($self->{insertion_mode} eq 'in head noscript') {
3400                  $title_el->manakai_append_text ($text);                  !!!parse-error (type => 'noscript in noscript');
               }  
                 
               $self->{content_model_flag} = 'PCDATA';  
                 
               if ($token->{type} eq 'end tag' and  
                   $token->{tag_name} eq 'title') {  
3401                  ## Ignore the token                  ## Ignore the token
3402                    redo B;
3403                } else {                } else {
3404                  !!!parse-error (type => 'in RCDATA:#'.$token->{type});                  #
                 ## ISSUE: And ignore?  
3405                }                }
3406                } elsif ($token->{tag_name} eq 'head' and
3407                         $self->{insertion_mode} ne 'after head') {
3408                  !!!parse-error (type => 'in head:head'); # or in head noscript
3409                  ## Ignore the token
3410                !!!next-token;                !!!next-token;
3411                redo B;                redo B;
3412              } elsif ($token->{tag_name} eq 'style') {              } elsif ($self->{insertion_mode} ne 'in head noscript' and
3413                $style_start_tag->();                       $token->{tag_name} eq 'script') {
3414                redo B;                if ($self->{insertion_mode} eq 'after head') {
3415              } elsif ($token->{tag_name} eq 'script') {                  !!!parse-error (type => 'after head:'.$token->{tag_name});
3416                $script_start_tag->();                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
3417                  }
3418                  ## NOTE: There is a "as if in head" code clone.
3419                  $script_start_tag->($insert_to_current);
3420                  pop @{$self->{open_elements}}
3421                      if $self->{insertion_mode} eq 'after head';
3422                redo B;                redo B;
3423              } elsif ({base => 1, link => 1, meta => 1}->{$token->{tag_name}}) {              } elsif ($self->{insertion_mode} eq 'after head' and
3424                ## NOTE: There are "as if in head" code clones                       $token->{tag_name} eq 'body') {
3425                my $el;                !!!insert-element ('body', $token->{attributes});
3426                !!!create-element ($el, $token->{tag_name}, $token->{attributes});                $self->{insertion_mode} = 'in body';
               (defined $self->{head_element} ? $self->{head_element} : $self->{open_elements}->[-1]->[0])  
                 ->append_child ($el);  
   
3427                !!!next-token;                !!!next-token;
3428                redo B;                redo B;
3429              } elsif ($token->{tag_name} eq 'head') {              } elsif ($self->{insertion_mode} eq 'after head' and
3430                !!!parse-error (type => 'in head:head');                       $token->{tag_name} eq 'frameset') {
3431                ## Ignore the token                !!!insert-element ('frameset', $token->{attributes});
3432                  $self->{insertion_mode} = 'in frameset';
3433                !!!next-token;                !!!next-token;
3434                redo B;                redo B;
3435              } else {              } else {
3436                #                #
3437              }              }
3438            } elsif ($token->{type} eq 'end tag') {            } elsif ($token->{type} eq 'end tag') {
3439              if ($token->{tag_name} eq 'head') {              if ($self->{insertion_mode} eq 'in head' and
3440                if ($self->{open_elements}->[-1]->[1] eq 'head') {                  $token->{tag_name} eq 'head') {
3441                  pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
               } else {  
                 !!!parse-error (type => 'unmatched end tag:head');  
               }  
3442                $self->{insertion_mode} = 'after head';                $self->{insertion_mode} = 'after head';
3443                !!!next-token;                !!!next-token;
3444                redo B;                redo B;
3445              } elsif ($token->{tag_name} eq 'html') {              } elsif ($self->{insertion_mode} eq 'in head noscript' and
3446                    $token->{tag_name} eq 'noscript') {
3447                  pop @{$self->{open_elements}};
3448                  $self->{insertion_mode} = 'in head';
3449                  !!!next-token;
3450                  redo B;
3451                } elsif ($self->{insertion_mode} eq 'in head' and
3452                         ($token->{tag_name} eq 'body' or
3453                          $token->{tag_name} eq 'html')) {
3454                #                #
3455              } else {              } elsif ($self->{insertion_mode} ne 'after head') {
3456                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3457                ## Ignore the token                ## Ignore the token
3458                !!!next-token;                !!!next-token;
3459                redo B;                redo B;
3460                } else {
3461                  #
3462              }              }
3463            } else {            } else {
3464              #              #
3465            }            }
3466    
3467            if ($self->{open_elements}->[-1]->[1] eq 'head') {            ## As if </head> or </noscript> or <body>
3468              ## As if </head>            if ($self->{insertion_mode} eq 'in head') {
3469                pop @{$self->{open_elements}};
3470                $self->{insertion_mode} = 'after head';
3471              } elsif ($self->{insertion_mode} eq 'in head noscript') {
3472              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
3473                !!!parse-error (type => 'in noscript:'.(defined $token->{tag_name} ? ($token->{type} eq 'end tag' ? '/' : '') . $token->{tag_name} : '#' . $token->{type}));
3474                $self->{insertion_mode} = 'in head';
3475              } else { # 'after head'
3476                !!!insert-element ('body');
3477                $self->{insertion_mode} = 'in body';
3478            }            }
           $self->{insertion_mode} = 'after head';  
3479            ## reprocess            ## reprocess
3480            redo B;            redo B;
3481    
3482            ## ISSUE: An issue in the spec.            ## ISSUE: An issue in the spec.
         } elsif ($self->{insertion_mode} eq 'after head') {  
           if ($token->{type} eq 'character') {  
             if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {  
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);  
               unless (length $token->{data}) {  
                 !!!next-token;  
                 redo B;  
               }  
             }  
               
             #  
           } elsif ($token->{type} eq 'comment') {  
             my $comment = $self->{document}->create_comment ($token->{data});  
             $self->{open_elements}->[-1]->[0]->append_child ($comment);  
             !!!next-token;  
             redo B;  
           } elsif ($token->{type} eq 'start tag') {  
             if ($token->{tag_name} eq 'body') {  
               !!!insert-element ('body', $token->{attributes});  
               $self->{insertion_mode} = 'in body';  
               !!!next-token;  
               redo B;  
             } elsif ($token->{tag_name} eq 'frameset') {  
               !!!insert-element ('frameset', $token->{attributes});  
               $self->{insertion_mode} = 'in frameset';  
               !!!next-token;  
               redo B;  
             } elsif ({  
                       base => 1, link => 1, meta => 1,  
                       script => 1, style => 1, title => 1,  
                      }->{$token->{tag_name}}) {  
               !!!parse-error (type => 'after head:'.$token->{tag_name});  
               $self->{insertion_mode} = 'in head';  
               ## reprocess  
               redo B;  
             } else {  
               #  
             }  
           } else {  
             #  
           }  
             
           ## As if <body>  
           !!!insert-element ('body');  
           $self->{insertion_mode} = 'in body';  
           ## reprocess  
           redo B;  
3483          } elsif ($self->{insertion_mode} eq 'in body') {          } elsif ($self->{insertion_mode} eq 'in body') {
3484            if ($token->{type} eq 'character') {            if ($token->{type} eq 'character') {
3485              ## NOTE: There is a code clone of "character in body".              ## NOTE: There is a code clone of "character in body".
# Line 5245  sub get_inner_html ($$$) { Line 5291  sub get_inner_html ($$$) {
5291            
5292      my $nt = $child->node_type;      my $nt = $child->node_type;
5293      if ($nt == 1) { # Element      if ($nt == 1) { # Element
5294        my $tag_name = lc $child->tag_name; ## ISSUE: Definition of "lowercase"        my $tag_name = $child->tag_name; ## TODO: manakai_tag_name
5295        $s .= '<' . $tag_name;        $s .= '<' . $tag_name;
5296          ## NOTE: Non-HTML case:
5297        ## ISSUE: Non-html elements        ## <http://permalink.gmane.org/gmane.org.w3c.whatwg.discuss/11191>
5298    
5299        my @attrs = @{$child->attributes}; # sort order MUST be stable        my @attrs = @{$child->attributes}; # sort order MUST be stable
5300        for my $attr (@attrs) { # order is implementation dependent        for my $attr (@attrs) { # order is implementation dependent
5301          my $attr_name = lc $attr->name; ## ISSUE: Definition of "lowercase"          my $attr_name = $attr->name; ## TODO: manakai_name
5302          $s .= ' ' . $attr_name . '="';          $s .= ' ' . $attr_name . '="';
5303          my $attr_value = $attr->value;          my $attr_value = $attr->value;
5304          ## escape          ## escape
# Line 5271  sub get_inner_html ($$$) { Line 5317  sub get_inner_html ($$$) {
5317          spacer => 1, wbr => 1,          spacer => 1, wbr => 1,
5318        }->{$tag_name};        }->{$tag_name};
5319    
5320          $s .= "\x0A" if $tag_name eq 'pre' or $tag_name eq 'textarea';
5321    
5322        if (not $in_cdata and {        if (not $in_cdata and {
5323          style => 1, script => 1, xmp => 1, iframe => 1,          style => 1, script => 1, xmp => 1, iframe => 1,
5324          noembed => 1, noframes => 1, noscript => 1,          noembed => 1, noframes => 1, noscript => 1,
5325            plaintext => 1,
5326        }->{$tag_name}) {        }->{$tag_name}) {
5327          unshift @node, 'cdata-out';          unshift @node, 'cdata-out';
5328          $in_cdata = 1;          $in_cdata = 1;

Legend:
Removed from v.1.18  
changed lines
  Added in v.1.27

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24