/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.24 by wakaba, Sat Jun 23 16:42:43 2007 UTC revision 1.30 by wakaba, Sat Jun 30 13:12:32 2007 UTC
# Line 247  sub _get_next_token ($) { Line 247  sub _get_next_token ($) {
247      } elsif ($self->{state} eq 'entity data') {      } elsif ($self->{state} eq 'entity data') {
248        ## (cannot happen in CDATA state)        ## (cannot happen in CDATA state)
249                
250        my $token = $self->_tokenize_attempt_to_consume_an_entity;        my $token = $self->_tokenize_attempt_to_consume_an_entity (0);
251    
252        $self->{state} = 'data';        $self->{state} = 'data';
253        # next-input-character is already done        # next-input-character is already done
# Line 327  sub _get_next_token ($) { Line 327  sub _get_next_token ($) {
327        if ($self->{content_model_flag} eq 'RCDATA' or        if ($self->{content_model_flag} eq 'RCDATA' or
328            $self->{content_model_flag} eq 'CDATA') {            $self->{content_model_flag} eq 'CDATA') {
329          if (defined $self->{last_emitted_start_tag_name}) {          if (defined $self->{last_emitted_start_tag_name}) {
330              ## NOTE: <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>
331            my @next_char;            my @next_char;
332            TAGNAME: for (my $i = 0; $i < length $self->{last_emitted_start_tag_name}; $i++) {            TAGNAME: for (my $i = 0; $i < length $self->{last_emitted_start_tag_name}; $i++) {
333              push @next_char, $self->{next_input_character};              push @next_char, $self->{next_input_character};
# Line 418  sub _get_next_token ($) { Line 419  sub _get_next_token ($) {
419          redo A;          redo A;
420        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
421          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
422              $self->{current_token}->{first_start_tag}
423                  = not defined $self->{last_emitted_start_tag_name};
424            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
425          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
426            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 443  sub _get_next_token ($) { Line 446  sub _get_next_token ($) {
446        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
447          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
448          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
449              $self->{current_token}->{first_start_tag}
450                  = not defined $self->{last_emitted_start_tag_name};
451            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
452          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
453            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 489  sub _get_next_token ($) { Line 494  sub _get_next_token ($) {
494          redo A;          redo A;
495        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
496          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
497              $self->{current_token}->{first_start_tag}
498                  = not defined $self->{last_emitted_start_tag_name};
499            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
500          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
501            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 527  sub _get_next_token ($) { Line 534  sub _get_next_token ($) {
534        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
535          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
536          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
537              $self->{current_token}->{first_start_tag}
538                  = not defined $self->{last_emitted_start_tag_name};
539            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
540          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
541            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 578  sub _get_next_token ($) { Line 587  sub _get_next_token ($) {
587        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
588          $before_leave->();          $before_leave->();
589          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
590              $self->{current_token}->{first_start_tag}
591                  = not defined $self->{last_emitted_start_tag_name};
592            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
593          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
594            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 617  sub _get_next_token ($) { Line 628  sub _get_next_token ($) {
628          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
629          $before_leave->();          $before_leave->();
630          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
631              $self->{current_token}->{first_start_tag}
632                  = not defined $self->{last_emitted_start_tag_name};
633            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
634          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
635            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 653  sub _get_next_token ($) { Line 666  sub _get_next_token ($) {
666          redo A;          redo A;
667        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
668          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
669              $self->{current_token}->{first_start_tag}
670                  = not defined $self->{last_emitted_start_tag_name};
671            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
672          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
673            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 691  sub _get_next_token ($) { Line 706  sub _get_next_token ($) {
706        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
707          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
708          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
709              $self->{current_token}->{first_start_tag}
710                  = not defined $self->{last_emitted_start_tag_name};
711            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
712          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
713            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 736  sub _get_next_token ($) { Line 753  sub _get_next_token ($) {
753          redo A;          redo A;
754        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
755          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
756              $self->{current_token}->{first_start_tag}
757                  = not defined $self->{last_emitted_start_tag_name};
758            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
759          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
760            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 754  sub _get_next_token ($) { Line 773  sub _get_next_token ($) {
773        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
774          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
775          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
776              $self->{current_token}->{first_start_tag}
777                  = not defined $self->{last_emitted_start_tag_name};
778            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
779          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
780            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 788  sub _get_next_token ($) { Line 809  sub _get_next_token ($) {
809        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
810          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
811          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
812              $self->{current_token}->{first_start_tag}
813                  = not defined $self->{last_emitted_start_tag_name};
814            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
815          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
816            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 822  sub _get_next_token ($) { Line 845  sub _get_next_token ($) {
845        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
846          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
847          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
848              $self->{current_token}->{first_start_tag}
849                  = not defined $self->{last_emitted_start_tag_name};
850            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
851          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
852            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 859  sub _get_next_token ($) { Line 884  sub _get_next_token ($) {
884          redo A;          redo A;
885        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
886          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
887              $self->{current_token}->{first_start_tag}
888                  = not defined $self->{last_emitted_start_tag_name};
889            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
890          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
891            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 877  sub _get_next_token ($) { Line 904  sub _get_next_token ($) {
904        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
905          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
906          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
907              $self->{current_token}->{first_start_tag}
908                  = not defined $self->{last_emitted_start_tag_name};
909            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
910          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
911            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 899  sub _get_next_token ($) { Line 928  sub _get_next_token ($) {
928          redo A;          redo A;
929        }        }
930      } elsif ($self->{state} eq 'entity in attribute value') {      } elsif ($self->{state} eq 'entity in attribute value') {
931        my $token = $self->_tokenize_attempt_to_consume_an_entity;        my $token = $self->_tokenize_attempt_to_consume_an_entity (1);
932    
933        unless (defined $token) {        unless (defined $token) {
934          $self->{current_attribute}->{value} .= '&';          $self->{current_attribute}->{value} .= '&';
# Line 990  sub _get_next_token ($) { Line 1019  sub _get_next_token ($) {
1019          }          }
1020        }        }
1021    
1022        !!!parse-error (type => 'bogus comment open');        !!!parse-error (type => 'bogus comment');
1023        $self->{next_input_character} = shift @next_char;        $self->{next_input_character} = shift @next_char;
1024        !!!back-next-input-character (@next_char);        !!!back-next-input-character (@next_char);
1025        $self->{state} = 'bogus comment';        $self->{state} = 'bogus comment';
# Line 1409  sub _get_next_token ($) { Line 1438  sub _get_next_token ($) {
1438          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
1439    
1440          $self->{state} = 'data';          $self->{state} = 'data';
1441          ## recomsume          ## reconsume
1442    
1443          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1444          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
# Line 1452  sub _get_next_token ($) { Line 1481  sub _get_next_token ($) {
1481          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
1482    
1483          $self->{state} = 'data';          $self->{state} = 'data';
1484          ## recomsume          ## reconsume
1485    
1486          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1487          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
1488    
1489          redo A;          redo A;
1490        } else {        } else {
1491          !!!parse-error (type => 'string after PUBLIC literal');          !!!parse-error (type => 'string after SYSTEM');
1492          $self->{state} = 'bogus DOCTYPE';          $self->{state} = 'bogus DOCTYPE';
1493          !!!next-input-character;          !!!next-input-character;
1494          redo A;          redo A;
# Line 1527  sub _get_next_token ($) { Line 1556  sub _get_next_token ($) {
1556          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
1557    
1558          $self->{state} = 'data';          $self->{state} = 'data';
1559          ## recomsume          ## reconsume
1560    
1561          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1562          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
# Line 1570  sub _get_next_token ($) { Line 1599  sub _get_next_token ($) {
1599    die "$0: _get_next_token: unexpected case";    die "$0: _get_next_token: unexpected case";
1600  } # _get_next_token  } # _get_next_token
1601    
1602  sub _tokenize_attempt_to_consume_an_entity ($) {  sub _tokenize_attempt_to_consume_an_entity ($$) {
1603    my $self = shift;    my ($self, $in_attr) = @_;
1604    
1605    if ({    if ({
1606         0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, # HT, LF, VT, FF,         0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, # HT, LF, VT, FF,
# Line 1584  sub _tokenize_attempt_to_consume_an_enti Line 1613  sub _tokenize_attempt_to_consume_an_enti
1613      !!!next-input-character;      !!!next-input-character;
1614      if ($self->{next_input_character} == 0x0078 or # x      if ($self->{next_input_character} == 0x0078 or # x
1615          $self->{next_input_character} == 0x0058) { # X          $self->{next_input_character} == 0x0058) { # X
1616        my $num;        my $code;
1617        X: {        X: {
1618          my $x_char = $self->{next_input_character};          my $x_char = $self->{next_input_character};
1619          !!!next-input-character;          !!!next-input-character;
1620          if (0x0030 <= $self->{next_input_character} and          if (0x0030 <= $self->{next_input_character} and
1621              $self->{next_input_character} <= 0x0039) { # 0..9              $self->{next_input_character} <= 0x0039) { # 0..9
1622            $num ||= 0;            $code ||= 0;
1623            $num *= 0x10;            $code *= 0x10;
1624            $num += $self->{next_input_character} - 0x0030;            $code += $self->{next_input_character} - 0x0030;
1625            redo X;            redo X;
1626          } elsif (0x0061 <= $self->{next_input_character} and          } elsif (0x0061 <= $self->{next_input_character} and
1627                   $self->{next_input_character} <= 0x0066) { # a..f                   $self->{next_input_character} <= 0x0066) { # a..f
1628            ## ISSUE: the spec says U+0078, which is apparently incorrect            $code ||= 0;
1629            $num ||= 0;            $code *= 0x10;
1630            $num *= 0x10;            $code += $self->{next_input_character} - 0x0060 + 9;
           $num += $self->{next_input_character} - 0x0060 + 9;  
1631            redo X;            redo X;
1632          } elsif (0x0041 <= $self->{next_input_character} and          } elsif (0x0041 <= $self->{next_input_character} and
1633                   $self->{next_input_character} <= 0x0046) { # A..F                   $self->{next_input_character} <= 0x0046) { # A..F
1634            ## ISSUE: the spec says U+0058, which is apparently incorrect            $code ||= 0;
1635            $num ||= 0;            $code *= 0x10;
1636            $num *= 0x10;            $code += $self->{next_input_character} - 0x0040 + 9;
           $num += $self->{next_input_character} - 0x0040 + 9;  
1637            redo X;            redo X;
1638          } elsif (not defined $num) { # no hexadecimal digit          } elsif (not defined $code) { # no hexadecimal digit
1639            !!!parse-error (type => 'bare hcro');            !!!parse-error (type => 'bare hcro');
1640            $self->{next_input_character} = 0x0023; # #            $self->{next_input_character} = 0x0023; # #
1641            !!!back-next-input-character ($x_char);            !!!back-next-input-character ($x_char);
# Line 1619  sub _tokenize_attempt_to_consume_an_enti Line 1646  sub _tokenize_attempt_to_consume_an_enti
1646            !!!parse-error (type => 'no refc');            !!!parse-error (type => 'no refc');
1647          }          }
1648    
1649          ## TODO: check the definition for |a valid Unicode character|.          if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {
1650          ## <http://lists.whatwg.org/pipermail/whatwg-whatwg.org/2006-December/thread.html#8189>            !!!parse-error (type => sprintf 'invalid character reference:U+%04X', $code);
1651          if ($num > 1114111 or $num == 0) {            $code = 0xFFFD;
1652            $num = 0xFFFD; # REPLACEMENT CHARACTER          } elsif ($code > 0x10FFFF) {
1653            ## ISSUE: Why this is not an error?            !!!parse-error (type => sprintf 'invalid character reference:U-%08X', $code);
1654          } elsif (0x80 <= $num and $num <= 0x9F) {            $code = 0xFFFD;
1655            !!!parse-error (type => sprintf 'c1 entity:U+%04X', $num);          } elsif ($code == 0x000D) {
1656            $num = $c1_entity_char->{$num};            !!!parse-error (type => 'CR character reference');
1657              $code = 0x000A;
1658            } elsif (0x80 <= $code and $code <= 0x9F) {
1659              !!!parse-error (type => sprintf 'C1 character reference:U+%04X', $code);
1660              $code = $c1_entity_char->{$code};
1661          }          }
1662    
1663          return {type => 'character', data => chr $num};          return {type => 'character', data => chr $code};
1664        } # X        } # X
1665      } elsif (0x0030 <= $self->{next_input_character} and      } elsif (0x0030 <= $self->{next_input_character} and
1666               $self->{next_input_character} <= 0x0039) { # 0..9               $self->{next_input_character} <= 0x0039) { # 0..9
# Line 1650  sub _tokenize_attempt_to_consume_an_enti Line 1681  sub _tokenize_attempt_to_consume_an_enti
1681          !!!parse-error (type => 'no refc');          !!!parse-error (type => 'no refc');
1682        }        }
1683    
1684        ## TODO: check the definition for |a valid Unicode character|.        if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {
1685        if ($code > 1114111 or $code == 0) {          !!!parse-error (type => sprintf 'invalid character reference:U+%04X', $code);
1686          $code = 0xFFFD; # REPLACEMENT CHARACTER          $code = 0xFFFD;
1687          ## ISSUE: Why this is not an error?        } elsif ($code > 0x10FFFF) {
1688            !!!parse-error (type => sprintf 'invalid character reference:U-%08X', $code);
1689            $code = 0xFFFD;
1690          } elsif ($code == 0x000D) {
1691            !!!parse-error (type => 'CR character reference');
1692            $code = 0x000A;
1693        } elsif (0x80 <= $code and $code <= 0x9F) {        } elsif (0x80 <= $code and $code <= 0x9F) {
1694          !!!parse-error (type => sprintf 'c1 entity:U+%04X', $code);          !!!parse-error (type => sprintf 'C1 character reference:U+%04X', $code);
1695          $code = $c1_entity_char->{$code};          $code = $c1_entity_char->{$code};
1696        }        }
1697                
# Line 1689  sub _tokenize_attempt_to_consume_an_enti Line 1725  sub _tokenize_attempt_to_consume_an_enti
1725              $self->{next_input_character} == 0x003B)) { # ;              $self->{next_input_character} == 0x003B)) { # ;
1726        $entity_name .= chr $self->{next_input_character};        $entity_name .= chr $self->{next_input_character};
1727        if (defined $EntityChar->{$entity_name}) {        if (defined $EntityChar->{$entity_name}) {
         $value = $EntityChar->{$entity_name};  
1728          if ($self->{next_input_character} == 0x003B) { # ;          if ($self->{next_input_character} == 0x003B) { # ;
1729              $value = $EntityChar->{$entity_name};
1730            $match = 1;            $match = 1;
1731            !!!next-input-character;            !!!next-input-character;
1732            last;            last;
1733          } else {          } elsif (not $in_attr) {
1734              $value = $EntityChar->{$entity_name};
1735            $match = -1;            $match = -1;
1736            } else {
1737              $value .= chr $self->{next_input_character};
1738          }          }
1739        } else {        } else {
1740          $value .= chr $self->{next_input_character};          $value .= chr $self->{next_input_character};
# Line 1706  sub _tokenize_attempt_to_consume_an_enti Line 1745  sub _tokenize_attempt_to_consume_an_enti
1745      if ($match > 0) {      if ($match > 0) {
1746        return {type => 'character', data => $value};        return {type => 'character', data => $value};
1747      } elsif ($match < 0) {      } elsif ($match < 0) {
1748        !!!parse-error (type => 'refc');        !!!parse-error (type => 'no refc');
1749        return {type => 'character', data => $value};        return {type => 'character', data => $value};
1750      } else {      } else {
1751        !!!parse-error (type => 'bare ero');        !!!parse-error (type => 'bare ero');
1752        ## NOTE: No characters are consumed in the spec.        ## NOTE: No characters are consumed in the spec.
1753        !!!back-token ({type => 'character', data => $value});        return {type => 'character', data => '&'.$value};
       return undef;  
1754      }      }
1755    } else {    } else {
1756      ## no characters are consumed      ## no characters are consumed
# Line 1907  sub _tree_construction_initial ($) { Line 1945  sub _tree_construction_initial ($) {
1945      } elsif ($token->{type} eq 'character') {      } elsif ($token->{type} eq 'character') {
1946        if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D        if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D
1947          ## Ignore the token          ## Ignore the token
1948    
1949          unless (length $token->{data}) {          unless (length $token->{data}) {
1950            ## Stay in the phase            ## Stay in the phase
1951            !!!next-token;            !!!next-token;
# Line 1949  sub _tree_construction_root_element ($) Line 1988  sub _tree_construction_root_element ($)
1988          !!!next-token;          !!!next-token;
1989          redo B;          redo B;
1990        } elsif ($token->{type} eq 'character') {        } elsif ($token->{type} eq 'character') {
1991          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D
1992            $self->{document}->manakai_append_text ($1);            ## Ignore the token.
1993            ## ISSUE: DOM3 Core does not allow Document > Text  
1994            unless (length $token->{data}) {            unless (length $token->{data}) {
1995              ## Stay in the phase              ## Stay in the phase
1996              !!!next-token;              !!!next-token;
# Line 1991  sub _reset_insertion_mode ($) { Line 2030  sub _reset_insertion_mode ($) {
2030            
2031      ## Step 3      ## Step 3
2032      S3: {      S3: {
2033        $last = 1 if $self->{open_elements}->[0]->[0] eq $node->[0];        ## ISSUE: Oops! "If node is the first node in the stack of open
2034        if (defined $self->{inner_html_node}) {        ## elements, then set last to true. If the context element of the
2035          if ($self->{inner_html_node}->[1] eq 'td' or        ## HTML fragment parsing algorithm is neither a td element nor a
2036              $self->{inner_html_node}->[1] eq 'th') {        ## th element, then set node to the context element. (fragment case)":
2037            #        ## The second "if" is in the scope of the first "if"!?
2038          } else {        if ($self->{open_elements}->[0]->[0] eq $node->[0]) {
2039            $node = $self->{inner_html_node};          $last = 1;
2040            if (defined $self->{inner_html_node}) {
2041              if ($self->{inner_html_node}->[1] eq 'td' or
2042                  $self->{inner_html_node}->[1] eq 'th') {
2043                #
2044              } else {
2045                $node = $self->{inner_html_node};
2046              }
2047          }          }
2048        }        }
2049            
# Line 2128  sub _tree_construction_main ($) { Line 2174  sub _tree_construction_main ($) {
2174      }      }
2175    }; # $clear_up_to_marker    }; # $clear_up_to_marker
2176    
2177    my $style_start_tag = sub {    my $parse_rcdata = sub ($$) {
2178      my $style_el; !!!create-element ($style_el, 'style', $token->{attributes});      my ($content_model_flag, $insert) = @_;
2179      ## $self->{insertion_mode} eq 'in head' and ... (always true)  
2180      (($self->{insertion_mode} eq 'in head' and defined $self->{head_element})      ## Step 1
2181       ? $self->{head_element} : $self->{open_elements}->[-1]->[0])      my $start_tag_name = $token->{tag_name};
2182        ->append_child ($style_el);      my $el;
2183      $self->{content_model_flag} = 'CDATA';      !!!create-element ($el, $start_tag_name, $token->{attributes});
2184    
2185        ## Step 2
2186        $insert->($el); # /context node/->append_child ($el)
2187    
2188        ## Step 3
2189        $self->{content_model_flag} = $content_model_flag; # CDATA or RCDATA
2190      delete $self->{escape}; # MUST      delete $self->{escape}; # MUST
2191                  
2192        ## Step 4
2193      my $text = '';      my $text = '';
2194      !!!next-token;      !!!next-token;
2195      while ($token->{type} eq 'character') {      while ($token->{type} eq 'character') { # or until stop tokenizing
2196        $text .= $token->{data};        $text .= $token->{data};
2197        !!!next-token;        !!!next-token;
2198      } # stop if non-character token or tokenizer stops tokenising      }
2199    
2200        ## Step 5
2201      if (length $text) {      if (length $text) {
2202        $style_el->manakai_append_text ($text);        my $text = $self->{document}->create_text_node ($text);
2203          $el->append_child ($text);
2204      }      }
2205        
2206        ## Step 6
2207      $self->{content_model_flag} = 'PCDATA';      $self->{content_model_flag} = 'PCDATA';
2208                  
2209      if ($token->{type} eq 'end tag' and $token->{tag_name} eq 'style') {      ## Step 7
2210        if ($token->{type} eq 'end tag' and $token->{tag_name} eq $start_tag_name) {
2211        ## Ignore the token        ## Ignore the token
2212      } else {      } else {
2213        !!!parse-error (type => 'in CDATA:#'.$token->{type});        !!!parse-error (type => 'in '.$content_model_flag.':#'.$token->{type});
       ## ISSUE: And ignore?  
2214      }      }
2215      !!!next-token;      !!!next-token;
2216    }; # $style_start_tag    }; # $parse_rcdata
2217    
2218    my $script_start_tag = sub {    my $script_start_tag = sub ($) {
2219        my $insert = $_[0];
2220      my $script_el;      my $script_el;
2221      !!!create-element ($script_el, 'script', $token->{attributes});      !!!create-element ($script_el, 'script', $token->{attributes});
2222      ## TODO: mark as "parser-inserted"      ## TODO: mark as "parser-inserted"
# Line 2192  sub _tree_construction_main ($) { Line 2250  sub _tree_construction_main ($) {
2250      } else {      } else {
2251        ## TODO: $old_insertion_point = current insertion point        ## TODO: $old_insertion_point = current insertion point
2252        ## TODO: insertion point = just before the next input character        ## TODO: insertion point = just before the next input character
2253          
2254        (($self->{insertion_mode} eq 'in head' and defined $self->{head_element})        $insert->($script_el);
        ? $self->{head_element} : $self->{open_elements}->[-1]->[0])->append_child ($script_el);  
2255                
2256        ## TODO: insertion point = $old_insertion_point (might be "undefined")        ## TODO: insertion point = $old_insertion_point (might be "undefined")
2257                
# Line 2388  sub _tree_construction_main ($) { Line 2445  sub _tree_construction_main ($) {
2445    }; # $formatting_end_tag    }; # $formatting_end_tag
2446    
2447    my $insert_to_current = sub {    my $insert_to_current = sub {
2448      $self->{open_elements}->[-1]->[0]->append_child (shift);      $self->{open_elements}->[-1]->[0]->append_child ($_[0]);
2449    }; # $insert_to_current    }; # $insert_to_current
2450    
2451    my $insert_to_foster = sub {    my $insert_to_foster = sub {
# Line 2426  sub _tree_construction_main ($) { Line 2483  sub _tree_construction_main ($) {
2483      my $insert = shift;      my $insert = shift;
2484      if ($token->{type} eq 'start tag') {      if ($token->{type} eq 'start tag') {
2485        if ($token->{tag_name} eq 'script') {        if ($token->{tag_name} eq 'script') {
2486          $script_start_tag->();          ## NOTE: This is an "as if in head" code clone
2487            $script_start_tag->($insert);
2488          return;          return;
2489        } elsif ($token->{tag_name} eq 'style') {        } elsif ($token->{tag_name} eq 'style') {
2490          $style_start_tag->();          ## NOTE: This is an "as if in head" code clone
2491            $parse_rcdata->('CDATA', $insert);
2492          return;          return;
2493        } elsif ({        } elsif ({
2494                  base => 1, link => 1, meta => 1,                  base => 1, link => 1, meta => 1,
2495                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
2496          ## NOTE: This is an "as if in head" code clone          ## NOTE: This is an "as if in head" code clone, only "-t" differs
2497          my $el;          !!!insert-element-t ($token->{tag_name}, $token->{attributes});
2498          !!!create-element ($el, $token->{tag_name}, $token->{attributes});          pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
         if ($self->{insertion_mode} eq 'in head' and  
             defined $self->{head_element}) {  
           $self->{head_element}->append_child ($el);  
         } else {  
           $insert->($el);  
         }  
           
2499          !!!next-token;          !!!next-token;
2500            ## TODO: Extracting |charset| from |meta|.
2501          return;          return;
2502        } elsif ($token->{tag_name} eq 'title') {        } elsif ($token->{tag_name} eq 'title') {
2503          !!!parse-error (type => 'in body:title');          !!!parse-error (type => 'in body:title');
2504          ## NOTE: There is an "as if in head" code clone          ## NOTE: This is an "as if in head" code clone
2505          my $title_el;          $parse_rcdata->('RCDATA', $insert);
         !!!create-element ($title_el, 'title', $token->{attributes});  
         (defined $self->{head_element} ? $self->{head_element} : $self->{open_elements}->[-1]->[0])  
           ->append_child ($title_el);  
         $self->{content_model_flag} = 'RCDATA';  
         delete $self->{escape}; # MUST  
           
         my $text = '';  
         !!!next-token;  
         while ($token->{type} eq 'character') {  
           $text .= $token->{data};  
           !!!next-token;  
         }  
         if (length $text) {  
           $title_el->manakai_append_text ($text);  
         }  
           
         $self->{content_model_flag} = 'PCDATA';  
           
         if ($token->{type} eq 'end tag' and  
             $token->{tag_name} eq 'title') {  
           ## Ignore the token  
         } else {  
           !!!parse-error (type => 'in RCDATA:#'.$token->{type});  
           ## ISSUE: And ignore?  
         }  
         !!!next-token;  
2506          return;          return;
2507        } elsif ($token->{tag_name} eq 'body') {        } elsif ($token->{tag_name} eq 'body') {
2508          !!!parse-error (type => 'in body:body');          !!!parse-error (type => 'in body:body');
# Line 2578  sub _tree_construction_main ($) { Line 2605  sub _tree_construction_main ($) {
2605              if ($i != -1) {              if ($i != -1) {
2606                !!!parse-error (type => 'end tag missing:'.                !!!parse-error (type => 'end tag missing:'.
2607                                $self->{open_elements}->[-1]->[1]);                                $self->{open_elements}->[-1]->[1]);
               ## TODO: test  
2608              }              }
2609              splice @{$self->{open_elements}}, $i;              splice @{$self->{open_elements}}, $i;
2610              last LI;              last LI;
# Line 2626  sub _tree_construction_main ($) { Line 2652  sub _tree_construction_main ($) {
2652              if ($i != -1) {              if ($i != -1) {
2653                !!!parse-error (type => 'end tag missing:'.                !!!parse-error (type => 'end tag missing:'.
2654                                $self->{open_elements}->[-1]->[1]);                                $self->{open_elements}->[-1]->[1]);
               ## TODO: test  
2655              }              }
2656              splice @{$self->{open_elements}}, $i;              splice @{$self->{open_elements}}, $i;
2657              last LI;              last LI;
# Line 2821  sub _tree_construction_main ($) { Line 2846  sub _tree_construction_main ($) {
2846          return;          return;
2847        } elsif ($token->{tag_name} eq 'xmp') {        } elsif ($token->{tag_name} eq 'xmp') {
2848          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
2849                    $parse_rcdata->('CDATA', $insert);
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
           
         $self->{content_model_flag} = 'CDATA';  
         delete $self->{escape}; # MUST  
           
         !!!next-token;  
2850          return;          return;
2851        } elsif ($token->{tag_name} eq 'table') {        } elsif ($token->{tag_name} eq 'table') {
2852          ## has a p element in scope          ## has a p element in scope
# Line 2936  sub _tree_construction_main ($) { Line 2955  sub _tree_construction_main ($) {
2955            !!!back-token (@tokens);            !!!back-token (@tokens);
2956            return;            return;
2957          }          }
2958        } elsif ({        } elsif ($token->{tag_name} eq 'textarea') {
                 textarea => 1,  
                 iframe => 1,  
                 noembed => 1,  
                 noframes => 1,  
                 noscript => 0, ## TODO: 1 if scripting is enabled  
                }->{$token->{tag_name}}) {  
2959          my $tag_name = $token->{tag_name};          my $tag_name = $token->{tag_name};
2960          my $el;          my $el;
2961          !!!create-element ($el, $token->{tag_name}, $token->{attributes});          !!!create-element ($el, $token->{tag_name}, $token->{attributes});
2962                    
2963          if ($token->{tag_name} eq 'textarea') {          ## TODO: $self->{form_element} if defined
2964            ## TODO: $self->{form_element} if defined          $self->{content_model_flag} = 'RCDATA';
           $self->{content_model_flag} = 'RCDATA';  
         } else {  
           $self->{content_model_flag} = 'CDATA';  
         }  
2965          delete $self->{escape}; # MUST          delete $self->{escape}; # MUST
2966                    
2967          $insert->($el);          $insert->($el);
2968                    
2969          my $text = '';          my $text = '';
2970          if ($token->{tag_name} eq 'textarea') {          !!!next-token;
2971            !!!next-token;          if ($token->{type} eq 'character') {
2972            if ($token->{type} eq 'character') {            $token->{data} =~ s/^\x0A//;
2973              $token->{data} =~ s/^\x0A//;            unless (length $token->{data}) {
2974              unless (length $token->{data}) {              !!!next-token;
               !!!next-token;  
             }  
2975            }            }
         } else {  
           !!!next-token;  
2976          }          }
2977          while ($token->{type} eq 'character') {          while ($token->{type} eq 'character') {
2978            $text .= $token->{data};            $text .= $token->{data};
# Line 2983  sub _tree_construction_main ($) { Line 2988  sub _tree_construction_main ($) {
2988              $token->{tag_name} eq $tag_name) {              $token->{tag_name} eq $tag_name) {
2989            ## Ignore the token            ## Ignore the token
2990          } else {          } else {
2991            if ($token->{tag_name} eq 'textarea') {            !!!parse-error (type => 'in RCDATA:#'.$token->{type});
             !!!parse-error (type => 'in RCDATA:#'.$token->{type});  
           } else {  
             !!!parse-error (type => 'in CDATA:#'.$token->{type});  
           }  
           ## ISSUE: And ignore?  
2992          }          }
2993          !!!next-token;          !!!next-token;
2994          return;          return;
2995          } elsif ({
2996                    iframe => 1,
2997                    noembed => 1,
2998                    noframes => 1,
2999                    noscript => 0, ## TODO: 1 if scripting is enabled
3000                   }->{$token->{tag_name}}) {
3001            $parse_rcdata->('CDATA', $insert);
3002            return;
3003        } elsif ($token->{tag_name} eq 'select') {        } elsif ($token->{tag_name} eq 'select') {
3004          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
3005                    
# Line 3238  sub _tree_construction_main ($) { Line 3246  sub _tree_construction_main ($) {
3246                  #not $phrasing_category->{$node->[1]} and                  #not $phrasing_category->{$node->[1]} and
3247                  ($special_category->{$node->[1]} or                  ($special_category->{$node->[1]} or
3248                   $scoping_category->{$node->[1]})) {                   $scoping_category->{$node->[1]})) {
3249                !!!parse-error (type => 'not closed:'.$node->[1]);                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3250                ## Ignore the token                ## Ignore the token
3251                !!!next-token;                !!!next-token;
3252                last S2;                last S2;
# Line 3267  sub _tree_construction_main ($) { Line 3275  sub _tree_construction_main ($) {
3275          redo B;          redo B;
3276        } elsif ($token->{type} eq 'start tag' and        } elsif ($token->{type} eq 'start tag' and
3277                 $token->{tag_name} eq 'html') {                 $token->{tag_name} eq 'html') {
3278          ## TODO: unless it is the first start tag token, parse-error  ## ISSUE: "aa<html>" is not a parse error.
3279    ## ISSUE: "<html>" in fragment is not a parse error.
3280            unless ($token->{first_start_tag}) {
3281              !!!parse-error (type => 'not first start tag');
3282            }
3283          my $top_el = $self->{open_elements}->[0]->[0];          my $top_el = $self->{open_elements}->[0]->[0];
3284          for my $attr_name (keys %{$token->{attributes}}) {          for my $attr_name (keys %{$token->{attributes}}) {
3285            unless ($top_el->has_attribute_ns (undef, $attr_name)) {            unless ($top_el->has_attribute_ns (undef, $attr_name)) {
# Line 3358  sub _tree_construction_main ($) { Line 3370  sub _tree_construction_main ($) {
3370            } else {            } else {
3371              die "$0: $token->{type}: Unknown type";              die "$0: $token->{type}: Unknown type";
3372            }            }
3373          } elsif ($self->{insertion_mode} eq 'in head') {          } elsif ($self->{insertion_mode} eq 'in head' or
3374                     $self->{insertion_mode} eq 'in head noscript' or
3375                     $self->{insertion_mode} eq 'after head') {
3376            if ($token->{type} eq 'character') {            if ($token->{type} eq 'character') {
3377              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {
3378                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
# Line 3375  sub _tree_construction_main ($) { Line 3389  sub _tree_construction_main ($) {
3389              !!!next-token;              !!!next-token;
3390              redo B;              redo B;
3391            } elsif ($token->{type} eq 'start tag') {            } elsif ($token->{type} eq 'start tag') {
3392              if ($token->{tag_name} eq 'title') {              if ({base => ($self->{insertion_mode} eq 'in head' or
3393                ## NOTE: There is an "as if in head" code clone                            $self->{insertion_mode} eq 'after head'),
3394                my $title_el;                   link => 1, meta => 1}->{$token->{tag_name}}) {
3395                !!!create-element ($title_el, 'title', $token->{attributes});                ## NOTE: There is a "as if in head" code clone.
3396                (defined $self->{head_element} ? $self->{head_element} : $self->{open_elements}->[-1]->[0])                if ($self->{insertion_mode} eq 'after head') {
3397                  ->append_child ($title_el);                  !!!parse-error (type => 'after head:'.$token->{tag_name});
3398                $self->{content_model_flag} = 'RCDATA';                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
3399                delete $self->{escape}; # MUST                }
3400                  !!!insert-element ($token->{tag_name}, $token->{attributes});
3401                my $text = '';                pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
3402                  ## TODO: Extracting |charset| from |meta|.
3403                  pop @{$self->{open_elements}}
3404                      if $self->{insertion_mode} eq 'after head';
3405                !!!next-token;                !!!next-token;
3406                while ($token->{type} eq 'character') {                redo B;
3407                  $text .= $token->{data};              } elsif ($token->{tag_name} eq 'title' and
3408                         $self->{insertion_mode} eq 'in head') {
3409                  ## NOTE: There is a "as if in head" code clone.
3410                  if ($self->{insertion_mode} eq 'after head') {
3411                    !!!parse-error (type => 'after head:'.$token->{tag_name});
3412                    push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
3413                  }
3414                  $parse_rcdata->('RCDATA', $insert_to_current);
3415                  pop @{$self->{open_elements}}
3416                      if $self->{insertion_mode} eq 'after head';
3417                  redo B;
3418                } elsif ($token->{tag_name} eq 'style') {
3419                  ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and
3420                  ## insertion mode 'in head')
3421                  ## NOTE: There is a "as if in head" code clone.
3422                  if ($self->{insertion_mode} eq 'after head') {
3423                    !!!parse-error (type => 'after head:'.$token->{tag_name});
3424                    push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
3425                  }
3426                  $parse_rcdata->('CDATA', $insert_to_current);
3427                  pop @{$self->{open_elements}}
3428                      if $self->{insertion_mode} eq 'after head';
3429                  redo B;
3430                } elsif ($token->{tag_name} eq 'noscript') {
3431                  if ($self->{insertion_mode} eq 'in head') {
3432                    ## NOTE: and scripting is disalbed
3433                    !!!insert-element ($token->{tag_name}, $token->{attributes});
3434                    $self->{insertion_mode} = 'in head noscript';
3435                  !!!next-token;                  !!!next-token;
3436                }                  redo B;
3437                if (length $text) {                } elsif ($self->{insertion_mode} eq 'in head noscript') {
3438                  $title_el->manakai_append_text ($text);                  !!!parse-error (type => 'in noscript:noscript');
               }  
                 
               $self->{content_model_flag} = 'PCDATA';  
                 
               if ($token->{type} eq 'end tag' and  
                   $token->{tag_name} eq 'title') {  
3439                  ## Ignore the token                  ## Ignore the token
3440                    redo B;
3441                } else {                } else {
3442                  !!!parse-error (type => 'in RCDATA:#'.$token->{type});                  #
                 ## ISSUE: And ignore?  
3443                }                }
3444                } elsif ($token->{tag_name} eq 'head' and
3445                         $self->{insertion_mode} ne 'after head') {
3446                  !!!parse-error (type => 'in head:head'); # or in head noscript
3447                  ## Ignore the token
3448                !!!next-token;                !!!next-token;
3449                redo B;                redo B;
3450              } elsif ($token->{tag_name} eq 'style') {              } elsif ($self->{insertion_mode} ne 'in head noscript' and
3451                $style_start_tag->();                       $token->{tag_name} eq 'script') {
3452                redo B;                if ($self->{insertion_mode} eq 'after head') {
3453              } elsif ($token->{tag_name} eq 'script') {                  !!!parse-error (type => 'after head:'.$token->{tag_name});
3454                $script_start_tag->();                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
3455                  }
3456                  ## NOTE: There is a "as if in head" code clone.
3457                  $script_start_tag->($insert_to_current);
3458                  pop @{$self->{open_elements}}
3459                      if $self->{insertion_mode} eq 'after head';
3460                redo B;                redo B;
3461              } elsif ({base => 1, link => 1, meta => 1}->{$token->{tag_name}}) {              } elsif ($self->{insertion_mode} eq 'after head' and
3462                ## NOTE: There are "as if in head" code clones                       $token->{tag_name} eq 'body') {
3463                my $el;                !!!insert-element ('body', $token->{attributes});
3464                !!!create-element ($el, $token->{tag_name}, $token->{attributes});                $self->{insertion_mode} = 'in body';
               if ($self->{insertion_mode} eq 'in head' and  
                   defined $self->{head_element}) {  
                 $self->{head_element}->append_child ($el);  
               } else {  
                 $self->{open_elements}->[-1]->[0]->append_child ($el);  
               }  
   
3465                !!!next-token;                !!!next-token;
3466                redo B;                redo B;
3467              } elsif ($token->{tag_name} eq 'head') {              } elsif ($self->{insertion_mode} eq 'after head' and
3468                !!!parse-error (type => 'in head:head');                       $token->{tag_name} eq 'frameset') {
3469                ## Ignore the token                !!!insert-element ('frameset', $token->{attributes});
3470                  $self->{insertion_mode} = 'in frameset';
3471                !!!next-token;                !!!next-token;
3472                redo B;                redo B;
3473              } else {              } else {
3474                #                #
3475              }              }
3476            } elsif ($token->{type} eq 'end tag') {            } elsif ($token->{type} eq 'end tag') {
3477              if ($token->{tag_name} eq 'head') {              if ($self->{insertion_mode} eq 'in head' and
3478                if ($self->{open_elements}->[-1]->[1] eq 'head') {                  $token->{tag_name} eq 'head') {
3479                  pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
               } else {  
                 !!!parse-error (type => 'unmatched end tag:head');  
               }  
3480                $self->{insertion_mode} = 'after head';                $self->{insertion_mode} = 'after head';
3481                !!!next-token;                !!!next-token;
3482                redo B;                redo B;
3483              } elsif ($token->{tag_name} eq 'body' or              } elsif ($self->{insertion_mode} eq 'in head noscript' and
3484                       $token->{tag_name} eq 'html') {                  $token->{tag_name} eq 'noscript') {
3485                  pop @{$self->{open_elements}};
3486                  $self->{insertion_mode} = 'in head';
3487                  !!!next-token;
3488                  redo B;
3489                } elsif ($self->{insertion_mode} eq 'in head' and
3490                         ($token->{tag_name} eq 'body' or
3491                          $token->{tag_name} eq 'html')) {
3492                #                #
3493              } else {              } elsif ($self->{insertion_mode} ne 'after head') {
3494                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3495                ## Ignore the token                ## Ignore the token
3496                !!!next-token;                !!!next-token;
3497                redo B;                redo B;
3498                } else {
3499                  #
3500              }              }
3501            } else {            } else {
3502              #              #
3503            }            }
3504    
3505            if ($self->{open_elements}->[-1]->[1] eq 'head') {            ## As if </head> or </noscript> or <body>
3506              ## As if </head>            if ($self->{insertion_mode} eq 'in head') {
3507                pop @{$self->{open_elements}};
3508                $self->{insertion_mode} = 'after head';
3509              } elsif ($self->{insertion_mode} eq 'in head noscript') {
3510              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
3511                !!!parse-error (type => 'in noscript:'.(defined $token->{tag_name} ? ($token->{type} eq 'end tag' ? '/' : '') . $token->{tag_name} : '#' . $token->{type}));
3512                $self->{insertion_mode} = 'in head';
3513              } else { # 'after head'
3514                !!!insert-element ('body');
3515                $self->{insertion_mode} = 'in body';
3516            }            }
           $self->{insertion_mode} = 'after head';  
3517            ## reprocess            ## reprocess
3518            redo B;            redo B;
3519    
3520            ## ISSUE: An issue in the spec.            ## ISSUE: An issue in the spec.
         } elsif ($self->{insertion_mode} eq 'after head') {  
           if ($token->{type} eq 'character') {  
             if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {  
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);  
               unless (length $token->{data}) {  
                 !!!next-token;  
                 redo B;  
               }  
             }  
               
             #  
           } elsif ($token->{type} eq 'comment') {  
             my $comment = $self->{document}->create_comment ($token->{data});  
             $self->{open_elements}->[-1]->[0]->append_child ($comment);  
             !!!next-token;  
             redo B;  
           } elsif ($token->{type} eq 'start tag') {  
             if ($token->{tag_name} eq 'body') {  
               !!!insert-element ('body', $token->{attributes});  
               $self->{insertion_mode} = 'in body';  
               !!!next-token;  
               redo B;  
             } elsif ($token->{tag_name} eq 'frameset') {  
               !!!insert-element ('frameset', $token->{attributes});  
               $self->{insertion_mode} = 'in frameset';  
               !!!next-token;  
               redo B;  
             } elsif ({  
                       base => 1, link => 1, meta => 1,  
                       script => 1, style => 1, title => 1,  
                      }->{$token->{tag_name}}) {  
               !!!parse-error (type => 'after head:'.$token->{tag_name});  
               $self->{insertion_mode} = 'in head';  
               ## reprocess  
               redo B;  
             } else {  
               #  
             }  
           } else {  
             #  
           }  
             
           ## As if <body>  
           !!!insert-element ('body');  
           $self->{insertion_mode} = 'in body';  
           ## reprocess  
           redo B;  
3521          } elsif ($self->{insertion_mode} eq 'in body') {          } elsif ($self->{insertion_mode} eq 'in body') {
3522            if ($token->{type} eq 'character') {            if ($token->{type} eq 'character') {
3523              ## NOTE: There is a code clone of "character in body".              ## NOTE: There is a code clone of "character in body".
# Line 5019  sub _tree_construction_main ($) { Line 5026  sub _tree_construction_main ($) {
5026            }            }
5027                        
5028            if (defined $token->{tag_name}) {            if (defined $token->{tag_name}) {
5029              !!!parse-error (type => 'in frameset:'.$token->{tag_name});              !!!parse-error (type => 'in frameset:'.($token->{type} eq 'end tag' ? '/' : '').$token->{tag_name});
5030            } else {            } else {
5031              !!!parse-error (type => 'in frameset:#'.$token->{type});              !!!parse-error (type => 'in frameset:#'.$token->{type});
5032            }            }
# Line 5063  sub _tree_construction_main ($) { Line 5070  sub _tree_construction_main ($) {
5070            }            }
5071                        
5072            if (defined $token->{tag_name}) {            if (defined $token->{tag_name}) {
5073              !!!parse-error (type => 'after frameset:'.$token->{tag_name});              !!!parse-error (type => 'after frameset:'.($token->{tag_name} eq 'end tag' ? '/' : '').$token->{tag_name});
5074            } else {            } else {
5075              !!!parse-error (type => 'after frameset:#'.$token->{type});              !!!parse-error (type => 'after frameset:#'.$token->{type});
5076            }            }
# Line 5113  sub _tree_construction_main ($) { Line 5120  sub _tree_construction_main ($) {
5120          redo B;          redo B;
5121        } elsif ($token->{type} eq 'start tag' or        } elsif ($token->{type} eq 'start tag' or
5122                 $token->{type} eq 'end tag') {                 $token->{type} eq 'end tag') {
5123          !!!parse-error (type => 'after html:'.$token->{tag_name});          !!!parse-error (type => 'after html:'.($token->{type} eq 'end tag' ? '/' : '').$token->{tag_name});
5124          $phase = 'main';          $phase = 'main';
5125          ## reprocess          ## reprocess
5126          redo B;          redo B;
# Line 5322  sub get_inner_html ($$$) { Line 5329  sub get_inner_html ($$$) {
5329            
5330      my $nt = $child->node_type;      my $nt = $child->node_type;
5331      if ($nt == 1) { # Element      if ($nt == 1) { # Element
5332        my $tag_name = lc $child->tag_name; ## ISSUE: Definition of "lowercase"        my $tag_name = $child->tag_name; ## TODO: manakai_tag_name
5333        $s .= '<' . $tag_name;        $s .= '<' . $tag_name;
5334          ## NOTE: Non-HTML case:
5335        ## ISSUE: Non-html elements        ## <http://permalink.gmane.org/gmane.org.w3c.whatwg.discuss/11191>
5336    
5337        my @attrs = @{$child->attributes}; # sort order MUST be stable        my @attrs = @{$child->attributes}; # sort order MUST be stable
5338        for my $attr (@attrs) { # order is implementation dependent        for my $attr (@attrs) { # order is implementation dependent
5339          my $attr_name = lc $attr->name; ## ISSUE: Definition of "lowercase"          my $attr_name = $attr->name; ## TODO: manakai_name
5340          $s .= ' ' . $attr_name . '="';          $s .= ' ' . $attr_name . '="';
5341          my $attr_value = $attr->value;          my $attr_value = $attr->value;
5342          ## escape          ## escape
# Line 5353  sub get_inner_html ($$$) { Line 5360  sub get_inner_html ($$$) {
5360        if (not $in_cdata and {        if (not $in_cdata and {
5361          style => 1, script => 1, xmp => 1, iframe => 1,          style => 1, script => 1, xmp => 1, iframe => 1,
5362          noembed => 1, noframes => 1, noscript => 1,          noembed => 1, noframes => 1, noscript => 1,
5363            plaintext => 1,
5364        }->{$tag_name}) {        }->{$tag_name}) {
5365          unshift @node, 'cdata-out';          unshift @node, 'cdata-out';
5366          $in_cdata = 1;          $in_cdata = 1;

Legend:
Removed from v.1.24  
changed lines
  Added in v.1.30

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24