/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src
Suika

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory Parent Directory | Revision Log Revision Log | View Patch Patch

revision 1.24 by wakaba, Sat Jun 23 16:42:43 2007 UTC revision 1.28 by wakaba, Mon Jun 25 00:14:40 2007 UTC
# Line 247  sub _get_next_token ($) { Line 247  sub _get_next_token ($) {
247      } elsif ($self->{state} eq 'entity data') {      } elsif ($self->{state} eq 'entity data') {
248        ## (cannot happen in CDATA state)        ## (cannot happen in CDATA state)
249                
250        my $token = $self->_tokenize_attempt_to_consume_an_entity;        my $token = $self->_tokenize_attempt_to_consume_an_entity (0);
251    
252        $self->{state} = 'data';        $self->{state} = 'data';
253        # next-input-character is already done        # next-input-character is already done
# Line 418  sub _get_next_token ($) { Line 418  sub _get_next_token ($) {
418          redo A;          redo A;
419        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
420          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
421              $self->{current_token}->{first_start_tag}
422                  = not defined $self->{last_emitted_start_tag_name};
423            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
424          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
425            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 443  sub _get_next_token ($) { Line 445  sub _get_next_token ($) {
445        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
446          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
447          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
448              $self->{current_token}->{first_start_tag}
449                  = not defined $self->{last_emitted_start_tag_name};
450            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
451          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
452            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 489  sub _get_next_token ($) { Line 493  sub _get_next_token ($) {
493          redo A;          redo A;
494        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
495          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
496              $self->{current_token}->{first_start_tag}
497                  = not defined $self->{last_emitted_start_tag_name};
498            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
499          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
500            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 527  sub _get_next_token ($) { Line 533  sub _get_next_token ($) {
533        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
534          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
535          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
536              $self->{current_token}->{first_start_tag}
537                  = not defined $self->{last_emitted_start_tag_name};
538            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
539          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
540            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 578  sub _get_next_token ($) { Line 586  sub _get_next_token ($) {
586        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
587          $before_leave->();          $before_leave->();
588          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
589              $self->{current_token}->{first_start_tag}
590                  = not defined $self->{last_emitted_start_tag_name};
591            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
592          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
593            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 617  sub _get_next_token ($) { Line 627  sub _get_next_token ($) {
627          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
628          $before_leave->();          $before_leave->();
629          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
630              $self->{current_token}->{first_start_tag}
631                  = not defined $self->{last_emitted_start_tag_name};
632            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
633          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
634            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 653  sub _get_next_token ($) { Line 665  sub _get_next_token ($) {
665          redo A;          redo A;
666        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
667          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
668              $self->{current_token}->{first_start_tag}
669                  = not defined $self->{last_emitted_start_tag_name};
670            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
671          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
672            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 691  sub _get_next_token ($) { Line 705  sub _get_next_token ($) {
705        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
706          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
707          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
708              $self->{current_token}->{first_start_tag}
709                  = not defined $self->{last_emitted_start_tag_name};
710            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
711          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
712            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 736  sub _get_next_token ($) { Line 752  sub _get_next_token ($) {
752          redo A;          redo A;
753        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
754          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
755              $self->{current_token}->{first_start_tag}
756                  = not defined $self->{last_emitted_start_tag_name};
757            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
758          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
759            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 754  sub _get_next_token ($) { Line 772  sub _get_next_token ($) {
772        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
773          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
774          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
775              $self->{current_token}->{first_start_tag}
776                  = not defined $self->{last_emitted_start_tag_name};
777            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
778          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
779            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 788  sub _get_next_token ($) { Line 808  sub _get_next_token ($) {
808        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
809          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
810          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
811              $self->{current_token}->{first_start_tag}
812                  = not defined $self->{last_emitted_start_tag_name};
813            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
814          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
815            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 822  sub _get_next_token ($) { Line 844  sub _get_next_token ($) {
844        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
845          !!!parse-error (type => 'unclosed attribute value');          !!!parse-error (type => 'unclosed attribute value');
846          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
847              $self->{current_token}->{first_start_tag}
848                  = not defined $self->{last_emitted_start_tag_name};
849            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
850          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
851            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 859  sub _get_next_token ($) { Line 883  sub _get_next_token ($) {
883          redo A;          redo A;
884        } elsif ($self->{next_input_character} == 0x003E) { # >        } elsif ($self->{next_input_character} == 0x003E) { # >
885          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
886              $self->{current_token}->{first_start_tag}
887                  = not defined $self->{last_emitted_start_tag_name};
888            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
889          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
890            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 877  sub _get_next_token ($) { Line 903  sub _get_next_token ($) {
903        } elsif ($self->{next_input_character} == -1) {        } elsif ($self->{next_input_character} == -1) {
904          !!!parse-error (type => 'unclosed tag');          !!!parse-error (type => 'unclosed tag');
905          if ($self->{current_token}->{type} eq 'start tag') {          if ($self->{current_token}->{type} eq 'start tag') {
906              $self->{current_token}->{first_start_tag}
907                  = not defined $self->{last_emitted_start_tag_name};
908            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};            $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
909          } elsif ($self->{current_token}->{type} eq 'end tag') {          } elsif ($self->{current_token}->{type} eq 'end tag') {
910            $self->{content_model_flag} = 'PCDATA'; # MUST            $self->{content_model_flag} = 'PCDATA'; # MUST
# Line 899  sub _get_next_token ($) { Line 927  sub _get_next_token ($) {
927          redo A;          redo A;
928        }        }
929      } elsif ($self->{state} eq 'entity in attribute value') {      } elsif ($self->{state} eq 'entity in attribute value') {
930        my $token = $self->_tokenize_attempt_to_consume_an_entity;        my $token = $self->_tokenize_attempt_to_consume_an_entity (1);
931    
932        unless (defined $token) {        unless (defined $token) {
933          $self->{current_attribute}->{value} .= '&';          $self->{current_attribute}->{value} .= '&';
# Line 1409  sub _get_next_token ($) { Line 1437  sub _get_next_token ($) {
1437          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
1438    
1439          $self->{state} = 'data';          $self->{state} = 'data';
1440          ## recomsume          ## reconsume
1441    
1442          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1443          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
# Line 1452  sub _get_next_token ($) { Line 1480  sub _get_next_token ($) {
1480          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
1481    
1482          $self->{state} = 'data';          $self->{state} = 'data';
1483          ## recomsume          ## reconsume
1484    
1485          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1486          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
# Line 1527  sub _get_next_token ($) { Line 1555  sub _get_next_token ($) {
1555          !!!parse-error (type => 'unclosed DOCTYPE');          !!!parse-error (type => 'unclosed DOCTYPE');
1556    
1557          $self->{state} = 'data';          $self->{state} = 'data';
1558          ## recomsume          ## reconsume
1559    
1560          delete $self->{current_token}->{correct};          delete $self->{current_token}->{correct};
1561          !!!emit ($self->{current_token}); # DOCTYPE          !!!emit ($self->{current_token}); # DOCTYPE
# Line 1570  sub _get_next_token ($) { Line 1598  sub _get_next_token ($) {
1598    die "$0: _get_next_token: unexpected case";    die "$0: _get_next_token: unexpected case";
1599  } # _get_next_token  } # _get_next_token
1600    
1601  sub _tokenize_attempt_to_consume_an_entity ($) {  sub _tokenize_attempt_to_consume_an_entity ($$) {
1602    my $self = shift;    my ($self, $in_attr) = @_;
1603    
1604    if ({    if ({
1605         0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, # HT, LF, VT, FF,         0x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, # HT, LF, VT, FF,
# Line 1584  sub _tokenize_attempt_to_consume_an_enti Line 1612  sub _tokenize_attempt_to_consume_an_enti
1612      !!!next-input-character;      !!!next-input-character;
1613      if ($self->{next_input_character} == 0x0078 or # x      if ($self->{next_input_character} == 0x0078 or # x
1614          $self->{next_input_character} == 0x0058) { # X          $self->{next_input_character} == 0x0058) { # X
1615        my $num;        my $code;
1616        X: {        X: {
1617          my $x_char = $self->{next_input_character};          my $x_char = $self->{next_input_character};
1618          !!!next-input-character;          !!!next-input-character;
1619          if (0x0030 <= $self->{next_input_character} and          if (0x0030 <= $self->{next_input_character} and
1620              $self->{next_input_character} <= 0x0039) { # 0..9              $self->{next_input_character} <= 0x0039) { # 0..9
1621            $num ||= 0;            $code ||= 0;
1622            $num *= 0x10;            $code *= 0x10;
1623            $num += $self->{next_input_character} - 0x0030;            $code += $self->{next_input_character} - 0x0030;
1624            redo X;            redo X;
1625          } elsif (0x0061 <= $self->{next_input_character} and          } elsif (0x0061 <= $self->{next_input_character} and
1626                   $self->{next_input_character} <= 0x0066) { # a..f                   $self->{next_input_character} <= 0x0066) { # a..f
1627            ## ISSUE: the spec says U+0078, which is apparently incorrect            $code ||= 0;
1628            $num ||= 0;            $code *= 0x10;
1629            $num *= 0x10;            $code += $self->{next_input_character} - 0x0060 + 9;
           $num += $self->{next_input_character} - 0x0060 + 9;  
1630            redo X;            redo X;
1631          } elsif (0x0041 <= $self->{next_input_character} and          } elsif (0x0041 <= $self->{next_input_character} and
1632                   $self->{next_input_character} <= 0x0046) { # A..F                   $self->{next_input_character} <= 0x0046) { # A..F
1633            ## ISSUE: the spec says U+0058, which is apparently incorrect            $code ||= 0;
1634            $num ||= 0;            $code *= 0x10;
1635            $num *= 0x10;            $code += $self->{next_input_character} - 0x0040 + 9;
           $num += $self->{next_input_character} - 0x0040 + 9;  
1636            redo X;            redo X;
1637          } elsif (not defined $num) { # no hexadecimal digit          } elsif (not defined $code) { # no hexadecimal digit
1638            !!!parse-error (type => 'bare hcro');            !!!parse-error (type => 'bare hcro');
1639            $self->{next_input_character} = 0x0023; # #            $self->{next_input_character} = 0x0023; # #
1640            !!!back-next-input-character ($x_char);            !!!back-next-input-character ($x_char);
# Line 1619  sub _tokenize_attempt_to_consume_an_enti Line 1645  sub _tokenize_attempt_to_consume_an_enti
1645            !!!parse-error (type => 'no refc');            !!!parse-error (type => 'no refc');
1646          }          }
1647    
1648          ## TODO: check the definition for |a valid Unicode character|.          if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {
1649          ## <http://lists.whatwg.org/pipermail/whatwg-whatwg.org/2006-December/thread.html#8189>            !!!parse-error (type => sprintf 'invalid character reference:U+%04X', $code);
1650          if ($num > 1114111 or $num == 0) {            $code = 0xFFFD;
1651            $num = 0xFFFD; # REPLACEMENT CHARACTER          } elsif ($code > 0x10FFFF) {
1652            ## ISSUE: Why this is not an error?            !!!parse-error (type => sprintf 'invalid character reference:U-%08X', $code);
1653          } elsif (0x80 <= $num and $num <= 0x9F) {            $code = 0xFFFD;
1654            !!!parse-error (type => sprintf 'c1 entity:U+%04X', $num);          } elsif ($code == 0x000D) {
1655            $num = $c1_entity_char->{$num};            !!!parse-error (type => 'CR character reference');
1656              $code = 0x000A;
1657            } elsif (0x80 <= $code and $code <= 0x9F) {
1658              !!!parse-error (type => sprintf 'c1 entity:U+%04X', $code);
1659              $code = $c1_entity_char->{$code};
1660          }          }
1661    
1662          return {type => 'character', data => chr $num};          return {type => 'character', data => chr $code};
1663        } # X        } # X
1664      } elsif (0x0030 <= $self->{next_input_character} and      } elsif (0x0030 <= $self->{next_input_character} and
1665               $self->{next_input_character} <= 0x0039) { # 0..9               $self->{next_input_character} <= 0x0039) { # 0..9
# Line 1650  sub _tokenize_attempt_to_consume_an_enti Line 1680  sub _tokenize_attempt_to_consume_an_enti
1680          !!!parse-error (type => 'no refc');          !!!parse-error (type => 'no refc');
1681        }        }
1682    
1683        ## TODO: check the definition for |a valid Unicode character|.        if ($code == 0 or (0xD800 <= $code and $code <= 0xDFFF)) {
1684        if ($code > 1114111 or $code == 0) {          !!!parse-error (type => sprintf 'invalid character reference:U+%04X', $code);
1685          $code = 0xFFFD; # REPLACEMENT CHARACTER          $code = 0xFFFD;
1686          ## ISSUE: Why this is not an error?        } elsif ($code > 0x10FFFF) {
1687            !!!parse-error (type => sprintf 'invalid character reference:U-%08X', $code);
1688            $code = 0xFFFD;
1689          } elsif ($code == 0x000D) {
1690            !!!parse-error (type => 'CR character reference');
1691            $code = 0x000A;
1692        } elsif (0x80 <= $code and $code <= 0x9F) {        } elsif (0x80 <= $code and $code <= 0x9F) {
1693          !!!parse-error (type => sprintf 'c1 entity:U+%04X', $code);          !!!parse-error (type => sprintf 'c1 entity:U+%04X', $code);
1694          $code = $c1_entity_char->{$code};          $code = $c1_entity_char->{$code};
# Line 1689  sub _tokenize_attempt_to_consume_an_enti Line 1724  sub _tokenize_attempt_to_consume_an_enti
1724              $self->{next_input_character} == 0x003B)) { # ;              $self->{next_input_character} == 0x003B)) { # ;
1725        $entity_name .= chr $self->{next_input_character};        $entity_name .= chr $self->{next_input_character};
1726        if (defined $EntityChar->{$entity_name}) {        if (defined $EntityChar->{$entity_name}) {
         $value = $EntityChar->{$entity_name};  
1727          if ($self->{next_input_character} == 0x003B) { # ;          if ($self->{next_input_character} == 0x003B) { # ;
1728              $value = $EntityChar->{$entity_name};
1729            $match = 1;            $match = 1;
1730            !!!next-input-character;            !!!next-input-character;
1731            last;            last;
1732          } else {          } elsif (not $in_attr) {
1733              $value = $EntityChar->{$entity_name};
1734            $match = -1;            $match = -1;
1735            } else {
1736              $value .= chr $self->{next_input_character};
1737          }          }
1738        } else {        } else {
1739          $value .= chr $self->{next_input_character};          $value .= chr $self->{next_input_character};
# Line 1711  sub _tokenize_attempt_to_consume_an_enti Line 1749  sub _tokenize_attempt_to_consume_an_enti
1749      } else {      } else {
1750        !!!parse-error (type => 'bare ero');        !!!parse-error (type => 'bare ero');
1751        ## NOTE: No characters are consumed in the spec.        ## NOTE: No characters are consumed in the spec.
1752        !!!back-token ({type => 'character', data => $value});        return {type => 'character', data => '&'.$value};
       return undef;  
1753      }      }
1754    } else {    } else {
1755      ## no characters are consumed      ## no characters are consumed
# Line 1907  sub _tree_construction_initial ($) { Line 1944  sub _tree_construction_initial ($) {
1944      } elsif ($token->{type} eq 'character') {      } elsif ($token->{type} eq 'character') {
1945        if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D        if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D
1946          ## Ignore the token          ## Ignore the token
1947    
1948          unless (length $token->{data}) {          unless (length $token->{data}) {
1949            ## Stay in the phase            ## Stay in the phase
1950            !!!next-token;            !!!next-token;
# Line 1949  sub _tree_construction_root_element ($) Line 1987  sub _tree_construction_root_element ($)
1987          !!!next-token;          !!!next-token;
1988          redo B;          redo B;
1989        } elsif ($token->{type} eq 'character') {        } elsif ($token->{type} eq 'character') {
1990          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {          if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) { # \x0D
1991            $self->{document}->manakai_append_text ($1);            ## Ignore the token.
1992            ## ISSUE: DOM3 Core does not allow Document > Text  
1993            unless (length $token->{data}) {            unless (length $token->{data}) {
1994              ## Stay in the phase              ## Stay in the phase
1995              !!!next-token;              !!!next-token;
# Line 2128  sub _tree_construction_main ($) { Line 2166  sub _tree_construction_main ($) {
2166      }      }
2167    }; # $clear_up_to_marker    }; # $clear_up_to_marker
2168    
2169    my $style_start_tag = sub {    my $parse_rcdata = sub ($$) {
2170      my $style_el; !!!create-element ($style_el, 'style', $token->{attributes});      my ($content_model_flag, $insert) = @_;
2171      ## $self->{insertion_mode} eq 'in head' and ... (always true)  
2172      (($self->{insertion_mode} eq 'in head' and defined $self->{head_element})      ## Step 1
2173       ? $self->{head_element} : $self->{open_elements}->[-1]->[0])      my $start_tag_name = $token->{tag_name};
2174        ->append_child ($style_el);      my $el;
2175      $self->{content_model_flag} = 'CDATA';      !!!create-element ($el, $start_tag_name, $token->{attributes});
2176    
2177        ## Step 2
2178        $insert->($el); # /context node/->append_child ($el)
2179    
2180        ## Step 3
2181        $self->{content_model_flag} = $content_model_flag; # CDATA or RCDATA
2182      delete $self->{escape}; # MUST      delete $self->{escape}; # MUST
2183                  
2184        ## Step 4
2185      my $text = '';      my $text = '';
2186      !!!next-token;      !!!next-token;
2187      while ($token->{type} eq 'character') {      while ($token->{type} eq 'character') { # or until stop tokenizing
2188        $text .= $token->{data};        $text .= $token->{data};
2189        !!!next-token;        !!!next-token;
2190      } # stop if non-character token or tokenizer stops tokenising      }
2191    
2192        ## Step 5
2193      if (length $text) {      if (length $text) {
2194        $style_el->manakai_append_text ($text);        my $text = $self->{document}->create_text_node ($text);
2195          $el->append_child ($text);
2196      }      }
2197        
2198        ## Step 6
2199      $self->{content_model_flag} = 'PCDATA';      $self->{content_model_flag} = 'PCDATA';
2200                  
2201      if ($token->{type} eq 'end tag' and $token->{tag_name} eq 'style') {      ## Step 7
2202        if ($token->{type} eq 'end tag' and $token->{tag_name} eq $start_tag_name) {
2203        ## Ignore the token        ## Ignore the token
2204      } else {      } else {
2205        !!!parse-error (type => 'in CDATA:#'.$token->{type});        !!!parse-error (type => 'in '.$content_model_flag.':#'.$token->{type});
       ## ISSUE: And ignore?  
2206      }      }
2207      !!!next-token;      !!!next-token;
2208    }; # $style_start_tag    }; # $parse_rcdata
2209    
2210    my $script_start_tag = sub {    my $script_start_tag = sub ($) {
2211        my $insert = $_[0];
2212      my $script_el;      my $script_el;
2213      !!!create-element ($script_el, 'script', $token->{attributes});      !!!create-element ($script_el, 'script', $token->{attributes});
2214      ## TODO: mark as "parser-inserted"      ## TODO: mark as "parser-inserted"
# Line 2192  sub _tree_construction_main ($) { Line 2242  sub _tree_construction_main ($) {
2242      } else {      } else {
2243        ## TODO: $old_insertion_point = current insertion point        ## TODO: $old_insertion_point = current insertion point
2244        ## TODO: insertion point = just before the next input character        ## TODO: insertion point = just before the next input character
2245          
2246        (($self->{insertion_mode} eq 'in head' and defined $self->{head_element})        $insert->($script_el);
        ? $self->{head_element} : $self->{open_elements}->[-1]->[0])->append_child ($script_el);  
2247                
2248        ## TODO: insertion point = $old_insertion_point (might be "undefined")        ## TODO: insertion point = $old_insertion_point (might be "undefined")
2249                
# Line 2388  sub _tree_construction_main ($) { Line 2437  sub _tree_construction_main ($) {
2437    }; # $formatting_end_tag    }; # $formatting_end_tag
2438    
2439    my $insert_to_current = sub {    my $insert_to_current = sub {
2440      $self->{open_elements}->[-1]->[0]->append_child (shift);      $self->{open_elements}->[-1]->[0]->append_child ($_[0]);
2441    }; # $insert_to_current    }; # $insert_to_current
2442    
2443    my $insert_to_foster = sub {    my $insert_to_foster = sub {
# Line 2426  sub _tree_construction_main ($) { Line 2475  sub _tree_construction_main ($) {
2475      my $insert = shift;      my $insert = shift;
2476      if ($token->{type} eq 'start tag') {      if ($token->{type} eq 'start tag') {
2477        if ($token->{tag_name} eq 'script') {        if ($token->{tag_name} eq 'script') {
2478          $script_start_tag->();          ## NOTE: This is an "as if in head" code clone
2479            $script_start_tag->($insert);
2480          return;          return;
2481        } elsif ($token->{tag_name} eq 'style') {        } elsif ($token->{tag_name} eq 'style') {
2482          $style_start_tag->();          ## NOTE: This is an "as if in head" code clone
2483            $parse_rcdata->('CDATA', $insert);
2484          return;          return;
2485        } elsif ({        } elsif ({
2486                  base => 1, link => 1, meta => 1,                  base => 1, link => 1, meta => 1,
2487                 }->{$token->{tag_name}}) {                 }->{$token->{tag_name}}) {
2488          ## NOTE: This is an "as if in head" code clone          ## NOTE: This is an "as if in head" code clone, only "-t" differs
2489          my $el;          !!!insert-element-t ($token->{tag_name}, $token->{attributes});
2490          !!!create-element ($el, $token->{tag_name}, $token->{attributes});          pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
         if ($self->{insertion_mode} eq 'in head' and  
             defined $self->{head_element}) {  
           $self->{head_element}->append_child ($el);  
         } else {  
           $insert->($el);  
         }  
           
2491          !!!next-token;          !!!next-token;
2492            ## TODO: Extracting |charset| from |meta|.
2493          return;          return;
2494        } elsif ($token->{tag_name} eq 'title') {        } elsif ($token->{tag_name} eq 'title') {
2495          !!!parse-error (type => 'in body:title');          !!!parse-error (type => 'in body:title');
2496          ## NOTE: There is an "as if in head" code clone          ## NOTE: This is an "as if in head" code clone
2497          my $title_el;          $parse_rcdata->('RCDATA', $insert);
         !!!create-element ($title_el, 'title', $token->{attributes});  
         (defined $self->{head_element} ? $self->{head_element} : $self->{open_elements}->[-1]->[0])  
           ->append_child ($title_el);  
         $self->{content_model_flag} = 'RCDATA';  
         delete $self->{escape}; # MUST  
           
         my $text = '';  
         !!!next-token;  
         while ($token->{type} eq 'character') {  
           $text .= $token->{data};  
           !!!next-token;  
         }  
         if (length $text) {  
           $title_el->manakai_append_text ($text);  
         }  
           
         $self->{content_model_flag} = 'PCDATA';  
           
         if ($token->{type} eq 'end tag' and  
             $token->{tag_name} eq 'title') {  
           ## Ignore the token  
         } else {  
           !!!parse-error (type => 'in RCDATA:#'.$token->{type});  
           ## ISSUE: And ignore?  
         }  
         !!!next-token;  
2498          return;          return;
2499        } elsif ($token->{tag_name} eq 'body') {        } elsif ($token->{tag_name} eq 'body') {
2500          !!!parse-error (type => 'in body:body');          !!!parse-error (type => 'in body:body');
# Line 2578  sub _tree_construction_main ($) { Line 2597  sub _tree_construction_main ($) {
2597              if ($i != -1) {              if ($i != -1) {
2598                !!!parse-error (type => 'end tag missing:'.                !!!parse-error (type => 'end tag missing:'.
2599                                $self->{open_elements}->[-1]->[1]);                                $self->{open_elements}->[-1]->[1]);
               ## TODO: test  
2600              }              }
2601              splice @{$self->{open_elements}}, $i;              splice @{$self->{open_elements}}, $i;
2602              last LI;              last LI;
# Line 2626  sub _tree_construction_main ($) { Line 2644  sub _tree_construction_main ($) {
2644              if ($i != -1) {              if ($i != -1) {
2645                !!!parse-error (type => 'end tag missing:'.                !!!parse-error (type => 'end tag missing:'.
2646                                $self->{open_elements}->[-1]->[1]);                                $self->{open_elements}->[-1]->[1]);
               ## TODO: test  
2647              }              }
2648              splice @{$self->{open_elements}}, $i;              splice @{$self->{open_elements}}, $i;
2649              last LI;              last LI;
# Line 2821  sub _tree_construction_main ($) { Line 2838  sub _tree_construction_main ($) {
2838          return;          return;
2839        } elsif ($token->{tag_name} eq 'xmp') {        } elsif ($token->{tag_name} eq 'xmp') {
2840          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
2841                    $parse_rcdata->('CDATA', $insert);
         !!!insert-element-t ($token->{tag_name}, $token->{attributes});  
           
         $self->{content_model_flag} = 'CDATA';  
         delete $self->{escape}; # MUST  
           
         !!!next-token;  
2842          return;          return;
2843        } elsif ($token->{tag_name} eq 'table') {        } elsif ($token->{tag_name} eq 'table') {
2844          ## has a p element in scope          ## has a p element in scope
# Line 2936  sub _tree_construction_main ($) { Line 2947  sub _tree_construction_main ($) {
2947            !!!back-token (@tokens);            !!!back-token (@tokens);
2948            return;            return;
2949          }          }
2950        } elsif ({        } elsif ($token->{tag_name} eq 'textarea') {
                 textarea => 1,  
                 iframe => 1,  
                 noembed => 1,  
                 noframes => 1,  
                 noscript => 0, ## TODO: 1 if scripting is enabled  
                }->{$token->{tag_name}}) {  
2951          my $tag_name = $token->{tag_name};          my $tag_name = $token->{tag_name};
2952          my $el;          my $el;
2953          !!!create-element ($el, $token->{tag_name}, $token->{attributes});          !!!create-element ($el, $token->{tag_name}, $token->{attributes});
2954                    
2955          if ($token->{tag_name} eq 'textarea') {          ## TODO: $self->{form_element} if defined
2956            ## TODO: $self->{form_element} if defined          $self->{content_model_flag} = 'RCDATA';
           $self->{content_model_flag} = 'RCDATA';  
         } else {  
           $self->{content_model_flag} = 'CDATA';  
         }  
2957          delete $self->{escape}; # MUST          delete $self->{escape}; # MUST
2958                    
2959          $insert->($el);          $insert->($el);
2960                    
2961          my $text = '';          my $text = '';
2962          if ($token->{tag_name} eq 'textarea') {          !!!next-token;
2963            !!!next-token;          if ($token->{type} eq 'character') {
2964            if ($token->{type} eq 'character') {            $token->{data} =~ s/^\x0A//;
2965              $token->{data} =~ s/^\x0A//;            unless (length $token->{data}) {
2966              unless (length $token->{data}) {              !!!next-token;
               !!!next-token;  
             }  
2967            }            }
         } else {  
           !!!next-token;  
2968          }          }
2969          while ($token->{type} eq 'character') {          while ($token->{type} eq 'character') {
2970            $text .= $token->{data};            $text .= $token->{data};
# Line 2983  sub _tree_construction_main ($) { Line 2980  sub _tree_construction_main ($) {
2980              $token->{tag_name} eq $tag_name) {              $token->{tag_name} eq $tag_name) {
2981            ## Ignore the token            ## Ignore the token
2982          } else {          } else {
2983            if ($token->{tag_name} eq 'textarea') {            !!!parse-error (type => 'in RCDATA:#'.$token->{type});
             !!!parse-error (type => 'in RCDATA:#'.$token->{type});  
           } else {  
             !!!parse-error (type => 'in CDATA:#'.$token->{type});  
           }  
           ## ISSUE: And ignore?  
2984          }          }
2985          !!!next-token;          !!!next-token;
2986          return;          return;
2987          } elsif ({
2988                    iframe => 1,
2989                    noembed => 1,
2990                    noframes => 1,
2991                    noscript => 0, ## TODO: 1 if scripting is enabled
2992                   }->{$token->{tag_name}}) {
2993            $parse_rcdata->('CDATA', $insert);
2994            return;
2995        } elsif ($token->{tag_name} eq 'select') {        } elsif ($token->{tag_name} eq 'select') {
2996          $reconstruct_active_formatting_elements->($insert_to_current);          $reconstruct_active_formatting_elements->($insert_to_current);
2997                    
# Line 3238  sub _tree_construction_main ($) { Line 3238  sub _tree_construction_main ($) {
3238                  #not $phrasing_category->{$node->[1]} and                  #not $phrasing_category->{$node->[1]} and
3239                  ($special_category->{$node->[1]} or                  ($special_category->{$node->[1]} or
3240                   $scoping_category->{$node->[1]})) {                   $scoping_category->{$node->[1]})) {
3241                !!!parse-error (type => 'not closed:'.$node->[1]);                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3242                ## Ignore the token                ## Ignore the token
3243                !!!next-token;                !!!next-token;
3244                last S2;                last S2;
# Line 3267  sub _tree_construction_main ($) { Line 3267  sub _tree_construction_main ($) {
3267          redo B;          redo B;
3268        } elsif ($token->{type} eq 'start tag' and        } elsif ($token->{type} eq 'start tag' and
3269                 $token->{tag_name} eq 'html') {                 $token->{tag_name} eq 'html') {
3270          ## TODO: unless it is the first start tag token, parse-error  ## ISSUE: "aa<html>" is not a parse error.
3271    ## ISSUE: "<html>" in fragment is not a parse error.
3272            unless ($token->{first_start_tag}) {
3273              !!!parse-error (type => 'not first start tag');
3274            }
3275          my $top_el = $self->{open_elements}->[0]->[0];          my $top_el = $self->{open_elements}->[0]->[0];
3276          for my $attr_name (keys %{$token->{attributes}}) {          for my $attr_name (keys %{$token->{attributes}}) {
3277            unless ($top_el->has_attribute_ns (undef, $attr_name)) {            unless ($top_el->has_attribute_ns (undef, $attr_name)) {
# Line 3358  sub _tree_construction_main ($) { Line 3362  sub _tree_construction_main ($) {
3362            } else {            } else {
3363              die "$0: $token->{type}: Unknown type";              die "$0: $token->{type}: Unknown type";
3364            }            }
3365          } elsif ($self->{insertion_mode} eq 'in head') {          } elsif ($self->{insertion_mode} eq 'in head' or
3366                     $self->{insertion_mode} eq 'in head noscript' or
3367                     $self->{insertion_mode} eq 'after head') {
3368            if ($token->{type} eq 'character') {            if ($token->{type} eq 'character') {
3369              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {              if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {
3370                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);                $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);
# Line 3375  sub _tree_construction_main ($) { Line 3381  sub _tree_construction_main ($) {
3381              !!!next-token;              !!!next-token;
3382              redo B;              redo B;
3383            } elsif ($token->{type} eq 'start tag') {            } elsif ($token->{type} eq 'start tag') {
3384              if ($token->{tag_name} eq 'title') {              if ({base => ($self->{insertion_mode} eq 'in head' or
3385                ## NOTE: There is an "as if in head" code clone                            $self->{insertion_mode} eq 'after head'),
3386                my $title_el;                   link => 1, meta => 1}->{$token->{tag_name}}) {
3387                !!!create-element ($title_el, 'title', $token->{attributes});                ## NOTE: There is a "as if in head" code clone.
3388                (defined $self->{head_element} ? $self->{head_element} : $self->{open_elements}->[-1]->[0])                if ($self->{insertion_mode} eq 'after head') {
3389                  ->append_child ($title_el);                  !!!parse-error (type => 'after head:'.$token->{tag_name});
3390                $self->{content_model_flag} = 'RCDATA';                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
3391                delete $self->{escape}; # MUST                }
3392                  !!!insert-element ($token->{tag_name}, $token->{attributes});
3393                my $text = '';                pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
3394                  ## TODO: Extracting |charset| from |meta|.
3395                  pop @{$self->{open_elements}}
3396                      if $self->{insertion_mode} eq 'after head';
3397                !!!next-token;                !!!next-token;
3398                while ($token->{type} eq 'character') {                redo B;
3399                  $text .= $token->{data};              } elsif ($token->{tag_name} eq 'title' and
3400                         $self->{insertion_mode} eq 'in head') {
3401                  ## NOTE: There is a "as if in head" code clone.
3402                  if ($self->{insertion_mode} eq 'after head') {
3403                    !!!parse-error (type => 'after head:'.$token->{tag_name});
3404                    push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
3405                  }
3406                  $parse_rcdata->('RCDATA', $insert_to_current);
3407                  pop @{$self->{open_elements}}
3408                      if $self->{insertion_mode} eq 'after head';
3409                  redo B;
3410                } elsif ($token->{tag_name} eq 'style') {
3411                  ## NOTE: Or (scripting is enabled and tag_name eq 'noscript' and
3412                  ## insertion mode 'in head')
3413                  ## NOTE: There is a "as if in head" code clone.
3414                  if ($self->{insertion_mode} eq 'after head') {
3415                    !!!parse-error (type => 'after head:'.$token->{tag_name});
3416                    push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
3417                  }
3418                  $parse_rcdata->('CDATA', $insert_to_current);
3419                  pop @{$self->{open_elements}}
3420                      if $self->{insertion_mode} eq 'after head';
3421                  redo B;
3422                } elsif ($token->{tag_name} eq 'noscript') {
3423                  if ($self->{insertion_mode} eq 'in head') {
3424                    ## NOTE: and scripting is disalbed
3425                    !!!insert-element ($token->{tag_name}, $token->{attributes});
3426                    $self->{insertion_mode} = 'in head noscript';
3427                  !!!next-token;                  !!!next-token;
3428                }                  redo B;
3429                if (length $text) {                } elsif ($self->{insertion_mode} eq 'in head noscript') {
3430                  $title_el->manakai_append_text ($text);                  !!!parse-error (type => 'noscript in noscript');
               }  
                 
               $self->{content_model_flag} = 'PCDATA';  
                 
               if ($token->{type} eq 'end tag' and  
                   $token->{tag_name} eq 'title') {  
3431                  ## Ignore the token                  ## Ignore the token
3432                    redo B;
3433                } else {                } else {
3434                  !!!parse-error (type => 'in RCDATA:#'.$token->{type});                  #
                 ## ISSUE: And ignore?  
3435                }                }
3436                } elsif ($token->{tag_name} eq 'head' and
3437                         $self->{insertion_mode} ne 'after head') {
3438                  !!!parse-error (type => 'in head:head'); # or in head noscript
3439                  ## Ignore the token
3440                !!!next-token;                !!!next-token;
3441                redo B;                redo B;
3442              } elsif ($token->{tag_name} eq 'style') {              } elsif ($self->{insertion_mode} ne 'in head noscript' and
3443                $style_start_tag->();                       $token->{tag_name} eq 'script') {
3444                redo B;                if ($self->{insertion_mode} eq 'after head') {
3445              } elsif ($token->{tag_name} eq 'script') {                  !!!parse-error (type => 'after head:'.$token->{tag_name});
3446                $script_start_tag->();                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
3447                  }
3448                  ## NOTE: There is a "as if in head" code clone.
3449                  $script_start_tag->($insert_to_current);
3450                  pop @{$self->{open_elements}}
3451                      if $self->{insertion_mode} eq 'after head';
3452                redo B;                redo B;
3453              } elsif ({base => 1, link => 1, meta => 1}->{$token->{tag_name}}) {              } elsif ($self->{insertion_mode} eq 'after head' and
3454                ## NOTE: There are "as if in head" code clones                       $token->{tag_name} eq 'body') {
3455                my $el;                !!!insert-element ('body', $token->{attributes});
3456                !!!create-element ($el, $token->{tag_name}, $token->{attributes});                $self->{insertion_mode} = 'in body';
               if ($self->{insertion_mode} eq 'in head' and  
                   defined $self->{head_element}) {  
                 $self->{head_element}->append_child ($el);  
               } else {  
                 $self->{open_elements}->[-1]->[0]->append_child ($el);  
               }  
   
3457                !!!next-token;                !!!next-token;
3458                redo B;                redo B;
3459              } elsif ($token->{tag_name} eq 'head') {              } elsif ($self->{insertion_mode} eq 'after head' and
3460                !!!parse-error (type => 'in head:head');                       $token->{tag_name} eq 'frameset') {
3461                ## Ignore the token                !!!insert-element ('frameset', $token->{attributes});
3462                  $self->{insertion_mode} = 'in frameset';
3463                !!!next-token;                !!!next-token;
3464                redo B;                redo B;
3465              } else {              } else {
3466                #                #
3467              }              }
3468            } elsif ($token->{type} eq 'end tag') {            } elsif ($token->{type} eq 'end tag') {
3469              if ($token->{tag_name} eq 'head') {              if ($self->{insertion_mode} eq 'in head' and
3470                if ($self->{open_elements}->[-1]->[1] eq 'head') {                  $token->{tag_name} eq 'head') {
3471                  pop @{$self->{open_elements}};                pop @{$self->{open_elements}};
               } else {  
                 !!!parse-error (type => 'unmatched end tag:head');  
               }  
3472                $self->{insertion_mode} = 'after head';                $self->{insertion_mode} = 'after head';
3473                !!!next-token;                !!!next-token;
3474                redo B;                redo B;
3475              } elsif ($token->{tag_name} eq 'body' or              } elsif ($self->{insertion_mode} eq 'in head noscript' and
3476                       $token->{tag_name} eq 'html') {                  $token->{tag_name} eq 'noscript') {
3477                  pop @{$self->{open_elements}};
3478                  $self->{insertion_mode} = 'in head';
3479                  !!!next-token;
3480                  redo B;
3481                } elsif ($self->{insertion_mode} eq 'in head' and
3482                         ($token->{tag_name} eq 'body' or
3483                          $token->{tag_name} eq 'html')) {
3484                #                #
3485              } else {              } elsif ($self->{insertion_mode} ne 'after head') {
3486                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});                !!!parse-error (type => 'unmatched end tag:'.$token->{tag_name});
3487                ## Ignore the token                ## Ignore the token
3488                !!!next-token;                !!!next-token;
3489                redo B;                redo B;
3490                } else {
3491                  #
3492              }              }
3493            } else {            } else {
3494              #              #
3495            }            }
3496    
3497            if ($self->{open_elements}->[-1]->[1] eq 'head') {            ## As if </head> or </noscript> or <body>
3498              ## As if </head>            if ($self->{insertion_mode} eq 'in head') {
3499                pop @{$self->{open_elements}};
3500                $self->{insertion_mode} = 'after head';
3501              } elsif ($self->{insertion_mode} eq 'in head noscript') {
3502              pop @{$self->{open_elements}};              pop @{$self->{open_elements}};
3503                !!!parse-error (type => 'in noscript:'.(defined $token->{tag_name} ? ($token->{type} eq 'end tag' ? '/' : '') . $token->{tag_name} : '#' . $token->{type}));
3504                $self->{insertion_mode} = 'in head';
3505              } else { # 'after head'
3506                !!!insert-element ('body');
3507                $self->{insertion_mode} = 'in body';
3508            }            }
           $self->{insertion_mode} = 'after head';  
3509            ## reprocess            ## reprocess
3510            redo B;            redo B;
3511    
3512            ## ISSUE: An issue in the spec.            ## ISSUE: An issue in the spec.
         } elsif ($self->{insertion_mode} eq 'after head') {  
           if ($token->{type} eq 'character') {  
             if ($token->{data} =~ s/^([\x09\x0A\x0B\x0C\x20]+)//) {  
               $self->{open_elements}->[-1]->[0]->manakai_append_text ($1);  
               unless (length $token->{data}) {  
                 !!!next-token;  
                 redo B;  
               }  
             }  
               
             #  
           } elsif ($token->{type} eq 'comment') {  
             my $comment = $self->{document}->create_comment ($token->{data});  
             $self->{open_elements}->[-1]->[0]->append_child ($comment);  
             !!!next-token;  
             redo B;  
           } elsif ($token->{type} eq 'start tag') {  
             if ($token->{tag_name} eq 'body') {  
               !!!insert-element ('body', $token->{attributes});  
               $self->{insertion_mode} = 'in body';  
               !!!next-token;  
               redo B;  
             } elsif ($token->{tag_name} eq 'frameset') {  
               !!!insert-element ('frameset', $token->{attributes});  
               $self->{insertion_mode} = 'in frameset';  
               !!!next-token;  
               redo B;  
             } elsif ({  
                       base => 1, link => 1, meta => 1,  
                       script => 1, style => 1, title => 1,  
                      }->{$token->{tag_name}}) {  
               !!!parse-error (type => 'after head:'.$token->{tag_name});  
               $self->{insertion_mode} = 'in head';  
               ## reprocess  
               redo B;  
             } else {  
               #  
             }  
           } else {  
             #  
           }  
             
           ## As if <body>  
           !!!insert-element ('body');  
           $self->{insertion_mode} = 'in body';  
           ## reprocess  
           redo B;  
3513          } elsif ($self->{insertion_mode} eq 'in body') {          } elsif ($self->{insertion_mode} eq 'in body') {
3514            if ($token->{type} eq 'character') {            if ($token->{type} eq 'character') {
3515              ## NOTE: There is a code clone of "character in body".              ## NOTE: There is a code clone of "character in body".
# Line 5322  sub get_inner_html ($$$) { Line 5321  sub get_inner_html ($$$) {
5321            
5322      my $nt = $child->node_type;      my $nt = $child->node_type;
5323      if ($nt == 1) { # Element      if ($nt == 1) { # Element
5324        my $tag_name = lc $child->tag_name; ## ISSUE: Definition of "lowercase"        my $tag_name = $child->tag_name; ## TODO: manakai_tag_name
5325        $s .= '<' . $tag_name;        $s .= '<' . $tag_name;
5326          ## NOTE: Non-HTML case:
5327        ## ISSUE: Non-html elements        ## <http://permalink.gmane.org/gmane.org.w3c.whatwg.discuss/11191>
5328    
5329        my @attrs = @{$child->attributes}; # sort order MUST be stable        my @attrs = @{$child->attributes}; # sort order MUST be stable
5330        for my $attr (@attrs) { # order is implementation dependent        for my $attr (@attrs) { # order is implementation dependent
5331          my $attr_name = lc $attr->name; ## ISSUE: Definition of "lowercase"          my $attr_name = $attr->name; ## TODO: manakai_name
5332          $s .= ' ' . $attr_name . '="';          $s .= ' ' . $attr_name . '="';
5333          my $attr_value = $attr->value;          my $attr_value = $attr->value;
5334          ## escape          ## escape
# Line 5353  sub get_inner_html ($$$) { Line 5352  sub get_inner_html ($$$) {
5352        if (not $in_cdata and {        if (not $in_cdata and {
5353          style => 1, script => 1, xmp => 1, iframe => 1,          style => 1, script => 1, xmp => 1, iframe => 1,
5354          noembed => 1, noframes => 1, noscript => 1,          noembed => 1, noframes => 1, noscript => 1,
5355            plaintext => 1,
5356        }->{$tag_name}) {        }->{$tag_name}) {
5357          unshift @node, 'cdata-out';          unshift @node, 'cdata-out';
5358          $in_cdata = 1;          $in_cdata = 1;

Legend:
Removed from v.1.24  
changed lines
  Added in v.1.28

[email protected]
ViewVC Help
Powered by ViewVC 1.1.24