Parent Directory
|
Revision Log
|
Patch
| revision 1.31 by wakaba, Sat Jun 30 14:13:19 2007 UTC | revision 1.79 by wakaba, Mon Mar 3 13:15:54 2008 UTC | |
|---|---|---|
| # | Line 1 | Line 1 |
| 1 | package Whatpm::HTML; | package Whatpm::HTML; |
| 2 | use strict; | use strict; |
| 3 | our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r}; | our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r}; |
| 4 | use Error qw(:try); | |
| 5 | ||
| 6 | ## ISSUE: | ## ISSUE: |
| 7 | ## var doc = implementation.createDocument (null, null, null); | ## var doc = implementation.createDocument (null, null, null); |
| 8 | ## doc.write (''); | ## doc.write (''); |
| 9 | ## alert (doc.compatMode); | ## alert (doc.compatMode); |
| 10 | ||
| 11 | ## ISSUE: HTML5 revision 967 says that the encoding layer MUST NOT | ## TODO: Control charcters and noncharacters are not allowed (HTML5 revision 1263) |
| 12 | ## strip BOM and the HTML layer MUST ignore it. Whether we can do it | ## TODO: 1252 parse error (revision 1264) |
| 13 | ## is not yet clear. | ## TODO: 8859-11 = 874 (revision 1271) |
| ## "{U+FEFF}..." in UTF-16BE/UTF-16LE is three or four characters? | ||
| ## "{U+FEFF}..." in GB18030? | ||
| 14 | ||
| 15 | my $permitted_slash_tag_name = { | my $permitted_slash_tag_name = { |
| 16 | base => 1, | base => 1, |
| # | Line 19 my $permitted_slash_tag_name = { | Line 18 my $permitted_slash_tag_name = { |
| 18 | meta => 1, | meta => 1, |
| 19 | hr => 1, | hr => 1, |
| 20 | br => 1, | br => 1, |
| 21 | img=> 1, | img => 1, |
| 22 | embed => 1, | embed => 1, |
| 23 | param => 1, | param => 1, |
| 24 | area => 1, | area => 1, |
| # | Line 84 my $formatting_category = { | Line 83 my $formatting_category = { |
| 83 | }; | }; |
| 84 | # $phrasing_category: all other elements | # $phrasing_category: all other elements |
| 85 | ||
| 86 | sub parse_byte_string ($$$$;$) { | |
| 87 | my $self = ref $_[0] ? shift : shift->new; | |
| 88 | my $charset = shift; | |
| 89 | my $bytes_s = ref $_[0] ? $_[0] : \($_[0]); | |
| 90 | my $s; | |
| 91 | ||
| 92 | if (defined $charset) { | |
| 93 | require Encode; ## TODO: decode(utf8) don't delete BOM | |
| 94 | $s = \ (Encode::decode ($charset, $$bytes_s)); | |
| 95 | $self->{input_encoding} = lc $charset; ## TODO: normalize name | |
| 96 | $self->{confident} = 1; | |
| 97 | } else { | |
| 98 | ## TODO: Implement HTML5 detection algorithm | |
| 99 | require Whatpm::Charset::UniversalCharDet; | |
| 100 | $charset = Whatpm::Charset::UniversalCharDet->detect_byte_string | |
| 101 | (substr ($$bytes_s, 0, 1024)); | |
| 102 | $charset ||= 'windows-1252'; | |
| 103 | $s = \ (Encode::decode ($charset, $$bytes_s)); | |
| 104 | $self->{input_encoding} = $charset; | |
| 105 | $self->{confident} = 0; | |
| 106 | } | |
| 107 | ||
| 108 | $self->{change_encoding} = sub { | |
| 109 | my $self = shift; | |
| 110 | my $charset = lc shift; | |
| 111 | ## TODO: if $charset is supported | |
| 112 | ## TODO: normalize charset name | |
| 113 | ||
| 114 | ## "Change the encoding" algorithm: | |
| 115 | ||
| 116 | ## Step 1 | |
| 117 | if ($charset eq 'utf-16') { ## ISSUE: UTF-16BE -> UTF-8? UTF-16LE -> UTF-8? | |
| 118 | $charset = 'utf-8'; | |
| 119 | } | |
| 120 | ||
| 121 | ## Step 2 | |
| 122 | if (defined $self->{input_encoding} and | |
| 123 | $self->{input_encoding} eq $charset) { | |
| 124 | $self->{confident} = 1; | |
| 125 | return; | |
| 126 | } | |
| 127 | ||
| 128 | !!!parse-error (type => 'charset label detected:'.$self->{input_encoding}. | |
| 129 | ':'.$charset, level => 'w'); | |
| 130 | ||
| 131 | ## Step 3 | |