Parent Directory
|
Revision Log
|
Patch
| revision 1.42 by wakaba, Sat Jul 21 06:59:16 2007 UTC | revision 1.232 by wakaba, Sun Sep 6 01:30:08 2009 UTC | |
|---|---|---|
| # | Line 1 | Line 1 |
| 1 | package Whatpm::HTML; | package Whatpm::HTML; |
| 2 | use strict; | use strict; |
| 3 | our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r}; | our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r}; |
| 4 | use Error qw(:try); | |
| 5 | ||
| 6 | use Whatpm::HTML::Tokenizer; | |
| 7 | ||
| 8 | ## NOTE: This module don't check all HTML5 parse errors; character | |
| 9 | ## encoding related parse errors are expected to be handled by relevant | |
| 10 | ## modules. | |
| 11 | ## Parse errors for control characters that are not allowed in HTML5 | |
| 12 | ## documents, for surrogate code points, and for noncharacter code | |
| 13 | ## points, as well as U+FFFD substitions for characters whose code points | |
| 14 | ## is higher than U+10FFFF may be detected by combining the parser with | |
| 15 | ## the checker implemented by Whatpm::Charset::UnicodeChecker (for its | |
| 16 | ## usage example, see |t/HTML-tree.t| in the Whatpm package or the | |
| 17 | ## WebHACC::Language::HTML module in the WebHACC package). | |
| 18 | ||
| 19 | ## ISSUE: | ## ISSUE: |
| 20 | ## var doc = implementation.createDocument (null, null, null); | ## var doc = implementation.createDocument (null, null, null); |
| 21 | ## doc.write (''); | ## doc.write (''); |
| 22 | ## alert (doc.compatMode); | ## alert (doc.compatMode); |
| 23 | ||
| 24 | ## ISSUE: HTML5 revision 967 says that the encoding layer MUST NOT | require IO::Handle; |
| ## strip BOM and the HTML layer MUST ignore it. Whether we can do it | ||
| ## is not yet clear. | ||
| ## "{U+FEFF}..." in UTF-16BE/UTF-16LE is three or four characters? | ||
| ## "{U+FEFF}..." in GB18030? | ||
| my $permitted_slash_tag_name = { | ||
| base => 1, | ||
| link => 1, | ||
| meta => 1, | ||
| hr => 1, | ||
| br => 1, | ||
| img=> 1, | ||
| embed => 1, | ||
| param => 1, | ||
| area => 1, | ||
| col => 1, | ||
| input => 1, | ||
| }; | ||
| 25 | ||
| 26 | my $c1_entity_char = { | ## Namespace URLs |
| 27 | 0x80 => 0x20AC, | |
| 28 | 0x81 => 0xFFFD, | my $HTML_NS = q<http://www.w3.org/1999/xhtml>; |
| 29 | 0x82 => 0x201A, | my $MML_NS = q<http://www.w3.org/1998/Math/MathML>; |
| 30 | 0x83 => 0x0192, | my $SVG_NS = q<http://www.w3.org/2000/svg>; |
| 31 | 0x84 => 0x201E, | my $XLINK_NS = q<http://www.w3.org/1999/xlink>; |
| 32 | 0x85 => 0x2026, | my $XML_NS = q<http://www.w3.org/XML/1998/namespace>; |
| 33 | 0x86 => 0x2020, | my $XMLNS_NS = q<http://www.w3.org/2000/xmlns/>; |
| 34 | 0x87 => 0x2021, | |
| 35 | 0x88 => 0x02C6, | ## Element categories |
| 36 | 0x89 => 0x2030, | |
| 37 | 0x8A => 0x0160, | ## Bits 12-15 |
| 38 | 0x8B => 0x2039, | sub SPECIAL_EL () { 0b1_000000000000000 } |
| 39 | 0x8C => 0x0152, | sub SCOPING_EL () { 0b1_00000000000000 } |
| 40 | 0x8D => 0xFFFD, | sub FORMATTING_EL () { 0b1_0000000000000 } |
| 41 | 0x8E => 0x017D, | sub PHRASING_EL () { 0b1_000000000000 } |
| 42 | 0x8F => 0xFFFD, | |
| 43 | 0x90 => 0xFFFD, | ## Bits 10-11 |
| 44 | 0x91 => 0x2018, | #sub FOREIGN_EL () { 0b1_00000000000 } # see Whatpm::HTML::Tokenizer |
| 45 | 0x92 => 0x2019, | sub FOREIGN_FLOW_CONTENT_EL () { 0b1_0000000000 } |
| 46 | 0x93 => 0x201C, | |
| 47 | 0x94 => 0x201D, | ## Bits 6-9 |
| 48 | 0x95 => 0x2022, | sub TABLE_SCOPING_EL () { 0b1_000000000 } |
| 49 | 0x96 => 0x2013, | sub TABLE_ROWS_SCOPING_EL () { 0b1_00000000 } |
| 50 | 0x97 => 0x2014, | sub TABLE_ROW_SCOPING_EL () { 0b1_0000000 } |
| 51 | 0x98 => 0x02DC, | sub TABLE_ROWS_EL () { 0b1_000000 } |
| 52 | 0x99 => 0x2122, | |
| 53 | 0x9A => 0x0161, | ## Bit 5 |
| 54 | 0x9B => 0x203A, | sub ADDRESS_DIV_P_EL () { 0b1_00000 } |
| 55 | 0x9C => 0x0153, | |
| 56 | 0x9D => 0xFFFD, | ## NOTE: Used in </body> and EOF algorithms. |
| 57 | 0x9E => 0x017E, | ## Bit 4 |
| 58 | 0x9F => 0x0178, | sub ALL_END_TAG_OPTIONAL_EL () { 0b1_0000 } |
| 59 | }; # $c1_entity_char | |
| 60 | ## NOTE: Used in "generate implied end tags" algorithm. | |
| 61 | my $special_category = { | ## NOTE: There is a code where a modified version of |
| 62 | address => 1, area => 1, base => 1, basefont => 1, bgsound => 1, | ## END_TAG_OPTIONAL_EL is used in "generate implied end tags" |
| 63 | blockquote => 1, body => 1, br => 1, center => 1, col => 1, colgroup => 1, | ## implementation (search for the algorithm name). |
| 64 | dd => 1, dir => 1, div => 1, dl => 1, dt => 1, embed => 1, fieldset => 1, | ## Bit 3 |
| 65 | form => 1, frame => 1, frameset => 1, h1 => 1, h2 => 1, h3 => 1, | sub END_TAG_OPTIONAL_EL () { 0b1_000 } |
| 66 | h4 => 1, h5 => 1, h6 => 1, head => 1, hr => 1, iframe => 1, image => 1, | |
| 67 | img => 1, input => 1, isindex => 1, li => 1, link => 1, listing => 1, | ## Bits 0-2 |
| 68 | menu => 1, meta => 1, noembed => 1, noframes => 1, noscript => 1, | |
| 69 | ol => 1, optgroup => 1, option => 1, p => 1, param => 1, plaintext => 1, | sub MISC_SPECIAL_EL () { SPECIAL_EL | 0b000 } |
| 70 | pre => 1, script => 1, select => 1, spacer => 1, style => 1, tbody => 1, | sub FORM_EL () { SPECIAL_EL | 0b001 } |
| 71 | textarea => 1, tfoot => 1, thead => 1, title => 1, tr => 1, ul => 1, wbr => 1, | sub FRAMESET_EL () { SPECIAL_EL | 0b010 } |
| 72 | }; | sub HEADING_EL () { SPECIAL_EL | 0b011 } |
| 73 | my $scoping_category = { | sub SELECT_EL () { SPECIAL_EL | 0b100 } |
| 74 | button => 1, caption => 1, html => 1, marquee => 1, object => 1, | sub SCRIPT_EL () { SPECIAL_EL | 0b101 } |
| 75 | table => 1, td => 1, th => 1, | |
| 76 | sub ADDRESS_DIV_EL () { SPECIAL_EL | ADDRESS_DIV_P_EL | 0b001 } | |
| 77 | sub BODY_EL () { SPECIAL_EL | ALL_END_TAG_OPTIONAL_EL | 0b001 } | |
| 78 | ||
| 79 | sub DTDD_EL () { | |
| 80 | SPECIAL_EL | | |
| 81 | END_TAG_OPTIONAL_EL | | |
| 82 | ALL_END_TAG_OPTIONAL_EL | | |
| 83 | 0b010 | |
| 84 | } | |
| 85 | sub LI_EL () { | |
| 86 | SPECIAL_EL | | |
| 87 | END_TAG_OPTIONAL_EL | | |
| 88 | ALL_END_TAG_OPTIONAL_EL | | |
| 89 | 0b100 | |
| 90 | } | |
| 91 | sub P_EL () { | |
| 92 | SPECIAL_EL | | |
| 93 | ADDRESS_DIV_P_EL | | |
| 94 | END_TAG_OPTIONAL_EL | | |
| 95 | ALL_END_TAG_OPTIONAL_EL | | |
| 96 | 0b001 | |
| 97 | } | |
| 98 | ||
| 99 | sub TABLE_ROW_EL () { | |
| 100 | SPECIAL_EL | | |
| 101 | TABLE_ROWS_EL | | |
| 102 | TABLE_ROW_SCOPING_EL | | |
| 103 | ALL_END_TAG_OPTIONAL_EL | | |
| 104 | 0b001 | |
| 105 | } | |
| 106 | sub TABLE_ROW_GROUP_EL () { | |
| 107 | SPECIAL_EL | | |
| 108 | TABLE_ROWS_EL | | |
| 109 | TABLE_ROWS_SCOPING_EL | | |
| 110 | ALL_END_TAG_OPTIONAL_EL | | |
| 111 | 0b001 | |
| 112 | } | |
| 113 | ||
| 114 | sub MISC_SCOPING_EL () { SCOPING_EL | 0b000 } | |
| 115 | sub BUTTON_EL () { SCOPING_EL | 0b001 } | |
| 116 | sub CAPTION_EL () { SCOPING_EL | 0b010 } | |
| 117 | sub HTML_EL () { | |
| 118 | SCOPING_EL | | |
| 119 | TABLE_SCOPING_EL | | |
| 120 | TABLE_ROWS_SCOPING_EL | | |
| 121 | TABLE_ROW_SCOPING_EL | | |
| 122 | ALL_END_TAG_OPTIONAL_EL | | |
| 123 | 0b001 | |
| 124 | } | |
| 125 | sub TABLE_EL () { | |
| 126 | SCOPING_EL | | |
| 127 | TABLE_ROWS_EL | | |
| 128 | TABLE_SCOPING_EL | | |
| 129 | 0b001 | |
| 130 | } | |
| 131 | sub TABLE_CELL_EL () { | |
| 132 | SCOPING_EL | | |
| 133 | TABLE_ROW_SCOPING_EL | | |
| 134 | ALL_END_TAG_OPTIONAL_EL | | |
| 135 | 0b001 | |
| 136 | } | |
| 137 | ||
| 138 | sub MISC_FORMATTING_EL () { FORMATTING_EL | 0b000 } | |
| 139 | sub A_EL () { FORMATTING_EL | 0b001 } | |
| 140 | sub NOBR_EL () { FORMATTING_EL | 0b010 } | |
| 141 | ||
| 142 | sub RUBY_EL () { PHRASING_EL | 0b001 } | |
| 143 | ||
| 144 | ## ISSUE: ALL_END_TAG_OPTIONAL_EL? | |
| 145 | sub OPTGROUP_EL () { PHRASING_EL | END_TAG_OPTIONAL_EL | 0b001 } | |
| 146 | sub OPTION_EL () { PHRASING_EL | END_TAG_OPTIONAL_EL | 0b010 } | |
| 147 | sub RUBY_COMPONENT_EL () { PHRASING_EL | END_TAG_OPTIONAL_EL | 0b100 } | |
| 148 | ||
| 149 | sub MML_AXML_EL () { PHRASING_EL | FOREIGN_EL | 0b001 } | |
| 150 | ||
| 151 | my $el_category = { | |
| 152 | a => A_EL, | |
| 153 | address => ADDRESS_DIV_EL, | |
| 154 | applet => MISC_SCOPING_EL, | |
| 155 | area => MISC_SPECIAL_EL, | |
| 156 | article => MISC_SPECIAL_EL, | |
| 157 | aside => MISC_SPECIAL_EL, | |
| 158 | b => FORMATTING_EL, | |
| 159 | base => MISC_SPECIAL_EL, | |
| 160 | basefont => MISC_SPECIAL_EL, | |
| 161 | bgsound => MISC_SPECIAL_EL, | |
| 162 | big => FORMATTING_EL, | |
| 163 | blockquote => MISC_SPECIAL_EL, | |
| 164 | body => BODY_EL, | |
| 165 | br => MISC_SPECIAL_EL, | |
| 166 | button => BUTTON_EL, | |
| 167 | caption => CAPTION_EL, | |
| 168 | center => MISC_SPECIAL_EL, | |
| 169 | col => MISC_SPECIAL_EL, | |
| 170 | colgroup => MISC_SPECIAL_EL, | |
| 171 | command => MISC_SPECIAL_EL, | |
| 172 | datagrid => MISC_SPECIAL_EL, | |
| 173 | dd => DTDD_EL, | |
| 174 | details => MISC_SPECIAL_EL, | |
| 175 | dialog => MISC_SPECIAL_EL, | |
| 176 | dir => MISC_SPECIAL_EL, | |
| 177 | div => ADDRESS_DIV_EL, | |
| 178 | dl => MISC_SPECIAL_EL, | |
| 179 | dt => DTDD_EL, | |
| 180 | em => FORMATTING_EL, | |
| 181 | embed => MISC_SPECIAL_EL, | |
| 182 | fieldset => MISC_SPECIAL_EL, | |
| 183 | figure => MISC_SPECIAL_EL, | |
| 184 | font => FORMATTING_EL, | |
| 185 | footer => MISC_SPECIAL_EL, | |
| 186 | form => FORM_EL, | |
| 187 | frame => MISC_SPECIAL_EL, | |
| 188 | frameset => FRAMESET_EL, | |
| 189 | h1 => HEADING_EL, | |
| 190 | h2 => HEADING_EL, | |
| 191 | h3 => HEADING_EL, | |
| 192 | h4 => HEADING_EL, | |
| 193 | h5 => HEADING_EL, | |
| 194 | h6 => HEADING_EL, | |
| 195 | head => MISC_SPECIAL_EL, | |
| 196 | header => MISC_SPECIAL_EL, | |
| 197 | hr => MISC_SPECIAL_EL, | |
| 198 | html => HTML_EL, | |
| 199 | i => FORMATTING_EL, | |
| 200 | iframe => MISC_SPECIAL_EL, | |
| 201 | img => MISC_SPECIAL_EL, | |
| 202 | #image => MISC_SPECIAL_EL, ## NOTE: Commented out in the spec. | |
| 203 | input => MISC_SPECIAL_EL, | |
| 204 | isindex => MISC_SPECIAL_EL, | |
| 205 | ## XXX keygen? (Whether a void element is in Special or not does not | |
| 206 | ## affect to the processing, however.) | |
| 207 | li => LI_EL, | |
| 208 | link => MISC_SPECIAL_EL, | |
| 209 | listing => MISC_SPECIAL_EL, | |
| 210 | marquee => MISC_SCOPING_EL, | |
| 211 | menu => MISC_SPECIAL_EL, | |
| 212 | meta => MISC_SPECIAL_EL, | |
| 213 | nav => MISC_SPECIAL_EL, | |
| 214 | nobr => NOBR_EL, | |
| 215 | noembed => MISC_SPECIAL_EL, | |
| 216 | noframes => MISC_SPECIAL_EL, | |
| 217 | noscript => MISC_SPECIAL_EL, | |
| 218 | object => MISC_SCOPING_EL, | |
| 219 | ol => MISC_SPECIAL_EL, | |
| 220 | optgroup => OPTGROUP_EL, | |
| 221 | option => OPTION_EL, | |
| 222 | p => P_EL, | |
| 223 | param => MISC_SPECIAL_EL, | |
| 224 | plaintext => MISC_SPECIAL_EL, | |
| 225 | pre => MISC_SPECIAL_EL, | |
| 226 | rp => RUBY_COMPONENT_EL, | |
| 227 | rt => RUBY_COMPONENT_EL, | |
| 228 | ruby => RUBY_EL, | |
| 229 | s => FORMATTING_EL, | |
| 230 | script => MISC_SPECIAL_EL, | |
| 231 | select => SELECT_EL, | |
| 232 | section => MISC_SPECIAL_EL, | |
| 233 | small => FORMATTING_EL, | |
| 234 | spacer => MISC_SPECIAL_EL, | |
| 235 | strike => FORMATTING_EL, | |
| 236 | strong => FORMATTING_EL, | |
| 237 | style => MISC_SPECIAL_EL, | |
| 238 | table => TABLE_EL, | |
| 239 | tbody => TABLE_ROW_GROUP_EL, | |
| 240 | td => TABLE_CELL_EL, | |
| 241 | textarea => MISC_SPECIAL_EL, | |
| 242 | tfoot => TABLE_ROW_GROUP_EL, | |
| 243 | th => TABLE_CELL_EL, | |
| 244 | thead => TABLE_ROW_GROUP_EL, | |
| 245 | title => MISC_SPECIAL_EL, | |
| 246 | tr => TABLE_ROW_EL, | |
| 247 | tt => FORMATTING_EL, | |
| 248 | u => FORMATTING_EL, | |
| 249 | ul => MISC_SPECIAL_EL, | |
| 250 | wbr => MISC_SPECIAL_EL, | |
| 251 | }; | }; |
| 252 | my $formatting_category = { | |
| 253 | a => 1, b => 1, big => 1, em => 1, font => 1, i => 1, nobr => 1, | my $el_category_f = { |
| 254 | s => 1, small => 1, strile => 1, strong => 1, tt => 1, u => 1, | $MML_NS => { |
| 255 | 'annotation-xml' => MML_AXML_EL, | |
| 256 | mi => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL, | |
| 257 | mo => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL, | |
| 258 | mn => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL, | |
| 259 | ms => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL, | |
| 260 | mtext => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL, | |
| 261 | }, | |
| 262 | $SVG_NS => { | |
| 263 | foreignObject => SCOPING_EL | FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL, | |
| 264 | desc => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL, | |
| 265 | title => FOREIGN_EL | FOREIGN_FLOW_CONTENT_EL, | |
| 266 | }, | |
| 267 | ## NOTE: In addition, FOREIGN_EL is set to non-HTML elements. | |
| 268 | }; | }; |
| # $phrasing_category: all other elements | ||
| 269 | ||
| 270 | sub parse_string ($$$;$) { | my $svg_attr_name = { |
| 271 | my $self = shift->new; | attributename => 'attributeName', |
| 272 | my $s = \$_[0]; | attributetype => 'attributeType', |
| 273 | $self->{document} = $_[1]; | basefrequency => 'baseFrequency', |
| 274 | baseprofile => 'baseProfile', | |
| 275 | calcmode => 'calcMode', | |
| 276 | clippathunits => 'clipPathUnits', | |
| 277 | contentscripttype => 'contentScriptType', | |
| 278 | contentstyletype => 'contentStyleType', | |
| 279 | diffuseconstant => 'diffuseConstant', | |
| 280 | edgemode => 'edgeMode', | |
| 281 | externalresourcesrequired => 'externalResourcesRequired', | |
| 282 | filterres => 'filterRes', | |
| 283 | filterunits => 'filterUnits', | |
| 284 | glyphref => 'glyphRef', | |
| 285 | gradienttransform => 'gradientTransform', | |
| 286 | gradientunits => 'gradientUnits', | |
| 287 | kernelmatrix => 'kernelMatrix', | |
| 288 | kernelunitlength => 'kernelUnitLength', | |
| 289 | keypoints => 'keyPoints', | |
| 290 | keysplines => 'keySplines', | |
| 291 | keytimes => 'keyTimes', | |
| 292 | lengthadjust => 'lengthAdjust', | |
| 293 | limitingconeangle => 'limitingConeAngle', | |
| 294 | markerheight => 'markerHeight', | |
| 295 | markerunits => 'markerUnits', | |
| 296 | markerwidth => 'markerWidth', | |
| 297 | maskcontentunits => 'maskContentUnits', | |
| 298 | maskunits => 'maskUnits', | |
| 299 | numoctaves => 'numOctaves', | |
| 300 | pathlength => 'pathLength', | |
| 301 | patterncontentunits => 'patternContentUnits', | |
| 302 | patterntransform => 'patternTransform', | |
| 303 | patternunits => 'patternUnits', | |
| 304 | pointsatx => 'pointsAtX', | |
| 305 | pointsaty => 'pointsAtY', | |
| 306 | pointsatz => 'pointsAtZ', | |
| 307 | preservealpha => 'preserveAlpha', | |
| 308 | preserveaspectratio => 'preserveAspectRatio', | |
| 309 | primitiveunits => 'primitiveUnits', | |
| 310 | refx => 'refX', | |
| 311 | refy => 'refY', | |
| 312 | repeatcount => 'repeatCount', | |
| 313 | repeatdur => 'repeatDur', | |
| 314 | requiredextensions => 'requiredExtensions', | |
| 315 | requiredfeatures => 'requiredFeatures', | |
| 316 | specularconstant => 'specularConstant', | |
| 317 | specularexponent => 'specularExponent', | |
| 318 | spreadmethod => 'spreadMethod', | |
| 319 | startoffset => 'startOffset', | |
| 320 | stddeviation => 'stdDeviation', | |
| 321 | stitchtiles => 'stitchTiles', | |
| 322 | surfacescale => 'surfaceScale', | |
| 323 | systemlanguage => 'systemLanguage', | |
| 324 | tablevalues => 'tableValues', | |
| 325 | targetx => 'targetX', | |
| 326 | targety => 'targetY', | |
| 327 | textlength => 'textLength', | |
| 328 | viewbox => 'viewBox', | |
| 329 | viewtarget => 'viewTarget', | |
| 330 | xchannelselector => 'xChannelSelector', | |
| 331 | ychannelselector => 'yChannelSelector', | |
| 332 | zoomandpan => 'zoomAndPan', | |
| 333 | }; | |
| 334 | ||
| 335 | ## NOTE: |set_inner_html| copies most of this method's code | my $foreign_attr_xname = { |
| 336 | 'xlink:actuate' => [$XLINK_NS, ['xlink', 'actuate']], | |
| 337 | 'xlink:arcrole' => [$XLINK_NS, ['xlink', 'arcrole']], | |
| 338 | 'xlink:href' => [$XLINK_NS, ['xlink', 'href']], | |
| 339 | 'xlink:role' => [$XLINK_NS, ['xlink', 'role']], | |
| 340 | 'xlink:show' => [$XLINK_NS, ['xlink', 'show']], | |
| 341 | 'xlink:title' => [$XLINK_NS, ['xlink', 'title']], | |
| 342 | 'xlink:type' => [$XLINK_NS, ['xlink', 'type']], | |
| 343 | 'xml:base' => [$XML_NS, ['xml', 'base']], | |
| 344 | 'xml:lang' => [$XML_NS, ['xml', 'lang']], | |
| 345 | 'xml:space' => [$XML_NS, ['xml', 'space']], | |
| 346 | 'xmlns' => [$XMLNS_NS, [undef, 'xmlns']], | |
| 347 | 'xmlns:xlink' => [$XMLNS_NS, ['xmlns', 'xlink']], | |
| 348 | }; | |
| 349 | ||
| 350 | my $i = 0; | ## ISSUE: xmlns:xlink="non-xlink-ns" is not an error. |
| my $line = 1; | ||
| my $column = 0; | ||
| $self->{set_next_input_character} = sub { | ||
| my $self = shift; | ||
| 351 | ||
| 352 | pop @{$self->{prev_input_character}}; | ## TODO: Invoke the reset algorithm when a resettable element is |
| 353 | unshift @{$self->{prev_input_character}}, $self->{next_input_character}; | ## created (cf. HTML5 revision 2259). |
| 354 | ||
| 355 | $self->{next_input_character} = -1 and return if $i >= length $$s; | sub parse_byte_string ($$$$;$) { |
| 356 | $self->{next_input_character} = ord substr $$s, $i++, 1; | my $self = shift; |
| 357 | $column++; | my $charset_name = shift; |
| 358 | open my $input, '<', ref $_[0] ? $_[0] : \($_[0]); | |
| 359 | if ($self->{next_input_character} == 0x000A) { # LF | return $self->parse_byte_stream ($charset_name, $input, @_[1..$#_]); |
| 360 | $line++; | } # parse_byte_string |
| 361 | $column = 0; | |
| 362 | } elsif ($self->{next_input_character} == 0x000D) { # CR | sub parse_byte_stream ($$$$;$$) { |
| 363 | $i++ if substr ($$s, $i, 1) eq "\x0A"; | # my ($self, $charset_name, $byte_stream, $doc, $onerror, $get_wrapper) = @_; |
| 364 | $self->{next_input_character} = 0x000A; # LF # MUST | my $self = ref $_[0] ? shift : shift->new; |
| 365 | $line++; | my $charset_name = shift; |
| 366 | $column = 0; | my $byte_stream = $_[0]; |
| } elsif ($self->{next_input_character} > 0x10FFFF) { | ||
| $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST | ||
| } elsif ($self->{next_input_character} == 0x0000) { # NULL | ||
| !!!parse-error (type => 'NULL'); | ||
| $self->{next_input_character} = 0xFFFD; # REPLACEMENT CHARACTER # MUST | ||
| } | ||
| }; | ||
| $self->{prev_input_character} = [-1, -1, -1]; | ||
| $self->{next_input_character} = -1; | ||
| 367 | ||
| 368 | my $onerror = $_[2] || sub { | my $onerror = $_[2] || sub { |
| 369 | my (%opt) = @_; | my (%opt) = @_; |
| 370 | warn "Parse error ($opt{type}) at line $opt{line} column $opt{column}\n"; | warn "Parse error ($opt{type})\n"; |
| }; | ||
| $self->{parse_error} = sub { | ||
| $onerror->(@_, line => $line, column => $column); | ||
| 371 | }; | }; |
| 372 | $self->{parse_error} = $onerror; # updated later by parse_char_string | |
| 373 | ||
| 374 | $self->_initialize_tokenizer; | my $get_wrapper = $_[3] || sub ($) { |
| 375 | $self->_initialize_tree_constructor; | return $_[0]; # $_[0] = byte stream handle, returned = arg to char handle |
| $self->_construct_tree; | ||
| $self->_terminate_tree_constructor; | ||
| return $self->{document}; | ||
| } # parse_string | ||
| sub new ($) { | ||
| my $class = shift; | ||
| my $self = bless {}, $class; | ||
| $self->{set_next_input_character} = sub { | ||
| $self->{next_input_character} = -1; | ||
| }; | ||
| $self->{parse_error} = sub { | ||
| # | ||
| 376 | }; | }; |
| return $self; | ||
| } # new | ||
| 377 | ||
| 378 | sub CM_ENTITY () { 0b001 } # & markup in data | ## HTML5 encoding sniffing algorithm |
| 379 | sub CM_LIMITED_MARKUP () { 0b010 } # < markup in data (limited) | require Message::Charset::Info; |
| 380 | sub CM_FULL_MARKUP () { 0b100 } # < markup in data (any) | my $charset; |
| 381 | my $buffer; | |
| 382 | sub PLAINTEXT_CONTENT_MODEL () { 0 } | my ($char_stream, $e_status); |
| 383 | sub CDATA_CONTENT_MODEL () { CM_LIMITED_MARKUP } | |
| 384 | sub RCDATA_CONTENT_MODEL () { CM_ENTITY | CM_LIMITED_MARKUP } | SNIFFING: { |
| 385 | sub PCDATA_CONTENT_MODEL () { CM_ENTITY | CM_FULL_MARKUP } | ## NOTE: By setting |allow_fallback| option true when the |
| 386 | ## |get_decode_handle| method is invoked, we ignore what the HTML5 | |
| 387 | ## spec requires, i.e. unsupported encoding should be ignored. | |
| 388 | ## TODO: We should not do this unless the parser is invoked | |
| 389 | ## in the conformance checking mode, in which this behavior | |
| 390 | ## would be useful. | |
| 391 | ||
| 392 | ## Implementations MUST act as if state machine in the spec | ## Step 1 |
| 393 | if (defined $charset_name) { | |
| 394 | $charset = Message::Charset::Info->get_by_html_name ($charset_name); | |
| 395 | ## TODO: Is this ok? Transfer protocol's parameter should be | |
| 396 | ## interpreted in its semantics? | |
| 397 | ||
| 398 | ($char_stream, $e_status) = $charset->get_decode_handle | |
| 399 | ($byte_stream, allow_error_reporting => 1, | |
| 400 | allow_fallback => 1); | |
| 401 | if ($char_stream) { | |
| 402 | $self->{confident} = 1; | |
| 403 | last SNIFFING; | |
| 404 | } else { | |
| 405 | !!!parse-error (type => 'charset:not supported', | |
| 406 | layer => 'encode', | |
| 407 | line => 1, column => 1, | |
| 408 | value => $charset_name, | |
| 409 | level => $self->{level}->{uncertain}); | |
| 410 | } | |
| 411 | } | |
| 412 | ||
| 413 | sub _initialize_tokenizer ($) { | ## Step 2 |
| 414 | my $self = shift; | my $byte_buffer = ''; |
| 415 | $self->{state} = 'data'; # MUST | for (1..1024) { |
| 416 | $self->{content_model} = PCDATA_CONTENT_MODEL; # be | my $char = $byte_stream->getc; |
| 417 | undef $self->{current_token}; # start tag, end tag, comment, or DOCTYPE | last unless defined $char; |
| 418 | undef $self->{current_attribute}; | $byte_buffer .= $char; |
| 419 | undef $self->{last_emitted_start_tag_name}; | } ## TODO: timeout |
| undef $self->{last_attribute_value_state}; | ||
| $self->{char} = []; | ||
| # $self->{next_input_character} | ||
| !!!next-input-character; | ||
| $self->{token} = []; | ||
| # $self->{escape} | ||
| } # _initialize_tokenizer | ||
| ## A token has: | ||
| ## ->{type} eq 'DOCTYPE', 'start tag', 'end tag', 'comment', | ||
| ## 'character', or 'end-of-file' | ||
| ## ->{name} (DOCTYPE, start tag (tag name), end tag (tag name)) | ||
| ## ->{public_identifier} (DOCTYPE) | ||
| ## ->{system_identifier} (DOCTYPE) | ||
| ## ->{correct} == 1 or 0 (DOCTYPE) | ||
| ## ->{attributes} isa HASH (start tag, end tag) | ||
| ## ->{data} (comment, character) | ||
| ## Emitted token MUST immediately be handled by the tree construction state. | ||
| ## Before each step, UA MAY check to see if either one of the scripts in | ||
| ## "list of scripts that will execute as soon as possible" or the first | ||
| ## script in the "list of scripts that will execute asynchronously", | ||
| ## has completed loading. If one has, then it MUST be executed | ||
| ## and removed from the list. | ||
| 420 | ||
| 421 | sub _get_next_token ($) { | ## Step 3 |
| 422 | my $self = shift; | if ($byte_buffer =~ /^\xFE\xFF/) { |
| 423 | if (@{$self->{token}}) { | $charset = Message::Charset::Info->get_by_html_name ('utf-16be'); |
| 424 | return shift @{$self->{token}}; | ($char_stream, $e_status) = $charset->get_decode_handle |
| 425 | } | ($byte_stream, allow_error_reporting => 1, |
| 426 | allow_fallback => 1, byte_buffer => \$byte_buffer); | |
| 427 | $self->{confident} = 1; | |
| 428 | last SNIFFING; | |
| 429 | } elsif ($byte_buffer =~ /^\xFF\xFE/) { | |
| 430 | $charset = Message::Charset::Info->get_by_html_name ('utf-16le'); | |
| 431 | ($char_stream, $e_status) = $charset->get_decode_handle | |
| 432 | ($byte_stream, allow_error_reporting => 1, | |
| 433 | allow_fallback => 1, byte_buffer => \$byte_buffer); | |
| 434 | $self->{confident} = 1; | |
| 435 | last SNIFFING; | |
| 436 | } elsif ($byte_buffer =~ /^\xEF\xBB\xBF/) { | |
| 437 | $charset = Message::Charset::Info->get_by_html_name ('utf-8'); | |
| 438 | ($char_stream, $e_status) = $charset->get_decode_handle | |
| 439 | ($byte_stream, allow_error_reporting => 1, | |
| 440 | allow_fallback => 1, byte_buffer => \$byte_buffer); | |
| 441 | $self->{confident} = 1; | |
| 442 | last SNIFFING; | |
| 443 | } | |
| 444 | ||
| 445 | A: { | ## Step 4 |
| 446 | if ($self->{state} eq 'data') { | ## TODO: <meta charset> |
| if ($self->{next_input_character} == 0x0026) { # & | ||
| if ($self->{content_model} & CM_ENTITY) { # PCDATA | RCDATA | ||
| $self->{state} = 'entity data'; | ||
| !!!next-input-character; | ||
| redo A; | ||
| } else { | ||
| # | ||
| } | ||
| } elsif ($self->{next_input_character} == 0x002D) { # - | ||
| if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA | ||
| unless ($self->{escape}) { | ||
| if ($self->{prev_input_character}->[0] == 0x002D and # - |