Parent Directory
|
Revision Log
|
Patch
| revision 1.11 by wakaba, Sat Jun 23 03:53:35 2007 UTC | revision 1.243 by wakaba, Sun Sep 6 13:52:06 2009 UTC | |
|---|---|---|
| # | Line 1 | Line 1 |
| 1 | package Whatpm::HTML; | package Whatpm::HTML; |
| 2 | use strict; | use strict; |
| 3 | our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r}; | our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r}; |
| 4 | use Error qw(:try); | |
| 5 | ||
| 6 | ## This is an early version of an HTML parser. | use Whatpm::HTML::Tokenizer; |
| 7 | ||
| 8 | my $permitted_slash_tag_name = { | ## NOTE: This module don't check all HTML5 parse errors; character |
| 9 | base => 1, | ## encoding related parse errors are expected to be handled by relevant |
| 10 | link => 1, | ## modules. |
| 11 | meta => 1, | ## Parse errors for control characters that are not allowed in HTML5 |
| 12 | hr => 1, | ## documents, for surrogate code points, and for noncharacter code |
| 13 | br => 1, | ## points, as well as U+FFFD substitions for characters whose code points |
| 14 | img=> 1, | ## is higher than U+10FFFF may be detected by combining the parser with |
| 15 | embed => 1, | ## the checker implemented by Whatpm::Charset::UnicodeChecker (for its |
| 16 | param => 1, | ## usage example, see |t/HTML-tree.t| in the Whatpm package or the |
| 17 | area |