/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory | Revision Log | View Patch Patch

-revision 1.61 by wakaba,
Sun Nov  4 04:15:06 2007 UTC
+revision 1.72 by wakaba,
Sun Mar  2 14:32:26 2008 UTC
 Line 1
  package Whatpm::HTML;
  use strict;
  our $VERSION=do{my @r=(q$Revision$=~/\d+/g);sprintf "%d."."%02d" x $#r,@r};
+ use Error qw(:try);
  ## ISSUE:
  ## var doc = implementation.createDocument (null, null, null);
  ## doc.write ('');
  ## alert (doc.compatMode);
- ## ISSUE: HTML5 revision 967 says that the encoding layer MUST NOT
+ ## TODO: Control charcters and noncharacters are not allowed (HTML5 revision 1263)
- ## strip BOM and the HTML layer MUST ignore it.  Whether we can do it
+ ## TODO: 1252 parse error (revision 1264)
- ## is not yet clear.
+ ## TODO: 8859-11 = 874 (revision 1271)
- ## "{U+FEFF}..." in UTF-16BE/UTF-16LE is three or four characters?
- ## "{U+FEFF}..." in GB18030?
  my $permitted_slash_tag_name = {
    base => 1,
-Line 19 
 my $permitted_slash_tag_name = {
+Line 18 
 my $permitted_slash_tag_name = {
    meta => 1,
    hr => 1,
    br => 1,
-   img=> 1,
+   img => 1,
    embed => 1,
    param => 1,
    area => 1,
-Line 84 
 my $formatting_category = {
+Line 83 
 my $formatting_category = {
  };
  # $phrasing_category: all other elements
+ sub parse_byte_string ($$$$;$) {
+   my $self = ref $_[0] ? shift : shift->new;
+   my $charset = shift;
+   my $bytes_s = ref $_[0] ? $_[0] : \($_[0]);
+   my $s;
+   if (defined $charset) {
+     require Encode; ## TODO: decode(utf8) don't delete BOM
+     $s = \ (Encode::decode ($charset, $$bytes_s));
+     $self->{input_encoding} = lc $charset; ## TODO: normalize name
+     $self->{confident} = 1;
+   } else {
+     ## TODO: Implement HTML5 detection algorithm
+     require Whatpm::Charset::UniversalCharDet;
+     $charset = Whatpm::Charset::UniversalCharDet->detect_byte_string
+         (substr ($$bytes_s, 0, 1024));
+     $charset ||= 'windows-1252';
+     $s = \ (Encode::decode ($charset, $$bytes_s));
+     $self->{input_encoding} = $charset;
+     $self->{confident} = 0;
+   }
+   $self->{change_encoding} = sub {
+     my $self = shift;
+     my $charset = lc shift;
+     ## TODO: if $charset is supported
+     ## TODO: normalize charset name
+     ## "Change the encoding" algorithm:
+     ## Step 1
+     if ($charset eq 'utf-16') { ## ISSUE: UTF-16BE -> UTF-8? UTF-16LE -> UTF-8?
+       $charset = 'utf-8';
+     }
+     ## Step 2
+     if (defined $self->{input_encoding} and
+         $self->{input_encoding} eq $charset) {
+       $self->{confident} = 1;
+       return;
+     }
+     !!!parse-error (type => 'charset label detected:'.$self->{input_encoding}.
+         ':'.$charset, level => 'w');
+     ## Step 3
+     # if (can) {
+       ## change the encoding on the fly.
+       #$self->{confident} = 1;
+       #return;
+     # }
+     ## Step 4
+     throw Whatpm::HTML::RestartParser (charset => $charset);
+   }; # $self->{change_encoding}
+   my @args = @_; shift @args; # $s
+   my $return;
+   try {
+     $return = $self->parse_char_string ($s, @args);
+   } catch Whatpm::HTML::RestartParser with {
+     my $charset = shift->{charset};
+     $s = \ (Encode::decode ($charset, $$bytes_s));
+     $self->{input_encoding} = $charset; ## TODO: normalize
+     $self->{confident} = 1;
+     $return = $self->parse_char_string ($s, @args);
+   };
+   return $return;
+ } # parse_byte_string
+ ## NOTE: HTML5 spec says that the encoding layer MUST NOT strip BOM
+ ## and the HTML layer MUST ignore it.  However, we does strip BOM in
+ ## the encoding layer and the HTML layer does not ignore any U+FEFF,
+ ## because the core part of our HTML parser expects a string of character,
+ ## not a string of bytes or code units or anything which might contain a BOM.
+ ## Therefore, any parser interface that accepts a string of bytes,
+ ## such as |parse_byte_string| in this module, must ensure that it does
+ ## strip the BOM and never strip any ZWNBSP.
+ *parse_char_string = \&parse_string;
  sub parse_string ($$$;$) {
-   my $self = shift->new;
+   my $self = ref $_[0] ? shift : shift->new;
-   my $s = \$_[0];
+   my $s = ref $_[0] ? $_[0] : \($_[0]);
    $self->{document} = $_[1];
+   @{$self->{document}->child_nodes} = ();
    ## NOTE: |set_inner_html| copies most of this method's code
+   $self->{confident} = 1 unless exists $self->{confident};
+   $self->{document}->input_encoding ($self->{input_encoding})
+       if defined $self->{input_encoding};
    my $i = 0;
    my $line = 1;
    my $column = 0;
-Line 147 
 sub new ($) {
+Line 232 
 sub new ($) {
    $self->{parse_error} = sub {
      #
    };
+   $self->{change_encoding} = sub {
+     # if ($_[0] is a supported encoding) {
+     #   run "change the encoding" algorithm;
+     #   throw Whatpm::HTML::RestartParser (charset => $new_encoding);
+     # }
+   };
    $self->{application_cache_selection} = sub {
      #
    };
-Line 195 
 sub DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUO
+Line 286 
 sub DOCTYPE_SYSTEM_IDENTIFIER_DOUBLE_QUO
  sub DOCTYPE_SYSTEM_IDENTIFIER_SINGLE_QUOTED_STATE () { 30 }
  sub AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE () { 31 }
  sub BOGUS_DOCTYPE_STATE () { 32 }
+ sub AFTER_ATTRIBUTE_VALUE_QUOTED_STATE () { 33 }
  sub DOCTYPE_TOKEN () { 1 }
  sub COMMENT_TOKEN () { 2 }
-Line 256 
 sub _initialize_tokenizer ($) {
+Line 348 
 sub _initialize_tokenizer ($) {
  ##   ->{system_identifier} (DOCTYPE_TOKEN)
  ##   ->{correct} == 1 or 0 (DOCTYPE_TOKEN)
  ##   ->{attributes} isa HASH (START_TAG_TOKEN, END_TAG_TOKEN)
+ ##        ->{name}
+ ##        ->{value}
+ ##        ->{has_reference} == 1 or 0
  ##   ->{data} (COMMENT_TOKEN, CHARACTER_TOKEN)
  ## Emitted token MUST immediately be handled by the tree construction state.
-Line 290 
 sub _get_next_token ($) {
+Line 385 
 sub _get_next_token ($) {
    A: {
      if ($self->{state} == DATA_STATE) {
        if ($self->{next_input_character} == 0x0026) { # &
-         if ($self->{content_model} & CM_ENTITY) { # PCDATA | RCDATA
+         if ($self->{content_model} & CM_ENTITY and # PCDATA | RCDATA
+             not $self->{escape}) {
            $self->{state} = ENTITY_DATA_STATE;
            !!!next-input-character;
            redo A;
-Line 345 
 sub _get_next_token ($) {
+Line 441 
 sub _get_next_token ($) {
      } elsif ($self->{state} == ENTITY_DATA_STATE) {
        ## (cannot happen in CDATA state)
-       my $token = $self->_tokenize_attempt_to_consume_an_entity (0);
+       my $token = $self->_tokenize_attempt_to_consume_an_entity (0, -1);
        $self->{state} = DATA_STATE;
        # next-input-character is already done
-Line 648 
 sub _get_next_token ($) {
+Line 744 
 sub _get_next_token ($) {
          redo A;
        } else {
+         if ({
+x0022 => 1, # "
+x0027 => 1, # '
+x003D => 1, # =
+             }->{$self->{next_input_character}}) {
+           !!!parse-error (type => 'bad attribute name');
+         }
          $self->{current_attribute} = {name => chr ($self->{next_input_character}),
                                value => ''};
          $self->{state} = ATTRIBUTE_NAME_STATE;
-Line 742 
 sub _get_next_token ($) {
+Line 845 
 sub _get_next_token ($) {
          redo A;
        } else {
+         if ($self->{next_input_character} == 0x0022 or # "
+             $self->{next_input_character} == 0x0027) { # '
+           !!!parse-error (type => 'bad attribute name');
+         }
          $self->{current_attribute}->{name} .= chr ($self->{next_input_character});
          ## Stay in the state
          !!!next-input-character;
-Line 888 
 sub _get_next_token ($) {
+Line 995 
 sub _get_next_token ($) {
          redo A;
        } else {
+         if ($self->{next_input_character} == 0x003D) { # =
+           !!!parse-error (type => 'bad attribute value');
+         }
          $self->{current_attribute}->{value} .= chr ($self->{next_input_character});
          $self->{state} = ATTRIBUTE_VALUE_UNQUOTED_STATE;
          !!!next-input-character;
-Line 895 
 sub _get_next_token ($) {
+Line 1005 
 sub _get_next_token ($) {
        }
      } elsif ($self->{state} == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE) {
        if ($self->{next_input_character} == 0x0022) { # "
-         $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
+         $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
          !!!next-input-character;
          redo A;
        } elsif ($self->{next_input_character} == 0x0026) { # &
-Line 931 
 sub _get_next_token ($) {
+Line 1041 
 sub _get_next_token ($) {
        }
      } elsif ($self->{state} == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE) {
        if ($self->{next_input_character} == 0x0027) { # '
-         $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
+         $self->{state} = AFTER_ATTRIBUTE_VALUE_QUOTED_STATE;
          !!!next-input-character;
          redo A;
        } elsif ($self->{next_input_character} == 0x0026) { # &
-Line 1019 
 sub _get_next_token ($) {
+Line 1129 
 sub _get_next_token ($) {
          redo A;
        } else {
+         if ({
+x0022 => 1, # "
+x0027 => 1, # '
+x003D => 1, # =
+             }->{$self->{next_input_character}}) {
+           !!!parse-error (type => 'bad attribute value');
+         }
          $self->{current_attribute}->{value} .= chr ($self->{next_input_character});
          ## Stay in the state
          !!!next-input-character;
          redo A;
        }
      } elsif ($self->{state} == ENTITY_IN_ATTRIBUTE_VALUE_STATE) {
-       my $token = $self->_tokenize_attempt_to_consume_an_entity (1);
+       my $token = $self->_tokenize_attempt_to_consume_an_entity
+           (1,
+            $self->{last_attribute_value_state}
+              == ATTRIBUTE_VALUE_DOUBLE_QUOTED_STATE ? 0x0022 : # "
+            $self->{last_attribute_value_state}
+              == ATTRIBUTE_VALUE_SINGLE_QUOTED_STATE ? 0x0027 : # '
+            -1);
        unless (defined $token) {
          $self->{current_attribute}->{value} .= '&';
        } else {
          $self->{current_attribute}->{value} .= $token->{data};
+         $self->{current_attribute}->{has_reference} = $token->{has_reference};
          ## ISSUE: spec says "append the returned character token to the current attribute's value"
        }
        $self->{state} = $self->{last_attribute_value_state};
        # next-input-character is already done
        redo A;
+     } elsif ($self->{state} == AFTER_ATTRIBUTE_VALUE_QUOTED_STATE) {
+       if ($self->{next_input_character} == 0x0009 or # HT
+           $self->{next_input_character} == 0x000A or # LF
+           $self->{next_input_character} == 0x000B or # VT
+           $self->{next_input_character} == 0x000C or # FF
+           $self->{next_input_character} == 0x0020) { # SP
+         $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
+         !!!next-input-character;
+         redo A;
+       } elsif ($self->{next_input_character} == 0x003E) { # >
+         if ($self->{current_token}->{type} == START_TAG_TOKEN) {
+           $self->{current_token}->{first_start_tag}
+               = not defined $self->{last_emitted_start_tag_name};
+           $self->{last_emitted_start_tag_name} = $self->{current_token}->{tag_name};
+         } elsif ($self->{current_token}->{type} == END_TAG_TOKEN) {
+           $self->{content_model} = PCDATA_CONTENT_MODEL; # MUST
+           if ($self->{current_token}->{attributes}) {
+             !!!parse-error (type => 'end tag attribute');
+           }
+         } else {
+           die "$0: $self->{current_token}->{type}: Unknown token type";
+         }
+         $self->{state} = DATA_STATE;
+         !!!next-input-character;
+         !!!emit ($self->{current_token}); # start tag or end tag
+         redo A;
+       } elsif ($self->{next_input_character} == 0x002F) { # /
+         !!!next-input-character;
+         if ($self->{next_input_character} == 0x003E and # >
+             $self->{current_token}->{type} == START_TAG_TOKEN and
+             $permitted_slash_tag_name->{$self->{current_token}->{tag_name}}) {
+           # permitted slash
+           #
+         } else {
+           !!!parse-error (type => 'nestc');
+         }
+         $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
+         # next-input-character is already done
+         redo A;
+       } else {
+         !!!parse-error (type => 'no space between attributes');
+         $self->{state} = BEFORE_ATTRIBUTE_NAME_STATE;
+         ## reconsume
+         redo A;
+       }
      } elsif ($self->{state} == BOGUS_COMMENT_STATE) {
        ## (only happen if PCDATA state)
-Line 1467 
 sub _get_next_token ($) {
+Line 1638 
 sub _get_next_token ($) {
          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
          !!!next-input-character;
          redo A;
+       } elsif ($self->{next_input_character} == 0x003E) { # >
+         !!!parse-error (type => 'unclosed PUBLIC literal');
+         $self->{state} = DATA_STATE;
+         !!!next-input-character;
+         delete $self->{current_token}->{correct};
+         !!!emit ($self->{current_token}); # DOCTYPE
+         redo A;
        } elsif ($self->{next_input_character} == -1) {
          !!!parse-error (type => 'unclosed PUBLIC literal');
-Line 1489 
 sub _get_next_token ($) {
+Line 1670 
 sub _get_next_token ($) {
          $self->{state} = AFTER_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
          !!!next-input-character;
          redo A;
+       } elsif ($self->{next_input_character} == 0x003E) { # >
+         !!!parse-error (type => 'unclosed PUBLIC literal');
+         $self->{state} = DATA_STATE;
+         !!!next-input-character;
+         delete $self->{current_token}->{correct};
+         !!!emit ($self->{current_token}); # DOCTYPE
+         redo A;
        } elsif ($self->{next_input_character} == -1) {
          !!!parse-error (type => 'unclosed PUBLIC literal');
-Line 1595 
 sub _get_next_token ($) {
+Line 1786 
 sub _get_next_token ($) {
          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
          !!!next-input-character;
          redo A;
+       } elsif ($self->{next_input_character} == 0x003E) { # >
+         !!!parse-error (type => 'unclosed PUBLIC literal');
+         $self->{state} = DATA_STATE;
+         !!!next-input-character;
+         delete $self->{current_token}->{correct};
+         !!!emit ($self->{current_token}); # DOCTYPE
+         redo A;
        } elsif ($self->{next_input_character} == -1) {
          !!!parse-error (type => 'unclosed SYSTEM literal');
-Line 1617 
 sub _get_next_token ($) {
+Line 1818 
 sub _get_next_token ($) {
          $self->{state} = AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
          !!!next-input-character;
          redo A;
+       } elsif ($self->{next_input_character} == 0x003E) { # >
+         !!!parse-error (type => 'unclosed PUBLIC literal');
+         $self->{state} = DATA_STATE;
+         !!!next-input-character;
+         delete $self->{current_token}->{correct};
+         !!!emit ($self->{current_token}); # DOCTYPE
+         redo A;
        } elsif ($self->{next_input_character} == -1) {
          !!!parse-error (type => 'unclosed SYSTEM literal');
-Line 1696 
 sub _get_next_token ($) {
+Line 1907 
 sub _get_next_token ($) {
    die "$0: _get_next_token: unexpected case";
  } # _get_next_token
- sub _tokenize_attempt_to_consume_an_entity ($$) {
+ sub _tokenize_attempt_to_consume_an_entity ($$$) {
-   my ($self, $in_attr) = @_;
+   my ($self, $in_attr, $additional) = @_;
    if ({
 x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, # HT, LF, VT, FF,
 x0020 => 1, 0x003C => 1, 0x0026 => 1, -1 => 1, # SP, <, & # 0x000D # CR
+        $additional => 1,
        }->{$self->{next_input_character}}) {
      ## Don't consume
      ## No error
-Line 1757 
 sub _tokenize_attempt_to_consume_an_enti
+Line 1969 
 sub _tokenize_attempt_to_consume_an_enti
            $code = $c1_entity_char->{$code};
          }
-         return {type => CHARACTER_TOKEN, data => chr $code};
+         return {type => CHARACTER_TOKEN, data => chr $code,
+                 has_reference => 1};
        } # X
      } elsif (0x0030 <= $self->{next_input_character} and
               $self->{next_input_character} <= 0x0039) { # 0..9
-Line 1792 
 sub _tokenize_attempt_to_consume_an_enti
+Line 2005 
 sub _tokenize_attempt_to_consume_an_enti
          $code = $c1_entity_char->{$code};
        }
-       return {type => CHARACTER_TOKEN, data => chr $code};
+       return {type => CHARACTER_TOKEN, data => chr $code, has_reference => 1};
      } else {
        !!!parse-error (type => 'bare nero');
        !!!back-next-input-character ($self->{next_input_character});
-Line 1840 
 sub _tokenize_attempt_to_consume_an_enti
+Line 2053 
 sub _tokenize_attempt_to_consume_an_enti
      }
      if ($match > 0) {
-       return {type => CHARACTER_TOKEN, data => $value};
+       return {type => CHARACTER_TOKEN, data => $value, has_reference => 1};
      } elsif ($match < 0) {
        !!!parse-error (type => 'no refc');
        if ($in_attr and $match < -1) {
          return {type => CHARACTER_TOKEN, data => '&'.$entity_name};
        } else {
-         return {type => CHARACTER_TOKEN, data => $value};
+         return {type => CHARACTER_TOKEN, data => $value, has_reference => 1};
        }
      } else {
        !!!parse-error (type => 'bare ero');
-       ## NOTE: No characters are consumed in the spec.
+       ## NOTE: "No characters are consumed" in the spec.
        return {type => CHARACTER_TOKEN, data => '&'.$value};
      }
    } else {
-Line 1987 
 sub _tree_construction_initial ($) {
+Line 2200 
 sub _tree_construction_initial ($) {
            "-//NETSCAPE COMM. CORP.//DTD STRICT HTML//EN" => 1,
            "-//O'REILLY AND ASSOCIATES//DTD HTML 2.0//EN" => 1,
            "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED 1.0//EN" => 1,
+           "-//O'REILLY AND ASSOCIATES//DTD HTML EXTENDED RELAXED 1.0//EN" => 1,
+           "-//SOFTQUAD SOFTWARE//DTD HOTMETAL PRO 6.0::19990601::EXTENSIONS TO HTML 4.0//EN" => 1,
+           "-//SOFTQUAD//DTD HOTMETAL PRO 4.0::19971010::EXTENSIONS TO HTML 4.0//EN" => 1,
            "-//SPYGLASS//DTD HTML 2.0 EXTENDED//EN" => 1,
            "-//SQ//DTD HTML 2.0 HOTMETAL + EXTENSIONS//EN" => 1,
            "-//SUN MICROSYSTEMS CORP.//DTD HOTJAVA HTML//EN" => 1,
-Line 2104 
 sub _tree_construction_root_element ($)
+Line 2320 
 sub _tree_construction_root_element ($)
          #
        } elsif ($token->{type} == START_TAG_TOKEN) {
          if ($token->{tag_name} eq 'html' and
-             $token->{attributes}->{manifest}) { ## ISSUE: Spec spells as "application"
+             $token->{attributes}->{manifest}) {
            $self->{application_cache_selection}
                 ->($token->{attributes}->{manifest}->{value});
            ## ISSUE: No relative reference resolution?
-Line 2782 
 sub _tree_construction_main ($) {
+Line 2998 
 sub _tree_construction_main ($) {
                  push @{$self->{open_elements}}, [$self->{head_element}, 'head'];
                }
                !!!insert-element ($token->{tag_name}, $token->{attributes});
-               pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
+               my $meta_el = pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
                unless ($self->{confident}) {
-                 my $charset;
                  if ($token->{attributes}->{charset}) { ## TODO: And if supported
-                   $charset = $token->{attributes}->{charset}->{value};
+                   $self->{change_encoding}
-                 }
+                       ->($self, $token->{attributes}->{charset}->{value});
-                 if ($token->{attributes}->{'http-equiv'}) {
+                   $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
+                       ->set_user_data (manakai_has_reference =>
+                                            $token->{attributes}->{charset}
+                                                ->{has_reference});
+                 } elsif ($token->{attributes}->{content}) {
                    ## ISSUE: Algorithm name in the spec was incorrect so that not linked to the definition.
-                   if ($token->{attributes}->{'http-equiv'}->{value}
+                   if ($token->{attributes}->{content}->{value}
-                       =~ /\A[^;]*;[\x09-\x0D\x20]*charset[\x09-\x0D\x20]*=
+                       =~ /\A[^;]*;[\x09-\x0D\x20]*[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
+                           [\x09-\x0D\x20]*=
                            [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
                            ([^"'\x09-\x0D\x20][^\x09-\x0D\x20]*))/x) {
-                     $charset = defined $1 ? $1 : defined $2 ? $2 : $3;
+                     $self->{change_encoding}
-                   } ## TODO: And if supported
+                         ->($self, defined $1 ? $1 : defined $2 ? $2 : $3);
+                     $meta_el->[0]->get_attribute_node_ns (undef, 'content')
+                         ->set_user_data (manakai_has_reference =>
+                                              $token->{attributes}->{content}
+                                                    ->{has_reference});
+                   }
+                 }
+               } else {
+                 if ($token->{attributes}->{charset}) {
+                   $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
+                       ->set_user_data (manakai_has_reference =>
+                                            $token->{attributes}->{charset}
+                                                ->{has_reference});
+                 }
+                 if ($token->{attributes}->{content}) {
+                   $meta_el->[0]->get_attribute_node_ns (undef, 'content')
+                       ->set_user_data (manakai_has_reference =>
+                                            $token->{attributes}->{content}
+                                                ->{has_reference});
                  }
-                 ## TODO: Change the encoding
                }
-               ## TODO: Extracting |charset| from |meta|.
                pop @{$self->{open_elements}}
                    if $self->{insertion_mode} == AFTER_HEAD_IM;
                !!!next-token;
-Line 4372 
 sub _tree_construction_main ($) {
+Line 4609 
 sub _tree_construction_main ($) {
        } elsif ($token->{tag_name} eq 'meta') {
          ## NOTE: This is an "as if in head" code clone, only "-t" differs
          !!!insert-element-t ($token->{tag_name}, $token->{attributes});
-         pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
+         my $meta_el = pop @{$self->{open_elements}}; ## ISSUE: This step is missing in the spec.
          unless ($self->{confident}) {
-           my $charset;
            if ($token->{attributes}->{charset}) { ## TODO: And if supported
-             $charset = $token->{attributes}->{charset}->{value};
+             $self->{change_encoding}
-           }
+                 ->($self, $token->{attributes}->{charset}->{value});
-           if ($token->{attributes}->{'http-equiv'}) {
+             $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
+                 ->set_user_data (manakai_has_reference =>
+                                      $token->{attributes}->{charset}
+                                          ->{has_reference});
+           } elsif ($token->{attributes}->{content}) {
              ## ISSUE: Algorithm name in the spec was incorrect so that not linked to the definition.
-             if ($token->{attributes}->{'http-equiv'}->{value}
+             if ($token->{attributes}->{content}->{value}
-                 =~ /\A[^;]*;[\x09-\x0D\x20]*charset[\x09-\x0D\x20]*=
+                 =~ /\A[^;]*;[\x09-\x0D\x20]*[Cc][Hh][Aa][Rr][Ss][Ee][Tt]
+                     [\x09-\x0D\x20]*=
                      [\x09-\x0D\x20]*(?>"([^"]*)"|'([^']*)'|
                      ([^"'\x09-\x0D\x20][^\x09-\x0D\x20]*))/x) {
-               $charset = defined $1 ? $1 : defined $2 ? $2 : $3;
+               $self->{change_encoding}
-             } ## TODO: And if supported
+                   ->($self, defined $1 ? $1 : defined $2 ? $2 : $3);
+               $meta_el->[0]->get_attribute_node_ns (undef, 'content')
+                   ->set_user_data (manakai_has_reference =>
+                                        $token->{attributes}->{content}
+                                              ->{has_reference});
+             }
+           }
+         } else {
+           if ($token->{attributes}->{charset}) {
+             $meta_el->[0]->get_attribute_node_ns (undef, 'charset')
+                 ->set_user_data (manakai_has_reference =>
+                                      $token->{attributes}->{charset}
+                                          ->{has_reference});
+           }
+           if ($token->{attributes}->{content}) {
+             $meta_el->[0]->get_attribute_node_ns (undef, 'content')
+                 ->set_user_data (manakai_has_reference =>
+                                      $token->{attributes}->{content}
+                                          ->{has_reference});
            }
-           ## TODO: Change the encoding
          }
          !!!next-token;
-Line 5214 
 sub set_inner_html ($$$) {
+Line 5473 
 sub set_inner_html ($$$) {
    my $s = \$_[0];
    my $onerror = $_[1];
+   ## ISSUE: Should {confident} be true?
    my $nt = $node->node_type;
    if ($nt == 9) {
      # MUST
-Line 5286 
 sub set_inner_html ($$$) {
+Line 5547 
 sub set_inner_html ($$$) {
      $p->_initialize_tree_constructor;
      ## Step 2
-     my $node_ln = $node->local_name;
+     my $node_ln = $node->manakai_local_name;
      $p->{content_model} = {
        title => RCDATA_CONTENT_MODEL,
        textarea => RCDATA_CONTENT_MODEL,
-Line 5326 
 sub set_inner_html ($$$) {
+Line 5587 
 sub set_inner_html ($$$) {
        if ($anode->node_type == 1) {
          my $nsuri = $anode->namespace_uri;
          if (defined $nsuri and $nsuri eq 'http://www.w3.org/1999/xhtml') {
-           if ($anode->local_name eq 'form') { ## TODO: case?
+           if ($anode->manakai_local_name eq 'form') {
              $p->{form_element} = $anode;
              last AN;
            }
-Line 5366 
 sub set_inner_html ($$$) {
+Line 5627 
 sub set_inner_html ($$$) {
  } # tree construction stage
- sub get_inner_html ($$$) {
+ package Whatpm::HTML::RestartParser;
-   my (undef, $node, $on_error) = @_;
+ push our @ISA, 'Error';
-   ## Step 1
-   my $s = '';
-   my $in_cdata;
-   my $parent = $node;
-   while (defined $parent) {
-     if ($parent->node_type == 1 and
-         $parent->namespace_uri eq 'http://www.w3.org/1999/xhtml' and
-         {
-           style => 1, script => 1, xmp => 1, iframe => 1,
-           noembed => 1, noframes => 1, noscript => 1,
-         }->{$parent->local_name}) { ## TODO: case thingy
-       $in_cdata = 1;
-     }
-     $parent = $parent->parent_node;
-   }
-   ## Step 2
-   my @node = @{$node->child_nodes};
-   C: while (@node) {
-     my $child = shift @node;
-     unless (ref $child) {
-       if ($child eq 'cdata-out') {
-         $in_cdata = 0;
-       } else {
-         $s .= $child; # end tag
-       }
-       next C;
-     }
-     my $nt = $child->node_type;
-     if ($nt == 1) { # Element
-       my $tag_name = $child->tag_name; ## TODO: manakai_tag_name
-       $s .= '<' . $tag_name;
-       ## NOTE: Non-HTML case:
-       ## <http://permalink.gmane.org/gmane.org.w3c.whatwg.discuss/11191>
-       my @attrs = @{$child->attributes}; # sort order MUST be stable
-       for my $attr (@attrs) { # order is implementation dependent
-         my $attr_name = $attr->name; ## TODO: manakai_name
-         $s .= ' ' . $attr_name . '="';
-         my $attr_value = $attr->value;
-         ## escape
-         $attr_value =~ s/&/&amp;/g;
-         $attr_value =~ s/</&lt;/g;
-         $attr_value =~ s/>/&gt;/g;
-         $attr_value =~ s/"/&quot;/g;
-         $s .= $attr_value . '"';
-       }
-       $s .= '>';
-       next C if {
-         area => 1, base => 1, basefont => 1, bgsound => 1,
-         br => 1, col => 1, embed => 1, frame => 1, hr => 1,
-         img => 1, input => 1, link => 1, meta => 1, param => 1,
-         spacer => 1, wbr => 1,
-       }->{$tag_name};
-       $s .= "\x0A" if $tag_name eq 'pre' or $tag_name eq 'textarea';
-       if (not $in_cdata and {
-         style => 1, script => 1, xmp => 1, iframe => 1,
-         noembed => 1, noframes => 1, noscript => 1,
-         plaintext => 1,
-       }->{$tag_name}) {
-         unshift @node, 'cdata-out';
-         $in_cdata = 1;
-       }
-       unshift @node, @{$child->child_nodes}, '</' . $tag_name . '>';
-     } elsif ($nt == 3 or $nt == 4) {
-       if ($in_cdata) {
-         $s .= $child->data;
-       } else {
-         my $value = $child->data;
-         $value =~ s/&/&amp;/g;
-         $value =~ s/</&lt;/g;
-         $value =~ s/>/&gt;/g;
-         $value =~ s/"/&quot;/g;
-         $s .= $value;
-       }
-     } elsif ($nt == 8) {
-       $s .= '<!--' . $child->data . '-->';
-     } elsif ($nt == 10) {
-       $s .= '<!DOCTYPE ' . $child->name . '>';
-     } elsif ($nt == 5) { # entrefs
-       push @node, @{$child->child_nodes};
-     } else {
-       $on_error->($child) if defined $on_error;
-     }
-     ## ISSUE: This code does not support PIs.
-   } # C
-   ## Step 3
-   return \$s;
- } # get_inner_html
 ;
  # $Date$

 Legend:



Removed from v.1.61
 


changed lines


 
Added in v.1.72
 Legend:



Removed from v.1.61
 


changed lines


 
Added in v.1.72
-Removed from v.1.61
+Added in v.1.72

admin@suikawiki.org	ViewVC Help
Powered by ViewVC 1.1.24