/[suikacvs]/markup/html/whatpm/Whatpm/HTML.pm.src

Diff of /markup/html/whatpm/Whatpm/HTML.pm.src

Parent Directory | Revision Log | View Patch Patch

-revision 1.158 by wakaba,
Sun Aug 31 12:11:42 2008 UTC
+revision 1.166 by wakaba,
Sat Sep 13 08:21:35 2008 UTC
 Line 354 
 sub parse_byte_string ($$$$;$) {
    return $self->parse_byte_stream ($charset_name, $input, @_[1..$#_]);
  } # parse_byte_string
- sub parse_byte_stream ($$$$;$) {
+ sub parse_byte_stream ($$$$;$$) {
+   # my ($self, $charset_name, $byte_stream, $doc, $onerror, $get_wrapper) = @_;
    my $self = ref $_[0] ? shift : shift->new;
    my $charset_name = shift;
    my $byte_stream = $_[0];
-Line 365 
 sub parse_byte_stream ($$$$;$) {
+Line 366 
 sub parse_byte_stream ($$$$;$) {
    };
    $self->{parse_error} = $onerror; # updated later by parse_char_string
+   my $get_wrapper = $_[3] || sub ($) {
+     return $_[0]; # $_[0] = byte stream handle, returned = arg to char handle
+   };
    ## HTML5 encoding sniffing algorithm
    require Message::Charset::Info;
    my $charset;
-Line 372 
 sub parse_byte_stream ($$$$;$) {
+Line 377 
 sub parse_byte_stream ($$$$;$) {
    my ($char_stream, $e_status);
    SNIFFING: {
+     ## NOTE: By setting |allow_fallback| option true when the
+     ## |get_decode_handle| method is invoked, we ignore what the HTML5
+     ## spec requires, i.e. unsupported encoding should be ignored.
+       ## TODO: We should not do this unless the parser is invoked
+       ## in the conformance checking mode, in which this behavior
+       ## would be useful.
      ## Step 1
      if (defined $charset_name) {
-       $charset = Message::Charset::Info->get_by_iana_name ($charset_name);
+       $charset = Message::Charset::Info->get_by_html_name ($charset_name);
+           ## TODO: Is this ok?  Transfer protocol's parameter should be
+           ## interpreted in its semantics?
        ## ISSUE: Unsupported encoding is not ignored according to the spec.
        ($char_stream, $e_status) = $charset->get_decode_handle
-Line 399 
 sub parse_byte_stream ($$$$;$) {
+Line 412 
 sub parse_byte_stream ($$$$;$) {
      ## Step 3
      if ($byte_buffer =~ /^\xFE\xFF/) {
-       $charset = Message::Charset::Info->get_by_iana_name ('utf-16be');
+       $charset = Message::Charset::Info->get_by_html_name ('utf-16be');
        ($char_stream, $e_status) = $charset->get_decode_handle
            ($byte_stream, allow_error_reporting => 1,
             allow_fallback => 1, byte_buffer => \$byte_buffer);
        $self->{confident} = 1;
        last SNIFFING;
      } elsif ($byte_buffer =~ /^\xFF\xFE/) {
-       $charset = Message::Charset::Info->get_by_iana_name ('utf-16le');
+       $charset = Message::Charset::Info->get_by_html_name ('utf-16le');
        ($char_stream, $e_status) = $charset->get_decode_handle
            ($byte_stream, allow_error_reporting => 1,
             allow_fallback => 1, byte_buffer => \$byte_buffer);
        $self->{confident} = 1;
        last SNIFFING;
      } elsif ($byte_buffer =~ /^\xEF\xBB\xBF/) {
-       $charset = Message::Charset::Info->get_by_iana_name ('utf-8');
+       $charset = Message::Charset::Info->get_by_html_name ('utf-8');
        ($char_stream, $e_status) = $charset->get_decode_handle
            ($byte_stream, allow_error_reporting => 1,
             allow_fallback => 1, byte_buffer => \$byte_buffer);
-Line 432 
 sub parse_byte_stream ($$$$;$) {
+Line 445 
 sub parse_byte_stream ($$$$;$) {
      $charset_name = Whatpm::Charset::UniversalCharDet->detect_byte_string
          ($byte_buffer);
      if (defined $charset_name) {
-       $charset = Message::Charset::Info->get_by_iana_name ($charset_name);
+       $charset = Message::Charset::Info->get_by_html_name ($charset_name);
        ## ISSUE: Unsupported encoding is not ignored according to the spec.
        require Whatpm::Charset::DecodeHandle;
-Line 455 
 sub parse_byte_stream ($$$$;$) {
+Line 468 
 sub parse_byte_stream ($$$$;$) {
      ## Step 7: default
      ## TODO: Make this configurable.
-     $charset = Message::Charset::Info->get_by_iana_name ('windows-1252');
+     $charset = Message::Charset::Info->get_by_html_name ('windows-1252');
          ## NOTE: We choose |windows-1252| here, since |utf-8| should be
          ## detectable in the step 6.
      require Whatpm::Charset::DecodeHandle;
-Line 475 
 sub parse_byte_stream ($$$$;$) {
+Line 488 
 sub parse_byte_stream ($$$$;$) {
      $self->{confident} = 0;
    } # SNIFFING
-   $self->{input_encoding} = $charset->get_iana_name;
    if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
+     $self->{input_encoding} = $charset->get_iana_name; ## TODO: Should we set actual charset decoder's encoding name?
      !!!parse-error (type => 'chardecode:fallback',
-                     text => $self->{input_encoding},
+                     #text => $self->{input_encoding},
                      level => $self->{level}->{uncertain},
                      line => 1, column => 1,
                      layer => 'encode');
    } elsif (not ($e_status &
                  Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) {
+     $self->{input_encoding} = $charset->get_iana_name;
      !!!parse-error (type => 'chardecode:no error',
                      text => $self->{input_encoding},
                      level => $self->{level}->{uncertain},
                      line => 1, column => 1,
                      layer => 'encode');
+   } else {
+     $self->{input_encoding} = $charset->get_iana_name;
    }
    $self->{change_encoding} = sub {
-Line 496 
 sub parse_byte_stream ($$$$;$) {
+Line 512 
 sub parse_byte_stream ($$$$;$) {
      $charset_name = shift;
      my $token = shift;
-     $charset = Message::Charset::Info->get_by_iana_name ($charset_name);
+     $charset = Message::Charset::Info->get_by_html_name ($charset_name);
      ($char_stream, $e_status) = $charset->get_decode_handle
          ($byte_stream, allow_error_reporting => 1, allow_fallback => 1,
           byte_buffer => \ $buffer->{buffer});
-Line 507 
 sub parse_byte_stream ($$$$;$) {
+Line 523 
 sub parse_byte_stream ($$$$;$) {
        ## Step 1
        if ($charset->{category} &
            Message::Charset::Info::CHARSET_CATEGORY_UTF16 ()) {
-         $charset = Message::Charset::Info->get_by_iana_name ('utf-8');
+         $charset = Message::Charset::Info->get_by_html_name ('utf-8');
          ($char_stream, $e_status) = $charset->get_decode_handle
              ($byte_stream,
               byte_buffer => \ $buffer->{buffer});
-Line 551 
 sub parse_byte_stream ($$$$;$) {
+Line 567 
 sub parse_byte_stream ($$$$;$) {
        ${$opt{octets}} = "\x{FFFD}"; # relacement character
      }
    };
-   $char_stream->onerror ($char_onerror);
+   my $wrapped_char_stream = $get_wrapper->($char_stream);
+   $wrapped_char_stream->onerror ($char_onerror);
    my @args = @_; shift @args; # $s
    my $return;
    try {
-     $return = $self->parse_char_stream ($char_stream, @args);
+     $return = $self->parse_char_stream ($wrapped_char_stream, @args);
    } catch Whatpm::HTML::RestartParser with {
      ## NOTE: Invoked after {change_encoding}.
-     $self->{input_encoding} = $charset->get_iana_name;
      if ($e_status & Message::Charset::Info::FALLBACK_ENCODING_IMPL ()) {
+       $self->{input_encoding} = $charset->get_iana_name; ## TODO: Should we set actual charset decoder's encoding name?
        !!!parse-error (type => 'chardecode:fallback',
-                       text => $self->{input_encoding},
                        level => $self->{level}->{uncertain},
+                       #text => $self->{input_encoding},
                        line => 1, column => 1,
                        layer => 'encode');
      } elsif (not ($e_status &
                    Message::Charset::Info::ERROR_REPORTING_ENCODING_IMPL())) {
+       $self->{input_encoding} = $charset->get_iana_name;
        !!!parse-error (type => 'chardecode:no error',
                        text => $self->{input_encoding},
                        level => $self->{level}->{uncertain},
                        line => 1, column => 1,
                        layer => 'encode');
+     } else {
+       $self->{input_encoding} = $charset->get_iana_name;
      }
      $self->{confident} = 1;
-     $char_stream->onerror ($char_onerror);
-     $return = $self->parse_char_stream ($char_stream, @args);
+     $wrapped_char_stream = $get_wrapper->($char_stream);
+     $wrapped_char_stream->onerror ($char_onerror);
+     $return = $self->parse_char_stream ($wrapped_char_stream, @args);
    };
    return $return;
  } # parse_byte_stream
-Line 591 
 sub parse_byte_stream ($$$$;$) {
+Line 615 
 sub parse_byte_stream ($$$$;$) {
  ## such as |parse_byte_string| in this module, must ensure that it does
  ## strip the BOM and never strip any ZWNBSP.
- sub parse_char_string ($$$;$) {
+ sub parse_char_string ($$$;$$) {
+   #my ($self, $s, $doc, $onerror, $get_wrapper) = @_;
    my $self = shift;
    require utf8;
    my $s = ref $_[0] ? $_[0] : \($_[0]);
    open my $input, '<' . (utf8::is_utf8 ($$s) ? ':utf8' : ''), $s;
+   if ($_[3]) {
+     $input = $_[3]->($input);
+   }
    return $self->parse_char_stream ($input, @_[1..$#_]);
  } # parse_char_string
- *parse_string = \&parse_char_string;
+ *parse_string = \&parse_char_string; ## NOTE: Alias for backward compatibility.
  sub parse_char_stream ($$$;$) {
    my $self = ref $_[0] ? shift : shift->new;
-Line 708 
 sub new ($) {
+Line 736 
 sub new ($) {
    my $class = shift;
    my $self = bless {
      level => {must => 'm',
+               should => 's',
                warn => 'w',
                info => 'i',
                uncertain => 'u'},
-Line 774 
 sub AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STAT
+Line 803 
 sub AFTER_DOCTYPE_SYSTEM_IDENTIFIER_STAT
  sub BOGUS_DOCTYPE_STATE () { 32 }
  sub AFTER_ATTRIBUTE_VALUE_QUOTED_STATE () { 33 }
  sub SELF_CLOSING_START_TAG_STATE () { 34 }
- sub CDATA_BLOCK_STATE () { 35 }
+ sub CDATA_SECTION_STATE () { 35 }
+ sub MD_HYPHEN_STATE () { 36 } # "markup declaration open state" in the spec
+ sub MD_DOCTYPE_STATE () { 37 } # "markup declaration open state" in the spec
+ sub MD_CDATA_STATE () { 38 } # "markup declaration open state" in the spec
+ sub CDATA_PCDATA_CLOSE_TAG_STATE () { 39 } # "close tag open state" in the spec
+ sub CDATA_SECTION_MSE1_STATE () { 40 } # "CDATA section state" in the spec
+ sub CDATA_SECTION_MSE2_STATE () { 41 } # "CDATA section state" in the spec
+ sub PUBLIC_STATE () { 42 } # "after DOCTYPE name state" in the spec
+ sub SYSTEM_STATE () { 43 } # "after DOCTYPE name state" in the spec
  sub DOCTYPE_TOKEN () { 1 }
  sub COMMENT_TOKEN () { 2 }
-Line 827 
 sub IN_COLUMN_GROUP_IM () { 0b10 }
+Line 864 
 sub IN_COLUMN_GROUP_IM () { 0b10 }
  sub _initialize_tokenizer ($) {
    my $self = shift;
    $self->{state} = DATA_STATE; # MUST
+   #$self->{state_keyword}; # initialized when used
    $self->{content_model} = PCDATA_CONTENT_MODEL; # be
-   undef $self->{current_token}; # start tag, end tag, comment, or DOCTYPE
+   undef $self->{current_token};
    undef $self->{current_attribute};
    undef $self->{last_emitted_start_tag_name};
    undef $self->{last_attribute_value_state};
-Line 1089 
 sub _get_next_token ($) {
+Line 1127 
 sub _get_next_token ($) {
          die "$0: $self->{content_model} in tag open";
        }
      } elsif ($self->{state} == CLOSE_TAG_OPEN_STATE) {
+       ## NOTE: The "close tag open state" in the spec is implemented as
+       ## |CLOSE_TAG_OPEN_STATE| and |CDATA_PCDATA_CLOSE_TAG_STATE|.
        my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1); # "<"of"</"
        if ($self->{content_model} & CM_LIMITED_MARKUP) { # RCDATA | CDATA
          if (defined $self->{last_emitted_start_tag_name}) {
+           $self->{state} = CDATA_PCDATA_CLOSE_TAG_STATE;
-           ## NOTE: <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>
+           $self->{state_keyword} = '';
-           my @next_char;
+           ## Reconsume.
-           TAGNAME: for (my $i = 0; $i < length $self->{last_emitted_start_tag_name}; $i++) {
+           redo A;
-             push @next_char, $self->{next_char};
-             my $c = ord substr ($self->{last_emitted_start_tag_name}, $i, 1);
-             my $C = 0x0061 <= $c && $c <= 0x007A ? $c - 0x0020 : $c;
-             if ($self->{next_char} == $c or $self->{next_char} == $C) {
-               !!!cp (24);
-               !!!next-input-character;
-               next TAGNAME;
-             } else {
-               !!!cp (25);
-               $self->{next_char} = shift @next_char; # reconsume
-               !!!back-next-input-character (@next_char);
-               $self->{state} = DATA_STATE;
-               !!!emit ({type => CHARACTER_TOKEN, data => '</',
-                         line => $l, column => $c,
-                        });
-               redo A;
-             }
-           }
-           push @next_char, $self->{next_char};
-           unless ($self->{next_char} == 0x0009 or # HT
-                   $self->{next_char} == 0x000A or # LF
-                   $self->{next_char} == 0x000B or # VT
-                   $self->{next_char} == 0x000C or # FF
-                   $self->{next_char} == 0x0020 or # SP
-                   $self->{next_char} == 0x003E or # >
-                   $self->{next_char} == 0x002F or # /
-                   $self->{next_char} == -1) {
-             !!!cp (26);
-             $self->{next_char} = shift @next_char; # reconsume
-             !!!back-next-input-character (@next_char);
-             $self->{state} = DATA_STATE;
-             !!!emit ({type => CHARACTER_TOKEN, data => '</',
-                       line => $l, column => $c,
-                      });
-             redo A;
-           } else {
-             !!!cp (27);
-             $self->{next_char} = shift @next_char;
-             !!!back-next-input-character (@next_char);
-             # and consume...
-           }
          } else {
            ## No start tag token has ever been emitted
+           ## NOTE: See <http://krijnhoetmer.nl/irc-logs/whatwg/20070626#l-564>.
            !!!cp (28);
-           # next-input-character is already done
            $self->{state} = DATA_STATE;
+           ## Reconsume.
            !!!emit ({type => CHARACTER_TOKEN, data => '</',
                      line => $l, column => $c,
                     });
            redo A;
          }
        }
        if (0x0041 <= $self->{next_char} and
            $self->{next_char} <= 0x005A) { # A..Z
          !!!cp (29);
-Line 1198 
 sub _get_next_token ($) {
+Line 1196 
 sub _get_next_token ($) {
                                    line => $self->{line_prev}, # "<" of "</"
                                    column => $self->{column_prev} - 1,
                                   };
-         ## $self->{next_char} is intentionally left as is
+         ## NOTE: $self->{next_char} is intentionally left as is.
-         redo A;
+         ## Although the "anything else" case of the spec not explicitly
+         ## states that the next input character is to be reconsumed,
+         ## it will be included to the |data| of the comment token
+         ## generated from the bogus end tag, as defined in the
+         ## "bogus comment state" entry.
+         redo A;
+       }
+     } elsif ($self->{state} == CDATA_PCDATA_CLOSE_TAG_STATE) {
+       my $ch = substr $self->{last_emitted_start_tag_name}, length $self->{state_keyword}, 1;
+       if (length $ch) {
+         my $CH = $ch;
+         $ch =~ tr/a-z/A-Z/;
+         my $nch = chr $self->{next_char};
+         if ($nch eq $ch or $nch eq $CH) {
+           !!!cp (24);
+           ## Stay in the state.
+           $self->{state_keyword} .= $nch;
+           !!!next-input-character;
+           redo A;
+         } else {
+           !!!cp (25);
+           $self->{state} = DATA_STATE;
+           ## Reconsume.
+           !!!emit ({type => CHARACTER_TOKEN,
+                     data => '</' . $self->{state_keyword},
+                     line => $self->{line_prev},
+                     column => $self->{column_prev} - 1 - length $self->{state_keyword},
+                    });
+           redo A;
+         }
+       } else { # after "<{tag-name}"
+         unless ({
+x0009 => 1, # HT
+x000A => 1, # LF
+x000B => 1, # VT
+x000C => 1, # FF
+x0020 => 1, # SP
+x003E => 1, # >
+x002F => 1, # /
+                  -1 => 1, # EOF
+                 }->{$self->{next_char}}) {
+           !!!cp (26);
+           ## Reconsume.
+           $self->{state} = DATA_STATE;
+           !!!emit ({type => CHARACTER_TOKEN,
+                     data => '</' . $self->{state_keyword},
+                     line => $self->{line_prev},
+                     column => $self->{column_prev} - 1 - length $self->{state_keyword},
+                    });
+           redo A;
+         } else {
+           !!!cp (27);
+           $self->{current_token}
+               = {type => END_TAG_TOKEN,
+                  tag_name => $self->{last_emitted_start_tag_name},
+                  line => $self->{line_prev},
+                  column => $self->{column_prev} - 1 - length $self->{state_keyword}};
+           $self->{state} = TAG_NAME_STATE;
+           ## Reconsume.
+           redo A;
+         }
        }
      } elsif ($self->{state} == TAG_NAME_STATE) {
        if ($self->{next_char} == 0x0009 or # HT
-Line 1972 
 sub _get_next_token ($) {
+Line 2030 
 sub _get_next_token ($) {
        die "$0: _get_next_token: unexpected case [BC]";
      } elsif ($self->{state} == MARKUP_DECLARATION_OPEN_STATE) {
        ## (only happen if PCDATA state)
-       my ($l, $c) = ($self->{line_prev}, $self->{column_prev} - 1);
-       my @next_char;
-       push @next_char, $self->{next_char};
        if ($self->{next_char} == 0x002D) { # -
+         !!!cp (133);
+         $self->{state} = MD_HYPHEN_STATE;
          !!!next-input-character;
-         push @next_char, $self->{next_char};
+         redo A;
-         if ($self->{next_char} == 0x002D) { # -
-           !!!cp (127);
-           $self->{current_token} = {type => COMMENT_TOKEN, data => '',
-                                     line => $l, column => $c,
-                                    };
-           $self->{state} = COMMENT_START_STATE;
-           !!!next-input-character;
-           redo A;
-         } else {
-           !!!cp (128);
-         }
        } elsif ($self->{next_char} == 0x0044 or # D
                 $self->{next_char} == 0x0064) { # d
+         ## ASCII case-insensitive.
+         !!!cp (130);
+         $self->{state} = MD_DOCTYPE_STATE;
+         $self->{state_keyword} = chr $self->{next_char};
          !!!next-input-character;
-         push @next_char, $self->{next_char};
+         redo A;
-         if ($self->{next_char} == 0x004F or # O
-             $self->{next_char} == 0x006F) { # o
-           !!!next-input-character;
-           push @next_char, $self->{next_char};
-           if ($self->{next_char} == 0x0043 or # C
-               $self->{next_char} == 0x0063) { # c
-             !!!next-input-character;
-             push @next_char, $self->{next_char};
-             if ($self->{next_char} == 0x0054 or # T
-                 $self->{next_char} == 0x0074) { # t
-               !!!next-input-character;
-               push @next_char, $self->{next_char};
-               if ($self->{next_char} == 0x0059 or # Y
-                   $self->{next_char} == 0x0079) { # y
-                 !!!next-input-character;
-                 push @next_char, $self->{next_char};
-                 if ($self->{next_char} == 0x0050 or # P
-                     $self->{next_char} == 0x0070) { # p
-                   !!!next-input-character;
-                   push @next_char, $self->{next_char};
-                   if ($self->{next_char} == 0x0045 or # E
-                       $self->{next_char} == 0x0065) { # e
-                     !!!cp (129);
-                     ## TODO: What a stupid code this is!
-                     $self->{state} = DOCTYPE_STATE;
-                     $self->{current_token} = {type => DOCTYPE_TOKEN,
-                                               quirks => 1,
-                                               line => $l, column => $c,
-                                              };
-                     !!!next-input-character;
-                     redo A;
-                   } else {
-                     !!!cp (130);
-                   }
-                 } else {
-                   !!!cp (131);
-                 }
-               } else {
-                 !!!cp (132);
-               }
-             } else {
-               !!!cp (133);
-             }
-           } else {
-             !!!cp (134);
-           }
-         } else {
-           !!!cp (135);
-         }
        } elsif ($self->{insertion_mode} & IN_FOREIGN_CONTENT_IM and
                 $self->{open_elements}->[-1]->[1] & FOREIGN_EL and
                 $self->{next_char} == 0x005B) { # [
+         !!!cp (135.4);
+         $self->{state} = MD_CDATA_STATE;
+         $self->{state_keyword} = '[';
          !!!next-input-character;
-         push @next_char, $self->{next_char};
+         redo A;
-         if ($self->{next_char} == 0x0043) { # C
-           !!!next-input-character;
-           push @next_char, $self->{next_char};
-           if ($self->{next_char} == 0x0044) { # D
-             !!!next-input-character;
-             push @next_char, $self->{next_char};
-             if ($self->{next_char} == 0x0041) { # A
-               !!!next-input-character;
-               push @next_char, $self->{next_char};
-               if ($self->{next_char} == 0x0054) { # T
-                 !!!next-input-character;
-                 push @next_char, $self->{next_char};
-                 if ($self->{next_char} == 0x0041) { # A
-                   !!!next-input-character;
-                   push @next_char, $self->{next_char};
-                   if ($self->{next_char} == 0x005B) { # [
-                     !!!cp (135.1);
-                     $self->{state} = CDATA_BLOCK_STATE;
-                     !!!next-input-character;
-                     redo A;
-                   } else {
-                     !!!cp (135.2);
-                   }
-                 } else {
-                   !!!cp (135.3);
-                 }
-               } else {
-                 !!!cp (135.4);
-               }
-             } else {
-               !!!cp (135.5);
-             }
-           } else {
-             !!!cp (135.6);
-           }
-         } else {
-           !!!cp (135.7);
-         }
        } else {
          !!!cp (136);
        }
-       !!!parse-error (type => 'bogus comment');
+       !!!parse-error (type => 'bogus comment',
-       $self->{next_char} = shift @next_char;
+                       line => $self->{line_prev},
-       !!!back-next-input-character (@next_char);
+                       column => $self->{column_prev} - 1);
+       ## Reconsume.
        $self->{state} = BOGUS_COMMENT_STATE;
        $self->{current_token} = {type => COMMENT_TOKEN, data => '',
-                                 line => $l, column => $c,
+                                 line => $self->{line_prev},
+                                 column => $self->{column_prev} - 1,
                                 };
        redo A;
+     } elsif ($self->{state} == MD_HYPHEN_STATE) {
-       ## ISSUE: typos in spec: chacacters, is is a parse error
+       if ($self->{next_char} == 0x002D) { # -
-       ## ISSUE: spec is somewhat unclear on "is the first character that will be in the comment"; what is "that will be in the comment" is what the algorithm defines, isn't it?
+         !!!cp (127);
+         $self->{current_token} = {type => COMMENT_TOKEN, data => '',
+                                   line => $self->{line_prev},
+                                   column => $self->{column_prev} - 2,
+                                  };
+         $self->{state} = COMMENT_START_STATE;
+         !!!next-input-character;
+         redo A;
+       } else {
+         !!!cp (128);
+         !!!parse-error (type => 'bogus comment',
+                         line => $self->{line_prev},
+                         column => $self->{column_prev} - 2);
+         $self->{state} = BOGUS_COMMENT_STATE;
+         ## Reconsume.
+         $self->{current_token} = {type => COMMENT_TOKEN,
+                                   data => '-',
+                                   line => $self->{line_prev},
+                                   column => $self->{column_prev} - 2,
+                                  };
+         redo A;
+       }
+     } elsif ($self->{state} == MD_DOCTYPE_STATE) {
+       ## ASCII case-insensitive.
+       if ($self->{next_char} == [
+             undef,
+x004F, # O
+x0043, # C
+x0054, # T
+x0059, # Y
+x0050, # P
+           ]->[length $self->{state_keyword}] or
+           $self->{next_char} == [
+             undef,
+x006F, # o
+x0063, # c
+x0074, # t
+x0079, # y
+x0070, # p
+           ]->[length $self->{state_keyword}]) {
+         !!!cp (131);
+         ## Stay in the state.
+         $self->{state_keyword} .= chr $self->{next_char};
+         !!!next-input-character;
+         redo A;
+       } elsif ((length $self->{state_keyword}) == 6 and
+                ($self->{next_char} == 0x0045 or # E
+                 $self->{next_char} == 0x0065)) { # e
+         !!!cp (129);
+         $self->{state} = DOCTYPE_STATE;
+         $self->{current_token} = {type => DOCTYPE_TOKEN,
+                                   quirks => 1,
+                                   line => $self->{line_prev},
+                                   column => $self->{column_prev} - 7,
+                                  };
+         !!!next-input-character;
+         redo A;
+       } else {
+         !!!cp (132);
+         !!!parse-error (type => 'bogus comment',
+                         line => $self->{line_prev},
+                         column => $self->{column_prev} - 1 - length $self->{state_keyword});
+         $self->{state} = BOGUS_COMMENT_STATE;
+         ## Reconsume.
+         $self->{current_token} = {type => COMMENT_TOKEN,
+                                   data => $self->{state_keyword},
+                                   line => $self->{line_prev},
+                                   column => $self->{column_prev} - 1 - length $self->{state_keyword},
+                                  };
+         redo A;
+       }
+     } elsif ($self->{state} == MD_CDATA_STATE) {
+       if ($self->{next_char} == {
+             '[' => 0x0043, # C
+             '[C' => 0x0044, # D
+             '[CD' => 0x0041, # A
+             '[CDA' => 0x0054, # T
+             '[CDAT' => 0x0041, # A
+           }->{$self->{state_keyword}}) {
+         !!!cp (135.1);
+         ## Stay in the state.
+         $self->{state_keyword} .= chr $self->{next_char};
+         !!!next-input-character;
+         redo A;
+       } elsif ($self->{state_keyword} eq '[CDATA' and
+                $self->{next_char} == 0x005B) { # [
+         !!!cp (135.2);
+         $self->{current_token} = {type => CHARACTER_TOKEN,
+                                   data => '',
+                                   line => $self->{line_prev},
+                                   column => $self->{column_prev} - 7};
+         $self->{state} = CDATA_SECTION_STATE;
+         !!!next-input-character;
+         redo A;
+       } else {
+         !!!cp (135.3);
+         !!!parse-error (type => 'bogus comment',
+                         line => $self->{line_prev},
+                         column => $self->{column_prev} - 1 - length $self->{state_keyword});
+         $self->{state} = BOGUS_COMMENT_STATE;
+         ## Reconsume.
+         $self->{current_token} = {type => COMMENT_TOKEN,
+                                   data => $self->{state_keyword},
+                                   line => $self->{line_prev},
+                                   column => $self->{column_prev} - 1 - length $self->{state_keyword},
+                                  };
+         redo A;
+       }
      } elsif ($self->{state} == COMMENT_START_STATE) {
        if ($self->{next_char} == 0x002D) { # -
          !!!cp (137);
-Line 2369 
 sub _get_next_token ($) {
+Line 2442 
 sub _get_next_token ($) {
          redo A;
        } elsif ($self->{next_char} == 0x0050 or # P
                 $self->{next_char} == 0x0070) { # p
+         $self->{state} = PUBLIC_STATE;
+         $self->{state_keyword} = chr $self->{next_char};
          !!!next-input-character;
-         if ($self->{next_char} == 0x0055 or # U
+         redo A;
-             $self->{next_char} == 0x0075) { # u
-           !!!next-input-character;
-           if ($self->{next_char} == 0x0042 or # B
-               $self->{next_char} == 0x0062) { # b
-             !!!next-input-character;
-             if ($self->{next_char} == 0x004C or # L
-                 $self->{next_char} == 0x006C) { # l
-               !!!next-input-character;
-               if ($self->{next_char} == 0x0049 or # I
-                   $self->{next_char} == 0x0069) { # i
-                 !!!next-input-character;
-                 if ($self->{next_char} == 0x0043 or # C
-                     $self->{next_char} == 0x0063) { # c
-                   !!!cp (168);
-                   $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
-                   !!!next-input-character;
-                   redo A;
-                 } else {
-                   !!!cp (169);
-                 }
-               } else {
-                 !!!cp (170);
-               }
-             } else {
-               !!!cp (171);
-             }
-           } else {
-             !!!cp (172);
-           }
-         } else {
-           !!!cp (173);
-         }
-         #
        } elsif ($self->{next_char} == 0x0053 or # S
                 $self->{next_char} == 0x0073) { # s
+         $self->{state} = SYSTEM_STATE;
+         $self->{state_keyword} = chr $self->{next_char};
          !!!next-input-character;
-         if ($self->{next_char} == 0x0059 or # Y
+         redo A;
-             $self->{next_char} == 0x0079) { # y
-           !!!next-input-character;
-           if ($self->{next_char} == 0x0053 or # S
-               $self->{next_char} == 0x0073) { # s
-             !!!next-input-character;
-             if ($self->{next_char} == 0x0054 or # T
-                 $self->{next_char} == 0x0074) { # t
-               !!!next-input-character;
-               if ($self->{next_char} == 0x0045 or # E
-                   $self->{next_char} == 0x0065) { # e
-                 !!!next-input-character;
-                 if ($self->{next_char} == 0x004D or # M
-                     $self->{next_char} == 0x006D) { # m
-                   !!!cp (174);
-                   $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
-                   !!!next-input-character;
-                   redo A;
-                 } else {
-                   !!!cp (175);
-                 }
-               } else {
-                 !!!cp (176);
-               }
-             } else {
-               !!!cp (177);
-             }
-           } else {
-             !!!cp (178);
-           }
-         } else {
-           !!!cp (179);
-         }
-         #
        } else {
          !!!cp (180);
+         !!!parse-error (type => 'string after DOCTYPE name');
+         $self->{current_token}->{quirks} = 1;
+         $self->{state} = BOGUS_DOCTYPE_STATE;
          !!!next-input-character;
-         #
+         redo A;
        }
+     } elsif ($self->{state} == PUBLIC_STATE) {
+       ## ASCII case-insensitive
+       if ($self->{next_char} == [
+             undef,
+x0055, # U
+x0042, # B
+x004C, # L
+x0049, # I
+           ]->[length $self->{state_keyword}] or
+           $self->{next_char} == [
+             undef,
+x0075, # u
+x0062, # b
+x006C, # l
+x0069, # i
+           ]->[length $self->{state_keyword}]) {
+         !!!cp (175);
+         ## Stay in the state.
+         $self->{state_keyword} .= chr $self->{next_char};
+         !!!next-input-character;
+         redo A;
+       } elsif ((length $self->{state_keyword}) == 5 and
+                ($self->{next_char} == 0x0043 or # C
+                 $self->{next_char} == 0x0063)) { # c
+         !!!cp (168);
+         $self->{state} = BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE;
+         !!!next-input-character;
+         redo A;
+       } else {
+         !!!cp (169);
+         !!!parse-error (type => 'string after DOCTYPE name',
+                         line => $self->{line_prev},
+                         column => $self->{column_prev} + 1 - length $self->{state_keyword});
+         $self->{current_token}->{quirks} = 1;
-       !!!parse-error (type => 'string after DOCTYPE name');
+         $self->{state} = BOGUS_DOCTYPE_STATE;
-       $self->{current_token}->{quirks} = 1;
+         ## Reconsume.
+         redo A;
+       }
+     } elsif ($self->{state} == SYSTEM_STATE) {
+       ## ASCII case-insensitive
+       if ($self->{next_char} == [
+             undef,
+x0059, # Y
+x0053, # S
+x0054, # T
+x0045, # E
+           ]->[length $self->{state_keyword}] or
+           $self->{next_char} == [
+             undef,
+x0079, # y
+x0073, # s
+x0074, # t
+x0065, # e
+           ]->[length $self->{state_keyword}]) {
+         !!!cp (170);
+         ## Stay in the state.
+         $self->{state_keyword} .= chr $self->{next_char};
+         !!!next-input-character;
+         redo A;
+       } elsif ((length $self->{state_keyword}) == 5 and
+                ($self->{next_char} == 0x004D or # M
+                 $self->{next_char} == 0x006D)) { # m
+         !!!cp (171);
+         $self->{state} = BEFORE_DOCTYPE_SYSTEM_IDENTIFIER_STATE;
+         !!!next-input-character;
+         redo A;
+       } else {
+         !!!cp (172);
+         !!!parse-error (type => 'string after DOCTYPE name',
+                         line => $self->{line_prev},
+                         column => $self->{column_prev} + 1 - length $self->{state_keyword});
+         $self->{current_token}->{quirks} = 1;
-       $self->{state} = BOGUS_DOCTYPE_STATE;
+         $self->{state} = BOGUS_DOCTYPE_STATE;
-       # next-input-character is already done
+         ## Reconsume.
-       redo A;
+         redo A;
+       }
      } elsif ($self->{state} == BEFORE_DOCTYPE_PUBLIC_IDENTIFIER_STATE) {
        if ({
 x0009 => 1, 0x000A => 1, 0x000B => 1, 0x000C => 1, 0x0020 => 1,
-Line 2811 
 sub _get_next_token ($) {
+Line 2895 
 sub _get_next_token ($) {
          !!!next-input-character;
          redo A;
        }
-     } elsif ($self->{state} == CDATA_BLOCK_STATE) {
+     } elsif ($self->{state} == CDATA_SECTION_STATE) {
-       my $s = '';
+       ## NOTE: "CDATA section state" in the state is jointly implemented
+       ## by three states, |CDATA_SECTION_STATE|, |CDATA_SECTION_MSE1_STATE|,
+       ## and |CDATA_SECTION_MSE2_STATE|.
-       my ($l, $c) = ($self->{line}, $self->{column});
+       if ($self->{next_char} == 0x005D) { # ]
+         !!!cp (221.1);
+         $self->{state} = CDATA_SECTION_MSE1_STATE;
+         !!!next-input-character;
+         redo A;
+       } elsif ($self->{next_char} == -1) {
+         $self->{state} = DATA_STATE;
+         !!!next-input-character;
+         if (length $self->{current_token}->{data}) { # character
+           !!!cp (221.2);
+           !!!emit ($self->{current_token}); # character
+         } else {
+           !!!cp (221.3);
+           ## No token to emit. $self->{current_token} is discarded.
+         }
+         redo A;
+       } else {
+         !!!cp (221.4);
+         $self->{current_token}->{data} .= chr $self->{next_char};
+         ## Stay in the state.
+         !!!next-input-character;
+         redo A;
+       }
-       CS: while ($self->{next_char} != -1) {
+       ## ISSUE: "text tokens" in spec.
-         if ($self->{next_char} == 0x005D) { # ]
+     } elsif ($self->{state} == CDATA_SECTION_MSE1_STATE) {
-           !!!next-input-character;
+       if ($self->{next_char} == 0x005D) { # ]
-           if ($self->{next_char} == 0x005D) { # ]
+         !!!cp (221.5);
-             !!!next-input-character;
+         $self->{state} = CDATA_SECTION_MSE2_STATE;
-             MDC: {
+         !!!next-input-character;
-               if ($self->{next_char} == 0x003E) { # >
+         redo A;
-                 !!!cp (221.1);
+       } else {
-                 !!!next-input-character;
+         !!!cp (221.6);
-                 last CS;
+         $self->{current_token}->{data} .= ']';
-               } elsif ($self->{next_char} == 0x005D) { # ]
+         $self->{state} = CDATA_SECTION_STATE;
-                 !!!cp (221.2);
+         ## Reconsume.
-                 $s .= ']';
+         redo A;
-                 !!!next-input-character;
+       }
-                 redo MDC;
+     } elsif ($self->{state} == CDATA_SECTION_MSE2_STATE) {
-               } else {
+       if ($self->{next_char} == 0x003E) { # >
-                 !!!cp (221.3);
+         $self->{state} = DATA_STATE;
-                 $s .= ']]';
+         !!!next-input-character;
-                 #
+         if (length $self->{current_token}->{data}) { # character
-               }
+           !!!cp (221.7);
-             } # MDC
+           !!!emit ($self->{current_token}); # character
-           } else {
-             !!!cp (221.4);
-             $s .= ']';
-             #
-           }
          } else {
-           !!!cp (221.5);
+           !!!cp (221.8);
-           #
+           ## No token to emit. $self->{current_token} is discarded.
          }
-         $s .= chr $self->{next_char};
+         redo A;
+       } elsif ($self->{next_char} == 0x005D) { # ]
+         !!!cp (221.9); # character
+         $self->{current_token}->{data} .= ']'; ## Add first "]" of "]]]".
+         ## Stay in the state.
          !!!next-input-character;
-       } # CS
+         redo A;
-       $self->{state} = DATA_STATE;
-       ## next-input-character done or EOF, which is reconsumed.
-       if (length $s) {
-         !!!cp (221.6);
-         !!!emit ({type => CHARACTER_TOKEN, data => $s,
-                   line => $l, column => $c});
        } else {
-         !!!cp (221.7);
+         !!!cp (221.11);
+         $self->{current_token}->{data} .= ']]'; # character
+         $self->{state} = CDATA_SECTION_STATE;
+         ## Reconsume.
+         redo A;
        }
-       redo A;
-       ## ISSUE: "text tokens" in spec.
-       ## TODO: Streaming support
      } else {
        die "$0: $self->{state}: Unknown state";
      }
-Line 3154 
 sub _tree_construction_initial ($) {
+Line 3252 
 sub _tree_construction_initial ($) {
        ## language.
        my $doctype_name = $token->{name};
        $doctype_name = '' unless defined $doctype_name;
-       $doctype_name =~ tr/a-z/A-Z/;
+       $doctype_name =~ tr/a-z/A-Z/; # ASCII case-insensitive
        if (not defined $token->{name} or # <!DOCTYPE>
-           defined $token->{public_identifier} or
            defined $token->{system_identifier}) {
          !!!cp ('t1');
          !!!parse-error (type => 'not HTML5', token => $token);
        } elsif ($doctype_name ne 'HTML') {
          !!!cp ('t2');
-         ## ISSUE: ASCII case-insensitive? (in fact it does not matter)
          !!!parse-error (type => 'not HTML5', token => $token);
+       } elsif (defined $token->{public_identifier}) {
+         if ($token->{public_identifier} eq 'XSLT-compat') {
+           !!!cp ('t1.2');
+           !!!parse-error (type => 'XSLT-compat', token => $token,
+                           level => $self->{level}->{should});
+         } else {
+           !!!parse-error (type => 'not HTML5', token => $token);
+         }
        } else {
          !!!cp ('t3');
+         #
        }
        my $doctype = $self->{document}->create_document_type_definition
-Line 7436 
 sub _tree_construction_main ($) {
+Line 7541 
 sub _tree_construction_main ($) {
    ## TODO: script stuffs
  } # _tree_construct_main
- sub set_inner_html ($$$) {
+ sub set_inner_html ($$$;$) {
    my $class = shift;
    my $node = shift;
    my $s = \$_[0];
    my $onerror = $_[1];
+   my $get_wrapper = $_[2] || sub ($) { return $_[0] };
    ## ISSUE: Should {confident} be true?
-Line 7459 
 sub set_inner_html ($$$) {
+Line 7565 
 sub set_inner_html ($$$) {
      }
      ## Step 3, 4, 5 # MUST
-     $class->parse_string ($$s => $node, $onerror);
+     $class->parse_char_string ($$s => $node, $onerror, $get_wrapper);
    } elsif ($nt == 1) {
      ## TODO: If non-html element
      ## NOTE: Most of this code is copied from |parse_string|
+ ## TODO: Support for $get_wrapper
      ## Step 1 # MUST
      my $this_doc = $node->owner_document;
      my $doc = $this_doc->implementation->create_document;

 Legend:



Removed from v.1.158
 


changed lines


 
Added in v.1.166
 Legend:



Removed from v.1.158
 


changed lines


 
Added in v.1.166
-Removed from v.1.158
+Added in v.1.166

admin@suikawiki.org	ViewVC Help
Powered by ViewVC 1.1.24