Added: 
nutch/branches/nutchbase/src/plugin/parse-html/src/java/org/apache/nutch/parse/html/HtmlParser.java
URL: 
http://svn.apache.org/viewvc/nutch/branches/nutchbase/src/plugin/parse-html/src/java/org/apache/nutch/parse/html/HtmlParser.java?rev=982184&view=auto
==============================================================================
--- 
nutch/branches/nutchbase/src/plugin/parse-html/src/java/org/apache/nutch/parse/html/HtmlParser.java
 (added)
+++ 
nutch/branches/nutchbase/src/plugin/parse-html/src/java/org/apache/nutch/parse/html/HtmlParser.java
 Wed Aug  4 10:01:08 2010
@@ -0,0 +1,330 @@
+/**
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements.  See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License.  You may obtain a copy of the License at
+ *
+ *     http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.nutch.parse.html;
+
+import java.io.ByteArrayInputStream;
+import java.io.DataInputStream;
+import java.io.File;
+import java.io.FileInputStream;
+import java.io.IOException;
+import java.io.UnsupportedEncodingException;
+import java.net.MalformedURLException;
+import java.net.URL;
+import java.nio.ByteBuffer;
+import java.nio.charset.Charset;
+import java.util.ArrayList;
+import java.util.Arrays;
+import java.util.Collection;
+import java.util.HashSet;
+import java.util.regex.Matcher;
+import java.util.regex.Pattern;
+
+import org.apache.avro.util.Utf8;
+import org.apache.commons.logging.Log;
+import org.apache.commons.logging.LogFactory;
+import org.apache.hadoop.conf.Configuration;
+import org.apache.html.dom.HTMLDocumentImpl;
+import org.apache.nutch.metadata.Metadata;
+import org.apache.nutch.metadata.Nutch;
+import org.apache.nutch.parse.HTMLMetaTags;
+import org.apache.nutch.parse.HtmlParseFilters;
+import org.apache.nutch.parse.Outlink;
+import org.apache.nutch.parse.Parse;
+import org.apache.nutch.parse.ParseStatusCodes;
+import org.apache.nutch.parse.ParseStatusUtils;
+import org.apache.nutch.parse.Parser;
+import org.apache.nutch.storage.ParseStatus;
+import org.apache.nutch.storage.WebPage;
+import org.apache.nutch.util.Bytes;
+import org.apache.nutch.util.EncodingDetector;
+import org.apache.nutch.util.LogUtil;
+import org.apache.nutch.util.NutchConfiguration;
+import org.apache.nutch.util.TableUtil;
+import org.cyberneko.html.parsers.DOMFragmentParser;
+import org.w3c.dom.DOMException;
+import org.w3c.dom.DocumentFragment;
+import org.xml.sax.InputSource;
+import org.xml.sax.SAXException;
+
+public class HtmlParser implements Parser {
+  public static final Log LOG = 
LogFactory.getLog("org.apache.nutch.parse.html");
+
+  // I used 1000 bytes at first, but  found that some documents have
+  // meta tag well past the first 1000 bytes.
+  // (e.g. http://cn.promo.yahoo.com/customcare/music.html)
+  private static final int CHUNK_SIZE = 2000;
+  private static Pattern metaPattern =
+    Pattern.compile("<meta\\s+([^>]*http-equiv=\"?content-type\"?[^>]*)>",
+        Pattern.CASE_INSENSITIVE);
+  private static Pattern charsetPattern =
+    Pattern.compile("charset=\\s*([a-z][_\\-0-9a-z]*)",
+        Pattern.CASE_INSENSITIVE);
+
+  private static Collection<WebPage.Field> FIELDS = new 
HashSet<WebPage.Field>();
+
+  static {
+    FIELDS.add(WebPage.Field.BASE_URL);
+  }
+
+  private String parserImpl;
+
+  /**
+   * Given a <code>byte[]</code> representing an html file of an
+   * <em>unknown</em> encoding,  read out 'charset' parameter in the meta tag
+   * from the first <code>CHUNK_SIZE</code> bytes.
+   * If there's no meta tag for Content-Type or no charset is specified,
+   * <code>null</code> is returned.  <br />
+   * FIXME: non-byte oriented character encodings (UTF-16, UTF-32)
+   * can't be handled with this.
+   * We need to do something similar to what's done by mozilla
+   * 
(http://lxr.mozilla.org/seamonkey/source/parser/htmlparser/src/nsParser.cpp#1993).
+   * See also http://www.w3.org/TR/REC-xml/#sec-guessing
+   * <br />
+   *
+   * @param content <code>byte[]</code> representation of an html file
+   */
+
+  private static String sniffCharacterEncoding(byte[] content) {
+    int length = content.length < CHUNK_SIZE ?
+        content.length : CHUNK_SIZE;
+
+    // We don't care about non-ASCII parts so that it's sufficient
+    // to just inflate each byte to a 16-bit value by padding.
+    // For instance, the sequence {0x41, 0x82, 0xb7} will be turned into
+    // {U+0041, U+0082, U+00B7}.
+    String str = "";
+    try {
+      str = new String(content, 0, length,
+          Charset.forName("ASCII").toString());
+    } catch (UnsupportedEncodingException e) {
+      // code should never come here, but just in case...
+      return null;
+    }
+
+    Matcher metaMatcher = metaPattern.matcher(str);
+    String encoding = null;
+    if (metaMatcher.find()) {
+      Matcher charsetMatcher = charsetPattern.matcher(metaMatcher.group(1));
+      if (charsetMatcher.find())
+        encoding = new String(charsetMatcher.group(1));
+    }
+
+    return encoding;
+  }
+
+  private String defaultCharEncoding;
+
+  private Configuration conf;
+
+  private DOMContentUtils utils;
+
+  private HtmlParseFilters htmlParseFilters;
+
+  private String cachingPolicy;
+
+  public Parse getParse(String url, WebPage page) {
+    HTMLMetaTags metaTags = new HTMLMetaTags();
+
+    String baseUrl = TableUtil.toString(page.getBaseUrl());
+    URL base;
+    try {
+      base = new URL(baseUrl);
+    } catch (MalformedURLException e) {
+      return ParseStatusUtils.getEmptyParse(e, getConf());
+    }
+
+    String text = "";
+    String title = "";
+    Outlink[] outlinks = new Outlink[0];
+    Metadata metadata = new Metadata();
+
+    // parse the content
+    DocumentFragment root;
+    try {
+      byte[] contentInOctets = page.getContent().array();
+      InputSource input = new InputSource(new 
ByteArrayInputStream(contentInOctets));
+
+      EncodingDetector detector = new EncodingDetector(conf);
+      detector.autoDetectClues(page, true);
+      detector.addClue(sniffCharacterEncoding(contentInOctets), "sniffed");
+      String encoding = detector.guessEncoding(page, defaultCharEncoding);
+
+      metadata.set(Metadata.ORIGINAL_CHAR_ENCODING, encoding);
+      metadata.set(Metadata.CHAR_ENCODING_FOR_CONVERSION, encoding);
+
+      input.setEncoding(encoding);
+      if (LOG.isTraceEnabled()) { LOG.trace("Parsing..."); }
+      root = parse(input);
+    } catch (IOException e) {
+      return ParseStatusUtils.getEmptyParse(e, getConf());
+    } catch (DOMException e) {
+      return ParseStatusUtils.getEmptyParse(e, getConf());
+    } catch (SAXException e) {
+      return ParseStatusUtils.getEmptyParse(e, getConf());
+    } catch (Exception e) {
+      e.printStackTrace(LogUtil.getWarnStream(LOG));
+      return ParseStatusUtils.getEmptyParse(e, getConf());
+    }
+
+    // get meta directives
+    HTMLMetaProcessor.getMetaTags(metaTags, root, base);
+    if (LOG.isTraceEnabled()) {
+      LOG.trace("Meta tags for " + base + ": " + metaTags.toString());
+    }
+    // check meta directives
+    if (!metaTags.getNoIndex()) {               // okay to index
+      StringBuilder sb = new StringBuilder();
+      if (LOG.isTraceEnabled()) { LOG.trace("Getting text..."); }
+      utils.getText(sb, root);          // extract text
+      text = sb.toString();
+      sb.setLength(0);
+      if (LOG.isTraceEnabled()) { LOG.trace("Getting title..."); }
+      utils.getTitle(sb, root);         // extract title
+      title = sb.toString().trim();
+    }
+
+    if (!metaTags.getNoFollow()) {              // okay to follow links
+      ArrayList<Outlink> l = new ArrayList<Outlink>();   // extract outlinks
+      URL baseTag = utils.getBase(root);
+      if (LOG.isTraceEnabled()) { LOG.trace("Getting links..."); }
+      utils.getOutlinks(baseTag!=null?baseTag:base, l, root);
+      outlinks = l.toArray(new Outlink[l.size()]);
+      if (LOG.isTraceEnabled()) {
+        LOG.trace("found "+outlinks.length+" outlinks in "+ url);
+      }
+    }
+
+    ParseStatus status = new ParseStatus();
+    status.setMajorCode(ParseStatusCodes.SUCCESS);
+    if (metaTags.getRefresh()) {
+      status.setMinorCode(ParseStatusCodes.SUCCESS_REDIRECT);
+      status.addToArgs(new Utf8(metaTags.getRefreshHref().toString()));
+      status.addToArgs(new Utf8(Integer.toString(metaTags.getRefreshTime())));
+    }
+
+    Parse parse = new Parse(text, title, outlinks, status);
+    parse = htmlParseFilters.filter(url, page, parse, metaTags, root);
+
+    if (metaTags.getNoCache()) {             // not okay to cache
+      page.putToMetadata(new Utf8(Nutch.CACHING_FORBIDDEN_KEY),
+          ByteBuffer.wrap(Bytes.toBytes(cachingPolicy)));
+    }
+
+    return parse;
+  }
+
+  private DocumentFragment parse(InputSource input) throws Exception {
+    if (parserImpl.equalsIgnoreCase("tagsoup"))
+      return parseTagSoup(input);
+    else return parseNeko(input);
+  }
+
+  private DocumentFragment parseTagSoup(InputSource input) throws Exception {
+    HTMLDocumentImpl doc = new HTMLDocumentImpl();
+    DocumentFragment frag = doc.createDocumentFragment();
+    DOMBuilder builder = new DOMBuilder(doc, frag);
+    org.ccil.cowan.tagsoup.Parser reader = new org.ccil.cowan.tagsoup.Parser();
+    reader.setContentHandler(builder);
+    reader.setFeature(org.ccil.cowan.tagsoup.Parser.ignoreBogonsFeature, true);
+    reader.setFeature(org.ccil.cowan.tagsoup.Parser.bogonsEmptyFeature, false);
+    reader.setProperty("http://xml.org/sax/properties/lexical-handler";, 
builder);
+    reader.parse(input);
+    return frag;
+  }
+
+  private DocumentFragment parseNeko(InputSource input) throws Exception {
+    DOMFragmentParser parser = new DOMFragmentParser();
+    try {
+      parser.setFeature("http://cyberneko.org/html/features/augmentations";,
+          true);
+      
parser.setProperty("http://cyberneko.org/html/properties/default-encoding";,
+          defaultCharEncoding);
+      
parser.setFeature("http://cyberneko.org/html/features/scanner/ignore-specified-charset";,
+          true);
+      
parser.setFeature("http://cyberneko.org/html/features/balance-tags/ignore-outside-content";,
+          false);
+      
parser.setFeature("http://cyberneko.org/html/features/balance-tags/document-fragment";,
+          true);
+      parser.setFeature("http://cyberneko.org/html/features/report-errors";,
+          LOG.isTraceEnabled());
+    } catch (SAXException e) {}
+    // convert Document to DocumentFragment
+    HTMLDocumentImpl doc = new HTMLDocumentImpl();
+    doc.setErrorChecking(false);
+    DocumentFragment res = doc.createDocumentFragment();
+    DocumentFragment frag = doc.createDocumentFragment();
+    parser.parse(input, frag);
+    res.appendChild(frag);
+
+    try {
+      while(true) {
+        frag = doc.createDocumentFragment();
+        parser.parse(input, frag);
+        if (!frag.hasChildNodes()) break;
+        if (LOG.isInfoEnabled()) {
+          LOG.info(" - new frag, " + frag.getChildNodes().getLength() + " 
nodes.");
+        }
+        res.appendChild(frag);
+      }
+    } catch (Exception x) { x.printStackTrace(LogUtil.getWarnStream(LOG));};
+    return res;
+  }
+
+  public void setConf(Configuration conf) {
+    this.conf = conf;
+    this.htmlParseFilters = new HtmlParseFilters(getConf());
+    this.parserImpl = getConf().get("parser.html.impl", "neko");
+    this.defaultCharEncoding = getConf().get(
+        "parser.character.encoding.default", "windows-1252");
+    this.utils = new DOMContentUtils(conf);
+    this.cachingPolicy = getConf().get("parser.caching.forbidden.policy",
+        Nutch.CACHING_FORBIDDEN_CONTENT);
+  }
+
+  public Configuration getConf() {
+    return this.conf;
+  }
+
+  @Override
+  public Collection<WebPage.Field> getFields() {
+    return FIELDS;
+  }
+
+  public static void main(String[] args) throws Exception {
+    //LOG.setLevel(Level.FINE);
+    String name = args[0];
+    String url = "file:"+name;
+    File file = new File(name);
+    byte[] bytes = new byte[(int)file.length()];
+    DataInputStream in = new DataInputStream(new FileInputStream(file));
+    in.readFully(bytes);
+    Configuration conf = NutchConfiguration.create();
+    HtmlParser parser = new HtmlParser();
+    parser.setConf(conf);
+    WebPage page = new WebPage();
+    page.setBaseUrl(new Utf8(url));
+    page.setContent(ByteBuffer.wrap(bytes));
+    page.setContentType(new Utf8("text/html"));
+    Parse parse = parser.getParse(url, page);
+    System.out.println("title: "+parse.getTitle());
+    System.out.println("text: "+parse.getText());
+    System.out.println("outlinks: " + Arrays.toString(parse.getOutlinks()));
+
+  }
+
+}

Added: 
nutch/branches/nutchbase/src/plugin/parse-html/src/java/org/apache/nutch/parse/html/XMLCharacterRecognizer.java
URL: 
http://svn.apache.org/viewvc/nutch/branches/nutchbase/src/plugin/parse-html/src/java/org/apache/nutch/parse/html/XMLCharacterRecognizer.java?rev=982184&view=auto
==============================================================================
--- 
nutch/branches/nutchbase/src/plugin/parse-html/src/java/org/apache/nutch/parse/html/XMLCharacterRecognizer.java
 (added)
+++ 
nutch/branches/nutchbase/src/plugin/parse-html/src/java/org/apache/nutch/parse/html/XMLCharacterRecognizer.java
 Wed Aug  4 10:01:08 2010
@@ -0,0 +1,113 @@
+/*
+ * XXX [email protected]: This class is copied verbatim from Xalan-J 2.6.0
+ * XXX distribution, org.apache.xml.utils.XMLCharacterRecognizer,
+ * XXX in order to avoid dependency on Xalan.
+ */
+
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements.  See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License.  You may obtain a copy of the License at
+ *
+ *     http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+/*
+ * $Id: XMLCharacterRecognizer.java,v 1.7 2004/02/17 04:21:14 minchau Exp $
+ */
+package org.apache.nutch.parse.html;
+
+/**
+ * Class used to verify whether the specified <var>ch</var> 
+ * conforms to the XML 1.0 definition of whitespace. 
+ */
+public class XMLCharacterRecognizer
+{
+
+  /**
+   * Returns whether the specified <var>ch</var> conforms to the XML 1.0 
definition
+   * of whitespace.  Refer to <A 
href="http://www.w3.org/TR/1998/REC-xml-19980210#NT-S";>
+   * the definition of <CODE>S</CODE></A> for details.
+   * @param ch Character to check as XML whitespace.
+   * @return =true if <var>ch</var> is XML whitespace; otherwise =false.
+   */
+  public static boolean isWhiteSpace(char ch)
+  {
+    return (ch == 0x20) || (ch == 0x09) || (ch == 0xD) || (ch == 0xA);
+  }
+
+  /**
+   * Tell if the string is whitespace.
+   *
+   * @param ch Character array to check as XML whitespace.
+   * @param start Start index of characters in the array
+   * @param length Number of characters in the array 
+   * @return True if the characters in the array are 
+   * XML whitespace; otherwise, false.
+   */
+  public static boolean isWhiteSpace(char ch[], int start, int length)
+  {
+
+    int end = start + length;
+
+    for (int s = start; s < end; s++)
+    {
+      if (!isWhiteSpace(ch[s]))
+        return false;
+    }
+
+    return true;
+  }
+
+  /**
+   * Tell if the string is whitespace.
+   *
+   * @param buf StringBuffer to check as XML whitespace.
+   * @return True if characters in buffer are XML whitespace, false otherwise
+   */
+  public static boolean isWhiteSpace(StringBuffer buf)
+  {
+
+    int n = buf.length();
+
+    for (int i = 0; i < n; i++)
+    {
+      if (!isWhiteSpace(buf.charAt(i)))
+        return false;
+    }
+
+    return true;
+  }
+  
+  /**
+   * Tell if the string is whitespace.
+   *
+   * @param s String to check as XML whitespace.
+   * @return True if characters in buffer are XML whitespace, false otherwise
+   */
+  public static boolean isWhiteSpace(String s)
+  {
+
+    if(null != s)
+    {
+      int n = s.length();
+  
+      for (int i = 0; i < n; i++)
+      {
+        if (!isWhiteSpace(s.charAt(i)))
+          return false;
+      }
+    }
+
+    return true;
+  }
+
+}

Added: 
nutch/branches/nutchbase/src/plugin/parse-html/src/java/org/apache/nutch/parse/html/package.html
URL: 
http://svn.apache.org/viewvc/nutch/branches/nutchbase/src/plugin/parse-html/src/java/org/apache/nutch/parse/html/package.html?rev=982184&view=auto
==============================================================================
--- 
nutch/branches/nutchbase/src/plugin/parse-html/src/java/org/apache/nutch/parse/html/package.html
 (added)
+++ 
nutch/branches/nutchbase/src/plugin/parse-html/src/java/org/apache/nutch/parse/html/package.html
 Wed Aug  4 10:01:08 2010
@@ -0,0 +1,5 @@
+<html>
+<body>
+<p>An HTML document parsing plugin.</p><p>This package relies on <a 
href="http://www.apache.org/~andyc/neko/doc/html/index.html";>NekoHTML</a>.</p>
+</body>
+</html>

Added: 
nutch/branches/nutchbase/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestDOMContentUtils.java
URL: 
http://svn.apache.org/viewvc/nutch/branches/nutchbase/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestDOMContentUtils.java?rev=982184&view=auto
==============================================================================
--- 
nutch/branches/nutchbase/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestDOMContentUtils.java
 (added)
+++ 
nutch/branches/nutchbase/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestDOMContentUtils.java
 Wed Aug  4 10:01:08 2010
@@ -0,0 +1,408 @@
+/**
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements.  See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License.  You may obtain a copy of the License at
+ *
+ *     http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.nutch.parse.html;
+
+import junit.framework.TestCase;
+
+import org.apache.nutch.parse.Outlink;
+import org.apache.hadoop.conf.Configuration;
+import org.apache.nutch.util.NutchConfiguration;
+
+import java.io.ByteArrayInputStream;
+import java.net.MalformedURLException;
+import java.net.URL;
+import java.util.ArrayList;
+import java.util.StringTokenizer;
+
+import org.cyberneko.html.parsers.*;
+import org.xml.sax.*;
+import org.w3c.dom.*;
+import org.apache.html.dom.*;
+
+/** 
+ * Unit tests for DOMContentUtils.
+ */
+public class TestDOMContentUtils extends TestCase {
+
+  private static final String[] testPages= { 
+    new String("<html><head><title> title </title><script> script </script>"
+               + "</head><body> body <a href=\"http://www.nutch.org\";>"
+               + " anchor </a><!--comment-->"
+               + "</body></html>"),
+    new String("<html><head><title> title </title><script> script </script>"
+               + "</head><body> body <a href=\"/\">"
+               + " home </a><!--comment-->"
+               + "<style> style </style>"
+               + " <a href=\"bot.html\">"
+               + " bots </a>"
+               + "</body></html>"),
+    new String("<html><head><title> </title>"
+               + "</head><body> "
+               + "<a href=\"/\"> separate this "
+               + "<a href=\"ok\"> from this"
+               + "</a></a>"
+               + "</body></html>"),
+    // this one relies on certain neko fixup behavior, possibly
+    // distributing the anchors into the LI's-but not the other
+    // anchors (outside of them, instead)!  So you get a tree that
+    // looks like:
+    // ... <li> <a href=/> home </a> </li>
+    //     <li> <a href=/> <a href="1"> 1 </a> </a> </li>
+    //     <li> <a href=/> <a href="1"> <a href="2"> 2 </a> </a> </a> </li>
+    new String("<html><head><title> my title </title>"
+               + "</head><body> body "
+               + "<ul>"
+               + "<li> <a href=\"/\"> home"
+               + "<li> <a href=\"1\"> 1"
+               + "<li> <a href=\"2\"> 2"
+               + "</ul>"
+               + "</body></html>"),
+    // test frameset link extraction. The invalid frame in the middle will be
+    // fixed to a third standalone frame.
+    new String("<html><head><title> my title </title>"
+               + "</head><frameset rows=\"20,*\"> "
+               + "<frame src=\"top.html\">"
+               + "</frame>"
+               + "<frameset cols=\"20,*\">"
+               + "<frame src=\"left.html\">"
+               + "<frame src=\"invalid.html\"/>"
+               + "</frame>"
+               + "<frame src=\"right.html\">"
+               + "</frame>"
+               + "</frameset>"
+               + "</frameset>"
+               + "</body></html>"),
+    // test <area> and <iframe> link extraction + url normalization
+    new String("<html><head><title> my title </title>"
+               + "</head><body>"
+               + "<img src=\"logo.gif\" usemap=\"#green\" border=\"0\">"
+                          + "<map name=\"green\">"
+                          + "<area shape=\"polygon\" coords=\"19,44,45,11,87\" 
href=\"../index.html\">"
+                          + "<area shape=\"rect\" coords=\"128,132,241,179\" 
href=\"#bottom\">"
+                          + "<area shape=\"circle\" coords=\"68,211,35\" 
href=\"../bot.html\">"
+                          + "</map>"
+               + "<a name=\"bottom\"/><h1> the bottom </h1> "
+               + "<iframe src=\"../docs/index.html\"/>"
+               + "</body></html>"),
+    // test whitespace processing for plain text extraction
+    new String("<html><head>\n <title> my\t\n  title\r\n </title>\n"
+               + " </head>\n"
+               + " <body>\n"
+               + "    <h1> Whitespace\ttest  </h1> \n"
+               + "\t<a href=\"../index.html\">\n  \twhitespace  test\r\n\t</a> 
 \t\n"
+               + "    <p> This is<span> a whitespace<span></span> test</span>. 
Newlines\n"
+               + "should appear as space too.</p><p>Tabs\tare spaces 
too.\n</p>"
+               + "    This\t<b>is a</b> break -&gt;<br>and the line after<i> 
break</i>.<br>\n"
+               + "<table>"
+               + "    <tr><td>one</td><td>two</td><td>three</td></tr>\n"
+               + "    <tr><td>space here </td><td> space there</td><td>no 
space</td></tr>"
+               + 
"\t<tr><td>one\r\ntwo</td><td>two\tthree</td><td>three\r\tfour</td></tr>\n"
+               + "</table>put some text here<Br>and there."
+               + "<h2>End\tthis\rmadness\n!</h2>\r\n"
+               + "         .        .        .         ."
+               + "</body>  </html>"),
+
+    // test that <a rel=nofollow> links are not returned
+    new String("<html><head></head><body>"
+               + "<a href=\"http://www.nutch.org\"; rel=\"nofollow\"> ignore 
</a>"
+               + "<a rel=\"nofollow\" href=\"http://www.nutch.org\";> ignore 
</a>"
+               + "</body></html>"),
+    // test that POST form actions are skipped
+    new String("<html><head></head><body>"
+            + "<form method='POST' action='/search.jsp'><input type=text>"
+            + "<input type=submit><p>test1</p></form>"
+            + "<form method='GET' action='/dummy.jsp'><input type=text>"
+            + "<input type=submit><p>test2</p></form></body></html>"),
+    // test that all form actions are skipped
+    new String("<html><head></head><body>"
+            + "<form method='POST' action='/search.jsp'><input type=text>"
+            + "<input type=submit><p>test1</p></form>"
+            + "<form method='GET' action='/dummy.jsp'><input type=text>"
+            + "<input type=submit><p>test2</p></form></body></html>"),
+    new String("<html><head><title> title </title>"
+      + "</head><body>"
+      + "<a href=\";x\">anchor1</a>"
+      + "<a href=\"g;x\">anchor2</a>"
+      + "<a href=\"g;x?y#s\">anchor3</a>"
+      + "</body></html>"),  
+    new String("<html><head><title> title </title>"
+        + "</head><body>"
+        + "<a href=\"g\">anchor1</a>"
+        + "<a href=\"g?y#s\">anchor2</a>"
+        + "<a href=\"?y=1\">anchor3</a>"
+        + "<a href=\"?y=1#s\">anchor4</a>"
+        + "<a href=\"?y=1;somethingelse\">anchor5</a>"
+        + "</body></html>"), 
+  };
+  
+  private static int SKIP = 9;
+
+  private static String[] testBaseHrefs= {
+    "http://www.nutch.org";,     
+    "http://www.nutch.org/docs/foo.html";,     
+    "http://www.nutch.org/docs/";,     
+    "http://www.nutch.org/docs/";,
+    "http://www.nutch.org/frames/";,     
+    "http://www.nutch.org/maps/";,
+    "http://www.nutch.org/whitespace/";,
+    "http://www.nutch.org//";,
+    "http://www.nutch.org/";,
+    "http://www.nutch.org/";,
+    "http://www.nutch.org/";,
+    "http://www.nutch.org/;something";
+  };
+    
+  private static final DocumentFragment testDOMs[]=
+    new DocumentFragment[testPages.length];
+
+  private static URL[] testBaseHrefURLs= 
+    new URL[testPages.length];
+
+
+  private static final String[] answerText= {
+    "title body anchor",
+    "title body home bots",
+    "separate this from this",
+    "my title body home 1 2",
+    "my title",
+    "my title the bottom",
+    "my title Whitespace test whitespace test "
+        + "This is a whitespace test . Newlines should appear as space too. "
+        + "Tabs are spaces too. This is a break -> and the line after break . "
+        + "one two three space here space there no space "
+        + "one two two three three four put some text here and there. "
+        + "End this madness ! . . . .",
+    "ignore ignore",
+    "test1 test2",
+    "test1 test2",
+    "title anchor1 anchor2 anchor3",
+    "title anchor1 anchor2 anchor3 anchor4 anchor5"
+  };
+
+  private static final String[] answerTitle= {
+    "title",
+    "title",
+    "",
+    "my title",
+    "my title",
+    "my title",
+    "my title",
+    "",
+    "",
+    "",
+    "title",
+    "title"
+  };
+
+  // note: should be in page-order
+  private static Outlink[][] answerOutlinks;
+  
+  private static Configuration conf;
+  private static DOMContentUtils utils = null;
+  
+  public TestDOMContentUtils(String name) { 
+    super(name); 
+  }
+
+  private static void setup() {
+    conf = NutchConfiguration.create();
+    conf.setBoolean("parser.html.form.use_action", true);
+    utils = new DOMContentUtils(conf);
+    DOMFragmentParser parser= new DOMFragmentParser();
+    for (int i= 0; i < testPages.length; i++) {
+        DocumentFragment node= 
+          new HTMLDocumentImpl().createDocumentFragment();
+        try {
+          parser.parse(
+            new InputSource( 
+              new ByteArrayInputStream(testPages[i].getBytes()) ),
+            node);
+          testBaseHrefURLs[i]= new URL(testBaseHrefs[i]);
+        } catch (Exception e) {
+          assertTrue("caught exception: " + e, false);
+        } 
+      testDOMs[i]= node;
+    }
+    try {
+    answerOutlinks = new Outlink[][]{ 
+        {
+          new Outlink("http://www.nutch.org";, "anchor"),
+        },
+        {
+          new Outlink("http://www.nutch.org/";, "home"),
+          new Outlink("http://www.nutch.org/docs/bot.html";, "bots"),
+        },
+        {
+          new Outlink("http://www.nutch.org/";, "separate this"),
+          new Outlink("http://www.nutch.org/docs/ok";, "from this"),
+        },
+        {
+          new Outlink("http://www.nutch.org/";, "home"),
+          new Outlink("http://www.nutch.org/docs/1";, "1"),
+          new Outlink("http://www.nutch.org/docs/2";, "2"),
+        },
+        {
+          new Outlink("http://www.nutch.org/frames/top.html";, ""),
+          new Outlink("http://www.nutch.org/frames/left.html";, ""),
+          new Outlink("http://www.nutch.org/frames/invalid.html";, ""),
+          new Outlink("http://www.nutch.org/frames/right.html";, ""),
+        },
+        {
+          new Outlink("http://www.nutch.org/maps/logo.gif";, ""),
+          new Outlink("http://www.nutch.org/index.html";, ""),
+          new Outlink("http://www.nutch.org/maps/#bottom";, ""),
+          new Outlink("http://www.nutch.org/bot.html";, ""),
+          new Outlink("http://www.nutch.org/docs/index.html";, ""),
+        },
+        {
+          new Outlink("http://www.nutch.org/index.html";, "whitespace test"),
+        },
+        {
+        },
+        {
+          new Outlink("http://www.nutch.org/dummy.jsp";, "test2"),
+        },
+        {
+        },
+        {
+          new Outlink("http://www.nutch.org/;x";, "anchor1"),
+          new Outlink("http://www.nutch.org/g;x";, "anchor2"),
+          new Outlink("http://www.nutch.org/g;x?y#s";, "anchor3")
+        },
+        {
+          new Outlink("http://www.nutch.org/g;something";, "anchor1"),
+          new Outlink("http://www.nutch.org/g;something?y#s";, "anchor2"),
+          new Outlink("http://www.nutch.org/;something?y=1";, "anchor3"),
+          new Outlink("http://www.nutch.org/;something?y=1#s";, "anchor4"),
+          new Outlink("http://www.nutch.org/?y=1;somethingelse";, "anchor5")
+        }
+    };
+
+    } catch (MalformedURLException e) {
+        
+  }
+  }
+
+  private static boolean equalsIgnoreWhitespace(String s1, String s2) {
+    StringTokenizer st1= new StringTokenizer(s1);
+    StringTokenizer st2= new StringTokenizer(s2);
+
+    while (st1.hasMoreTokens()) {
+      if (!st2.hasMoreTokens()) 
+        return false;
+      if ( ! st1.nextToken().equals(st2.nextToken()) )
+        return false;
+    }
+    if (st2.hasMoreTokens()) 
+      return false;
+    return true;
+  }
+
+  public void testGetText() {
+    if (testDOMs[0] == null) 
+      setup();
+    for (int i= 0; i < testPages.length; i++) {
+      StringBuilder sb= new StringBuilder();
+      utils.getText(sb, testDOMs[i]);
+      String text= sb.toString();
+      assertTrue("expecting text: " + answerText[i] 
+                 + System.getProperty("line.separator") 
+                 + System.getProperty("line.separator") 
+                 + "got text: "+ text, 
+                 equalsIgnoreWhitespace(answerText[i], text));
+    }
+  }
+
+  public void testGetTitle() {
+    if (testDOMs[0] == null) 
+      setup();
+    for (int i= 0; i < testPages.length; i++) {
+      StringBuilder sb= new StringBuilder();
+      utils.getTitle(sb, testDOMs[i]);
+      String text= sb.toString();
+      assertTrue("expecting text: " + answerText[i] 
+                 + System.getProperty("line.separator") 
+                 + System.getProperty("line.separator") 
+                 + "got text: "+ text, 
+                 equalsIgnoreWhitespace(answerTitle[i], text));
+    }
+  }
+
+  public void testGetOutlinks() {
+    if (testDOMs[0] == null) 
+      setup();
+    for (int i= 0; i < testPages.length; i++) {
+      ArrayList<Outlink> outlinks= new ArrayList<Outlink>();
+      if (i == SKIP) {
+        conf.setBoolean("parser.html.form.use_action", false);
+        utils.setConf(conf);
+      } else {
+        conf.setBoolean("parser.html.form.use_action", true);
+        utils.setConf(conf);
+      }
+      utils.getOutlinks(testBaseHrefURLs[i], outlinks, testDOMs[i]);
+      Outlink[] outlinkArr= new Outlink[outlinks.size()];
+      outlinkArr= outlinks.toArray(outlinkArr);
+      compareOutlinks(answerOutlinks[i], outlinkArr);
+    }
+  }
+
+  private static final void appendOutlinks(StringBuffer sb, Outlink[] o) {
+    for (int i= 0; i < o.length; i++) {
+      sb.append(o[i].toString());
+      sb.append(System.getProperty("line.separator"));
+    }
+  }
+
+  private static final String outlinksString(Outlink[] o) {
+    StringBuffer sb= new StringBuffer();
+    appendOutlinks(sb, o);
+    return sb.toString();
+  }
+
+  private static final void compareOutlinks(Outlink[] o1, Outlink[] o2) {
+    if (o1.length != o2.length) {
+      assertTrue("got wrong number of outlinks (expecting " + o1.length 
+                 + ", got " + o2.length + ")" 
+                 + System.getProperty("line.separator") 
+                 + "answer: " + System.getProperty("line.separator") 
+                 + outlinksString(o1) 
+                 + System.getProperty("line.separator") 
+                 + "got: " + System.getProperty("line.separator") 
+                 + outlinksString(o2)
+                 + System.getProperty("line.separator"),
+                 false
+        );
+    }
+
+    for (int i= 0; i < o1.length; i++) {
+      if (!o1[i].equals(o2[i])) {
+        assertTrue("got wrong outlinks at position " + i
+                   + System.getProperty("line.separator") 
+                   + "answer: " + System.getProperty("line.separator") 
+                   + "'" + o1[i].getToUrl() + "', anchor: '" + 
o1[i].getAnchor() + "'"
+                   + System.getProperty("line.separator") 
+                   + "got: " + System.getProperty("line.separator") 
+                   + "'" + o2[i].getToUrl() + "', anchor: '" + 
o2[i].getAnchor() + "'",
+                   false
+          );
+        
+      }
+    }
+  }
+}

Added: 
nutch/branches/nutchbase/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestRobotsMetaProcessor.java
URL: 
http://svn.apache.org/viewvc/nutch/branches/nutchbase/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestRobotsMetaProcessor.java?rev=982184&view=auto
==============================================================================
--- 
nutch/branches/nutchbase/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestRobotsMetaProcessor.java
 (added)
+++ 
nutch/branches/nutchbase/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestRobotsMetaProcessor.java
 Wed Aug  4 10:01:08 2010
@@ -0,0 +1,182 @@
+/**
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements.  See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License.  You may obtain a copy of the License at
+ *
+ *     http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package org.apache.nutch.parse.html;
+
+import junit.framework.TestCase;
+
+import org.apache.nutch.parse.HTMLMetaTags;
+
+import java.io.ByteArrayInputStream;
+import java.net.URL;
+
+import org.cyberneko.html.parsers.*;
+import org.xml.sax.*;
+import org.w3c.dom.*;
+import org.apache.html.dom.*;
+
+/** Unit tests for HTMLMetaProcessor. */
+public class TestRobotsMetaProcessor extends TestCase {
+  public TestRobotsMetaProcessor(String name) { 
+    super(name); 
+  }
+
+  /*
+
+  some sample tags:
+
+  <meta name="robots" content="index,follow">
+  <meta name="robots" content="noindex,follow">
+  <meta name="robots" content="index,nofollow">
+  <meta name="robots" content="noindex,nofollow">
+
+  <META HTTP-EQUIV="Pragma" CONTENT="no-cache">
+
+  */
+
+
+  public static String[] tests= 
+  {
+    "<html><head><title>test page</title>"
+    + "<META NAME=\"ROBOTS\" CONTENT=\"NONE\"> "
+    + "<META HTTP-EQUIV=\"PRAGMA\" CONTENT=\"NO-CACHE\"> "
+    + "</head><body>"
+    + " some text"
+    + "</body></html>",
+
+    "<html><head><title>test page</title>"
+    + "<meta name=\"robots\" content=\"all\"> "
+    + "<meta http-equiv=\"pragma\" content=\"no-cache\"> "
+    + "</head><body>"
+    + " some text"
+    + "</body></html>",
+
+    "<html><head><title>test page</title>"
+    + "<MeTa NaMe=\"RoBoTs\" CoNtEnT=\"nOnE\"> "
+    + "<MeTa HtTp-EqUiV=\"pRaGmA\" cOnTeNt=\"No-CaChE\"> "
+    + "</head><body>"
+    + " some text"
+    + "</body></html>",
+
+    "<html><head><title>test page</title>"
+    + "<meta name=\"robots\" content=\"none\"> "
+    + "</head><body>"
+    + " some text"
+    + "</body></html>",
+
+    "<html><head><title>test page</title>"
+    + "<meta name=\"robots\" content=\"noindex,nofollow\"> "
+    + "</head><body>"
+    + " some text"
+    + "</body></html>",
+
+    "<html><head><title>test page</title>"
+    + "<meta name=\"robots\" content=\"noindex,follow\"> "
+    + "</head><body>"
+    + " some text"
+    + "</body></html>",
+
+    "<html><head><title>test page</title>"
+    + "<meta name=\"robots\" content=\"index,nofollow\"> "
+    + "</head><body>"
+    + " some text"
+    + "</body></html>",
+
+    "<html><head><title>test page</title>"
+    + "<meta name=\"robots\" content=\"index,follow\"> "
+    + "<base href=\"http://www.nutch.org/\";>"
+    + "</head><body>"
+    + " some text"
+    + "</body></html>",
+
+    "<html><head><title>test page</title>"
+    + "<meta name=\"robots\"> "
+    + "<base href=\"http://www.nutch.org/base/\";>"
+    + "</head><body>"
+    + " some text"
+    + "</body></html>",
+
+  };
+
+  public static final boolean[][] answers= {
+    {true, true, true},     // NONE
+    {false, false, true},   // all
+    {true, true, true},     // nOnE
+    {true, true, false},    // none
+    {true, true, false},    // noindex,nofollow
+    {true, false, false},   // noindex,follow
+    {false, true, false},   // index,nofollow
+    {false, false, false},  // index,follow
+    {false, false, false},  // missing!
+  };
+
+  private URL[][] currURLsAndAnswers;
+
+  public void testRobotsMetaProcessor() {
+    DOMFragmentParser parser= new DOMFragmentParser();;
+
+    try { 
+      currURLsAndAnswers= new URL[][] {
+        {new URL("http://www.nutch.org";), null},
+        {new URL("http://www.nutch.org";), null},
+        {new URL("http://www.nutch.org";), null},
+        {new URL("http://www.nutch.org";), null},
+        {new URL("http://www.nutch.org";), null},
+        {new URL("http://www.nutch.org";), null},
+        {new URL("http://www.nutch.org";), null},
+        {new URL("http://www.nutch.org/foo/";), 
+         new URL("http://www.nutch.org/";)},
+        {new URL("http://www.nutch.org";), 
+         new URL("http://www.nutch.org/base/";)}
+      };
+    } catch (Exception e) {
+      assertTrue("couldn't make test URLs!", false);
+    }
+
+    for (int i= 0; i < tests.length; i++) {
+      byte[] bytes= tests[i].getBytes();
+
+      DocumentFragment node = new HTMLDocumentImpl().createDocumentFragment();
+
+      try {
+        parser.parse(new InputSource(new ByteArrayInputStream(bytes)), node);
+      } catch (Exception e) {
+        e.printStackTrace();
+      }
+
+      HTMLMetaTags robotsMeta= new HTMLMetaTags();
+      HTMLMetaProcessor.getMetaTags(robotsMeta, node, 
+                                                  currURLsAndAnswers[i][0]);
+
+      assertTrue("got index wrong on test " + i,
+                 robotsMeta.getNoIndex() == answers[i][0]);
+      assertTrue("got follow wrong on test " + i,
+                 robotsMeta.getNoFollow() == answers[i][1]);
+      assertTrue("got cache wrong on test " + i,
+                 robotsMeta.getNoCache() == answers[i][2]);
+      assertTrue("got base href wrong on test " + i + " (got "
+                 + robotsMeta.getBaseHref() + ")",
+                 ( (robotsMeta.getBaseHref() == null)
+                    && (currURLsAndAnswers[i][1] == null) )
+                 || ( (robotsMeta.getBaseHref() != null)
+                      && robotsMeta.getBaseHref().equals(
+                        currURLsAndAnswers[i][1]) ) );
+      
+    }
+  }
+
+}


Reply via email to