Added: nutch/branches/nutchbase/src/plugin/parse-html/src/java/org/apache/nutch/parse/html/HtmlParser.java URL: http://svn.apache.org/viewvc/nutch/branches/nutchbase/src/plugin/parse-html/src/java/org/apache/nutch/parse/html/HtmlParser.java?rev=982184&view=auto ============================================================================== --- nutch/branches/nutchbase/src/plugin/parse-html/src/java/org/apache/nutch/parse/html/HtmlParser.java (added) +++ nutch/branches/nutchbase/src/plugin/parse-html/src/java/org/apache/nutch/parse/html/HtmlParser.java Wed Aug 4 10:01:08 2010 @@ -0,0 +1,330 @@ +/** + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.nutch.parse.html; + +import java.io.ByteArrayInputStream; +import java.io.DataInputStream; +import java.io.File; +import java.io.FileInputStream; +import java.io.IOException; +import java.io.UnsupportedEncodingException; +import java.net.MalformedURLException; +import java.net.URL; +import java.nio.ByteBuffer; +import java.nio.charset.Charset; +import java.util.ArrayList; +import java.util.Arrays; +import java.util.Collection; +import java.util.HashSet; +import java.util.regex.Matcher; +import java.util.regex.Pattern; + +import org.apache.avro.util.Utf8; +import org.apache.commons.logging.Log; +import org.apache.commons.logging.LogFactory; +import org.apache.hadoop.conf.Configuration; +import org.apache.html.dom.HTMLDocumentImpl; +import org.apache.nutch.metadata.Metadata; +import org.apache.nutch.metadata.Nutch; +import org.apache.nutch.parse.HTMLMetaTags; +import org.apache.nutch.parse.HtmlParseFilters; +import org.apache.nutch.parse.Outlink; +import org.apache.nutch.parse.Parse; +import org.apache.nutch.parse.ParseStatusCodes; +import org.apache.nutch.parse.ParseStatusUtils; +import org.apache.nutch.parse.Parser; +import org.apache.nutch.storage.ParseStatus; +import org.apache.nutch.storage.WebPage; +import org.apache.nutch.util.Bytes; +import org.apache.nutch.util.EncodingDetector; +import org.apache.nutch.util.LogUtil; +import org.apache.nutch.util.NutchConfiguration; +import org.apache.nutch.util.TableUtil; +import org.cyberneko.html.parsers.DOMFragmentParser; +import org.w3c.dom.DOMException; +import org.w3c.dom.DocumentFragment; +import org.xml.sax.InputSource; +import org.xml.sax.SAXException; + +public class HtmlParser implements Parser { + public static final Log LOG = LogFactory.getLog("org.apache.nutch.parse.html"); + + // I used 1000 bytes at first, but found that some documents have + // meta tag well past the first 1000 bytes. + // (e.g. http://cn.promo.yahoo.com/customcare/music.html) + private static final int CHUNK_SIZE = 2000; + private static Pattern metaPattern = + Pattern.compile("<meta\\s+([^>]*http-equiv=\"?content-type\"?[^>]*)>", + Pattern.CASE_INSENSITIVE); + private static Pattern charsetPattern = + Pattern.compile("charset=\\s*([a-z][_\\-0-9a-z]*)", + Pattern.CASE_INSENSITIVE); + + private static Collection<WebPage.Field> FIELDS = new HashSet<WebPage.Field>(); + + static { + FIELDS.add(WebPage.Field.BASE_URL); + } + + private String parserImpl; + + /** + * Given a <code>byte[]</code> representing an html file of an + * <em>unknown</em> encoding, read out 'charset' parameter in the meta tag + * from the first <code>CHUNK_SIZE</code> bytes. + * If there's no meta tag for Content-Type or no charset is specified, + * <code>null</code> is returned. <br /> + * FIXME: non-byte oriented character encodings (UTF-16, UTF-32) + * can't be handled with this. + * We need to do something similar to what's done by mozilla + * (http://lxr.mozilla.org/seamonkey/source/parser/htmlparser/src/nsParser.cpp#1993). + * See also http://www.w3.org/TR/REC-xml/#sec-guessing + * <br /> + * + * @param content <code>byte[]</code> representation of an html file + */ + + private static String sniffCharacterEncoding(byte[] content) { + int length = content.length < CHUNK_SIZE ? + content.length : CHUNK_SIZE; + + // We don't care about non-ASCII parts so that it's sufficient + // to just inflate each byte to a 16-bit value by padding. + // For instance, the sequence {0x41, 0x82, 0xb7} will be turned into + // {U+0041, U+0082, U+00B7}. + String str = ""; + try { + str = new String(content, 0, length, + Charset.forName("ASCII").toString()); + } catch (UnsupportedEncodingException e) { + // code should never come here, but just in case... + return null; + } + + Matcher metaMatcher = metaPattern.matcher(str); + String encoding = null; + if (metaMatcher.find()) { + Matcher charsetMatcher = charsetPattern.matcher(metaMatcher.group(1)); + if (charsetMatcher.find()) + encoding = new String(charsetMatcher.group(1)); + } + + return encoding; + } + + private String defaultCharEncoding; + + private Configuration conf; + + private DOMContentUtils utils; + + private HtmlParseFilters htmlParseFilters; + + private String cachingPolicy; + + public Parse getParse(String url, WebPage page) { + HTMLMetaTags metaTags = new HTMLMetaTags(); + + String baseUrl = TableUtil.toString(page.getBaseUrl()); + URL base; + try { + base = new URL(baseUrl); + } catch (MalformedURLException e) { + return ParseStatusUtils.getEmptyParse(e, getConf()); + } + + String text = ""; + String title = ""; + Outlink[] outlinks = new Outlink[0]; + Metadata metadata = new Metadata(); + + // parse the content + DocumentFragment root; + try { + byte[] contentInOctets = page.getContent().array(); + InputSource input = new InputSource(new ByteArrayInputStream(contentInOctets)); + + EncodingDetector detector = new EncodingDetector(conf); + detector.autoDetectClues(page, true); + detector.addClue(sniffCharacterEncoding(contentInOctets), "sniffed"); + String encoding = detector.guessEncoding(page, defaultCharEncoding); + + metadata.set(Metadata.ORIGINAL_CHAR_ENCODING, encoding); + metadata.set(Metadata.CHAR_ENCODING_FOR_CONVERSION, encoding); + + input.setEncoding(encoding); + if (LOG.isTraceEnabled()) { LOG.trace("Parsing..."); } + root = parse(input); + } catch (IOException e) { + return ParseStatusUtils.getEmptyParse(e, getConf()); + } catch (DOMException e) { + return ParseStatusUtils.getEmptyParse(e, getConf()); + } catch (SAXException e) { + return ParseStatusUtils.getEmptyParse(e, getConf()); + } catch (Exception e) { + e.printStackTrace(LogUtil.getWarnStream(LOG)); + return ParseStatusUtils.getEmptyParse(e, getConf()); + } + + // get meta directives + HTMLMetaProcessor.getMetaTags(metaTags, root, base); + if (LOG.isTraceEnabled()) { + LOG.trace("Meta tags for " + base + ": " + metaTags.toString()); + } + // check meta directives + if (!metaTags.getNoIndex()) { // okay to index + StringBuilder sb = new StringBuilder(); + if (LOG.isTraceEnabled()) { LOG.trace("Getting text..."); } + utils.getText(sb, root); // extract text + text = sb.toString(); + sb.setLength(0); + if (LOG.isTraceEnabled()) { LOG.trace("Getting title..."); } + utils.getTitle(sb, root); // extract title + title = sb.toString().trim(); + } + + if (!metaTags.getNoFollow()) { // okay to follow links + ArrayList<Outlink> l = new ArrayList<Outlink>(); // extract outlinks + URL baseTag = utils.getBase(root); + if (LOG.isTraceEnabled()) { LOG.trace("Getting links..."); } + utils.getOutlinks(baseTag!=null?baseTag:base, l, root); + outlinks = l.toArray(new Outlink[l.size()]); + if (LOG.isTraceEnabled()) { + LOG.trace("found "+outlinks.length+" outlinks in "+ url); + } + } + + ParseStatus status = new ParseStatus(); + status.setMajorCode(ParseStatusCodes.SUCCESS); + if (metaTags.getRefresh()) { + status.setMinorCode(ParseStatusCodes.SUCCESS_REDIRECT); + status.addToArgs(new Utf8(metaTags.getRefreshHref().toString())); + status.addToArgs(new Utf8(Integer.toString(metaTags.getRefreshTime()))); + } + + Parse parse = new Parse(text, title, outlinks, status); + parse = htmlParseFilters.filter(url, page, parse, metaTags, root); + + if (metaTags.getNoCache()) { // not okay to cache + page.putToMetadata(new Utf8(Nutch.CACHING_FORBIDDEN_KEY), + ByteBuffer.wrap(Bytes.toBytes(cachingPolicy))); + } + + return parse; + } + + private DocumentFragment parse(InputSource input) throws Exception { + if (parserImpl.equalsIgnoreCase("tagsoup")) + return parseTagSoup(input); + else return parseNeko(input); + } + + private DocumentFragment parseTagSoup(InputSource input) throws Exception { + HTMLDocumentImpl doc = new HTMLDocumentImpl(); + DocumentFragment frag = doc.createDocumentFragment(); + DOMBuilder builder = new DOMBuilder(doc, frag); + org.ccil.cowan.tagsoup.Parser reader = new org.ccil.cowan.tagsoup.Parser(); + reader.setContentHandler(builder); + reader.setFeature(org.ccil.cowan.tagsoup.Parser.ignoreBogonsFeature, true); + reader.setFeature(org.ccil.cowan.tagsoup.Parser.bogonsEmptyFeature, false); + reader.setProperty("http://xml.org/sax/properties/lexical-handler", builder); + reader.parse(input); + return frag; + } + + private DocumentFragment parseNeko(InputSource input) throws Exception { + DOMFragmentParser parser = new DOMFragmentParser(); + try { + parser.setFeature("http://cyberneko.org/html/features/augmentations", + true); + parser.setProperty("http://cyberneko.org/html/properties/default-encoding", + defaultCharEncoding); + parser.setFeature("http://cyberneko.org/html/features/scanner/ignore-specified-charset", + true); + parser.setFeature("http://cyberneko.org/html/features/balance-tags/ignore-outside-content", + false); + parser.setFeature("http://cyberneko.org/html/features/balance-tags/document-fragment", + true); + parser.setFeature("http://cyberneko.org/html/features/report-errors", + LOG.isTraceEnabled()); + } catch (SAXException e) {} + // convert Document to DocumentFragment + HTMLDocumentImpl doc = new HTMLDocumentImpl(); + doc.setErrorChecking(false); + DocumentFragment res = doc.createDocumentFragment(); + DocumentFragment frag = doc.createDocumentFragment(); + parser.parse(input, frag); + res.appendChild(frag); + + try { + while(true) { + frag = doc.createDocumentFragment(); + parser.parse(input, frag); + if (!frag.hasChildNodes()) break; + if (LOG.isInfoEnabled()) { + LOG.info(" - new frag, " + frag.getChildNodes().getLength() + " nodes."); + } + res.appendChild(frag); + } + } catch (Exception x) { x.printStackTrace(LogUtil.getWarnStream(LOG));}; + return res; + } + + public void setConf(Configuration conf) { + this.conf = conf; + this.htmlParseFilters = new HtmlParseFilters(getConf()); + this.parserImpl = getConf().get("parser.html.impl", "neko"); + this.defaultCharEncoding = getConf().get( + "parser.character.encoding.default", "windows-1252"); + this.utils = new DOMContentUtils(conf); + this.cachingPolicy = getConf().get("parser.caching.forbidden.policy", + Nutch.CACHING_FORBIDDEN_CONTENT); + } + + public Configuration getConf() { + return this.conf; + } + + @Override + public Collection<WebPage.Field> getFields() { + return FIELDS; + } + + public static void main(String[] args) throws Exception { + //LOG.setLevel(Level.FINE); + String name = args[0]; + String url = "file:"+name; + File file = new File(name); + byte[] bytes = new byte[(int)file.length()]; + DataInputStream in = new DataInputStream(new FileInputStream(file)); + in.readFully(bytes); + Configuration conf = NutchConfiguration.create(); + HtmlParser parser = new HtmlParser(); + parser.setConf(conf); + WebPage page = new WebPage(); + page.setBaseUrl(new Utf8(url)); + page.setContent(ByteBuffer.wrap(bytes)); + page.setContentType(new Utf8("text/html")); + Parse parse = parser.getParse(url, page); + System.out.println("title: "+parse.getTitle()); + System.out.println("text: "+parse.getText()); + System.out.println("outlinks: " + Arrays.toString(parse.getOutlinks())); + + } + +}
Added: nutch/branches/nutchbase/src/plugin/parse-html/src/java/org/apache/nutch/parse/html/XMLCharacterRecognizer.java URL: http://svn.apache.org/viewvc/nutch/branches/nutchbase/src/plugin/parse-html/src/java/org/apache/nutch/parse/html/XMLCharacterRecognizer.java?rev=982184&view=auto ============================================================================== --- nutch/branches/nutchbase/src/plugin/parse-html/src/java/org/apache/nutch/parse/html/XMLCharacterRecognizer.java (added) +++ nutch/branches/nutchbase/src/plugin/parse-html/src/java/org/apache/nutch/parse/html/XMLCharacterRecognizer.java Wed Aug 4 10:01:08 2010 @@ -0,0 +1,113 @@ +/* + * XXX [email protected]: This class is copied verbatim from Xalan-J 2.6.0 + * XXX distribution, org.apache.xml.utils.XMLCharacterRecognizer, + * XXX in order to avoid dependency on Xalan. + */ + +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +/* + * $Id: XMLCharacterRecognizer.java,v 1.7 2004/02/17 04:21:14 minchau Exp $ + */ +package org.apache.nutch.parse.html; + +/** + * Class used to verify whether the specified <var>ch</var> + * conforms to the XML 1.0 definition of whitespace. + */ +public class XMLCharacterRecognizer +{ + + /** + * Returns whether the specified <var>ch</var> conforms to the XML 1.0 definition + * of whitespace. Refer to <A href="http://www.w3.org/TR/1998/REC-xml-19980210#NT-S"> + * the definition of <CODE>S</CODE></A> for details. + * @param ch Character to check as XML whitespace. + * @return =true if <var>ch</var> is XML whitespace; otherwise =false. + */ + public static boolean isWhiteSpace(char ch) + { + return (ch == 0x20) || (ch == 0x09) || (ch == 0xD) || (ch == 0xA); + } + + /** + * Tell if the string is whitespace. + * + * @param ch Character array to check as XML whitespace. + * @param start Start index of characters in the array + * @param length Number of characters in the array + * @return True if the characters in the array are + * XML whitespace; otherwise, false. + */ + public static boolean isWhiteSpace(char ch[], int start, int length) + { + + int end = start + length; + + for (int s = start; s < end; s++) + { + if (!isWhiteSpace(ch[s])) + return false; + } + + return true; + } + + /** + * Tell if the string is whitespace. + * + * @param buf StringBuffer to check as XML whitespace. + * @return True if characters in buffer are XML whitespace, false otherwise + */ + public static boolean isWhiteSpace(StringBuffer buf) + { + + int n = buf.length(); + + for (int i = 0; i < n; i++) + { + if (!isWhiteSpace(buf.charAt(i))) + return false; + } + + return true; + } + + /** + * Tell if the string is whitespace. + * + * @param s String to check as XML whitespace. + * @return True if characters in buffer are XML whitespace, false otherwise + */ + public static boolean isWhiteSpace(String s) + { + + if(null != s) + { + int n = s.length(); + + for (int i = 0; i < n; i++) + { + if (!isWhiteSpace(s.charAt(i))) + return false; + } + } + + return true; + } + +} Added: nutch/branches/nutchbase/src/plugin/parse-html/src/java/org/apache/nutch/parse/html/package.html URL: http://svn.apache.org/viewvc/nutch/branches/nutchbase/src/plugin/parse-html/src/java/org/apache/nutch/parse/html/package.html?rev=982184&view=auto ============================================================================== --- nutch/branches/nutchbase/src/plugin/parse-html/src/java/org/apache/nutch/parse/html/package.html (added) +++ nutch/branches/nutchbase/src/plugin/parse-html/src/java/org/apache/nutch/parse/html/package.html Wed Aug 4 10:01:08 2010 @@ -0,0 +1,5 @@ +<html> +<body> +<p>An HTML document parsing plugin.</p><p>This package relies on <a href="http://www.apache.org/~andyc/neko/doc/html/index.html">NekoHTML</a>.</p> +</body> +</html> Added: nutch/branches/nutchbase/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestDOMContentUtils.java URL: http://svn.apache.org/viewvc/nutch/branches/nutchbase/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestDOMContentUtils.java?rev=982184&view=auto ============================================================================== --- nutch/branches/nutchbase/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestDOMContentUtils.java (added) +++ nutch/branches/nutchbase/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestDOMContentUtils.java Wed Aug 4 10:01:08 2010 @@ -0,0 +1,408 @@ +/** + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.nutch.parse.html; + +import junit.framework.TestCase; + +import org.apache.nutch.parse.Outlink; +import org.apache.hadoop.conf.Configuration; +import org.apache.nutch.util.NutchConfiguration; + +import java.io.ByteArrayInputStream; +import java.net.MalformedURLException; +import java.net.URL; +import java.util.ArrayList; +import java.util.StringTokenizer; + +import org.cyberneko.html.parsers.*; +import org.xml.sax.*; +import org.w3c.dom.*; +import org.apache.html.dom.*; + +/** + * Unit tests for DOMContentUtils. + */ +public class TestDOMContentUtils extends TestCase { + + private static final String[] testPages= { + new String("<html><head><title> title </title><script> script </script>" + + "</head><body> body <a href=\"http://www.nutch.org\">" + + " anchor </a><!--comment-->" + + "</body></html>"), + new String("<html><head><title> title </title><script> script </script>" + + "</head><body> body <a href=\"/\">" + + " home </a><!--comment-->" + + "<style> style </style>" + + " <a href=\"bot.html\">" + + " bots </a>" + + "</body></html>"), + new String("<html><head><title> </title>" + + "</head><body> " + + "<a href=\"/\"> separate this " + + "<a href=\"ok\"> from this" + + "</a></a>" + + "</body></html>"), + // this one relies on certain neko fixup behavior, possibly + // distributing the anchors into the LI's-but not the other + // anchors (outside of them, instead)! So you get a tree that + // looks like: + // ... <li> <a href=/> home </a> </li> + // <li> <a href=/> <a href="1"> 1 </a> </a> </li> + // <li> <a href=/> <a href="1"> <a href="2"> 2 </a> </a> </a> </li> + new String("<html><head><title> my title </title>" + + "</head><body> body " + + "<ul>" + + "<li> <a href=\"/\"> home" + + "<li> <a href=\"1\"> 1" + + "<li> <a href=\"2\"> 2" + + "</ul>" + + "</body></html>"), + // test frameset link extraction. The invalid frame in the middle will be + // fixed to a third standalone frame. + new String("<html><head><title> my title </title>" + + "</head><frameset rows=\"20,*\"> " + + "<frame src=\"top.html\">" + + "</frame>" + + "<frameset cols=\"20,*\">" + + "<frame src=\"left.html\">" + + "<frame src=\"invalid.html\"/>" + + "</frame>" + + "<frame src=\"right.html\">" + + "</frame>" + + "</frameset>" + + "</frameset>" + + "</body></html>"), + // test <area> and <iframe> link extraction + url normalization + new String("<html><head><title> my title </title>" + + "</head><body>" + + "<img src=\"logo.gif\" usemap=\"#green\" border=\"0\">" + + "<map name=\"green\">" + + "<area shape=\"polygon\" coords=\"19,44,45,11,87\" href=\"../index.html\">" + + "<area shape=\"rect\" coords=\"128,132,241,179\" href=\"#bottom\">" + + "<area shape=\"circle\" coords=\"68,211,35\" href=\"../bot.html\">" + + "</map>" + + "<a name=\"bottom\"/><h1> the bottom </h1> " + + "<iframe src=\"../docs/index.html\"/>" + + "</body></html>"), + // test whitespace processing for plain text extraction + new String("<html><head>\n <title> my\t\n title\r\n </title>\n" + + " </head>\n" + + " <body>\n" + + " <h1> Whitespace\ttest </h1> \n" + + "\t<a href=\"../index.html\">\n \twhitespace test\r\n\t</a> \t\n" + + " <p> This is<span> a whitespace<span></span> test</span>. Newlines\n" + + "should appear as space too.</p><p>Tabs\tare spaces too.\n</p>" + + " This\t<b>is a</b> break -><br>and the line after<i> break</i>.<br>\n" + + "<table>" + + " <tr><td>one</td><td>two</td><td>three</td></tr>\n" + + " <tr><td>space here </td><td> space there</td><td>no space</td></tr>" + + "\t<tr><td>one\r\ntwo</td><td>two\tthree</td><td>three\r\tfour</td></tr>\n" + + "</table>put some text here<Br>and there." + + "<h2>End\tthis\rmadness\n!</h2>\r\n" + + " . . . ." + + "</body> </html>"), + + // test that <a rel=nofollow> links are not returned + new String("<html><head></head><body>" + + "<a href=\"http://www.nutch.org\" rel=\"nofollow\"> ignore </a>" + + "<a rel=\"nofollow\" href=\"http://www.nutch.org\"> ignore </a>" + + "</body></html>"), + // test that POST form actions are skipped + new String("<html><head></head><body>" + + "<form method='POST' action='/search.jsp'><input type=text>" + + "<input type=submit><p>test1</p></form>" + + "<form method='GET' action='/dummy.jsp'><input type=text>" + + "<input type=submit><p>test2</p></form></body></html>"), + // test that all form actions are skipped + new String("<html><head></head><body>" + + "<form method='POST' action='/search.jsp'><input type=text>" + + "<input type=submit><p>test1</p></form>" + + "<form method='GET' action='/dummy.jsp'><input type=text>" + + "<input type=submit><p>test2</p></form></body></html>"), + new String("<html><head><title> title </title>" + + "</head><body>" + + "<a href=\";x\">anchor1</a>" + + "<a href=\"g;x\">anchor2</a>" + + "<a href=\"g;x?y#s\">anchor3</a>" + + "</body></html>"), + new String("<html><head><title> title </title>" + + "</head><body>" + + "<a href=\"g\">anchor1</a>" + + "<a href=\"g?y#s\">anchor2</a>" + + "<a href=\"?y=1\">anchor3</a>" + + "<a href=\"?y=1#s\">anchor4</a>" + + "<a href=\"?y=1;somethingelse\">anchor5</a>" + + "</body></html>"), + }; + + private static int SKIP = 9; + + private static String[] testBaseHrefs= { + "http://www.nutch.org", + "http://www.nutch.org/docs/foo.html", + "http://www.nutch.org/docs/", + "http://www.nutch.org/docs/", + "http://www.nutch.org/frames/", + "http://www.nutch.org/maps/", + "http://www.nutch.org/whitespace/", + "http://www.nutch.org//", + "http://www.nutch.org/", + "http://www.nutch.org/", + "http://www.nutch.org/", + "http://www.nutch.org/;something" + }; + + private static final DocumentFragment testDOMs[]= + new DocumentFragment[testPages.length]; + + private static URL[] testBaseHrefURLs= + new URL[testPages.length]; + + + private static final String[] answerText= { + "title body anchor", + "title body home bots", + "separate this from this", + "my title body home 1 2", + "my title", + "my title the bottom", + "my title Whitespace test whitespace test " + + "This is a whitespace test . Newlines should appear as space too. " + + "Tabs are spaces too. This is a break -> and the line after break . " + + "one two three space here space there no space " + + "one two two three three four put some text here and there. " + + "End this madness ! . . . .", + "ignore ignore", + "test1 test2", + "test1 test2", + "title anchor1 anchor2 anchor3", + "title anchor1 anchor2 anchor3 anchor4 anchor5" + }; + + private static final String[] answerTitle= { + "title", + "title", + "", + "my title", + "my title", + "my title", + "my title", + "", + "", + "", + "title", + "title" + }; + + // note: should be in page-order + private static Outlink[][] answerOutlinks; + + private static Configuration conf; + private static DOMContentUtils utils = null; + + public TestDOMContentUtils(String name) { + super(name); + } + + private static void setup() { + conf = NutchConfiguration.create(); + conf.setBoolean("parser.html.form.use_action", true); + utils = new DOMContentUtils(conf); + DOMFragmentParser parser= new DOMFragmentParser(); + for (int i= 0; i < testPages.length; i++) { + DocumentFragment node= + new HTMLDocumentImpl().createDocumentFragment(); + try { + parser.parse( + new InputSource( + new ByteArrayInputStream(testPages[i].getBytes()) ), + node); + testBaseHrefURLs[i]= new URL(testBaseHrefs[i]); + } catch (Exception e) { + assertTrue("caught exception: " + e, false); + } + testDOMs[i]= node; + } + try { + answerOutlinks = new Outlink[][]{ + { + new Outlink("http://www.nutch.org", "anchor"), + }, + { + new Outlink("http://www.nutch.org/", "home"), + new Outlink("http://www.nutch.org/docs/bot.html", "bots"), + }, + { + new Outlink("http://www.nutch.org/", "separate this"), + new Outlink("http://www.nutch.org/docs/ok", "from this"), + }, + { + new Outlink("http://www.nutch.org/", "home"), + new Outlink("http://www.nutch.org/docs/1", "1"), + new Outlink("http://www.nutch.org/docs/2", "2"), + }, + { + new Outlink("http://www.nutch.org/frames/top.html", ""), + new Outlink("http://www.nutch.org/frames/left.html", ""), + new Outlink("http://www.nutch.org/frames/invalid.html", ""), + new Outlink("http://www.nutch.org/frames/right.html", ""), + }, + { + new Outlink("http://www.nutch.org/maps/logo.gif", ""), + new Outlink("http://www.nutch.org/index.html", ""), + new Outlink("http://www.nutch.org/maps/#bottom", ""), + new Outlink("http://www.nutch.org/bot.html", ""), + new Outlink("http://www.nutch.org/docs/index.html", ""), + }, + { + new Outlink("http://www.nutch.org/index.html", "whitespace test"), + }, + { + }, + { + new Outlink("http://www.nutch.org/dummy.jsp", "test2"), + }, + { + }, + { + new Outlink("http://www.nutch.org/;x", "anchor1"), + new Outlink("http://www.nutch.org/g;x", "anchor2"), + new Outlink("http://www.nutch.org/g;x?y#s", "anchor3") + }, + { + new Outlink("http://www.nutch.org/g;something", "anchor1"), + new Outlink("http://www.nutch.org/g;something?y#s", "anchor2"), + new Outlink("http://www.nutch.org/;something?y=1", "anchor3"), + new Outlink("http://www.nutch.org/;something?y=1#s", "anchor4"), + new Outlink("http://www.nutch.org/?y=1;somethingelse", "anchor5") + } + }; + + } catch (MalformedURLException e) { + + } + } + + private static boolean equalsIgnoreWhitespace(String s1, String s2) { + StringTokenizer st1= new StringTokenizer(s1); + StringTokenizer st2= new StringTokenizer(s2); + + while (st1.hasMoreTokens()) { + if (!st2.hasMoreTokens()) + return false; + if ( ! st1.nextToken().equals(st2.nextToken()) ) + return false; + } + if (st2.hasMoreTokens()) + return false; + return true; + } + + public void testGetText() { + if (testDOMs[0] == null) + setup(); + for (int i= 0; i < testPages.length; i++) { + StringBuilder sb= new StringBuilder(); + utils.getText(sb, testDOMs[i]); + String text= sb.toString(); + assertTrue("expecting text: " + answerText[i] + + System.getProperty("line.separator") + + System.getProperty("line.separator") + + "got text: "+ text, + equalsIgnoreWhitespace(answerText[i], text)); + } + } + + public void testGetTitle() { + if (testDOMs[0] == null) + setup(); + for (int i= 0; i < testPages.length; i++) { + StringBuilder sb= new StringBuilder(); + utils.getTitle(sb, testDOMs[i]); + String text= sb.toString(); + assertTrue("expecting text: " + answerText[i] + + System.getProperty("line.separator") + + System.getProperty("line.separator") + + "got text: "+ text, + equalsIgnoreWhitespace(answerTitle[i], text)); + } + } + + public void testGetOutlinks() { + if (testDOMs[0] == null) + setup(); + for (int i= 0; i < testPages.length; i++) { + ArrayList<Outlink> outlinks= new ArrayList<Outlink>(); + if (i == SKIP) { + conf.setBoolean("parser.html.form.use_action", false); + utils.setConf(conf); + } else { + conf.setBoolean("parser.html.form.use_action", true); + utils.setConf(conf); + } + utils.getOutlinks(testBaseHrefURLs[i], outlinks, testDOMs[i]); + Outlink[] outlinkArr= new Outlink[outlinks.size()]; + outlinkArr= outlinks.toArray(outlinkArr); + compareOutlinks(answerOutlinks[i], outlinkArr); + } + } + + private static final void appendOutlinks(StringBuffer sb, Outlink[] o) { + for (int i= 0; i < o.length; i++) { + sb.append(o[i].toString()); + sb.append(System.getProperty("line.separator")); + } + } + + private static final String outlinksString(Outlink[] o) { + StringBuffer sb= new StringBuffer(); + appendOutlinks(sb, o); + return sb.toString(); + } + + private static final void compareOutlinks(Outlink[] o1, Outlink[] o2) { + if (o1.length != o2.length) { + assertTrue("got wrong number of outlinks (expecting " + o1.length + + ", got " + o2.length + ")" + + System.getProperty("line.separator") + + "answer: " + System.getProperty("line.separator") + + outlinksString(o1) + + System.getProperty("line.separator") + + "got: " + System.getProperty("line.separator") + + outlinksString(o2) + + System.getProperty("line.separator"), + false + ); + } + + for (int i= 0; i < o1.length; i++) { + if (!o1[i].equals(o2[i])) { + assertTrue("got wrong outlinks at position " + i + + System.getProperty("line.separator") + + "answer: " + System.getProperty("line.separator") + + "'" + o1[i].getToUrl() + "', anchor: '" + o1[i].getAnchor() + "'" + + System.getProperty("line.separator") + + "got: " + System.getProperty("line.separator") + + "'" + o2[i].getToUrl() + "', anchor: '" + o2[i].getAnchor() + "'", + false + ); + + } + } + } +} Added: nutch/branches/nutchbase/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestRobotsMetaProcessor.java URL: http://svn.apache.org/viewvc/nutch/branches/nutchbase/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestRobotsMetaProcessor.java?rev=982184&view=auto ============================================================================== --- nutch/branches/nutchbase/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestRobotsMetaProcessor.java (added) +++ nutch/branches/nutchbase/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestRobotsMetaProcessor.java Wed Aug 4 10:01:08 2010 @@ -0,0 +1,182 @@ +/** + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.nutch.parse.html; + +import junit.framework.TestCase; + +import org.apache.nutch.parse.HTMLMetaTags; + +import java.io.ByteArrayInputStream; +import java.net.URL; + +import org.cyberneko.html.parsers.*; +import org.xml.sax.*; +import org.w3c.dom.*; +import org.apache.html.dom.*; + +/** Unit tests for HTMLMetaProcessor. */ +public class TestRobotsMetaProcessor extends TestCase { + public TestRobotsMetaProcessor(String name) { + super(name); + } + + /* + + some sample tags: + + <meta name="robots" content="index,follow"> + <meta name="robots" content="noindex,follow"> + <meta name="robots" content="index,nofollow"> + <meta name="robots" content="noindex,nofollow"> + + <META HTTP-EQUIV="Pragma" CONTENT="no-cache"> + + */ + + + public static String[] tests= + { + "<html><head><title>test page</title>" + + "<META NAME=\"ROBOTS\" CONTENT=\"NONE\"> " + + "<META HTTP-EQUIV=\"PRAGMA\" CONTENT=\"NO-CACHE\"> " + + "</head><body>" + + " some text" + + "</body></html>", + + "<html><head><title>test page</title>" + + "<meta name=\"robots\" content=\"all\"> " + + "<meta http-equiv=\"pragma\" content=\"no-cache\"> " + + "</head><body>" + + " some text" + + "</body></html>", + + "<html><head><title>test page</title>" + + "<MeTa NaMe=\"RoBoTs\" CoNtEnT=\"nOnE\"> " + + "<MeTa HtTp-EqUiV=\"pRaGmA\" cOnTeNt=\"No-CaChE\"> " + + "</head><body>" + + " some text" + + "</body></html>", + + "<html><head><title>test page</title>" + + "<meta name=\"robots\" content=\"none\"> " + + "</head><body>" + + " some text" + + "</body></html>", + + "<html><head><title>test page</title>" + + "<meta name=\"robots\" content=\"noindex,nofollow\"> " + + "</head><body>" + + " some text" + + "</body></html>", + + "<html><head><title>test page</title>" + + "<meta name=\"robots\" content=\"noindex,follow\"> " + + "</head><body>" + + " some text" + + "</body></html>", + + "<html><head><title>test page</title>" + + "<meta name=\"robots\" content=\"index,nofollow\"> " + + "</head><body>" + + " some text" + + "</body></html>", + + "<html><head><title>test page</title>" + + "<meta name=\"robots\" content=\"index,follow\"> " + + "<base href=\"http://www.nutch.org/\">" + + "</head><body>" + + " some text" + + "</body></html>", + + "<html><head><title>test page</title>" + + "<meta name=\"robots\"> " + + "<base href=\"http://www.nutch.org/base/\">" + + "</head><body>" + + " some text" + + "</body></html>", + + }; + + public static final boolean[][] answers= { + {true, true, true}, // NONE + {false, false, true}, // all + {true, true, true}, // nOnE + {true, true, false}, // none + {true, true, false}, // noindex,nofollow + {true, false, false}, // noindex,follow + {false, true, false}, // index,nofollow + {false, false, false}, // index,follow + {false, false, false}, // missing! + }; + + private URL[][] currURLsAndAnswers; + + public void testRobotsMetaProcessor() { + DOMFragmentParser parser= new DOMFragmentParser();; + + try { + currURLsAndAnswers= new URL[][] { + {new URL("http://www.nutch.org"), null}, + {new URL("http://www.nutch.org"), null}, + {new URL("http://www.nutch.org"), null}, + {new URL("http://www.nutch.org"), null}, + {new URL("http://www.nutch.org"), null}, + {new URL("http://www.nutch.org"), null}, + {new URL("http://www.nutch.org"), null}, + {new URL("http://www.nutch.org/foo/"), + new URL("http://www.nutch.org/")}, + {new URL("http://www.nutch.org"), + new URL("http://www.nutch.org/base/")} + }; + } catch (Exception e) { + assertTrue("couldn't make test URLs!", false); + } + + for (int i= 0; i < tests.length; i++) { + byte[] bytes= tests[i].getBytes(); + + DocumentFragment node = new HTMLDocumentImpl().createDocumentFragment(); + + try { + parser.parse(new InputSource(new ByteArrayInputStream(bytes)), node); + } catch (Exception e) { + e.printStackTrace(); + } + + HTMLMetaTags robotsMeta= new HTMLMetaTags(); + HTMLMetaProcessor.getMetaTags(robotsMeta, node, + currURLsAndAnswers[i][0]); + + assertTrue("got index wrong on test " + i, + robotsMeta.getNoIndex() == answers[i][0]); + assertTrue("got follow wrong on test " + i, + robotsMeta.getNoFollow() == answers[i][1]); + assertTrue("got cache wrong on test " + i, + robotsMeta.getNoCache() == answers[i][2]); + assertTrue("got base href wrong on test " + i + " (got " + + robotsMeta.getBaseHref() + ")", + ( (robotsMeta.getBaseHref() == null) + && (currURLsAndAnswers[i][1] == null) ) + || ( (robotsMeta.getBaseHref() != null) + && robotsMeta.getBaseHref().equals( + currURLsAndAnswers[i][1]) ) ); + + } + } + +}
