Examples of org.apache.nutch.parse.ParseUtil

org.apache.nutch.parse.ParseUtil
A Utility class containing methods to simply perform parsing utilities such as iterating through a preferred list of {@link Parser}s to obtain {@link Parse} objects. @author mattmann @author Jérôme Charron @author Sébastien Le Callonnec


  public String getTextContent(String fileName) throws ProtocolException, ParseException {
    String urlString = "file:" + sampleDir + fileSeparator + fileName;
    Protocol protocol = new ProtocolFactory(conf).getProtocol(urlString);
    Content content = protocol.getProtocolOutput(new Text(urlString), new CrawlDatum()).getContent();
    Parse parse = new ParseUtil(conf).parseByExtensionId("parse-tika", content).get(content.getUrl());
    return parse.getText();
  }

View Full Code Here


      if (sampleFiles[i].startsWith("ootest")==false) continue;
      
      protocol = factory.getProtocol(urlString);
      content = protocol.getProtocolOutput(new Text(urlString), new CrawlDatum()).getContent();
      parse = new ParseUtil(conf).parseByExtensionId("parse-tika", content).get(content.getUrl());
      
      String text = parse.getText().replaceAll("[ \t\r\n]+", " ").trim();


      // simply test for the presence of a text - the ordering of the elements may differ from what was expected
      // in the previous tests

View Full Code Here

    Configuration conf = NutchConfiguration.create();
    urlString = "file:" + sampleDir + fileSeparator + rtfFile;
    protocol = new ProtocolFactory(conf).getProtocol(urlString);
    content = protocol.getProtocolOutput(new Text(urlString),
        new CrawlDatum()).getContent();
    parse = new ParseUtil(conf).parseByExtensionId("parse-tika", content)
        .get(content.getUrl());
    String text = parse.getText();
    assertEquals("The quick brown fox jumps over the lazy dog", text.trim());


    String title = parse.getData().getTitle();

View Full Code Here

      urlString = "file:" + sampleDir + fileSeparator + sampleFiles[i];


      Configuration conf = NutchConfiguration.create();
      protocol = new ProtocolFactory(conf).getProtocol(urlString);
      content = protocol.getProtocolOutput(new Text(urlString), new CrawlDatum()).getContent();
      parse = new ParseUtil(conf).parseByExtensionId("parse-tika", content).get(content.getUrl());


      int index = parse.getText().indexOf(expectedText);
      assertTrue(index > 0);
    }
  }

View Full Code Here

      urlString = "file:" + sampleDir + fileSeparator + sampleFiles[i];


      protocol = new ProtocolFactory(conf).getProtocol(urlString);
      content = protocol.getProtocolOutput(new Text(urlString),
          new CrawlDatum()).getContent();
      parse = new ParseUtil(conf).parseByExtensionId("parse-tika",
          content).get(content.getUrl());


      // check that there are 2 outlinks:
      // unlike the original parse-rss
      // tika ignores the URL and description of the channel

View Full Code Here

   * Test parsing of language identifiers from html 
   **/
  public void testMetaHTMLParsing() {


    try {
      ParseUtil parser = new ParseUtil(NutchConfiguration.create());
      /* loop through the test documents and validate result */
      for (int t = 0; t < docs.length; t++) {
        Content content = getContent(docs[t]);
        Parse parse = parser.parse(content).get(content.getUrl());
        assertEquals(metalanguages[t], (String) parse.getData().getParseMeta().get(Metadata.LANGUAGE));
      }
    } catch (Exception e) {
      e.printStackTrace(System.out);
      fail(e.toString());

View Full Code Here

  private static String getUrlContent(String url, Configuration conf) {
    Protocol protocol;
    try {
      protocol = new ProtocolFactory(conf).getProtocol(url);
      Content content = protocol.getProtocolOutput(new Text(url), new CrawlDatum()).getContent();
      Parse parse = new ParseUtil(conf).parse(content).get(content.getUrl());
      System.out.println("text:" + parse.getText());
      return parse.getText();


    } catch (ProtocolNotFound e) {
      e.printStackTrace();

View Full Code Here

    byte[] bytes = out.toByteArray();
    Configuration conf = NutchConfiguration.create();


    Content content =
      new Content(url, url, bytes, contentType, new Metadata(), conf);
    Parse parse =  new ParseUtil(conf).parse(content).get(content.getUrl());
    
    Metadata metadata = parse.getData().getParseMeta();
    assertEquals(license, metadata.get("License-Url"));
    assertEquals(location, metadata.get("License-Location"));
    assertEquals(type, metadata.get("Work-Type"));

View Full Code Here

    if (LOG.isInfoEnabled()) {
      LOG.info("parsing: "+url);
      LOG.info("contentType: "+contentType);
    }


    ParseResult parseResult = new ParseUtil(conf).parse(content);


    for (java.util.Map.Entry<Text, Parse> entry : parseResult) {
      Parse parse = entry.getValue();
      System.out.print("---------\nUrl\n---------------\n");
      System.out.print(entry.getKey());

View Full Code Here


      protocol = new ProtocolFactory(conf).getProtocol(urlString);
      content = protocol.getProtocolOutput(new Text(urlString),
          new CrawlDatum()).getContent();


      parseResult = new ParseUtil(conf).parseByExtensionId("feed", content);


      assertEquals(3, parseResult.size());


      boolean hasLink1 = false, hasLink2 = false, hasLink3=false;

View Full Code Here

0 1 2 3 4 5 6 7 8 9

TOP

Related Classes of org.apache.nutch.parse.ParseUtil

com.google.common.util.concurrent.ThreadFactoryBuilder

org.apache.avro.util.Utf8

org.apache.nutch.analysis.lang.LanguageIdentifier

org.apache.nutch.analysis.lang.TestHTMLLanguageParser

org.apache.nutch.crawl.URLWebPage

org.apache.nutch.fetcher.FetcherReducer

org.apache.nutch.indexer.IndexingFiltersChecker

org.apache.nutch.microformats.reltag.TestRelTagParser

org.apache.nutch.net.URLFilters

org.apache.nutch.net.URLNormalizers

All source code are property of their respective owners. Java is a trademark of Sun Microsystems, Inc and owned by ORACLE Inc. Contact coftware#gmail.com.