<?xml version="1.0" encoding="UTF-8"?>
<oai_dc:dc xmlns:oai_dc="http://www.openarchives.org/OAI/2.0/oai_dc/" xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:schemaLocation="http://www.openarchives.org/OAI/2.0/oai_dc/ http://www.openarchives.org/OAI/2.0/oai_dc.xsd">
  <dc:title>Interface to the Boilerpipe Java Library</dc:title>
  <dc:title>R package boilerpipeR version 1.3.2</dc:title>
  <dc:subject>CRAN Task View: NaturalLanguageProcessing (https://CRAN.R-project.org/view=NaturalLanguageProcessing)</dc:subject>
  <dc:subject>CRAN Task View: WebTechnologies (https://CRAN.R-project.org/view=WebTechnologies)</dc:subject>
  <dc:description>Generic Extraction of main text content from HTML files; removal
    of ads, sidebars and headers using the boilerpipe 
    &lt;https://github.com/kohlschutter/boilerpipe&gt; Java library. The
    extraction heuristics from boilerpipe show a robust performance for a wide
    range of web site templates.</dc:description>
  <dc:type>Software</dc:type>
  <dc:relation>Imports: rJava</dc:relation>
  <dc:relation>Suggests: RCurl</dc:relation>
  <dc:creator>Mario Annau &lt;mario.annau@gmail.com&gt;</dc:creator>
  <dc:publisher>Comprehensive R Archive Network (CRAN)</dc:publisher>
  <dc:contributor>See AUTHORS file.</dc:contributor>
  <dc:rights>Apache License (== 2.0)</dc:rights>
  <dc:date>2021-05-19</dc:date>
  <dc:format>application/tgz</dc:format>
  <dc:identifier>https://CRAN.R-project.org/package=boilerpipeR</dc:identifier>
  <dc:identifier>doi:10.32614/CRAN.package.boilerpipeR</dc:identifier>
</oai_dc:dc>
