diff --git a/THIRD-PARTY.txt b/THIRD-PARTY.txt index cdbe76f03..79698f511 100644 --- a/THIRD-PARTY.txt +++ b/THIRD-PARTY.txt @@ -22,7 +22,7 @@ List of third-party dependencies grouped by their license type. * Apache Avro (org.apache.avro:avro:1.12.1 - https://avro.apache.org) * Apache Commons CLI (commons-cli:commons-cli:1.11.0 - https://commons.apache.org/proper/commons-cli/) * Apache Commons Codec (commons-codec:commons-codec:1.22.1 - https://commons.apache.org/proper/commons-codec/) - * Apache Commons Collections (org.apache.commons:commons-collections4:4.5.0 - https://commons.apache.org/proper/commons-collections/) + * Apache Commons Collections (org.apache.commons:commons-collections4:4.6.0 - https://commons.apache.org/proper/commons-collections/) * Apache Commons Compress (org.apache.commons:commons-compress:1.28.0 - https://commons.apache.org/proper/commons-compress/) * Apache Commons CSV (org.apache.commons:commons-csv:1.14.1 - https://commons.apache.org/proper/commons-csv/) * Apache Commons Exec (org.apache.commons:commons-exec:1.6.0 - https://commons.apache.org/proper/commons-exec/) @@ -49,7 +49,6 @@ List of third-party dependencies grouped by their license type. * Apache HttpCore NIO (org.apache.httpcomponents:httpcore-nio:4.4.16 - http://hc.apache.org/httpcomponents-core-ga) * Apache James :: Mime4j :: Core (org.apache.james:apache-mime4j-core:0.8.14 - http://james.apache.org/mime4j/apache-mime4j-core) * Apache James :: Mime4j :: DOM (org.apache.james:apache-mime4j-dom:0.8.14 - http://james.apache.org/mime4j/apache-mime4j-dom) - * Apache JempBox (org.apache.pdfbox:jempbox:1.8.17 - http://www.apache.org/pdfbox-parent/jempbox/) * Apache Log4j API (org.apache.logging.log4j:log4j-api:2.26.0 - https://logging.apache.org/log4j/2.x/) * Apache Log4j Core (org.apache.logging.log4j:log4j-core:2.26.0 - https://logging.apache.org/log4j/2.x/) * Apache Log4j JUL Adapter (org.apache.logging.log4j:log4j-jul:2.25.4 - https://logging.apache.org/log4j/2.x/) @@ -73,37 +72,40 @@ List of third-party dependencies grouped by their license type. * Apache POI (org.apache.poi:poi-scratchpad:5.5.1 - https://poi.apache.org/) * Apache POI - API based on OPC and OOXML schemas (org.apache.poi:poi-ooxml:5.5.1 - https://poi.apache.org/) * Apache POI - Common (org.apache.poi:poi:5.5.1 - https://poi.apache.org/) - * Apache POI - OOXML schemas (full) (org.apache.poi:poi-ooxml-full:5.5.1 - https://poi.apache.org/) * Apache Solr (module: api) (org.apache.solr:solr-api:10.0.0 - https://solr.apache.org/) * Apache Solr (module: solrj) (org.apache.solr:solr-solrj:10.0.0 - https://solr.apache.org/) * Apache Solr (module: solrj-jetty) (org.apache.solr:solr-solrj-jetty:10.0.0 - https://solr.apache.org/) * Apache Solr (module: solrj-zookeeper) (org.apache.solr:solr-solrj-zookeeper:10.0.0 - https://solr.apache.org/) - * Apache Tika Apple parser module (org.apache.tika:tika-parser-apple-module:3.3.2 - https://tika.apache.org/tika-parser-apple-module/) - * Apache Tika audiovideo parser module (org.apache.tika:tika-parser-audiovideo-module:3.3.2 - https://tika.apache.org/tika-parser-audiovideo-module/) - * Apache Tika cad parser module (org.apache.tika:tika-parser-cad-module:3.3.2 - https://tika.apache.org/tika-parser-cad-module/) - * Apache Tika code parser module (org.apache.tika:tika-parser-code-module:3.3.2 - https://tika.apache.org/tika-parser-code-module/) - * Apache Tika core (org.apache.tika:tika-core:3.3.2 - https://tika.apache.org/) - * Apache Tika crypto parser module (org.apache.tika:tika-parser-crypto-module:3.3.2 - https://tika.apache.org/tika-parser-crypto-module/) - * Apache Tika digest commons (org.apache.tika:tika-parser-digest-commons:3.3.2 - https://tika.apache.org/tika-parser-digest-commons/) - * Apache Tika font parser module (org.apache.tika:tika-parser-font-module:3.3.2 - https://tika.apache.org/tika-parser-font-module/) - * Apache Tika html parser module (org.apache.tika:tika-parser-html-module:3.3.2 - https://tika.apache.org/tika-parser-html-module/) - * Apache Tika image parser module (org.apache.tika:tika-parser-image-module:3.3.2 - https://tika.apache.org/tika-parser-image-module/) - * Apache Tika mail commons (org.apache.tika:tika-parser-mail-commons:3.3.2 - https://tika.apache.org/tika-parser-mail-commons/) - * Apache Tika mail parser module (org.apache.tika:tika-parser-mail-module:3.3.2 - https://tika.apache.org/tika-parser-mail-module/) - * Apache Tika Microsoft parser module (org.apache.tika:tika-parser-microsoft-module:3.3.2 - https://tika.apache.org/tika-parser-microsoft-module/) - * Apache Tika miscellaneous office format parser module (org.apache.tika:tika-parser-miscoffice-module:3.3.2 - https://tika.apache.org/tika-parser-miscoffice-module/) - * Apache Tika news parser module (org.apache.tika:tika-parser-news-module:3.3.2 - https://tika.apache.org/tika-parser-news-module/) - * Apache Tika OCR parser module (org.apache.tika:tika-parser-ocr-module:3.3.2 - https://tika.apache.org/tika-parser-ocr-module/) - * Apache Tika package parser module (org.apache.tika:tika-parser-pkg-module:3.3.2 - https://tika.apache.org/tika-parser-pkg-module/) - * Apache Tika PDF parser module (org.apache.tika:tika-parser-pdf-module:3.3.2 - https://tika.apache.org/tika-parser-pdf-module/) - * Apache Tika plugin for Ogg, Vorbis and FLAC (org.gagravarr:vorbis-java-tika:0.8 - https://github.com/Gagravarr/VorbisJava) - * Apache Tika standard parser package (org.apache.tika:tika-parsers-standard-package:3.3.2 - https://tika.apache.org/tika-parsers/tika-parsers-standard/tika-parsers-standard-package/) - * Apache Tika text parser module (org.apache.tika:tika-parser-text-module:3.3.2 - https://tika.apache.org/tika-parser-text-module/) - * Apache Tika WARC parser module (org.apache.tika:tika-parser-webarchive-module:3.3.2 - https://tika.apache.org/tika-parser-webarchive-module/) - * Apache Tika XML parser module (org.apache.tika:tika-parser-xml-module:3.3.2 - https://tika.apache.org/tika-parser-xml-module/) - * Apache Tika XMP commons (org.apache.tika:tika-parser-xmp-commons:3.3.2 - https://tika.apache.org/tika-parser-xmp-commons/) - * Apache Tika ZIP commons (org.apache.tika:tika-parser-zip-commons:3.3.2 - https://tika.apache.org/tika-parser-zip-commons/) - * Apache XmpBox (org.apache.pdfbox:xmpbox:3.0.8 - https://www.apache.org/pdfbox-parent/xmpbox/) + * Apache Tika Apple parser module (org.apache.tika:tika-parser-apple-module:4.0.0 - https://tika.apache.org/tika-parser-apple-module/) + * Apache Tika audiovideo parser module (org.apache.tika:tika-parser-audiovideo-module:4.0.0 - https://tika.apache.org/tika-parser-audiovideo-module/) + * Apache Tika cad parser module (org.apache.tika:tika-parser-cad-module:4.0.0 - https://tika.apache.org/tika-parser-cad-module/) + * Apache Tika code parser module (org.apache.tika:tika-parser-code-module:4.0.0 - https://tika.apache.org/tika-parser-code-module/) + * Apache Tika core (org.apache.tika:tika-core:4.0.0 - https://tika.apache.org/) + * Apache Tika crypto parser module (org.apache.tika:tika-parser-crypto-module:4.0.0 - https://tika.apache.org/tika-parser-crypto-module/) + * Apache Tika data URI commons (org.apache.tika:tika-parser-datauri-commons:4.0.0 - https://tika.apache.org/tika-parser-datauri-commons/) + * Apache Tika digest commons (org.apache.tika:tika-parser-digest-commons:4.0.0 - https://tika.apache.org/tika-parser-digest-commons/) + * Apache Tika font parser module (org.apache.tika:tika-parser-font-module:4.0.0 - https://tika.apache.org/tika-parser-font-module/) + * Apache Tika HTML encoding detector (org.apache.tika:tika-encoding-detector-html:4.0.0 - https://tika.apache.org/tika-encoding-detectors/tika-encoding-detector-html/) + * Apache Tika html parser module (org.apache.tika:tika-parser-html-module:4.0.0 - https://tika.apache.org/tika-parser-html-module/) + * Apache Tika image parser module (org.apache.tika:tika-parser-image-module:4.0.0 - https://tika.apache.org/tika-parser-image-module/) + * Apache Tika mail commons (org.apache.tika:tika-parser-mail-commons:4.0.0 - https://tika.apache.org/tika-parser-mail-commons/) + * Apache Tika mail parser module (org.apache.tika:tika-parser-mail-module:4.0.0 - https://tika.apache.org/tika-parser-mail-module/) + * Apache Tika Microsoft parser module (org.apache.tika:tika-parser-microsoft-module:4.0.0 - https://tika.apache.org/tika-parser-microsoft-module/) + * Apache Tika miscellaneous office format parser module (org.apache.tika:tika-parser-miscoffice-module:4.0.0 - https://tika.apache.org/tika-parser-miscoffice-module/) + * Apache Tika ML charset encoding detector (org.apache.tika:tika-encoding-detector-mojibuster:4.0.0 - https://tika.apache.org/tika-encoding-detectors/tika-encoding-detector-mojibuster/) + * Apache Tika ML core (no Tika dependencies) (org.apache.tika:tika-ml-core:4.0.0 - https://tika.apache.org/tika-ml-core/) + * Apache Tika ML junk detector — runtime and training tools (org.apache.tika:tika-ml-junkdetect:4.0.0 - https://tika.apache.org/tika-ml-junkdetect/) + * Apache Tika news parser module (org.apache.tika:tika-parser-news-module:4.0.0 - https://tika.apache.org/tika-parser-news-module/) + * Apache Tika OCR parser module (org.apache.tika:tika-parser-ocr-module:4.0.0 - https://tika.apache.org/tika-parser-ocr-module/) + * Apache Tika package parser module (org.apache.tika:tika-parser-pkg-module:4.0.0 - https://tika.apache.org/tika-parser-pkg-module/) + * Apache Tika PDF parser module (org.apache.tika:tika-parser-pdf-module:4.0.0 - https://tika.apache.org/tika-parser-pdf-module/) + * Apache Tika serialization (org.apache.tika:tika-serialization:4.0.0 - https://tika.apache.org) + * Apache Tika standard parser package (org.apache.tika:tika-parsers-standard-package:4.0.0 - https://tika.apache.org/tika-parsers/tika-parsers-standard/tika-parsers-standard-package/) + * Apache Tika text parser module (org.apache.tika:tika-parser-text-module:4.0.0 - https://tika.apache.org/tika-parser-text-module/) + * Apache Tika WARC parser module (org.apache.tika:tika-parser-webarchive-module:4.0.0 - https://tika.apache.org/tika-parser-webarchive-module/) + * Apache Tika XML parser module (org.apache.tika:tika-parser-xml-module:4.0.0 - https://tika.apache.org/tika-parser-xml-module/) + * Apache Tika XMP commons (org.apache.tika:tika-parser-xmp-commons:4.0.0 - https://tika.apache.org/tika-parser-xmp-commons/) + * Apache Tika ZIP commons (org.apache.tika:tika-parser-zip-commons:4.0.0 - https://tika.apache.org/tika-parser-zip-commons/) * Apache ZooKeeper - Jute (org.apache.zookeeper:zookeeper-jute:3.9.4 - http://zookeeper.apache.org/zookeeper-jute) * Apache ZooKeeper - Server (org.apache.zookeeper:zookeeper:3.9.4 - http://zookeeper.apache.org/zookeeper) * AWS Event Stream (software.amazon.eventstream:eventstream:1.0.1 - https://github.com/awslabs/aws-eventstream-java) @@ -141,7 +143,7 @@ List of third-party dependencies grouped by their license type. * AWS Java SDK :: Utilities (software.amazon.awssdk:utils:2.54.6 - https://aws.amazon.com/sdkforjava/utils) * AWS Java SDK :: Utils Lite (software.amazon.awssdk:utils-lite:2.54.6 - https://aws.amazon.com/sdkforjava) * Caffeine cache (com.github.ben-manes.caffeine:caffeine:3.2.4 - https://github.com/ben-manes/caffeine) - * com.drewnoakes:metadata-extractor (com.drewnoakes:metadata-extractor:2.20.0 - https://drewnoakes.com/code/exif/) + * com.drewnoakes:metadata-extractor (com.drewnoakes:metadata-extractor:2.21.0 - https://drewnoakes.com/code/exif/) * compiler (com.github.spullara.mustache.java:compiler:0.9.14 - http://github.com/spullara/mustache.java) * Crawler-commons (com.github.crawler-commons:crawler-commons:1.6 - https://github.com/crawler-commons/crawler-commons) * Curator Client (org.apache.curator:curator-client:5.9.0 - https://curator.apache.org/curator-client) @@ -305,7 +307,13 @@ List of third-party dependencies grouped by their license type. * Bouncy Castle JavaMail Jakarta S/MIME APIs (org.bouncycastle:bcjmail-jdk18on:1.85 - https://www.bouncycastle.org/download/bouncy-castle-java/) * Bouncy Castle PKIX, CMS, EAC, TSP, PKCS, OCSP, CMP, and CRMF APIs (org.bouncycastle:bcpkix-jdk18on:1.84 - https://www.bouncycastle.org/download/bouncy-castle-java/) * Bouncy Castle Provider (org.bouncycastle:bcprov-jdk18on:1.84 - https://www.bouncycastle.org/download/bouncy-castle-java/) - * Bouncy Castle Provider (org.bouncycastle:bcprov-jdk18on:1.85 - https://www.bouncycastle.org/download/bouncy-castle-java/) + * Bouncy Castle Provider (org.bouncycastle:bcprov-jdk18on:1.85.2 - https://www.bouncycastle.org/download/bouncy-castle-java/) + + BSD-2-Clause + + * commonmark-java core (org.commonmark:commonmark:0.30.0 - https://github.com/commonmark/commonmark-java/commonmark) + * commonmark-java extension for strikethrough (org.commonmark:commonmark-ext-gfm-strikethrough:0.30.0 - https://github.com/commonmark/commonmark-java/commonmark-ext-gfm-strikethrough) + * commonmark-java extension for tables (org.commonmark:commonmark-ext-gfm-tables:0.30.0 - https://github.com/commonmark/commonmark-java/commonmark-ext-gfm-tables) BSD-2-Clause, Public Domain, per Creative Commons CC0 @@ -331,7 +339,7 @@ List of third-party dependencies grouped by their license type. CDDL, v1.0, LGPL, v2.1 or later - * JHighlight (org.codelibs:jhighlight:1.1.1 - https://github.com/codelibs/jhighlight) + * JHighlight (org.codelibs:jhighlight:2.0.0 - https://github.com/codelibs/jhighlight) CDDL/GPLv2+CE @@ -375,10 +383,6 @@ List of third-party dependencies grouped by their license type. * JSON-B API (jakarta.json.bind:jakarta.json.bind-api:2.0.0 - https://eclipse-ee4j.github.io/jsonb-api) * JSON-P Default Provider (org.glassfish:jakarta.json:2.0.0 - https://github.com/eclipse-ee4j/jsonp) - GENERAL PUBLIC LICENSE, version 3 (GPL-3.0), GNU LESSER GENERAL PUBLIC LICENSE, version 3 (LGPL-3.0), Mozilla Public License Version 1.1 - - * juniversalchardet (com.github.albfernandez:juniversalchardet:2.5.0 - https://github.com/albfernandez/juniversalchardet) - MIT-0 * reactive-streams (org.reactivestreams:reactive-streams:1.0.4 - http://www.reactive-streams.org/) @@ -386,7 +390,7 @@ List of third-party dependencies grouped by their license type. MIT License * Animal Sniffer Annotations (org.codehaus.mojo:animal-sniffer-annotations:1.24 - https://www.mojohaus.org/animal-sniffer/animal-sniffer-annotations) - * dd-plist (com.googlecode.plist:dd-plist:1.29 - http://www.github.com/3breadt/dd-plist) + * dd-plist (com.googlecode.plist:dd-plist:1.30 - http://www.github.com/3breadt/dd-plist) * JOpt Simple (net.sf.jopt-simple:jopt-simple:5.0.4 - http://jopt-simple.github.io/jopt-simple) * jsoup Java HTML Parser (org.jsoup:jsoup:1.23.2 - https://jsoup.org/) * JTokkit (com.knuddels:jtokkit:1.1.0 - https://github.com/knuddelsgmbh/jtokkit) @@ -415,4 +419,4 @@ List of third-party dependencies grouped by their license type. UnRar License - * Java Unrar (com.github.junrar:junrar:7.6.0 - https://github.com/junrar/junrar) + * Java Unrar (com.github.junrar:junrar:8.1.0 - https://github.com/junrar/junrar) diff --git a/archetype/src/main/resources/archetype-resources/crawler-conf.yaml b/archetype/src/main/resources/archetype-resources/crawler-conf.yaml index bc57e3714..bcae9b229 100644 --- a/archetype/src/main/resources/archetype-resources/crawler-conf.yaml +++ b/archetype/src/main/resources/archetype-resources/crawler-conf.yaml @@ -129,7 +129,7 @@ config: - application/.*pdf.* # Tika parser configuration file - parse.tika.config.file: "tika-config.xml" + parser.tika.config.file: "tika-config.json" # custom fetch interval to be used when a document has the key/value in its metadata # and has been fetched successfully (value in minutes) diff --git a/core/src/main/java/org/apache/stormcrawler/bolt/JSoupParserBolt.java b/core/src/main/java/org/apache/stormcrawler/bolt/JSoupParserBolt.java index 1b84a4652..43a591c74 100644 --- a/core/src/main/java/org/apache/stormcrawler/bolt/JSoupParserBolt.java +++ b/core/src/main/java/org/apache/stormcrawler/bolt/JSoupParserBolt.java @@ -19,9 +19,7 @@ import static org.apache.stormcrawler.Constants.StatusStreamName; -import java.io.ByteArrayInputStream; import java.io.IOException; -import java.io.InputStream; import java.lang.reflect.InvocationTargetException; import java.net.MalformedURLException; import java.net.URL; @@ -62,10 +60,12 @@ import org.apache.stormcrawler.util.RefreshTag; import org.apache.stormcrawler.util.RobotsTags; import org.apache.stormcrawler.util.URLUtil; -import org.apache.tika.config.TikaConfig; +import org.apache.tika.detect.DefaultDetector; import org.apache.tika.detect.Detector; +import org.apache.tika.io.TikaInputStream; import org.apache.tika.metadata.TikaCoreProperties; import org.apache.tika.mime.MediaType; +import org.apache.tika.parser.ParseContext; import org.jsoup.nodes.Element; import org.jsoup.parser.Parser; import org.jsoup.select.Elements; @@ -89,7 +89,7 @@ public class JSoupParserBolt extends StatusEmitterBolt { private JSoupFilter jsoupFilters = null; - private final Detector detector = TikaConfig.getDefaultConfig().getDetector(); + private final Detector detector = new DefaultDetector(); private boolean detectMimeType = true; @@ -559,17 +559,18 @@ public String guessMimeType(String url, String httpContentType, byte[] content) if (StringUtils.isNotBlank(httpContentType)) { // pass content type from server as a clue - metadata.set(org.apache.tika.metadata.Metadata.CONTENT_TYPE, httpContentType); + metadata.set(org.apache.tika.metadata.HttpHeaders.CONTENT_TYPE, httpContentType); } // use full URL as a clue metadata.set(TikaCoreProperties.RESOURCE_NAME_KEY, url); metadata.set( - org.apache.tika.metadata.Metadata.CONTENT_LENGTH, Integer.toString(content.length)); + org.apache.tika.metadata.HttpHeaders.CONTENT_LENGTH, + Integer.toString(content.length)); - try (InputStream stream = new ByteArrayInputStream(content)) { - MediaType mt = detector.detect(stream, metadata); + try (TikaInputStream stream = TikaInputStream.get(content)) { + MediaType mt = detector.detect(stream, metadata, new ParseContext()); return mt.toString(); } catch (IOException e) { throw new IllegalStateException("Unexpected IOException", e); diff --git a/docs/src/main/asciidoc/configuration.adoc b/docs/src/main/asciidoc/configuration.adoc index d5996a6fc..af8cf9735 100644 --- a/docs/src/main/asciidoc/configuration.adoc +++ b/docs/src/main/asciidoc/configuration.adoc @@ -530,7 +530,7 @@ See the link:https://github.com/apache/stormcrawler/tree/main/external/tika[tika | key | default value | description | parser.mimetype.whitelist | - | Regex list of allowed MIME types. If set, only matching types are parsed by the Tika ParserBolt. -| parser.tika.config.file | tika-config.xml | Path to the Tika configuration XML file. +| parser.tika.config.file | tika-config.json | Name of the classpath resource holding the Tika configuration (JSON format since Tika 4). |=== NOTE: When using the Tika `ParserBolt` alongside `JSoupParserBolt`, set `jsoup.treat.non.html.as.error` to `false` so that non-HTML content is passed through to the Tika parser rather than being treated as an error. diff --git a/external/opensearch/archetype/src/main/resources/archetype-resources/crawler-conf.yaml b/external/opensearch/archetype/src/main/resources/archetype-resources/crawler-conf.yaml index f62103faf..57ce81f75 100644 --- a/external/opensearch/archetype/src/main/resources/archetype-resources/crawler-conf.yaml +++ b/external/opensearch/archetype/src/main/resources/archetype-resources/crawler-conf.yaml @@ -123,7 +123,7 @@ config: - application/.*pdf.* # Tika parser configuration file - parse.tika.config.file: "tika-config.xml" + parser.tika.config.file: "tika-config.json" # custom fetch interval to be used when a document has the key/value in its metadata # and has been fetched successfully (value in minutes) diff --git a/external/solr/archetype/src/main/resources/archetype-resources/crawler-conf.yaml b/external/solr/archetype/src/main/resources/archetype-resources/crawler-conf.yaml index f62103faf..57ce81f75 100644 --- a/external/solr/archetype/src/main/resources/archetype-resources/crawler-conf.yaml +++ b/external/solr/archetype/src/main/resources/archetype-resources/crawler-conf.yaml @@ -123,7 +123,7 @@ config: - application/.*pdf.* # Tika parser configuration file - parse.tika.config.file: "tika-config.xml" + parser.tika.config.file: "tika-config.json" # custom fetch interval to be used when a document has the key/value in its metadata # and has been fetched successfully (value in minutes) diff --git a/external/tika/README.md b/external/tika/README.md index 99d3ca6ff..d7f912d5d 100644 --- a/external/tika/README.md +++ b/external/tika/README.md @@ -31,4 +31,6 @@ The next step is to use a [RedirectionBolt](https://github.com/apache/stormcrawl ## Configure Tika -The Tika parser bolt loads a Tika configuration file from the Java classpath. The default file name (path) is `tika-config.xml` and can be changed by the configuration `parser.tika.config.file`. See [configuring Tika](https://tika.apache.org/2.1.0/configuring.html) and the default configuration file [tika-config.xml](./src/main/resources/tika-config.xml). +The Tika parser bolt loads a Tika configuration file from the Java classpath. The default file name (path) is `tika-config.json` and can be changed by the configuration `parser.tika.config.file`. Since Tika 4, configurations are written in JSON instead of XML - see [configuring Tika](https://tika.apache.org/docs/4.0.x/configuration/index.html) and the default configuration file [tika-config.json](./src/main/resources/tika-config.json). + +Note that a configuration which is present on the classpath but invalid is treated as an error: the bolt fails to start instead of silently falling back to the default Tika configuration. diff --git a/external/tika/pom.xml b/external/tika/pom.xml index f1eba0922..c9d8fe4a7 100644 --- a/external/tika/pom.xml +++ b/external/tika/pom.xml @@ -47,10 +47,27 @@ under the License. + + + org.apache.tika + tika-core + ${tika.version} + + + + org.apache.tika + tika-serialization + ${tika.version} + + + org.apache.tika tika-parsers-standard-package ${tika.version} + pom asm diff --git a/external/tika/src/main/java/org/apache/stormcrawler/tika/ParserBolt.java b/external/tika/src/main/java/org/apache/stormcrawler/tika/ParserBolt.java index 3d814138e..3962ab427 100644 --- a/external/tika/src/main/java/org/apache/stormcrawler/tika/ParserBolt.java +++ b/external/tika/src/main/java/org/apache/stormcrawler/tika/ParserBolt.java @@ -19,10 +19,15 @@ import static org.apache.stormcrawler.Constants.StatusStreamName; -import java.io.ByteArrayInputStream; import java.io.IOException; +import java.io.InputStream; import java.net.MalformedURLException; +import java.net.URISyntaxException; import java.net.URL; +import java.nio.file.Files; +import java.nio.file.Path; +import java.nio.file.Paths; +import java.nio.file.StandardCopyOption; import java.util.ArrayList; import java.util.HashMap; import java.util.LinkedList; @@ -56,8 +61,11 @@ import org.apache.stormcrawler.util.MetadataTransfer; import org.apache.stormcrawler.util.URLUtil; import org.apache.tika.Tika; -import org.apache.tika.config.TikaConfig; +import org.apache.tika.config.loader.TikaLoader; +import org.apache.tika.exception.TikaConfigException; +import org.apache.tika.io.TikaInputStream; import org.apache.tika.metadata.TikaCoreProperties; +import org.apache.tika.parser.EmptyParser; import org.apache.tika.parser.ParseContext; import org.apache.tika.parser.Parser; import org.apache.tika.parser.html.HtmlMapper; @@ -77,6 +85,9 @@ public class ParserBolt extends BaseRichBolt { private Tika tika; + /** ParseContext configured from the "parse-context" section of the Tika configuration. */ + private ParseContext configuredParseContext = new ParseContext(); + private URLFilters urlFilters = null; private ParseFilter parseFilters = null; @@ -194,14 +205,13 @@ public void execute(Tuple tuple) { long start = System.currentTimeMillis(); - ByteArrayInputStream bais = new ByteArrayInputStream(content); org.apache.tika.metadata.Metadata md = new org.apache.tika.metadata.Metadata(); // provide the mime-type as a clue for guessing String httpCT = metadata.getFirstValue(HttpHeaders.CONTENT_TYPE, this.protocolMDprefix); if (StringUtils.isNotBlank(httpCT)) { // pass content type from server as a clue - md.set(org.apache.tika.metadata.Metadata.CONTENT_TYPE, httpCT); + md.set(org.apache.tika.metadata.HttpHeaders.CONTENT_TYPE, httpCT); } // as well as the filename @@ -215,10 +225,16 @@ public void execute(Tuple tuple) { LinkContentHandler linkHandler = new LinkContentHandler(); ContentHandler textHandler = new BodyContentHandler(-1); TeeContentHandler teeHandler = new TeeContentHandler(linkHandler, textHandler); - ParseContext parseContext = new ParseContext(); + // seed the context with the components configured in the + // "parse-context" section of the Tika configuration + ParseContext parseContext = createParseContext(); if (extractEmbedded) { parseContext.set(Parser.class, tika.getParser()); + } else { + // the AutoDetectParser sets itself on the context unless a parser + // is present, which would parse the embedded documents anyway + parseContext.set(Parser.class, EmptyParser.INSTANCE); } try { @@ -242,18 +258,12 @@ public void execute(Tuple tuple) { // parse String text; - try { - tika.getParser().parse(bais, teeHandler, md, parseContext); + try (TikaInputStream tis = TikaInputStream.get(content)) { + tika.getParser().parse(tis, teeHandler, md, parseContext); text = textHandler.toString(); } catch (Throwable e) { handleException(url, e, metadata, tuple, "parse error"); return; - } finally { - try { - bais.close(); - } catch (IOException e) { - LOG.error("Exception while closing stream", e); - } } // add parse md to metadata @@ -327,31 +337,63 @@ private static boolean isEmptyDocument(ParseData parseDoc) { } private Tika instantiateTika(Map conf) { - Tika tika = null; String tikaConfigFile = - ConfUtils.getString(conf, "parser.tika.config.file", "tika-config.xml"); + ConfUtils.getString(conf, "parser.tika.config.file", "tika-config.json"); long start = System.currentTimeMillis(); URL tikaConfigUrl = getClass().getClassLoader().getResource(tikaConfigFile); if (tikaConfigUrl == null) { - LOG.error("Tika configuration file {} not found on classpath", tikaConfigFile); - } else { - LOG.info("Instantiating Tika using custom configuration {}", tikaConfigUrl); - try { - TikaConfig tikaConfig = new TikaConfig(tikaConfigUrl, getClass().getClassLoader()); - tika = new Tika(tikaConfig); - } catch (Exception e) { - LOG.error( - "Failed to instantiate Tika using custom configuration {}", - tikaConfigUrl, - e); - } + // fail fast: silently falling back to the default configuration + // would activate parsers the configuration excluded + throw new IllegalStateException( + "Tika configuration file " + tikaConfigFile + " not found on classpath"); } - if (tika == null) { - LOG.info("Instantiating Tika with default configuration"); - tika = new Tika(); + LOG.info("Instantiating Tika using custom configuration {}", tikaConfigUrl); + Path configPath = null; + boolean temporary = false; + try { + if ("file".equals(tikaConfigUrl.getProtocol())) { + configPath = Paths.get(tikaConfigUrl.toURI()); + } else { + // TikaLoader can only read configurations from the filesystem: + // copy the resource to a temporary file and delete it as soon + // as the configuration has been loaded + configPath = Files.createTempFile("tika-config", ".json"); + temporary = true; + try (InputStream is = tikaConfigUrl.openStream()) { + Files.copy(is, configPath, StandardCopyOption.REPLACE_EXISTING); + } + } + TikaLoader tikaLoader = TikaLoader.load(configPath, getClass().getClassLoader()); + configuredParseContext = tikaLoader.loadParseContext(); + Tika tika = new Tika(tikaLoader.loadDetectors(), tikaLoader.loadAutoDetectParser()); + LOG.debug("Tika loaded in {} msec", System.currentTimeMillis() - start); + return tika; + } catch (IOException | TikaConfigException | URISyntaxException e) { + throw new IllegalStateException( + "Failed to instantiate Tika using custom configuration " + tikaConfigUrl, e); + } finally { + if (temporary && configPath != null) { + try { + Files.deleteIfExists(configPath); + } catch (IOException e) { + LOG.warn("Failed to delete temporary Tika configuration {}", configPath, e); + } + } } - long end = System.currentTimeMillis(); - LOG.debug("Tika loaded in {} msec", end - start); + } + + /** + * Returns a ParseContext seeded with the components configured in the "parse-context" section + * of the Tika configuration. + */ + ParseContext createParseContext() { + ParseContext parseContext = new ParseContext(); + parseContext.copyFrom(configuredParseContext); + return parseContext; + } + + /** Returns the Tika instance used by this bolt. Exposed for tests. */ + Tika getTika() { return tika; } diff --git a/external/tika/src/main/resources/tika-config.json b/external/tika/src/main/resources/tika-config.json new file mode 100644 index 000000000..a17d7cab1 --- /dev/null +++ b/external/tika/src/main/resources/tika-config.json @@ -0,0 +1,9 @@ +{ + "parsers": [ + { + "default-parser": { + "exclude": ["tesseract-ocr-parser"] + } + } + ] +} diff --git a/external/tika/src/main/resources/tika-config.xml b/external/tika/src/main/resources/tika-config.xml deleted file mode 100644 index c55bc3626..000000000 --- a/external/tika/src/main/resources/tika-config.xml +++ /dev/null @@ -1,36 +0,0 @@ - - - - - - - - - - - - - - - - - - - diff --git a/external/tika/src/test/java/org/apache/stormcrawler/tika/ParserBoltTest.java b/external/tika/src/test/java/org/apache/stormcrawler/tika/ParserBoltTest.java index fa28878bb..b8234fc4c 100644 --- a/external/tika/src/test/java/org/apache/stormcrawler/tika/ParserBoltTest.java +++ b/external/tika/src/test/java/org/apache/stormcrawler/tika/ParserBoltTest.java @@ -137,4 +137,28 @@ void testEmptiedParentDocumentIsStillEmitted() throws IOException { List> outTuples = output.getEmitted(); Assertions.assertEquals(1, outTuples.size()); } + + /** + * Checks that embedded documents are only parsed when parser.extract.embedded is set: unless an + * EmptyParser is bound in the ParseContext, the AutoDetectParser sets itself as the embedded + * parser and the embedded content would be extracted anyway. + */ + @Test + void testEmbeddedNotParsedByDefault() throws IOException { + Map conf = new HashMap<>(); + conf.put("parser.extract.embedded", false); + bolt.prepare(conf, TestUtil.getMockedTopologyContext(), new OutputCollector(output)); + parse( + "https://stormcrawler.apache.org/test_recursive_embedded.docx", + "test_recursive_embedded.docx"); + List> outTuples = output.getEmitted(); + Assertions.assertEquals(1, outTuples.size()); + Assertions.assertFalse( + outTuples + .get(0) + .get(3) + .toString() + .contains("Life, Liberty and the pursuit of Happiness"), + "embedded documents should not be parsed when parser.extract.embedded is false"); + } } diff --git a/external/tika/src/test/java/org/apache/stormcrawler/tika/TikaConfigTest.java b/external/tika/src/test/java/org/apache/stormcrawler/tika/TikaConfigTest.java new file mode 100644 index 000000000..83478ab79 --- /dev/null +++ b/external/tika/src/test/java/org/apache/stormcrawler/tika/TikaConfigTest.java @@ -0,0 +1,90 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to you under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.stormcrawler.tika; + +import java.util.HashMap; +import java.util.Map; +import org.apache.stormcrawler.TestUtil; +import org.apache.tika.config.OutputLimits; +import org.apache.tika.parser.CompositeParser; +import org.apache.tika.parser.ParseContext; +import org.apache.tika.parser.Parser; +import org.apache.tika.parser.ocr.TesseractOCRParser; +import org.junit.jupiter.api.Assertions; +import org.junit.jupiter.api.Test; + +/** Verifies what the bolt actually loads from the Tika configuration. */ +class TikaConfigTest { + + @Test + void testTesseractOCRExcludedByDefaultConfig() { + ParserBolt bolt = new ParserBolt(); + bolt.prepare(new HashMap<>(), TestUtil.getMockedTopologyContext(), null); + // the bundled configuration excludes the OCR parser: prove it rather + // than assume the configuration was picked up + Parser parser = bolt.getTika().getParser(); + Assertions.assertTrue( + parser instanceof CompositeParser, "the loaded parser should be a CompositeParser"); + for (Parser p : ((CompositeParser) parser).getAllComponentParsers()) { + Assertions.assertFalse( + p instanceof TesseractOCRParser, + "TesseractOCRParser should be excluded by the default configuration"); + } + } + + @Test + void testParseContextSeededFromConfig() { + Map conf = new HashMap<>(); + conf.put("parser.tika.config.file", "test-tika-config.json"); + ParserBolt bolt = new ParserBolt(); + bolt.prepare(conf, TestUtil.getMockedTopologyContext(), null); + + // the "parse-context" section of the configuration must reach the + // context used for each parse + ParseContext parseContext = bolt.createParseContext(); + Assertions.assertEquals( + 100000, + OutputLimits.get(parseContext).getWriteLimit(), + "writeLimit from the parse-context section of the configuration"); + // no limits set by default + Assertions.assertEquals( + OutputLimits.UNLIMITED, OutputLimits.get(new ParseContext()).getWriteLimit()); + } + + @Test + void testBrokenConfigFailsFast() { + Map conf = new HashMap<>(); + conf.put("parser.tika.config.file", "broken-tika-config.json"); + ParserBolt bolt = new ParserBolt(); + // an invalid configuration must abort the topology instead of + // silently degrading to the default Tika configuration + Assertions.assertThrows( + IllegalStateException.class, + () -> bolt.prepare(conf, TestUtil.getMockedTopologyContext(), null)); + } + + @Test + void testMissingConfigFailsFast() { + Map conf = new HashMap<>(); + conf.put("parser.tika.config.file", "does-not-exist.json"); + ParserBolt bolt = new ParserBolt(); + Assertions.assertThrows( + IllegalStateException.class, + () -> bolt.prepare(conf, TestUtil.getMockedTopologyContext(), null)); + } +} diff --git a/external/tika/src/test/resources/broken-tika-config.json b/external/tika/src/test/resources/broken-tika-config.json new file mode 100644 index 000000000..063574fa6 --- /dev/null +++ b/external/tika/src/test/resources/broken-tika-config.json @@ -0,0 +1,9 @@ +{ + "parsers": [ + { + "default-parser": { + "thisKeyDoesNotExist": true + } + } + ] +} diff --git a/external/tika/src/test/resources/test-tika-config.json b/external/tika/src/test/resources/test-tika-config.json new file mode 100644 index 000000000..5c23f311a --- /dev/null +++ b/external/tika/src/test/resources/test-tika-config.json @@ -0,0 +1,14 @@ +{ + "parsers": [ + { + "default-parser": { + "exclude": ["tesseract-ocr-parser"] + } + } + ], + "parse-context": { + "output-limits": { + "writeLimit": 100000 + } + } +} diff --git a/pom.xml b/pom.xml index cfd6686b2..00183f7fd 100644 --- a/pom.xml +++ b/pom.xml @@ -69,7 +69,7 @@ under the License. 2.22 2.22.0 - 3.3.2 + 4.0.0 5.23.0 2.0.18 26.1.0