Search Options

Results per page
Sort
Preferred Languages
Advance

Results 1 - 8 of 8 for tikaExtractor (0.07 sec)

  1. fess-crawler/src/main/java/org/codelibs/fess/crawler/extractor/impl/TikaExtractor.java

     * processing.
     * </p>
     *
     */
    public class TikaExtractor extends PasswordBasedExtractor {
    
        private static final Logger logger = LogManager.getLogger(TikaExtractor.class);
    
        /**
         * Tesseract config file path.
         */
        public static final String TIKA_TESSERACT_CONFIG = "tika.tesseract.config";
    
    Registered: Sun Sep 21 03:50:09 UTC 2025
    - Last Modified: Thu Aug 07 02:55:08 UTC 2025
    - 30.7K bytes
    - Viewed (0)
  2. fess-crawler/src/test/java/org/codelibs/fess/crawler/extractor/impl/TikaExtractorTest.java

                        TikaExtractor tikaExtractor = container.getComponent("tikaExtractor");
                        factory.addExtractor("text/plain", tikaExtractor);
                        factory.addExtractor("text/html", tikaExtractor);
                    })//
            ;
    
            tikaExtractor = container.getComponent("tikaExtractor");
        }
    
        public void test_getTika_text() {
    Registered: Sun Sep 21 03:50:09 UTC 2025
    - Last Modified: Thu Aug 07 02:55:08 UTC 2025
    - 30.6K bytes
    - Viewed (0)
  3. src/main/java/org/codelibs/fess/helper/DocumentHelper.java

                    tikaExtractor.setMaxSymbolTermSize(getMaxSymbolTermSize());
                    tikaExtractor.setReplaceDuplication(isDuplicateTermRemoved());
                    tikaExtractor.setSpaceChars(getSpaceChars());
                }
            } catch (final ComponentNotFoundException e) {
                if (logger.isDebugEnabled()) {
    Registered: Thu Sep 04 12:52:25 UTC 2025
    - Last Modified: Thu Aug 07 03:06:29 UTC 2025
    - 17.2K bytes
    - Viewed (0)
  4. fess-crawler/src/test/java/org/codelibs/fess/crawler/CrawlerTest.java

                    })
                    .singleton("tikaExtractor", TikaExtractor.class)
                    .<ExtractorFactory> singleton("extractorFactory", ExtractorFactory.class, factory -> {
                        TikaExtractor tikaExtractor = container.getComponent("tikaExtractor");
                        factory.addExtractor("text/plain", tikaExtractor);
                        factory.addExtractor("text/html", tikaExtractor);
                    })//
    Registered: Sun Sep 21 03:50:09 UTC 2025
    - Last Modified: Sat Sep 06 04:15:37 UTC 2025
    - 19.1K bytes
    - Viewed (0)
  5. src/test/java/org/codelibs/fess/helper/DocumentHelperTest.java

            ResponseData responseData = new ResponseData();
            Map<String, Object> dataMap = new HashMap<>();
    
            responseData.getMetaDataMap().put(TikaExtractor.class.getSimpleName(), new TikaExtractor());
    
            String content = " Test Content ";
            assertEquals("Test Content", documentHelper.getContent(null, responseData, content, dataMap));
        }
    
    Registered: Thu Sep 04 12:52:25 UTC 2025
    - Last Modified: Thu Jul 10 13:41:04 UTC 2025
    - 13K bytes
    - Viewed (0)
  6. README.md

    });
    
    // Configure content extraction
    container.singleton("tikaExtractor", TikaExtractor.class);
    container.singleton("extractorFactory", ExtractorFactory.class, factory -> {
        factory.addExtractor("text/html", container.getComponent("tikaExtractor"));
        factory.addExtractor("application/pdf", container.getComponent("tikaExtractor"));
    });
    
    Crawler crawler = container.getComponent("crawler");
    Registered: Sun Sep 21 03:50:09 UTC 2025
    - Last Modified: Sun Aug 31 05:32:52 UTC 2025
    - 15.3K bytes
    - Viewed (0)
  7. fess-crawler/src/main/java/org/codelibs/fess/crawler/extractor/ExtractorBuilder.java

        private String filename;
    
        /** Cache file size threshold in bytes */
        private int cacheFileSize = 1_000_000;
    
        /** Name of the extractor to use */
        private String extractorName = "tikaExtractor";
    
        /** Maximum content length allowed */
        private long maxContentLength = -1;
    
        /**
         * Constructs a new ExtractorBuilder.
         *
         * @param crawlerContainer the crawler container
    Registered: Sun Sep 21 03:50:09 UTC 2025
    - Last Modified: Sun Jul 06 02:13:03 UTC 2025
    - 10.1K bytes
    - Viewed (0)
  8. src/main/java/org/codelibs/fess/crawler/transformer/AbstractFessFileTransformer.java

    import org.codelibs.fess.crawler.exception.CrawlerSystemException;
    import org.codelibs.fess.crawler.exception.CrawlingAccessException;
    import org.codelibs.fess.crawler.extractor.Extractor;
    import org.codelibs.fess.crawler.extractor.impl.TikaExtractor;
    import org.codelibs.fess.crawler.serializer.DataSerializer;
    import org.codelibs.fess.crawler.transformer.impl.AbstractTransformer;
    import org.codelibs.fess.crawler.util.CrawlingParameterUtil;
    Registered: Thu Sep 04 12:52:25 UTC 2025
    - Last Modified: Thu Aug 07 03:06:29 UTC 2025
    - 25.6K bytes
    - Viewed (0)
Back to top