From 679a3b3201596af8e550b73399bd8a3e2c083cc8 Mon Sep 17 00:00:00 2001 From: lewismc Date: Fri, 2 Oct 2026 07:26:57 -0700 Subject: [PATCH] NUTCH-3215 Remove trailing whitespace reported by Yetus blanks on master --- .yetus/blanks-tabs.txt | 8 +- conf/parse-plugins.xml.template | 148 +++++------ conf/subcollections.xml.template | 12 +- ivy/mvn.template | 2 +- .../LICENSE-bouncy-castle-licence.txt | 6 +- licenses-binary/LICENSE-bsd.txt | 22 +- ...ENSE-eclipse-distribution-license-v1.0.txt | 40 +-- .../LICENSE-unicode-icu-license.txt | 2 +- src/java/org/apache/nutch/crawl/CrawlDb.java | 6 +- .../nutch/fetcher/FetcherThreadEvent.java | 12 +- .../nutch/fetcher/FetcherThreadPublisher.java | 4 +- .../org/apache/nutch/indexer/IndexingJob.java | 4 +- src/java/org/apache/nutch/metadata/Nutch.java | 56 ++-- .../org/apache/nutch/parse/ParseSegment.java | 2 +- .../nutch/protocol/RobotRulesParser.java | 6 +- .../nutch/publisher/NutchPublishers.java | 4 +- .../nutch/scoring/webgraph/NodeDumper.java | 6 +- .../apache/nutch/segment/SegmentChecker.java | 2 +- .../apache/nutch/tools/CommonCrawlConfig.java | 240 +++++++++--------- .../nutch/tools/CommonCrawlFormatFactory.java | 24 +- .../nutch/tools/CommonCrawlFormatJackson.java | 144 +++++------ .../tools/CommonCrawlFormatJettinson.java | 170 ++++++------- .../nutch/tools/CommonCrawlFormatSimple.java | 186 +++++++------- .../nutch/util/CrawlCompletionStats.java | 4 +- .../org/apache/nutch/util/DumpFileUtil.java | 84 +++--- .../nutch/util/ProtocolStatusStatistics.java | 8 +- src/plugin/build-plugin.xml | 2 +- src/plugin/feed/build.xml | 28 +- src/plugin/feed/plugin.xml | 30 +-- src/plugin/feed/sample/rsstest.rss | 28 +- .../arbitrary/ArbitraryIndexingFilter.java | 54 ++-- .../nutch/indexer/arbitrary/Multiplier.java | 4 +- .../indexer/arbitrary/PopularityGauge.java | 10 +- src/plugin/index-replace/build.xml | 68 ++--- .../nutch/indexer/replace/ReplaceIndexer.java | 2 +- src/plugin/indexer-cloudsearch/ivy.xml | 2 +- src/plugin/indexer-solr/ivy.xml | 88 +++---- src/plugin/indexer-solr/plugin.xml | 96 +++---- src/plugin/indexer-solr/schema.xml | 48 ++-- .../analysis/lang/LanguageIndexingFilter.java | 2 +- .../protocol/htmlunit/HtmlUnitWebDriver.java | 40 +-- src/plugin/lib-xml/build.xml | 20 +- src/plugin/parse-metatags/build.xml | 26 +- src/plugin/parse-tika/sample/nutch.html | 24 +- src/plugin/parse-tika/sample/rsstest.rss | 28 +- .../tika/BoilerpipeExtractorRepository.java | 2 +- .../scoring/depth/DepthScoringFilter.java | 2 +- .../apache/nutch/parse/parse-plugin-test.xml | 8 +- 48 files changed, 910 insertions(+), 904 deletions(-) diff --git a/.yetus/blanks-tabs.txt b/.yetus/blanks-tabs.txt index 080c828d16..dee87e8e85 100644 --- a/.yetus/blanks-tabs.txt +++ b/.yetus/blanks-tabs.txt @@ -4,4 +4,10 @@ CHANGES.md # some configuration files define key-value mappings using tabs ^conf/.*\.txt\.template -^src/test/host-protocol-mapping\.txt \ No newline at end of file +^src/test/host-protocol-mapping\.txt +# Tab-delimited data or a documented tab layout, not indentation. +^src/plugin/urlnormalizer-protocol/data/protocols\.txt +^src/plugin/parsefilter-regex/data/regex-parsefilter\.txt +^src/plugin/parsefilter-regex/README\.txt +# Fixture bytes, including a leading BOM. +^src/plugin/parse-tika/sample/ootest\.txt diff --git a/conf/parse-plugins.xml.template b/conf/parse-plugins.xml.template index f085084d96..f900e630a2 100644 --- a/conf/parse-plugins.xml.template +++ b/conf/parse-plugins.xml.template @@ -1,96 +1,96 @@ - - - + + + - - - - + + + + - - - - + + + + - - - - + + + + - - - + + + - - - + + + - - - + + + - - - + + + - - + + - - - - + + + + - - - - - - - - - - - - - - - - - - + + + + + + + + + + + + + + + + + + diff --git a/conf/subcollections.xml.template b/conf/subcollections.xml.template index 7b8805d50a..dca67ef0ff 100644 --- a/conf/subcollections.xml.template +++ b/conf/subcollections.xml.template @@ -16,13 +16,13 @@ limitations under the License. --> - - nutch - nutch - + + nutch + nutch + http://lucene.apache.org/nutch/ http://wiki.apache.org/nutch/ - - + + diff --git a/ivy/mvn.template b/ivy/mvn.template index f4653f739b..73fe239c88 100644 --- a/ivy/mvn.template +++ b/ivy/mvn.template @@ -72,7 +72,7 @@ markus Markus Jelsma markus@apache.org - + fenglu Feng Lu diff --git a/licenses-binary/LICENSE-bouncy-castle-licence.txt b/licenses-binary/LICENSE-bouncy-castle-licence.txt index 16db5c5253..2fe6b7bdae 100644 --- a/licenses-binary/LICENSE-bouncy-castle-licence.txt +++ b/licenses-binary/LICENSE-bouncy-castle-licence.txt @@ -11,7 +11,7 @@ Please note this should be read in the same way as the MIT license. Please also note this licensing model is made possible through funding from donations and the sale of support contracts. License Copyright (c) 2000 - 2021 The Legion of the Bouncy Castle Inc. (https://www.bouncycastle.org) - Permission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the "Software"), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so, subject to the following conditions: - The above copyright notice and this permission notice shall be included in all copies or substantial portions of the Software. - THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + Permission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the "Software"), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so, subject to the following conditions: + The above copyright notice and this permission notice shall be included in all copies or substantial portions of the Software. + THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. diff --git a/licenses-binary/LICENSE-bsd.txt b/licenses-binary/LICENSE-bsd.txt index c2e90c495a..e28c5645bb 100644 --- a/licenses-binary/LICENSE-bsd.txt +++ b/licenses-binary/LICENSE-bsd.txt @@ -6,19 +6,19 @@ BSD license definition -The BSD license is a class of extremely simple and very liberal licenses for computer software that was originally developed at the University of California at Berkeley (UCB). It was first used in 1980 for the Berkeley Source Distribution (BSD), also known as BSD UNIX, an enhanced version of the original UNIX operating system that was first written in 1969 by Ken Thompson at Bell Labs. +The BSD license is a class of extremely simple and very liberal licenses for computer software that was originally developed at the University of California at Berkeley (UCB). It was first used in 1980 for the Berkeley Source Distribution (BSD), also known as BSD UNIX, an enhanced version of the original UNIX operating system that was first written in 1969 by Ken Thompson at Bell Labs. -The only restrictions placed on users of software released under a typical BSD license are that if they redistribute such software in any form, with or without modification, they must include in the redistribution (1) the original copyright notice, (2) a list of two simple restrictions and (3) a disclaimer of liability. These restrictions can be summarized as (1) one should not claim that they wrote the software if they did not write it and (2) one should not sue the developer if the software does not function as expected or as desired. Some BSD licenses additionally include a clause that restricts the use of the name of the project (or the names of its contributors) for endorsing or promoting derivative works. +The only restrictions placed on users of software released under a typical BSD license are that if they redistribute such software in any form, with or without modification, they must include in the redistribution (1) the original copyright notice, (2) a list of two simple restrictions and (3) a disclaimer of liability. These restrictions can be summarized as (1) one should not claim that they wrote the software if they did not write it and (2) one should not sue the developer if the software does not function as expected or as desired. Some BSD licenses additionally include a clause that restricts the use of the name of the project (or the names of its contributors) for endorsing or promoting derivative works. -The most basic definition of a derivative work is a product that is based on, or incorporates, one or more already existing works. This can become a complex issue, particularly with regard to software, but the primary indicator that a software program is a derivative of another program is if it includes source code from the original program, even if the source code has been modified, including improving, extending, reordering or translating it into another programming language. +The most basic definition of a derivative work is a product that is based on, or incorporates, one or more already existing works. This can become a complex issue, particularly with regard to software, but the primary indicator that a software program is a derivative of another program is if it includes source code from the original program, even if the source code has been modified, including improving, extending, reordering or translating it into another programming language. Source code is the version of software (usually an application program or an operating system) as it is originally written (i.e., typed into a computer) by a human in plain text (i.e., human readable alphanumeric characters). Source code can be written in any of hundreds of programming languages, some of the most popular of which are C, C++ and Java. -Due to the extremely minimal restrictions of BSD-style licenses, software released under such licenses can be freely modified and used in proprietary (i.e., commercial) software for which the source code is kept secret. +Due to the extremely minimal restrictions of BSD-style licenses, software released under such licenses can be freely modified and used in proprietary (i.e., commercial) software for which the source code is kept secret. It is possible for a product to be distributed under a BSD-style license and for some other license to apply as well. This was, in fact, the case with very early versions of BSD UNIX, which included both new code written at UCB and code from the original versions of UNIX written at Bell Labs. @@ -29,8 +29,8 @@ BSD-style licenses have been very successful, and they are now widely used for a BSD Licenses Versus the GPL - -The GPL (GNU General Public License) is by far the most widely used license for free software (i.e., software whose source code is available at no cost for anyone to use for any purpose). The Linux kernel (i.e., the core of the operating system) as well as much of the other software generally included in Linux distributions have been released under the terms of the GPL. + +The GPL (GNU General Public License) is by far the most widely used license for free software (i.e., software whose source code is available at no cost for anyone to use for any purpose). The Linux kernel (i.e., the core of the operating system) as well as much of the other software generally included in Linux distributions have been released under the terms of the GPL. Although far fewer programs are released under BSD-style licenses, this class of licenses is disproportionately important because of the widespread use of BSD-licensed code in both free and proprietary operating systems. @@ -69,7 +69,7 @@ The original version of the BSD license contained the so called advertising clau One of the problems with this clause arose from the fact that people who made changes to the source code often wanted to have their names added to the acknowledgment. This could easily result in large and cumbersome acknowledgments for products with numerous contributors and for software distributions consisting of multiple individual projects. -A second problem was legal incompatibility with the terms of the GPL. This is because the GPL prohibits the addition of restrictions beyond those that it already imposes. Thus it was necessary to segregate GPL and BSD-licensed software within projects. +A second problem was legal incompatibility with the terms of the GPL. This is because the GPL prohibits the addition of restrictions beyond those that it already imposes. Thus it was necessary to segregate GPL and BSD-licensed software within projects. Initially, the "obnoxious BSD advertising clause," as it was referred to by GPL advocates, was used only for the BSD UNIX license. That did not cause any major problems because it was only necessary to include a single sentence of acknowledgment in any advertisement. @@ -78,7 +78,7 @@ Initially, the "obnoxious BSD advertising clause," as it was referred to by GPL However, the fact that other software developers did not copy the clause verbatim, but replaced the phrase "University of California" with the name of their own organization or persons involved in it, resulted in a proliferation of slightly different licenses and a consequently serious problem when many such programs were assembled to form a larger work or an operating system. For example, if an operating system or other program required fifty slightly different acknowledgment sentences, each naming a different developer or group of developers, such advertising alone might require a full page. Not only would this be very tedious reading, but it could also be costly. -In June 1999, after two years of discussion, the Office of Technology Licensing at UCB finally proclaimed: "Effective immediately, licensees and distributors are no longer required to include the acknowledgment within advertising materials. Accordingly, the foregoing paragraph of those BSD Unix files containing it is hereby deleted in its entirety." +In June 1999, after two years of discussion, the Office of Technology Licensing at UCB finally proclaimed: "Effective immediately, licensees and distributors are no longer required to include the acknowledgment within advertising materials. Accordingly, the foregoing paragraph of those BSD Unix files containing it is hereby deleted in its entirety." This was clearly very useful. However, it could not eliminate the legacy of the advertising clause, as similar clauses still exist in the licenses of many programs that followed the old BSD license; only the developers of such packages can change them. @@ -96,7 +96,7 @@ Below are three examples of BSD-style licenses: (1) the BSD license as it is use Copyright 1994-2004 The FreeBSD Project. All rights reserved. - + Redistribution and use in source and binary forms, with or without modification, are permitted provided that the following conditions are met: @@ -140,7 +140,7 @@ or implied, of the FreeBSD Project. - + Sudo License Sudo is distributed under the following BSD-style license: @@ -182,7 +182,7 @@ Additionally, lsearch.c, fnmatch.c, getcwd.c, snprintf.c strcasecmp.c and fnmatc -(3) A template for a BSD-style license. [YEAR], [COPYRIGHT OWNER] and [LICENSOR] are to be replaced by the actual year of copyright, the owner of the copyright and the licensor. The copyright owner and licensor may be the same, as in the case of the license for FreeBSD (as shown above). +(3) A template for a BSD-style license. [YEAR], [COPYRIGHT OWNER] and [LICENSOR] are to be replaced by the actual year of copyright, the owner of the copyright and the licensor. The copyright owner and licensor may be the same, as in the case of the license for FreeBSD (as shown above). diff --git a/licenses-binary/LICENSE-eclipse-distribution-license-v1.0.txt b/licenses-binary/LICENSE-eclipse-distribution-license-v1.0.txt index 347787f3ea..83c4cd7f15 100644 --- a/licenses-binary/LICENSE-eclipse-distribution-license-v1.0.txt +++ b/licenses-binary/LICENSE-eclipse-distribution-license-v1.0.txt @@ -5,26 +5,26 @@ Eclipse Distribution License | The Eclipse Foundation Eclipse Distribution License - v 1.0 -Copyright (c) 2007, Eclipse Foundation, Inc. and its licensors. +Copyright (c) 2007, Eclipse Foundation, Inc. and its licensors. All rights reserved. -Redistribution and use in source and binary forms, with or without modification, - are permitted provided that the following conditions are met: -Redistributions of source code must retain the above copyright notice, - this list of conditions and the following disclaimer. -Redistributions in binary form must reproduce the above copyright notice, - this list of conditions and the following disclaimer in the documentation - and/or other materials provided with the distribution. -Neither the name of the Eclipse Foundation, Inc. nor the names of its - contributors may be used to endorse or promote products derived from - this software without specific prior written permission. -THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" -AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED -WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. -IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, -INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT -NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR -PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, -WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) -ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE +Redistribution and use in source and binary forms, with or without modification, + are permitted provided that the following conditions are met: +Redistributions of source code must retain the above copyright notice, + this list of conditions and the following disclaimer. +Redistributions in binary form must reproduce the above copyright notice, + this list of conditions and the following disclaimer in the documentation + and/or other materials provided with the distribution. +Neither the name of the Eclipse Foundation, Inc. nor the names of its + contributors may be used to endorse or promote products derived from + this software without specific prior written permission. +THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" +AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED +WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED. +IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, +INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT +NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR +PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, +WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) +ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. diff --git a/licenses-binary/LICENSE-unicode-icu-license.txt b/licenses-binary/LICENSE-unicode-icu-license.txt index 17ff27e7b5..68c30f4c9e 100644 --- a/licenses-binary/LICENSE-unicode-icu-license.txt +++ b/licenses-binary/LICENSE-unicode-icu-license.txt @@ -312,7 +312,7 @@ Lao Word Break Dictionary Data (laodict.txt) # License: https://github.com/rober42539/lao-dictionary/LICENSE.txt # (copied below) # - # This file is derived from the above dictionary version of Nov 22, 2020 + # This file is derived from the above dictionary version of Nov 22, 2020 # ---------------------------------------------------------------------- # Copyright (C) 2013 Brian Eugene Wilson, Robert Martin Campbell. # All rights reserved. diff --git a/src/java/org/apache/nutch/crawl/CrawlDb.java b/src/java/org/apache/nutch/crawl/CrawlDb.java index e27fba6731..bdd3dfc034 100644 --- a/src/java/org/apache/nutch/crawl/CrawlDb.java +++ b/src/java/org/apache/nutch/crawl/CrawlDb.java @@ -342,12 +342,12 @@ else if(args.containsKey(Nutch.ARG_SEGMENTS)) { Object segments = args.get(Nutch.ARG_SEGMENTS); ArrayList segmentList = new ArrayList<>(); if(segments instanceof ArrayList) { - segmentList = (ArrayList)segments; + segmentList = (ArrayList)segments; } else if(segments instanceof Path){ - segmentList.add(segments.toString()); + segmentList.add(segments.toString()); } - + for(String segment: segmentList) { dirs.add(new Path(segment)); } diff --git a/src/java/org/apache/nutch/fetcher/FetcherThreadEvent.java b/src/java/org/apache/nutch/fetcher/FetcherThreadEvent.java index 2d1d4b5583..b634b84918 100644 --- a/src/java/org/apache/nutch/fetcher/FetcherThreadEvent.java +++ b/src/java/org/apache/nutch/fetcher/FetcherThreadEvent.java @@ -93,15 +93,15 @@ public String getUrl() { /** * Set URL of this event (fetched page) - * @param url URL of the fetched page + * @param url URL of the fetched page */ public void setUrl(String url) { this.url = url; } /** * Add new data to the eventData object. - * @param key A key to refer to the data being added to this event - * @param value Data to be stored in the event referenced by the above key + * @param key A key to refer to the data being added to this event + * @param value Data to be stored in the event referenced by the above key */ public void addEventData(String key, Object value) { if(eventData == null) { @@ -113,8 +113,8 @@ public void addEventData(String key, Object value) { /** * Given a collection of lists this method will add it * the oultink metadata - * @param links A collection of outlinks generating from the fetched page - * this event refers to + * @param links A collection of outlinks generating from the fetched page + * this event refers to */ public void addOutlinksToEventData(Collection links) { ArrayList> outlinkList = new ArrayList<>(); @@ -137,7 +137,7 @@ public Long getTimestamp() { /** * Set timestamp for this event - * @param timestamp Timestamp of the occurrence of this event + * @param timestamp Timestamp of the occurrence of this event */ public void setTimestamp(Long timestamp) { this.timestamp = timestamp; diff --git a/src/java/org/apache/nutch/fetcher/FetcherThreadPublisher.java b/src/java/org/apache/nutch/fetcher/FetcherThreadPublisher.java index 863c0c07f5..5933259c19 100644 --- a/src/java/org/apache/nutch/fetcher/FetcherThreadPublisher.java +++ b/src/java/org/apache/nutch/fetcher/FetcherThreadPublisher.java @@ -46,8 +46,8 @@ public FetcherThreadPublisher(Configuration conf) { /** * Publish event to all registered publishers - * @param event {@link org.apache.nutch.fetcher.FetcherThreadEvent Event} to be published - * @param conf {@link org.apache.hadoop.conf.Configuration Configuration} to be used + * @param event {@link org.apache.nutch.fetcher.FetcherThreadEvent Event} to be published + * @param conf {@link org.apache.hadoop.conf.Configuration Configuration} to be used */ public void publish(FetcherThreadEvent event, Configuration conf) { if(publisher!=null) { diff --git a/src/java/org/apache/nutch/indexer/IndexingJob.java b/src/java/org/apache/nutch/indexer/IndexingJob.java index cfbd759198..c65c7e71bf 100644 --- a/src/java/org/apache/nutch/indexer/IndexingJob.java +++ b/src/java/org/apache/nutch/indexer/IndexingJob.java @@ -378,13 +378,13 @@ public Map run(Map args, String crawlId) throws Object segmentsFromArg = args.get(Nutch.ARG_SEGMENTS); ArrayList segmentList = new ArrayList(); if(segmentsFromArg instanceof ArrayList) { - segmentList = (ArrayList)segmentsFromArg; } + segmentList = (ArrayList)segmentsFromArg; } else if(segmentsFromArg instanceof Path){ segmentList.add(segmentsFromArg.toString()); } for(String segment: segmentList) { - segments.add(new Path(segment)); + segments.add(new Path(segment)); } } diff --git a/src/java/org/apache/nutch/metadata/Nutch.java b/src/java/org/apache/nutch/metadata/Nutch.java index 2542b35ee5..53c44c8bcc 100644 --- a/src/java/org/apache/nutch/metadata/Nutch.java +++ b/src/java/org/apache/nutch/metadata/Nutch.java @@ -81,32 +81,32 @@ public interface Nutch { public static final Text WRITABLE_FIXED_INTERVAL_KEY = new Text( FIXED_INTERVAL_KEY); - /** For progress of job (programmatic / tooling). */ - public static final String STAT_PROGRESS = "progress"; - /** Crawl id key for programmatic jobs. */ - public static final String CRAWL_ID_KEY = "storage.crawl.id"; - /** Argument key for seed URL directory path. */ - public static final String ARG_SEEDDIR = "url_dir"; - /** Argument key for crawldb location in programmatic jobs. */ - public static final String ARG_CRAWLDB = "crawldb"; - /** Argument key for linkdb location in programmatic jobs. */ - public static final String ARG_LINKDB = "linkdb"; - /** Name of the key used in the result map from {@link org.apache.nutch.util.NutchTool#run}. */ - public static final String VAL_RESULT = "result"; - /** Argument key for a directory of segments; similar to the -dir option in bin/nutch. */ - public static final String ARG_SEGMENTDIR = "segment_dir"; - /** Argument key for one segment or a list of segments (job-dependent). */ - public static final String ARG_SEGMENTS = "segment"; - /** Argument key for hostdb location in programmatic jobs. */ - public static final String ARG_HOSTDB = "hostdb"; - /** Title key in the Pub/Sub event metadata for the title of the parsed page*/ - public static final String FETCH_EVENT_TITLE = "title"; - /** Content-type key in the Pub/Sub event metadata for the content-type of the parsed page*/ - public static final String FETCH_EVENT_CONTENTTYPE = "content-type"; - /** Score key in the Pub/Sub event metadata for the score of the parsed page*/ - public static final String FETCH_EVENT_SCORE = "score"; - /** Fetch time key in the Pub/Sub event metadata for the fetch time of the parsed page*/ - public static final String FETCH_EVENT_FETCHTIME = "fetchTime"; - /** Content-lanueage key in the Pub/Sub event metadata for the content-language of the parsed page*/ - public static final String FETCH_EVENT_CONTENTLANG = "content-language"; + /** For progress of job (programmatic / tooling). */ + public static final String STAT_PROGRESS = "progress"; + /** Crawl id key for programmatic jobs. */ + public static final String CRAWL_ID_KEY = "storage.crawl.id"; + /** Argument key for seed URL directory path. */ + public static final String ARG_SEEDDIR = "url_dir"; + /** Argument key for crawldb location in programmatic jobs. */ + public static final String ARG_CRAWLDB = "crawldb"; + /** Argument key for linkdb location in programmatic jobs. */ + public static final String ARG_LINKDB = "linkdb"; + /** Name of the key used in the result map from {@link org.apache.nutch.util.NutchTool#run}. */ + public static final String VAL_RESULT = "result"; + /** Argument key for a directory of segments; similar to the -dir option in bin/nutch. */ + public static final String ARG_SEGMENTDIR = "segment_dir"; + /** Argument key for one segment or a list of segments (job-dependent). */ + public static final String ARG_SEGMENTS = "segment"; + /** Argument key for hostdb location in programmatic jobs. */ + public static final String ARG_HOSTDB = "hostdb"; + /** Title key in the Pub/Sub event metadata for the title of the parsed page*/ + public static final String FETCH_EVENT_TITLE = "title"; + /** Content-type key in the Pub/Sub event metadata for the content-type of the parsed page*/ + public static final String FETCH_EVENT_CONTENTTYPE = "content-type"; + /** Score key in the Pub/Sub event metadata for the score of the parsed page*/ + public static final String FETCH_EVENT_SCORE = "score"; + /** Fetch time key in the Pub/Sub event metadata for the fetch time of the parsed page*/ + public static final String FETCH_EVENT_FETCHTIME = "fetchTime"; + /** Content-lanueage key in the Pub/Sub event metadata for the content-language of the parsed page*/ + public static final String FETCH_EVENT_CONTENTLANG = "content-language"; } diff --git a/src/java/org/apache/nutch/parse/ParseSegment.java b/src/java/org/apache/nutch/parse/ParseSegment.java index 944d1c0736..461769f9ea 100644 --- a/src/java/org/apache/nutch/parse/ParseSegment.java +++ b/src/java/org/apache/nutch/parse/ParseSegment.java @@ -407,7 +407,7 @@ public Map run(Map args, String crawlId) throws } } else { - String segment_dir = crawlId+"/segments"; + String segment_dir = crawlId+"/segments"; File segmentsDir = new File(segment_dir); File[] segmentsList = segmentsDir.listFiles(); Arrays.sort(segmentsList, (f1, f2) -> { diff --git a/src/java/org/apache/nutch/protocol/RobotRulesParser.java b/src/java/org/apache/nutch/protocol/RobotRulesParser.java index 6f3b513842..cc4bbf8be6 100644 --- a/src/java/org/apache/nutch/protocol/RobotRulesParser.java +++ b/src/java/org/apache/nutch/protocol/RobotRulesParser.java @@ -162,8 +162,8 @@ public void setConf(Configuration conf) { } else { for (int i = 0; i < confAllowList.length; i++) { if (confAllowList[i].isEmpty()) { - LOG.info("Empty allowlisted URL skipped!"); - continue; + LOG.info("Empty allowlisted URL skipped!"); + continue; } allowList.add(confAllowList[i]); } @@ -193,7 +193,7 @@ public boolean isAllowListed(URL url) { String urlString = url.getHost(); if (matcher != null) { - match = matcher.matches(urlString); + match = matcher.matches(urlString); } return match; diff --git a/src/java/org/apache/nutch/publisher/NutchPublishers.java b/src/java/org/apache/nutch/publisher/NutchPublishers.java index 18076424c0..e93330b343 100644 --- a/src/java/org/apache/nutch/publisher/NutchPublishers.java +++ b/src/java/org/apache/nutch/publisher/NutchPublishers.java @@ -32,7 +32,7 @@ public class NutchPublishers extends Configured implements NutchPublisher{ private Configuration conf; public NutchPublishers(Configuration conf) { - this.conf = conf; + this.conf = conf; this.publishers = (NutchPublisher[])PluginRepository.get(conf). getOrderedPlugins(NutchPublisher.class, NutchPublisher.X_POINT_ID, "publisher.order"); @@ -77,6 +77,6 @@ public Configuration getConf() { @Override public void setConf(Configuration arg0) { - + } } diff --git a/src/java/org/apache/nutch/scoring/webgraph/NodeDumper.java b/src/java/org/apache/nutch/scoring/webgraph/NodeDumper.java index 4f75f2ccc9..c6b44e431b 100644 --- a/src/java/org/apache/nutch/scoring/webgraph/NodeDumper.java +++ b/src/java/org/apache/nutch/scoring/webgraph/NodeDumper.java @@ -100,8 +100,8 @@ public static class SorterMapper extends @Override public void setup(Mapper.Context context) { conf = context.getConfiguration(); - inlinks = conf.getBoolean("inlinks", false); - outlinks = conf.getBoolean("outlinks", false); + inlinks = conf.getBoolean("inlinks", false); + outlinks = conf.getBoolean("outlinks", false); } @Override @@ -185,7 +185,7 @@ public static class DumperMapper extends @Override public void setup(Mapper.Context context) { conf = context.getConfiguration(); - inlinks = conf.getBoolean("inlinks", false); + inlinks = conf.getBoolean("inlinks", false); outlinks = conf.getBoolean("outlinks", false); host = conf.getBoolean("host", false); } diff --git a/src/java/org/apache/nutch/segment/SegmentChecker.java b/src/java/org/apache/nutch/segment/SegmentChecker.java index 18cfec597e..a852393047 100644 --- a/src/java/org/apache/nutch/segment/SegmentChecker.java +++ b/src/java/org/apache/nutch/segment/SegmentChecker.java @@ -151,7 +151,7 @@ public static boolean isParsed(Path segment, FileSystem fs) throws IOException { if (fs.exists(new Path(segment, CrawlDatum.PARSE_DIR_NAME))){ - return true; + return true; } return false; } diff --git a/src/java/org/apache/nutch/tools/CommonCrawlConfig.java b/src/java/org/apache/nutch/tools/CommonCrawlConfig.java index 49d9c3143b..26daee5ce7 100644 --- a/src/java/org/apache/nutch/tools/CommonCrawlConfig.java +++ b/src/java/org/apache/nutch/tools/CommonCrawlConfig.java @@ -23,124 +23,124 @@ public class CommonCrawlConfig implements Serializable { - /** - * Serial version UID - */ - private static final long serialVersionUID = 5235013733207799661L; - - // Prefix for key value in the output format - private String keyPrefix = ""; - - private boolean simpleDateFormat = false; - - private boolean jsonArray = false; - - private boolean reverseKey = false; - - private String reverseKeyValue = ""; - - private boolean compressed = false; - - private long warcSize = 0; - - private String outputDir; - - /** - * Default constructor - */ - public CommonCrawlConfig() { - // TODO init(this.getClass().getResourceAsStream("CommonCrawlConfig.properties")); - } - - public CommonCrawlConfig(InputStream stream) { - init(stream); - } - - private void init(InputStream stream) { - if (stream == null) { - return; - } - Properties properties = new Properties(); - - try { - properties.load(stream); - } catch (IOException e) { - // TODO - } finally { - try { - stream.close(); - } catch (IOException e) { - // TODO - } - } - - setKeyPrefix(properties.getProperty("keyPrefix", "")); - setSimpleDateFormat(Boolean.parseBoolean(properties.getProperty("simpleDateFormat", "False"))); - setJsonArray(Boolean.parseBoolean(properties.getProperty("jsonArray", "False"))); - setReverseKey(Boolean.parseBoolean(properties.getProperty("reverseKey", "False"))); - } - - public void setKeyPrefix(String keyPrefix) { - this.keyPrefix = keyPrefix; - } - - public void setSimpleDateFormat(boolean simpleDateFormat) { - this.simpleDateFormat = simpleDateFormat; - } - - public void setJsonArray(boolean jsonArray) { - this.jsonArray = jsonArray; - } - - public void setReverseKey(boolean reverseKey) { - this.reverseKey = reverseKey; - } - - public void setReverseKeyValue(String reverseKeyValue) { - this.reverseKeyValue = reverseKeyValue; - } - - public String getKeyPrefix() { - return this.keyPrefix; - } - - public boolean getSimpleDateFormat() { - return this.simpleDateFormat; - } - - public boolean getJsonArray() { - return this.jsonArray; - } - - public boolean getReverseKey() { - return this.reverseKey; - } - - public String getReverseKeyValue() { - return this.reverseKeyValue; - } - - public boolean isCompressed() { - return compressed; - } - - public void setCompressed(boolean compressed) { - this.compressed = compressed; - } - - public long getWarcSize() { - return warcSize; - } - - public void setWarcSize(long warcSize) { - this.warcSize = warcSize; - } - - public String getOutputDir() { - return outputDir; - } - - public void setOutputDir(String outputDir) { - this.outputDir = outputDir; - } + /** + * Serial version UID + */ + private static final long serialVersionUID = 5235013733207799661L; + + // Prefix for key value in the output format + private String keyPrefix = ""; + + private boolean simpleDateFormat = false; + + private boolean jsonArray = false; + + private boolean reverseKey = false; + + private String reverseKeyValue = ""; + + private boolean compressed = false; + + private long warcSize = 0; + + private String outputDir; + + /** + * Default constructor + */ + public CommonCrawlConfig() { + // TODO init(this.getClass().getResourceAsStream("CommonCrawlConfig.properties")); + } + + public CommonCrawlConfig(InputStream stream) { + init(stream); + } + + private void init(InputStream stream) { + if (stream == null) { + return; + } + Properties properties = new Properties(); + + try { + properties.load(stream); + } catch (IOException e) { + // TODO + } finally { + try { + stream.close(); + } catch (IOException e) { + // TODO + } + } + + setKeyPrefix(properties.getProperty("keyPrefix", "")); + setSimpleDateFormat(Boolean.parseBoolean(properties.getProperty("simpleDateFormat", "False"))); + setJsonArray(Boolean.parseBoolean(properties.getProperty("jsonArray", "False"))); + setReverseKey(Boolean.parseBoolean(properties.getProperty("reverseKey", "False"))); + } + + public void setKeyPrefix(String keyPrefix) { + this.keyPrefix = keyPrefix; + } + + public void setSimpleDateFormat(boolean simpleDateFormat) { + this.simpleDateFormat = simpleDateFormat; + } + + public void setJsonArray(boolean jsonArray) { + this.jsonArray = jsonArray; + } + + public void setReverseKey(boolean reverseKey) { + this.reverseKey = reverseKey; + } + + public void setReverseKeyValue(String reverseKeyValue) { + this.reverseKeyValue = reverseKeyValue; + } + + public String getKeyPrefix() { + return this.keyPrefix; + } + + public boolean getSimpleDateFormat() { + return this.simpleDateFormat; + } + + public boolean getJsonArray() { + return this.jsonArray; + } + + public boolean getReverseKey() { + return this.reverseKey; + } + + public String getReverseKeyValue() { + return this.reverseKeyValue; + } + + public boolean isCompressed() { + return compressed; + } + + public void setCompressed(boolean compressed) { + this.compressed = compressed; + } + + public long getWarcSize() { + return warcSize; + } + + public void setWarcSize(long warcSize) { + this.warcSize = warcSize; + } + + public String getOutputDir() { + return outputDir; + } + + public void setOutputDir(String outputDir) { + this.outputDir = outputDir; + } } diff --git a/src/java/org/apache/nutch/tools/CommonCrawlFormatFactory.java b/src/java/org/apache/nutch/tools/CommonCrawlFormatFactory.java index c4cb57056e..f1d5db562e 100644 --- a/src/java/org/apache/nutch/tools/CommonCrawlFormatFactory.java +++ b/src/java/org/apache/nutch/tools/CommonCrawlFormatFactory.java @@ -25,17 +25,17 @@ * */ public class CommonCrawlFormatFactory { - - // The format should not depend on variable attributes, essentially this - // should be one for the full job - public static CommonCrawlFormat getCommonCrawlFormat(String formatType, Configuration nutchConf, CommonCrawlConfig config) throws IOException { - if (formatType.equalsIgnoreCase("WARC")) { - return new CommonCrawlFormatWARC(nutchConf, config); - } - if (formatType.equalsIgnoreCase("JACKSON")) { - return new CommonCrawlFormatJackson( nutchConf, config); - } - return null; - } + // The format should not depend on variable attributes, essentially this + // should be one for the full job + public static CommonCrawlFormat getCommonCrawlFormat(String formatType, Configuration nutchConf, CommonCrawlConfig config) throws IOException { + if (formatType.equalsIgnoreCase("WARC")) { + return new CommonCrawlFormatWARC(nutchConf, config); + } + + if (formatType.equalsIgnoreCase("JACKSON")) { + return new CommonCrawlFormatJackson( nutchConf, config); + } + return null; + } } diff --git a/src/java/org/apache/nutch/tools/CommonCrawlFormatJackson.java b/src/java/org/apache/nutch/tools/CommonCrawlFormatJackson.java index 78fcd322f0..8001d4a016 100644 --- a/src/java/org/apache/nutch/tools/CommonCrawlFormatJackson.java +++ b/src/java/org/apache/nutch/tools/CommonCrawlFormatJackson.java @@ -33,77 +33,77 @@ */ public class CommonCrawlFormatJackson extends AbstractCommonCrawlFormat { - private ByteArrayOutputStream out; - - private JsonGenerator generator; - - public CommonCrawlFormatJackson(Configuration nutchConf, - CommonCrawlConfig config) throws IOException { - super(null, null, null, nutchConf, config); - - JsonFactory factory = new JsonFactory(); - this.out = new ByteArrayOutputStream(); - this.generator = factory.createGenerator(out); - - this.generator.useDefaultPrettyPrinter(); // INDENTED OUTPUT - } - - public CommonCrawlFormatJackson(String url, Content content, Metadata metadata, Configuration nutchConf, CommonCrawlConfig config) throws IOException { - super(url, content, metadata, nutchConf, config); - - JsonFactory factory = new JsonFactory(); - this.out = new ByteArrayOutputStream(); - this.generator = factory.createGenerator(out); - - this.generator.useDefaultPrettyPrinter(); // INDENTED OUTPUT - } - - @Override - protected void writeKeyValue(String key, String value) throws IOException { - generator.writeFieldName(key); - generator.writeString(value); - } - - @Override - protected void writeKeyNull(String key) throws IOException { - generator.writeFieldName(key); - generator.writeNull(); - } - - @Override - protected void startArray(String key, boolean nested, boolean newline) throws IOException { - if (key != null) { - generator.writeFieldName(key); - } - generator.writeStartArray(); - } - - @Override - protected void closeArray(String key, boolean nested, boolean newline) throws IOException { - generator.writeEndArray(); - } - - @Override - protected void writeArrayValue(String value) throws IOException { - generator.writeString(value); - } - - @Override - protected void startObject(String key) throws IOException { - if (key != null) { - generator.writeFieldName(key); - } - generator.writeStartObject(); - } - - @Override - protected void closeObject(String key) throws IOException { - generator.writeEndObject(); - } - - @Override - protected String generateJson() throws IOException { - this.generator.flush(); + private ByteArrayOutputStream out; + + private JsonGenerator generator; + + public CommonCrawlFormatJackson(Configuration nutchConf, + CommonCrawlConfig config) throws IOException { + super(null, null, null, nutchConf, config); + + JsonFactory factory = new JsonFactory(); + this.out = new ByteArrayOutputStream(); + this.generator = factory.createGenerator(out); + + this.generator.useDefaultPrettyPrinter(); // INDENTED OUTPUT + } + + public CommonCrawlFormatJackson(String url, Content content, Metadata metadata, Configuration nutchConf, CommonCrawlConfig config) throws IOException { + super(url, content, metadata, nutchConf, config); + + JsonFactory factory = new JsonFactory(); + this.out = new ByteArrayOutputStream(); + this.generator = factory.createGenerator(out); + + this.generator.useDefaultPrettyPrinter(); // INDENTED OUTPUT + } + + @Override + protected void writeKeyValue(String key, String value) throws IOException { + generator.writeFieldName(key); + generator.writeString(value); + } + + @Override + protected void writeKeyNull(String key) throws IOException { + generator.writeFieldName(key); + generator.writeNull(); + } + + @Override + protected void startArray(String key, boolean nested, boolean newline) throws IOException { + if (key != null) { + generator.writeFieldName(key); + } + generator.writeStartArray(); + } + + @Override + protected void closeArray(String key, boolean nested, boolean newline) throws IOException { + generator.writeEndArray(); + } + + @Override + protected void writeArrayValue(String value) throws IOException { + generator.writeString(value); + } + + @Override + protected void startObject(String key) throws IOException { + if (key != null) { + generator.writeFieldName(key); + } + generator.writeStartObject(); + } + + @Override + protected void closeObject(String key) throws IOException { + generator.writeEndObject(); + } + + @Override + protected String generateJson() throws IOException { + this.generator.flush(); return this.out.toString(StandardCharsets.UTF_8); - } + } } diff --git a/src/java/org/apache/nutch/tools/CommonCrawlFormatJettinson.java b/src/java/org/apache/nutch/tools/CommonCrawlFormatJettinson.java index 499530f415..da08ac837c 100644 --- a/src/java/org/apache/nutch/tools/CommonCrawlFormatJettinson.java +++ b/src/java/org/apache/nutch/tools/CommonCrawlFormatJettinson.java @@ -32,90 +32,90 @@ * */ public class CommonCrawlFormatJettinson extends AbstractCommonCrawlFormat { - - private Deque stackObjects; - - private Deque stackArrays; - public CommonCrawlFormatJettinson(String url, Content content, Metadata metadata, Configuration nutchConf, CommonCrawlConfig config) throws IOException { - super(url, content, metadata, nutchConf, config); - - stackObjects = new ArrayDeque<>(); - stackArrays = new ArrayDeque<>(); - } - - @Override - protected void writeKeyValue(String key, String value) throws IOException { - try { - stackObjects.getFirst().put(key, value); - } catch (JSONException jsone) { - throw new IOException(jsone.getMessage()); - } - } - - @Override - protected void writeKeyNull(String key) throws IOException { - try { - stackObjects.getFirst().put(key, JSONObject.NULL); - } catch (JSONException jsone) { - throw new IOException(jsone.getMessage()); - } - } - - @Override - protected void startArray(String key, boolean nested, boolean newline) throws IOException { - JSONArray array = new JSONArray(); - stackArrays.push(array); - } - - @Override - protected void closeArray(String key, boolean nested, boolean newline) throws IOException { - try { - if (stackArrays.size() > 1) { - JSONArray array = stackArrays.pop(); - if (nested) { - stackArrays.getFirst().put(array); - } - else { - stackObjects.getFirst().put(key, array); - } - } - } catch (JSONException jsone) { - throw new IOException(jsone.getMessage()); - } - } - - @Override - protected void writeArrayValue(String value) throws IOException { - if (stackArrays.size() > 1) { - stackArrays.getFirst().put(value); - } - } - - @Override - protected void startObject(String key) throws IOException { - JSONObject object = new JSONObject(); - stackObjects.push(object); - } - - @Override - protected void closeObject(String key) throws IOException { - try { - if (stackObjects.size() > 1) { - JSONObject object = stackObjects.pop(); - stackObjects.getFirst().put(key, object); - } - } catch (JSONException jsone) { - throw new IOException(jsone.getMessage()); - } - } - - @Override - protected String generateJson() throws IOException { - try { - return stackObjects.getFirst().toString(2); - } catch (JSONException jsone) { - throw new IOException(jsone.getMessage()); - } - } + private Deque stackObjects; + + private Deque stackArrays; + + public CommonCrawlFormatJettinson(String url, Content content, Metadata metadata, Configuration nutchConf, CommonCrawlConfig config) throws IOException { + super(url, content, metadata, nutchConf, config); + + stackObjects = new ArrayDeque<>(); + stackArrays = new ArrayDeque<>(); + } + + @Override + protected void writeKeyValue(String key, String value) throws IOException { + try { + stackObjects.getFirst().put(key, value); + } catch (JSONException jsone) { + throw new IOException(jsone.getMessage()); + } + } + + @Override + protected void writeKeyNull(String key) throws IOException { + try { + stackObjects.getFirst().put(key, JSONObject.NULL); + } catch (JSONException jsone) { + throw new IOException(jsone.getMessage()); + } + } + + @Override + protected void startArray(String key, boolean nested, boolean newline) throws IOException { + JSONArray array = new JSONArray(); + stackArrays.push(array); + } + + @Override + protected void closeArray(String key, boolean nested, boolean newline) throws IOException { + try { + if (stackArrays.size() > 1) { + JSONArray array = stackArrays.pop(); + if (nested) { + stackArrays.getFirst().put(array); + } + else { + stackObjects.getFirst().put(key, array); + } + } + } catch (JSONException jsone) { + throw new IOException(jsone.getMessage()); + } + } + + @Override + protected void writeArrayValue(String value) throws IOException { + if (stackArrays.size() > 1) { + stackArrays.getFirst().put(value); + } + } + + @Override + protected void startObject(String key) throws IOException { + JSONObject object = new JSONObject(); + stackObjects.push(object); + } + + @Override + protected void closeObject(String key) throws IOException { + try { + if (stackObjects.size() > 1) { + JSONObject object = stackObjects.pop(); + stackObjects.getFirst().put(key, object); + } + } catch (JSONException jsone) { + throw new IOException(jsone.getMessage()); + } + } + + @Override + protected String generateJson() throws IOException { + try { + return stackObjects.getFirst().toString(2); + } catch (JSONException jsone) { + throw new IOException(jsone.getMessage()); + } + } } diff --git a/src/java/org/apache/nutch/tools/CommonCrawlFormatSimple.java b/src/java/org/apache/nutch/tools/CommonCrawlFormatSimple.java index 4a592a08d5..b0ac8c4b1e 100644 --- a/src/java/org/apache/nutch/tools/CommonCrawlFormatSimple.java +++ b/src/java/org/apache/nutch/tools/CommonCrawlFormatSimple.java @@ -28,95 +28,95 @@ * */ public class CommonCrawlFormatSimple extends AbstractCommonCrawlFormat { - - private StringBuilder sb; - - private int tabCount; - - public CommonCrawlFormatSimple(String url, Content content, Metadata metadata, Configuration nutchConf, CommonCrawlConfig config) throws IOException { - super(url, content, metadata, nutchConf, config); - - this.sb = new StringBuilder(); - this.tabCount = 0; - } - - @Override - protected void writeKeyValue(String key, String value) throws IOException { - sb.append(printTabs() + "\"" + key + "\": " + quote(value) + ",\n"); - } - - @Override - protected void writeKeyNull(String key) throws IOException { - sb.append(printTabs() + "\"" + key + "\": null,\n"); - } - - @Override - protected void startArray(String key, boolean nested, boolean newline) throws IOException { - String name = (key != null) ? "\"" + key + "\": " : ""; - String nl = (newline) ? "\n" : ""; - sb.append(printTabs() + name + "[" + nl); - if (newline) { - this.tabCount++; - } - } - - @Override - protected void closeArray(String key, boolean nested, boolean newline) throws IOException { - if (sb.charAt(sb.length()-1) == ',') { - sb.deleteCharAt(sb.length()-1); // delete comma - } - else if (sb.charAt(sb.length()-2) == ',') { - sb.deleteCharAt(sb.length()-2); // delete comma - } - String nl = (newline) ? printTabs() : ""; - if (newline) { - this.tabCount++; - } - sb.append(nl + "],\n"); - } - - @Override - protected void writeArrayValue(String value) { - sb.append("\"" + value + "\","); - } - - @Override + + private StringBuilder sb; + + private int tabCount; + + public CommonCrawlFormatSimple(String url, Content content, Metadata metadata, Configuration nutchConf, CommonCrawlConfig config) throws IOException { + super(url, content, metadata, nutchConf, config); + + this.sb = new StringBuilder(); + this.tabCount = 0; + } + + @Override + protected void writeKeyValue(String key, String value) throws IOException { + sb.append(printTabs() + "\"" + key + "\": " + quote(value) + ",\n"); + } + + @Override + protected void writeKeyNull(String key) throws IOException { + sb.append(printTabs() + "\"" + key + "\": null,\n"); + } + + @Override + protected void startArray(String key, boolean nested, boolean newline) throws IOException { + String name = (key != null) ? "\"" + key + "\": " : ""; + String nl = (newline) ? "\n" : ""; + sb.append(printTabs() + name + "[" + nl); + if (newline) { + this.tabCount++; + } + } + + @Override + protected void closeArray(String key, boolean nested, boolean newline) throws IOException { + if (sb.charAt(sb.length()-1) == ',') { + sb.deleteCharAt(sb.length()-1); // delete comma + } + else if (sb.charAt(sb.length()-2) == ',') { + sb.deleteCharAt(sb.length()-2); // delete comma + } + String nl = (newline) ? printTabs() : ""; + if (newline) { + this.tabCount++; + } + sb.append(nl + "],\n"); + } + + @Override + protected void writeArrayValue(String value) { + sb.append("\"" + value + "\","); + } + + @Override protected void startObject(String key) throws IOException { - String name = ""; - if (key != null) { - name = "\"" + key + "\": "; - } - sb.append(printTabs() + name + "{\n"); - this.tabCount++; - } - - @Override + String name = ""; + if (key != null) { + name = "\"" + key + "\": "; + } + sb.append(printTabs() + name + "{\n"); + this.tabCount++; + } + + @Override protected void closeObject(String key) throws IOException { - if (sb.charAt(sb.length()-2) == ',') { - sb.deleteCharAt(sb.length()-2); // delete comma - } - this.tabCount--; - sb.append(printTabs() + "},\n"); - } - - @Override + if (sb.charAt(sb.length()-2) == ',') { + sb.deleteCharAt(sb.length()-2); // delete comma + } + this.tabCount--; + sb.append(printTabs() + "},\n"); + } + + @Override protected String generateJson() throws IOException { - sb.deleteCharAt(sb.length()-1); // delete new line - sb.deleteCharAt(sb.length()-1); // delete comma - return sb.toString(); - } - - private String printTabs() { - StringBuilder sb = new StringBuilder(); - for (int i=0; i < this.tabCount ;i++) { - sb.append("\t"); - } - return sb.toString(); - } - + sb.deleteCharAt(sb.length()-1); // delete new line + sb.deleteCharAt(sb.length()-1); // delete comma + return sb.toString(); + } + + private String printTabs() { + StringBuilder sb = new StringBuilder(); + for (int i=0; i < this.tabCount ;i++) { + sb.append("\t"); + } + return sb.toString(); + } + private static String quote(String string) throws IOException { - StringBuilder sb = new StringBuilder(); - + StringBuilder sb = new StringBuilder(); + if (string == null || string.length() == 0) { sb.append("\"\""); return sb.toString(); @@ -140,34 +140,34 @@ private static String quote(String string) throws IOException { break; case '/': if (b == '<') { - sb.append('\\'); + sb.append('\\'); } sb.append(c); break; case '\b': - sb.append("\\b"); + sb.append("\\b"); break; case '\t': - sb.append("\\t"); + sb.append("\\t"); break; case '\n': - sb.append("\\n"); + sb.append("\\n"); break; case '\f': - sb.append("\\f"); + sb.append("\\f"); break; case '\r': - sb.append("\\r"); + sb.append("\\r"); break; default: if (c < ' ' || (c >= '\u0080' && c < '\u00a0') || (c >= '\u2000' && c < '\u2100')) { - sb.append("\\u"); + sb.append("\\u"); hhhh = Integer.toHexString(c); sb.append("0000", 0, 4 - hhhh.length()); sb.append(hhhh); } else { - sb.append(c); + sb.append(c); } } } diff --git a/src/java/org/apache/nutch/util/CrawlCompletionStats.java b/src/java/org/apache/nutch/util/CrawlCompletionStats.java index ac815f74f2..8bdda5ce87 100644 --- a/src/java/org/apache/nutch/util/CrawlCompletionStats.java +++ b/src/java/org/apache/nutch/util/CrawlCompletionStats.java @@ -52,8 +52,8 @@ * Extracts some simple crawl completion stats from the crawldb * * Stats will be sorted by host/domain and will be of the form: - * 1 www.spitzer.caltech.edu FETCHED - * 50 www.spitzer.caltech.edu UNFETCHED + * 1 www.spitzer.caltech.edu FETCHED + * 50 www.spitzer.caltech.edu UNFETCHED * */ public class CrawlCompletionStats extends Configured implements Tool { diff --git a/src/java/org/apache/nutch/util/DumpFileUtil.java b/src/java/org/apache/nutch/util/DumpFileUtil.java index 9461a01776..da4ad7934b 100644 --- a/src/java/org/apache/nutch/util/DumpFileUtil.java +++ b/src/java/org/apache/nutch/util/DumpFileUtil.java @@ -30,8 +30,8 @@ import org.slf4j.LoggerFactory; public class DumpFileUtil { - private static final Logger LOG = LoggerFactory - .getLogger(MethodHandles.lookup().lookupClass()); + private static final Logger LOG = LoggerFactory + .getLogger(MethodHandles.lookup().lookupClass()); private final static String DIR_PATTERN = "%s/%s/%s"; private final static String FILENAME_PATTERN = "%s_%s.%s"; @@ -57,12 +57,12 @@ public static String createTwoLevelsDirectory(String basePath, String md5, boole firstLevelDirName, secondLevelDirName); if (makeDir) { - try { - FileUtils.forceMkdir(new File(fullDirPath)); - } catch (IOException e) { - LOG.error("Failed to create dir: {}", fullDirPath); - fullDirPath = null; - } + try { + FileUtils.forceMkdir(new File(fullDirPath)); + } catch (IOException e) { + LOG.error("Failed to create dir: {}", fullDirPath); + fullDirPath = null; + } } return fullDirPath; @@ -83,7 +83,7 @@ public static String createFileName(String md5, String fileBaseName, String file fileExtension = StringUtils.substring(fileExtension, 0, MAX_LENGTH_OF_EXTENSION); } - // Added to prevent FileNotFoundException (Invalid Argument) - in *nix environment + // Added to prevent FileNotFoundException (Invalid Argument) - in *nix environment fileBaseName = fileBaseName.replaceAll("\\?", ""); fileExtension = fileExtension.replaceAll("\\?", ""); @@ -119,38 +119,38 @@ public static String createFileNameFromUrl(String basePath, return outputFullPath; } - public static String displayFileTypes(Map typeCounts, Map filteredCounts) { - StringBuilder builder = new StringBuilder(); - // print total stats - builder.append("\nTOTAL Stats:\n"); - builder.append("[\n"); - int mimetypeCount = 0; - for (String mimeType : typeCounts.keySet()) { - builder.append(" {\"mimeType\":\""); - builder.append(mimeType); - builder.append("\",\"count\":\""); - builder.append(typeCounts.get(mimeType)); - builder.append("\"}\n"); - mimetypeCount += typeCounts.get(mimeType); - } - builder.append("]\n"); - builder.append("Total count: " + mimetypeCount + "\n"); - // filtered types stats - mimetypeCount = 0; - if (!filteredCounts.isEmpty()) { - builder.append("\nFILTERED Stats:\n"); - builder.append("[\n"); - for (String mimeType : filteredCounts.keySet()) { - builder.append(" {\"mimeType\":\""); - builder.append(mimeType); - builder.append("\",\"count\":\""); - builder.append(filteredCounts.get(mimeType)); - builder.append("\"}\n"); - mimetypeCount += filteredCounts.get(mimeType); - } - builder.append("]\n"); - builder.append("Total filtered count: " + mimetypeCount + "\n"); - } - return builder.toString(); + public static String displayFileTypes(Map typeCounts, Map filteredCounts) { + StringBuilder builder = new StringBuilder(); + // print total stats + builder.append("\nTOTAL Stats:\n"); + builder.append("[\n"); + int mimetypeCount = 0; + for (String mimeType : typeCounts.keySet()) { + builder.append(" {\"mimeType\":\""); + builder.append(mimeType); + builder.append("\",\"count\":\""); + builder.append(typeCounts.get(mimeType)); + builder.append("\"}\n"); + mimetypeCount += typeCounts.get(mimeType); + } + builder.append("]\n"); + builder.append("Total count: " + mimetypeCount + "\n"); + // filtered types stats + mimetypeCount = 0; + if (!filteredCounts.isEmpty()) { + builder.append("\nFILTERED Stats:\n"); + builder.append("[\n"); + for (String mimeType : filteredCounts.keySet()) { + builder.append(" {\"mimeType\":\""); + builder.append(mimeType); + builder.append("\",\"count\":\""); + builder.append(filteredCounts.get(mimeType)); + builder.append("\"}\n"); + mimetypeCount += filteredCounts.get(mimeType); + } + builder.append("]\n"); + builder.append("Total filtered count: " + mimetypeCount + "\n"); + } + return builder.toString(); } } diff --git a/src/java/org/apache/nutch/util/ProtocolStatusStatistics.java b/src/java/org/apache/nutch/util/ProtocolStatusStatistics.java index 3c37887cc2..4d20dde8c2 100644 --- a/src/java/org/apache/nutch/util/ProtocolStatusStatistics.java +++ b/src/java/org/apache/nutch/util/ProtocolStatusStatistics.java @@ -50,10 +50,10 @@ * An example output run showing the number of encountered status * codes such as 200, 300, and a count of un-fetched record. * - * 38 200 - * 19 301 - * 2 302 - * 665 UNFETCHED + * 38 200 + * 19 301 + * 2 302 + * 665 UNFETCHED * */ public class ProtocolStatusStatistics extends Configured implements Tool { diff --git a/src/plugin/build-plugin.xml b/src/plugin/build-plugin.xml index 0dd4da342c..d8964c1717 100755 --- a/src/plugin/build-plugin.xml +++ b/src/plugin/build-plugin.xml @@ -185,7 +185,7 @@ - + diff --git a/src/plugin/feed/build.xml b/src/plugin/feed/build.xml index 2501561378..8f7c4d99d6 100644 --- a/src/plugin/feed/build.xml +++ b/src/plugin/feed/build.xml @@ -1,19 +1,19 @@ diff --git a/src/plugin/feed/plugin.xml b/src/plugin/feed/plugin.xml index 342bb28c76..eed25b5c4f 100644 --- a/src/plugin/feed/plugin.xml +++ b/src/plugin/feed/plugin.xml @@ -1,22 +1,22 @@ + provider-name="nutch.org"> diff --git a/src/plugin/feed/sample/rsstest.rss b/src/plugin/feed/sample/rsstest.rss index 83caae152e..cd6a1a12dc 100644 --- a/src/plugin/feed/sample/rsstest.rss +++ b/src/plugin/feed/sample/rsstest.rss @@ -1,19 +1,19 @@ diff --git a/src/plugin/index-arbitrary/src/java/org/apache/nutch/indexer/arbitrary/ArbitraryIndexingFilter.java b/src/plugin/index-arbitrary/src/java/org/apache/nutch/indexer/arbitrary/ArbitraryIndexingFilter.java index 901364985a..a5dca3f365 100644 --- a/src/plugin/index-arbitrary/src/java/org/apache/nutch/indexer/arbitrary/ArbitraryIndexingFilter.java +++ b/src/plugin/index-arbitrary/src/java/org/apache/nutch/indexer/arbitrary/ArbitraryIndexingFilter.java @@ -188,39 +188,39 @@ public NutchDocument filter(NutchDocument doc, Parse parse, Text url, } if (allFieldsAccess) { theConstructor = theClass.getDeclaredConstructor(String[].class, - NutchDocument.class, - Parse.class, - Text.class, - CrawlDatum.class, - Inlinks.class); - } else { + NutchDocument.class, + Parse.class, + Text.class, + CrawlDatum.class, + Inlinks.class); + } else { theConstructor = theClass.getDeclaredConstructor(String[].class); - } + } } catch (NoSuchMethodException nme) { LOG.error("Exception preparing reflection for constructor. className was {}", - String.valueOf(className)); + String.valueOf(className)); nme.printStackTrace(); continue; } catch (Exception e) { LOG.error("Exception preparing reflection tasks. className was {}", - String.valueOf(className)); + String.valueOf(className)); e.printStackTrace(); continue; } try { constrArgs = new String[userConstrArgs.length + 1]; System.arraycopy(userConstrArgs,0,constrArgs,1,userConstrArgs.length); - if (allFieldsAccess) { + if (allFieldsAccess) { instance = theConstructor.newInstance(constrArgs, - doc, - parse, - url, - datum, - inlinks); - } else { + doc, + parse, + url, + datum, + inlinks); + } else { constrArgs[0] = url.toString(); instance = theConstructor.newInstance(new Object[]{constrArgs}); - } + } if (methodArgs.length > 0) { result = theMethod.invoke(instance, new Object[]{methodArgs}); @@ -229,8 +229,8 @@ public NutchDocument filter(NutchDocument doc, Parse parse, Text url, } } catch (Exception e) { LOG.error("Exception in reflection trying to instantiate/invoke. " - + "url was {} & className was {}", - String.valueOf(url), String.valueOf(className)); + + "url was {} & className was {}", + String.valueOf(url), String.valueOf(className)); if (constrArgs.length > 1) { LOG.error("constrArgs[1] was {}", String.valueOf(constrArgs[1])); } @@ -239,19 +239,19 @@ public NutchDocument filter(NutchDocument doc, Parse parse, Text url, LOG.error("methodArgs[0] was {}", String.valueOf(methodArgs[0])); } e.printStackTrace(); - continue; + continue; } LOG.debug("{}.{}() returned {} for field {}.", className, - methodName, String.valueOf(result), String.valueOf(fieldName)); + methodName, String.valueOf(result), String.valueOf(fieldName)); // If user chose to overwrite, remove existing value if (overwrite) { - LOG.debug("overwrite == true for fieldName == {} ", fieldName); - if (doc.getFieldNames().contains(fieldName)) { - LOG.debug("Removing field '{}' from doc for overwrite", fieldName); - doc.removeField(fieldName); - } + LOG.debug("overwrite == true for fieldName == {} ", fieldName); + if (doc.getFieldNames().contains(fieldName)) { + LOG.debug("Removing field '{}' from doc for overwrite", fieldName); + doc.removeField(fieldName); + } } if (result == null) { LOG.debug("Call to {}.{} returned null", className, methodName); @@ -288,7 +288,7 @@ public void setIndexedConf(Configuration conf, int ndx) { LOG.debug("In setIndexedConf() where ndx was passed in as {}", String.valueOf(ndx)); fieldName = conf.get("index.arbitrary.fieldName.".concat(String.valueOf(ndx))); LOG.debug("Looking now for index.arbitrary.fieldname.{} which was: {}", - String.valueOf(ndx),String.valueOf(fieldName)); + String.valueOf(ndx),String.valueOf(fieldName)); if (fieldName == null || fieldName == "") { throw new RuntimeException ("Problem in configuration where the index.arbitrary.fieldName." diff --git a/src/plugin/index-arbitrary/src/test/org/apache/nutch/indexer/arbitrary/Multiplier.java b/src/plugin/index-arbitrary/src/test/org/apache/nutch/indexer/arbitrary/Multiplier.java index 749be508d2..5cfc0b35ea 100644 --- a/src/plugin/index-arbitrary/src/test/org/apache/nutch/indexer/arbitrary/Multiplier.java +++ b/src/plugin/index-arbitrary/src/test/org/apache/nutch/indexer/arbitrary/Multiplier.java @@ -41,7 +41,7 @@ public String getProduct(String args[]) { } public static void main(String[] args) { - Multiplier mp = new Multiplier(args); - out.println(mp.getProduct(args)); + Multiplier mp = new Multiplier(args); + out.println(mp.getProduct(args)); } } diff --git a/src/plugin/index-arbitrary/src/test/org/apache/nutch/indexer/arbitrary/PopularityGauge.java b/src/plugin/index-arbitrary/src/test/org/apache/nutch/indexer/arbitrary/PopularityGauge.java index 4b8df21243..07b66a6bef 100644 --- a/src/plugin/index-arbitrary/src/test/org/apache/nutch/indexer/arbitrary/PopularityGauge.java +++ b/src/plugin/index-arbitrary/src/test/org/apache/nutch/indexer/arbitrary/PopularityGauge.java @@ -39,7 +39,7 @@ public class PopularityGauge { private double popularityBoost; public PopularityGauge(String args[], - NutchDocument docIn, + NutchDocument docIn, Parse parseIn, Text urlIn, CrawlDatum datumIn, @@ -57,24 +57,24 @@ public double getPopularityBoost() { if(anchorSet.contains("dinosaur")){ popularityBoost = popularityBoost + 0.50; } else { - popularityBoost = popularityBoost - 0.20; + popularityBoost = popularityBoost - 0.20; } if (anchorSet.contains("baseball")) { popularityBoost = popularityBoost + 0.25; } else { - popularityBoost = popularityBoost - 0.15; + popularityBoost = popularityBoost - 0.15; } if (anchorSet.contains("source code")) { popularityBoost = popularityBoost + 0.25; } else { - popularityBoost = popularityBoost - 0.15; + popularityBoost = popularityBoost - 0.15; } } return popularityBoost; } public static void main(String[] args, - NutchDocument doc, + NutchDocument doc, Parse parse, Text url, CrawlDatum datum, diff --git a/src/plugin/index-replace/build.xml b/src/plugin/index-replace/build.xml index ea8c95d9f4..e0e4a41cdf 100644 --- a/src/plugin/index-replace/build.xml +++ b/src/plugin/index-replace/build.xml @@ -17,39 +17,39 @@ --> - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/src/plugin/index-replace/src/java/org/apache/nutch/indexer/replace/ReplaceIndexer.java b/src/plugin/index-replace/src/java/org/apache/nutch/indexer/replace/ReplaceIndexer.java index c21ae242d2..a882df516e 100644 --- a/src/plugin/index-replace/src/java/org/apache/nutch/indexer/replace/ReplaceIndexer.java +++ b/src/plugin/index-replace/src/java/org/apache/nutch/indexer/replace/ReplaceIndexer.java @@ -85,7 +85,7 @@ public class ReplaceIndexer implements IndexingFilter { private static final Logger LOG = LoggerFactory - .getLogger(MethodHandles.lookup().lookupClass()); + .getLogger(MethodHandles.lookup().lookupClass()); /** Special field name signifying the start of a host-specific match set */ private static final String HOSTMATCH = "hostmatch"; diff --git a/src/plugin/indexer-cloudsearch/ivy.xml b/src/plugin/indexer-cloudsearch/ivy.xml index a9af5f6c80..0b1018e133 100644 --- a/src/plugin/indexer-cloudsearch/ivy.xml +++ b/src/plugin/indexer-cloudsearch/ivy.xml @@ -37,7 +37,7 @@ - + diff --git a/src/plugin/indexer-solr/ivy.xml b/src/plugin/indexer-solr/ivy.xml index 5b2642ad76..0b0410bff3 100644 --- a/src/plugin/indexer-solr/ivy.xml +++ b/src/plugin/indexer-solr/ivy.xml @@ -17,52 +17,52 @@ --> - - - - - Apache Nutch - - + xsi:noNamespaceSchemaLocation="http://ant.apache.org/ivy/schemas/ivy.xsd" + xmlns:ns0="http://ant.apache.org/ivy/maven" version="2.0"> + + + + + Apache Nutch + + - - - + + + - - - - + + + + - - - - - - - - - - - - - - - - - - - - - - - + + + + + + + + + + + + + + + + + + + + + + + diff --git a/src/plugin/indexer-solr/plugin.xml b/src/plugin/indexer-solr/plugin.xml index be98131414..6b9a359b5f 100644 --- a/src/plugin/indexer-solr/plugin.xml +++ b/src/plugin/indexer-solr/plugin.xml @@ -16,56 +16,56 @@ limitations under the License. --> + provider-name="nutch.apache.org"> - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + - - - + + + - - - + + + diff --git a/src/plugin/indexer-solr/schema.xml b/src/plugin/indexer-solr/schema.xml index cf7759d92d..67e44217e2 100644 --- a/src/plugin/indexer-solr/schema.xml +++ b/src/plugin/indexer-solr/schema.xml @@ -111,9 +111,9 @@ + removes stop words from case-insensitive "stopwords.txt" + (empty by default), and down cases. At query time only, it + also applies synonyms. --> @@ -147,11 +147,11 @@ words="stopwords.txt" /> - + - + --> @@ -162,26 +162,26 @@ words="stopwords.txt" /> - + - + --> @@ -230,7 +230,7 @@ + each token, to enable more efficient leading wildcard queries. --> @@ -262,10 +262,10 @@ a token of "foo|1.4" would be indexed as "foo" with a payload of 1.4f Attributes of the DelimitedPayloadTokenFilterFactory : "delimiter" - a one character delimiter. Default is | (pipe) - "encoder" - how to encode the following value into a playload - float -> org.apache.lucene.analysis.payloads.FloatEncoder, - integer -> o.a.l.a.p.IntegerEncoder - identity -> o.a.l.a.p.IdentityEncoder + "encoder" - how to encode the following value into a playload + float -> org.apache.lucene.analysis.payloads.FloatEncoder, + integer -> o.a.l.a.p.IntegerEncoder + identity -> o.a.l.a.p.IdentityEncoder Fully Qualified class name implementing PayloadEncoder, Encoder must have a no arg constructor. --> diff --git a/src/plugin/language-identifier/src/java/org/apache/nutch/analysis/lang/LanguageIndexingFilter.java b/src/plugin/language-identifier/src/java/org/apache/nutch/analysis/lang/LanguageIndexingFilter.java index 754b92c7b8..02eade2090 100644 --- a/src/plugin/language-identifier/src/java/org/apache/nutch/analysis/lang/LanguageIndexingFilter.java +++ b/src/plugin/language-identifier/src/java/org/apache/nutch/analysis/lang/LanguageIndexingFilter.java @@ -75,7 +75,7 @@ public NutchDocument filter(NutchDocument doc, Parse parse, Text url, } if (!indexLangs.isEmpty() && !indexLangs.contains(lang)) { - return null; + return null; } doc.add("lang", lang); diff --git a/src/plugin/lib-htmlunit/src/java/org/apache/nutch/protocol/htmlunit/HtmlUnitWebDriver.java b/src/plugin/lib-htmlunit/src/java/org/apache/nutch/protocol/htmlunit/HtmlUnitWebDriver.java index 9b63a4fddc..9f26c207b6 100644 --- a/src/plugin/lib-htmlunit/src/java/org/apache/nutch/protocol/htmlunit/HtmlUnitWebDriver.java +++ b/src/plugin/lib-htmlunit/src/java/org/apache/nutch/protocol/htmlunit/HtmlUnitWebDriver.java @@ -56,7 +56,7 @@ protected WebClient modifyWebClient(WebClient client) { client.getOptions().setThrowExceptionOnScriptError(false); if(enableRedirect) client.addWebWindowListener(new HtmlUnitWebWindowListener(maxRedirects)); - return client; + return client; } public static WebDriver getDriverForPage(String url, Configuration conf) { @@ -67,9 +67,9 @@ public static WebDriver getDriverForPage(String url, Configuration conf) { int redirects = Integer.parseInt(conf.get("http.redirect.max", "0")); enableRedirect = redirects > 0; maxRedirects = redirects; - + WebDriver driver = null; - + try { driver = new HtmlUnitWebDriver(); driver.manage().timeouts().pageLoadTimeout(Duration.of(pageLoadTimout, @@ -77,8 +77,8 @@ public static WebDriver getDriverForPage(String url, Configuration conf) { driver.get(url); } catch(Exception e) { if(e instanceof TimeoutException) { - LOG.debug("HtmlUnit WebDriver: Timeout Exception: Capturing whatever loaded so far..."); - return driver; + LOG.debug("HtmlUnit WebDriver: Timeout Exception: Capturing whatever loaded so far..."); + return driver; } cleanUpDriver(driver); throw new RuntimeException(e); @@ -91,19 +91,19 @@ public static String getHTMLContent(WebDriver driver, Configuration conf) { try { if (conf.getBoolean("take.screenshot", false)) takeScreenshot(driver, conf); - + String innerHtml = ""; if(enableJavascript) { - WebElement body = driver.findElement(By.tagName("body")); - innerHtml = (String)((JavascriptExecutor)driver).executeScript("return arguments[0].innerHTML;", body); + WebElement body = driver.findElement(By.tagName("body")); + innerHtml = (String)((JavascriptExecutor)driver).executeScript("return arguments[0].innerHTML;", body); } else - innerHtml = driver.getPageSource().replaceAll("&", "&"); + innerHtml = driver.getPageSource().replaceAll("&", "&"); return innerHtml; } catch(Exception e) { - TemporaryFilesystem.getDefaultTmpFS().deleteTemporaryFiles(); - cleanUpDriver(driver); - throw new RuntimeException(e); + TemporaryFilesystem.getDefaultTmpFS().deleteTemporaryFiles(); + cleanUpDriver(driver); + throw new RuntimeException(e); } } @@ -135,19 +135,19 @@ public static String getHtmlPage(String url, Configuration conf) { try { if (conf.getBoolean("take.screenshot", false)) - takeScreenshot(driver, conf); + takeScreenshot(driver, conf); String innerHtml = ""; if(enableJavascript) { - WebElement body = driver.findElement(By.tagName("body")); - innerHtml = (String)((JavascriptExecutor)driver).executeScript("return arguments[0].innerHTML;", body); + WebElement body = driver.findElement(By.tagName("body")); + innerHtml = (String)((JavascriptExecutor)driver).executeScript("return arguments[0].innerHTML;", body); } else - innerHtml = driver.getPageSource().replaceAll("&", "&"); + innerHtml = driver.getPageSource().replaceAll("&", "&"); return innerHtml; } catch (Exception e) { - TemporaryFilesystem.getDefaultTmpFS().deleteTemporaryFiles(); + TemporaryFilesystem.getDefaultTmpFS().deleteTemporaryFiles(); throw new RuntimeException(e); } finally { cleanUpDriver(driver); @@ -161,7 +161,7 @@ private static void takeScreenshot(WebDriver driver, Configuration conf) { LOG.debug("In-memory screenshot taken of: {}", url); FileSystem fs = FileSystem.get(conf); if (conf.get("screenshot.location") != null) { - Path screenshotPath = new Path(conf.get("screenshot.location") + "/" + srcFile.getName()); + Path screenshotPath = new Path(conf.get("screenshot.location") + "/" + srcFile.getName()); OutputStream os = null; if (!fs.exists(screenshotPath)) { LOG.debug("No existing screenshot already exists... creating new file at {} {}.", screenshotPath, srcFile.getName()); @@ -175,8 +175,8 @@ private static void takeScreenshot(WebDriver driver, Configuration conf) { + "'screenshot.location' is absent from nutch-site.xml.", url); } } catch (Exception e) { - cleanUpDriver(driver); - throw new RuntimeException(e); + cleanUpDriver(driver); + throw new RuntimeException(e); } } } diff --git a/src/plugin/lib-xml/build.xml b/src/plugin/lib-xml/build.xml index 0f87c073eb..189f1ff325 100644 --- a/src/plugin/lib-xml/build.xml +++ b/src/plugin/lib-xml/build.xml @@ -17,20 +17,20 @@ --> - + - - + - + diff --git a/src/plugin/parse-metatags/build.xml b/src/plugin/parse-metatags/build.xml index e30292d92b..412ba3f298 100644 --- a/src/plugin/parse-metatags/build.xml +++ b/src/plugin/parse-metatags/build.xml @@ -17,21 +17,21 @@ --> - + - - - - - + + + + + - - - - - - - + + + + + + + diff --git a/src/plugin/parse-tika/sample/nutch.html b/src/plugin/parse-tika/sample/nutch.html index ec7a196915..2353896a69 100644 --- a/src/plugin/parse-tika/sample/nutch.html +++ b/src/plugin/parse-tika/sample/nutch.html @@ -237,7 +237,7 @@

Welcome to Nutch!

  • 09 February 2009 - Lucene at ApacheCon Europe 2009 in - Amsterdam + Amsterdam
  • 2 April 2007: Nutch 0.9 Released @@ -391,23 +391,23 @@

    23 March 2009 - Apache Nutch 1.0 Released

    here.

    09 February 2009 - Lucene at ApacheCon Europe 2009 in - Amsterdam

    + Amsterdam

    - + - ApacheCon EU 2009 Logo - + ApacheCon EU 2009 Logo + - Lucene will be extremely well represented at - ApacheCon EU 2009 - in Amsterdam, Netherlands this March 23-27, 2009: -

    + Lucene will be extremely well represented at + ApacheCon EU 2009 + in Amsterdam, Netherlands this March 23-27, 2009: +

      - +
    • - + Lucene Boot Camp - - A two day training session, March 23 & 24th
    • + - A two day training session, March 23 & 24th
    • Solr Boot Camp - A one day training session, March 24th
    • diff --git a/src/plugin/parse-tika/sample/rsstest.rss b/src/plugin/parse-tika/sample/rsstest.rss index 8fb241d62a..5384526391 100644 --- a/src/plugin/parse-tika/sample/rsstest.rss +++ b/src/plugin/parse-tika/sample/rsstest.rss @@ -1,19 +1,19 @@ diff --git a/src/plugin/parse-tika/src/java/org/apache/nutch/parse/tika/BoilerpipeExtractorRepository.java b/src/plugin/parse-tika/src/java/org/apache/nutch/parse/tika/BoilerpipeExtractorRepository.java index 5d65318b6e..1403546afa 100644 --- a/src/plugin/parse-tika/src/java/org/apache/nutch/parse/tika/BoilerpipeExtractorRepository.java +++ b/src/plugin/parse-tika/src/java/org/apache/nutch/parse/tika/BoilerpipeExtractorRepository.java @@ -26,7 +26,7 @@ class BoilerpipeExtractorRepository { private static final Logger LOG = LoggerFactory - .getLogger(MethodHandles.lookup().lookupClass()); + .getLogger(MethodHandles.lookup().lookupClass()); public static final HashMap extractorRepository = new HashMap<>(); /** diff --git a/src/plugin/scoring-depth/src/java/org/apache/nutch/scoring/depth/DepthScoringFilter.java b/src/plugin/scoring-depth/src/java/org/apache/nutch/scoring/depth/DepthScoringFilter.java index 561bdca1f4..88bdf9a579 100644 --- a/src/plugin/scoring-depth/src/java/org/apache/nutch/scoring/depth/DepthScoringFilter.java +++ b/src/plugin/scoring-depth/src/java/org/apache/nutch/scoring/depth/DepthScoringFilter.java @@ -48,7 +48,7 @@ public class DepthScoringFilter extends Configured implements ScoringFilter { private static final Logger LOG = LoggerFactory - .getLogger(MethodHandles.lookup().lookupClass()); + .getLogger(MethodHandles.lookup().lookupClass()); public static final String DEPTH_KEY = "_depth_"; public static final Text DEPTH_KEY_W = new Text(DEPTH_KEY); diff --git a/src/test/org/apache/nutch/parse/parse-plugin-test.xml b/src/test/org/apache/nutch/parse/parse-plugin-test.xml index f6a4267763..2eeb483915 100644 --- a/src/test/org/apache/nutch/parse/parse-plugin-test.xml +++ b/src/test/org/apache/nutch/parse/parse-plugin-test.xml @@ -26,7 +26,7 @@ - + - + - + @@ -53,6 +53,6 @@ + extension-id="org.apache.nutch.parse.tika.TikaParser" />