diff --git a/.github/workflows/master-build.yml b/.github/workflows/master-build.yml index 60e97a7349..4654a82cdb 100644 --- a/.github/workflows/master-build.yml +++ b/.github/workflows/master-build.yml @@ -65,6 +65,7 @@ jobs: java: ['11'] os: [ubuntu-latest, macos-latest] runs-on: ${{ matrix.os }} + timeout-minutes: 30 steps: - uses: actions/checkout@v4.2.2 - name: Set up JDK ${{ matrix.java }} diff --git a/build.xml b/build.xml index b24222ff2e..c4b0b0d661 100644 --- a/build.xml +++ b/build.xml @@ -485,12 +485,12 @@ - - - - + + + + @@ -502,9 +502,10 @@ - - - + + + + @@ -516,7 +517,6 @@ - Tests failed! diff --git a/default.properties b/default.properties index 4a79aad3f3..6008563bcf 100644 --- a/default.properties +++ b/default.properties @@ -38,7 +38,6 @@ test.build.lib.dir = ${test.build.dir}/lib test.build.data = ${test.build.dir}/data test.build.classes = ${test.build.dir}/classes test.build.javadoc = ${test.build.dir}/docs/api -test.junit.output.format = legacy-plain # Proxy Host and Port to use for building JavaDoc javadoc.proxy.host=-J-DproxyHost= diff --git a/ivy/ivy.xml b/ivy/ivy.xml index cf32e0bc08..a822f9c545 100644 --- a/ivy/ivy.xml +++ b/ivy/ivy.xml @@ -118,12 +118,12 @@ - - - + + - - + + + diff --git a/src/plugin/build-plugin.xml b/src/plugin/build-plugin.xml index 5dabc3a7c1..a4f737eda7 100755 --- a/src/plugin/build-plugin.xml +++ b/src/plugin/build-plugin.xml @@ -80,6 +80,7 @@ + @@ -208,28 +209,13 @@ - - - - - + + + + @@ -240,18 +226,19 @@ - - + + + + - + - Tests failed! diff --git a/src/plugin/creativecommons/src/test/org/creativecommons/nutch/TestCCParseFilter.java b/src/plugin/creativecommons/src/test/org/creativecommons/nutch/TestCCParseFilter.java index de189683b2..68f4a78d4f 100644 --- a/src/plugin/creativecommons/src/test/org/creativecommons/nutch/TestCCParseFilter.java +++ b/src/plugin/creativecommons/src/test/org/creativecommons/nutch/TestCCParseFilter.java @@ -22,17 +22,18 @@ import org.apache.nutch.protocol.Content; import org.apache.hadoop.conf.Configuration; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; import java.io.*; -public class TestCCParseFilter { +import static org.junit.jupiter.api.Assertions.assertEquals; + +class TestCCParseFilter { private static final File testDir = new File(System.getProperty("test.input")); @Test - public void testPages() throws Exception { + void testPages() throws Exception { pageTest(new File(testDir, "anchor.html"), "http://foo.com/", "http://creativecommons.org/licenses/by-nc-sa/1.0", "a", null); // Tika returns whereas parse-html returns @@ -65,8 +66,8 @@ public void pageTest(File file, String url, String license, String location, Parse parse = new ParseUtil(conf).parse(content).get(content.getUrl()); Metadata metadata = parse.getData().getParseMeta(); - Assert.assertEquals(license, metadata.get("License-Url")); - Assert.assertEquals(location, metadata.get("License-Location")); - Assert.assertEquals(type, metadata.get("Work-Type")); + assertEquals(license, metadata.get("License-Url")); + assertEquals(location, metadata.get("License-Location")); + assertEquals(type, metadata.get("Work-Type")); } } diff --git a/src/plugin/feed/src/test/org/apache/nutch/parse/feed/TestFeedParser.java b/src/plugin/feed/src/test/org/apache/nutch/parse/feed/TestFeedParser.java index 915ee54af0..1ea629172d 100644 --- a/src/plugin/feed/src/test/org/apache/nutch/parse/feed/TestFeedParser.java +++ b/src/plugin/feed/src/test/org/apache/nutch/parse/feed/TestFeedParser.java @@ -19,8 +19,7 @@ import java.util.Iterator; import java.util.Map; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; import org.apache.hadoop.conf.Configuration; import org.apache.hadoop.io.Text; @@ -35,6 +34,8 @@ import org.apache.nutch.protocol.ProtocolNotFound; import org.apache.nutch.util.NutchConfiguration; +import static org.junit.jupiter.api.Assertions.*; + /** * * @author mattmann @@ -86,7 +87,7 @@ public void testParseFetchChannel() throws ProtocolNotFound, ParseException { parseResult = new ParseUtil(conf).parseByExtensionId("feed", content); - Assert.assertEquals(3, parseResult.size()); + assertEquals(3, parseResult.size()); boolean hasLink1 = false, hasLink2 = false, hasLink3 = false; @@ -102,12 +103,12 @@ public void testParseFetchChannel() throws ProtocolNotFound, ParseException { hasLink3 = true; } - Assert.assertNotNull(entry.getValue()); - Assert.assertNotNull(entry.getValue().getData()); + assertNotNull(entry.getValue()); + assertNotNull(entry.getValue().getData()); } if (!hasLink1 || !hasLink2 || !hasLink3) { - Assert.fail("Outlinks read from sample rss file are not correct!"); + fail("Outlinks read from sample rss file are not correct!"); } } diff --git a/src/plugin/headings/src/test/org/apache/nutch/parse/headings/TestHeadingsParseFilter.java b/src/plugin/headings/src/test/org/apache/nutch/parse/headings/TestHeadingsParseFilter.java index 0795a9a05b..697e83858c 100644 --- a/src/plugin/headings/src/test/org/apache/nutch/parse/headings/TestHeadingsParseFilter.java +++ b/src/plugin/headings/src/test/org/apache/nutch/parse/headings/TestHeadingsParseFilter.java @@ -26,12 +26,13 @@ import java.io.ByteArrayInputStream; import java.io.IOException; +import org.junit.jupiter.api.Test; import org.w3c.dom.DocumentFragment; import org.xml.sax.InputSource; import org.xml.sax.SAXException; import org.cyberneko.html.parsers.DOMFragmentParser; -import org.junit.Assert; -import org.junit.Test; + +import static org.junit.jupiter.api.Assertions.assertEquals; public class TestHeadingsParseFilter { private static Configuration conf = NutchConfiguration.create(); @@ -59,9 +60,8 @@ public void testExtractHeadingFromNestedNodes() parseResult = filter.filter(content, parseResult, metaTags, node); - Assert.assertEquals( - "The h1 tag must include the content of the inner span node", - "header with span element", - parseResult.get(content.getUrl()).getData().getParseMeta().get("h1")); + assertEquals("header with span element", + parseResult.get(content.getUrl()).getData().getParseMeta().get("h1"), + "The h1 tag must include the content of the inner span node"); } } diff --git a/src/plugin/index-anchor/src/test/org/apache/nutch/indexer/anchor/TestAnchorIndexingFilter.java b/src/plugin/index-anchor/src/test/org/apache/nutch/indexer/anchor/TestAnchorIndexingFilter.java index 08a42f30bf..0974b11f41 100644 --- a/src/plugin/index-anchor/src/test/org/apache/nutch/indexer/anchor/TestAnchorIndexingFilter.java +++ b/src/plugin/index-anchor/src/test/org/apache/nutch/indexer/anchor/TestAnchorIndexingFilter.java @@ -25,8 +25,9 @@ import org.apache.nutch.parse.ParseData; import org.apache.nutch.parse.ParseImpl; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.*; /** * JUnit test case which tests 1. that anchor text is obtained 2. that anchor @@ -43,7 +44,7 @@ public void testDeduplicateAnchor() throws Exception { conf.setBoolean("anchorIndexingFilter.deduplicate", true); AnchorIndexingFilter filter = new AnchorIndexingFilter(); filter.setConf(conf); - Assert.assertNotNull(filter); + assertNotNull(filter); NutchDocument doc = new NutchDocument(); ParseImpl parse = new ParseImpl("foo bar", new ParseData()); Inlinks inlinks = new Inlinks(); @@ -55,13 +56,12 @@ public void testDeduplicateAnchor() throws Exception { new CrawlDatum(), inlinks); } catch (Exception e) { e.printStackTrace(); - Assert.fail(e.getMessage()); + fail(e.getMessage()); } - Assert.assertNotNull(doc); - Assert.assertTrue("test if there is an anchor at all", doc.getFieldNames() - .contains("anchor")); - Assert.assertEquals("test dedup, we expect 2", 2, doc.getField("anchor") - .getValues().size()); + assertNotNull(doc); + assertTrue(doc.getFieldNames().contains("anchor"), "test if there is an anchor at all"); + assertEquals(2, doc.getField("anchor").getValues().size(), + "test dedup, we expect 2"); } } diff --git a/src/plugin/index-arbitrary/src/test/org/apache/nutch/indexer/arbitrary/TestArbitraryIndexingFilter.java b/src/plugin/index-arbitrary/src/test/org/apache/nutch/indexer/arbitrary/TestArbitraryIndexingFilter.java index f11b43fa0a..4134950743 100644 --- a/src/plugin/index-arbitrary/src/test/org/apache/nutch/indexer/arbitrary/TestArbitraryIndexingFilter.java +++ b/src/plugin/index-arbitrary/src/test/org/apache/nutch/indexer/arbitrary/TestArbitraryIndexingFilter.java @@ -24,9 +24,10 @@ import org.apache.nutch.indexer.NutchDocument; import org.apache.nutch.parse.ParseImpl; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Before; -import org.junit.Test; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.*; /** * Tests that the index-arbitrary filter can add a new field with an arbitrary @@ -49,7 +50,7 @@ public class TestArbitraryIndexingFilter { ArbitraryIndexingFilter filter; NutchDocument doc; - @Before + @BeforeEach public void setUp() throws Exception { parse = new ParseImpl(); url = new Text("http://nutch.apache.org/index.html"); @@ -73,7 +74,7 @@ public void testAddingNewField() throws Exception { conf.set("index.arbitrary.methodName.0","getText"); filter = new ArbitraryIndexingFilter(); - Assert.assertNotNull("No filter exists for testAddingNewField",filter); + assertNotNull(filter, "No filter exists for testAddingNewField"); filter.setConf(conf); doc = new NutchDocument(); @@ -82,14 +83,13 @@ public void testAddingNewField() throws Exception { filter.filter(doc, parse, url, crawlDatum, inlinks); } catch (Exception e) { e.printStackTrace(); - Assert.fail(e.getMessage()); + fail(e.getMessage()); } - Assert.assertNotNull(doc); - Assert.assertFalse("test if doc is not empty", doc.getFieldNames() - .isEmpty()); - Assert.assertTrue("test if doc has new field with arbitrary value", doc.getField("foo") - .getValues().contains("Arbitrary text to add - bar")); + assertNotNull(doc); + assertFalse(doc.getFieldNames().isEmpty(), "test if doc is not empty"); + assertTrue(doc.getField("foo").getValues().contains("Arbitrary text to add - bar"), + "test if doc has new field with arbitrary value"); } /** @@ -113,42 +113,41 @@ public void testSupplementExistingField() throws Exception { conf.set("index.arbitrary.methodArgs.1","-1,3.14"); filter = new ArbitraryIndexingFilter(); - Assert.assertNotNull("No filter exists for testSupplementExistingField", filter); + assertNotNull(filter, "No filter exists for testSupplementExistingField"); filter.setConf(conf); doc = new NutchDocument(); - Assert.assertNotNull("doc doesn't exist", doc); + assertNotNull(doc, "doc doesn't exist"); doc.add("description","irrational"); - Assert.assertFalse("doc is empty", doc.getFieldNames().isEmpty()); + assertFalse(doc.getFieldNames().isEmpty(), "doc is empty"); - Assert.assertEquals("field description does not have exactly one value", 1, - doc.getField("description").getValues().size()); + assertEquals(1, doc.getField("description").getValues().size(), + "field description does not have exactly one value"); - Assert.assertTrue("field description does not have initial value 'irrational'", - doc.getField("description").getValues().contains("irrational")); + assertTrue(doc.getField("description").getValues().contains("irrational"), + "field description does not have initial value 'irrational'"); try { filter.filter(doc, parse, url, crawlDatum, inlinks); } catch (Exception e) { e.printStackTrace(); - Assert.fail(e.getMessage()); + fail(e.getMessage()); } - Assert.assertTrue("doc doesn't have new field with arbitrary value", - doc.getField("foo").getValues() - .contains("Arbitrary text to add - bar")); + assertTrue(doc.getField("foo").getValues().contains("Arbitrary text to add - bar"), + "doc doesn't have new field with arbitrary value"); - Assert.assertEquals("field description does not have 2 values", 2, - doc.getField("description").getValues().size()); + assertEquals(2, doc.getField("description").getValues().size(), + "field description does not have 2 values"); - Assert.assertTrue("field description original value gone", doc.getField("description") - .getValues().contains("irrational")); + assertTrue(doc.getField("description").getValues().contains("irrational"), + "field description original value gone"); - Assert.assertTrue("field description missing new value", doc.getField("description") - .getValues().contains("-3.14")); + assertTrue(doc.getField("description").getValues().contains("-3.14"), + "field description missing new value"); } @@ -176,47 +175,47 @@ public void testOverwritingExistingField() throws Exception { conf.set("index.arbitrary.overwrite.2","true"); filter = new ArbitraryIndexingFilter(); - Assert.assertNotNull("No filter exists for testOverwritingExistingField",filter); + assertNotNull(filter, "No filter exists for testOverwritingExistingField"); filter.setConf(conf); - Assert.assertNotNull("conf does not exist",conf); + assertNotNull(conf, "conf does not exist"); doc = new NutchDocument(); - Assert.assertNotNull("doc does not exist",doc); + assertNotNull(doc, "doc does not exist"); doc.add("description","irrational"); doc.add("philosopher","Socrates"); - Assert.assertEquals("field description does not have exactly one value", 1, doc.getField("description") - .getValues().size()); + assertEquals(1, doc.getField("description").getValues().size(), + "field description does not have exactly one value"); - Assert.assertEquals("field philosopher does not have exactly one value", 1, doc.getField("philosopher") - .getValues().size()); + assertEquals(1, doc.getField("philosopher").getValues().size(), + "field philosopher does not have exactly one value"); - Assert.assertTrue("field description does not have initial value 'irrational'", doc.getField("description") - .getValues().contains("irrational")); + assertTrue(doc.getField("description").getValues().contains("irrational"), + "field description does not have initial value 'irrational'"); - Assert.assertTrue("field philosopher does not have initial value 'Socrates'", doc.getField("philosopher") - .getValues().contains("Socrates")); + assertTrue(doc.getField("philosopher").getValues().contains("Socrates"), + "field philosopher does not have initial value 'Socrates'"); try { filter.filter(doc, parse, url, crawlDatum, inlinks); } catch (Exception e) { e.printStackTrace(System.out); - Assert.fail(e.getMessage()); + fail(e.getMessage()); } - Assert.assertNotNull(doc); + assertNotNull(doc); - Assert.assertEquals("field philosopher no longer has only one value", 1, doc.getField("philosopher") - .getValues().size()); + assertEquals(1, doc.getField("philosopher").getValues().size(), + "field philosopher no longer has only one value"); - Assert.assertFalse("field philosopher's original value 'Socrates' NOT overwritten", doc.getField("philosopher") - .getValues().contains("Socrates")); + assertFalse(doc.getField("philosopher").getValues().contains("Socrates"), + "field philosopher's original value 'Socrates' NOT overwritten"); - Assert.assertTrue("field philosopher does not have new value 'Popeye'", doc.getField("philosopher") - .getValues().contains("Popeye")); + assertTrue(doc.getField("philosopher").getValues().contains("Popeye"), + "field philosopher does not have new value 'Popeye'"); } /** @@ -247,34 +246,34 @@ public void testProcessingFieldAfterException() throws Exception { conf.set("index.arbitrary.overwrite.2","true"); filter = new ArbitraryIndexingFilter(); - Assert.assertNotNull("No filter exists for testProcessingFieldAfterException",filter); + assertNotNull(filter, "No filter exists for testProcessingFieldAfterException"); filter.setConf(conf); - Assert.assertNotNull("conf does not exist",conf); + assertNotNull(conf, "conf does not exist"); doc = new NutchDocument(); - Assert.assertNotNull("doc does not exist",doc); + assertNotNull(doc, "doc does not exist"); try { filter.filter(doc, parse, url, crawlDatum, inlinks); } catch (Exception e) { e.printStackTrace(System.out); - Assert.fail(e.getMessage()); + fail(e.getMessage()); } - Assert.assertNotNull(doc); + assertNotNull(doc); - Assert.assertTrue("field foo does not have 'first added value'", doc.getField("foo") - .getValues().contains("first added value")); + assertTrue(doc.getField("foo").getValues().contains("first added value"), + "field foo does not have 'first added value'"); - Assert.assertNull("field mangled has a value", doc.getField("mangled")); + assertNull(doc.getField("mangled"), "field mangled has a value"); - Assert.assertFalse("Value 'first added value' has leaked into field philospoher", doc.getField("philosopher") - .getValues().contains("first added value")); + assertFalse(doc.getField("philosopher").getValues().contains("first added value"), + "Value 'first added value' has leaked into field philospoher"); - Assert.assertTrue("field philosopher does not have new value 'last added value'", doc.getField("philosopher") - .getValues().contains("last added value")); + assertTrue(doc.getField("philosopher").getValues().contains("last added value"), + "field philosopher does not have new value 'last added value'"); } @@ -296,28 +295,28 @@ public void testAddingNewCalculatedField() throws Exception { conf.set("index.arbitrary.overwrite.0","true"); filter = new ArbitraryIndexingFilter(); - Assert.assertNotNull("No filter exists for testAddingCalculatedNewField",filter); + assertNotNull(filter, "No filter exists for testAddingCalculatedNewField"); filter.setConf(conf); doc = new NutchDocument(); Double boostVal = Double.valueOf("1.0"); doc.add("popularityBoost", boostVal); - Assert.assertFalse("doc is empty", doc.getFieldNames().isEmpty()); - Assert.assertTrue("test if doc has new field with arbitrary value", doc.getField("popularityBoost") - .getValues().contains(boostVal)); + assertFalse(doc.getFieldNames().isEmpty(), "doc is empty"); + assertTrue(doc.getField("popularityBoost").getValues().contains(boostVal), + "test if doc has new field with arbitrary value"); try { filter.filter(doc, parse, url, crawlDatum, inlinks); } catch (Exception e) { e.printStackTrace(); - Assert.fail(e.getMessage()); + fail(e.getMessage()); } - Assert.assertNotNull(doc); - Assert.assertFalse("doc is empty", doc.getFieldNames().isEmpty()); - Assert.assertTrue("test if unfetched doc has nonzero value in popularityBoost", doc.getField("popularityBoost") - .getValues().contains(1.0)); + assertNotNull(doc); + assertFalse(doc.getFieldNames().isEmpty(), "doc is empty"); + assertTrue(doc.getField("popularityBoost").getValues().contains(1.0), + "test if unfetched doc has nonzero value in popularityBoost"); inlinks.add(new Inlink("https://www.TeamSauropod.com/BullyForBrontosaurus","dinosaur")); inlinks.add(new Inlink("https://github.com/apache","source code")); @@ -329,11 +328,11 @@ public void testAddingNewCalculatedField() throws Exception { filter.filter(doc, parse, url, crawlDatum, inlinks); } catch (Exception e) { e.printStackTrace(); - Assert.fail(e.getMessage()); + fail(e.getMessage()); } - Assert.assertTrue("test if successfully fetched doc has expected value in popularityBoost", doc.getField("popularityBoost") - .getValues().contains(2.0)); + assertTrue(doc.getField("popularityBoost").getValues().contains(2.0), + "test if successfully fetched doc has expected value in popularityBoost"); } /** @@ -372,7 +371,7 @@ public void testUpdatingPOJOClass() throws Exception { conf.set("index.arbitrary.all.fields.access.3","false"); filter = new ArbitraryIndexingFilter(); - Assert.assertNotNull("No filter exists for testAddingNewField",filter); + assertNotNull(filter, "No filter exists for testAddingNewField"); filter.setConf(conf); doc = new NutchDocument(); @@ -381,19 +380,18 @@ public void testUpdatingPOJOClass() throws Exception { filter.filter(doc, parse, url, crawlDatum, inlinks); } catch (Exception e) { e.printStackTrace(); - Assert.fail(e.getMessage()); + fail(e.getMessage()); } - Assert.assertNotNull(doc); - Assert.assertFalse("test if doc is not empty", doc.getFieldNames() - .isEmpty()); - Assert.assertTrue("test if doc still has new field with arbitrary value running with new indexer", doc.getField("foo") - .getValues().contains("Original Echo class added 'bar'")); - Assert.assertTrue("test if updated POJO created new field with arbitrary value", doc.getField("bogusSite") - .getValues().contains("https://www.updatedNutchPluginJunitTest.com")); - Assert.assertTrue("test updated POJO set new value in existing field", doc.getField("description") - .getValues().contains("-3.14")); - Assert.assertTrue("test POJO with both constructor styles supports old calls", doc.getField("summary") - .getValues().contains("1025.0")); + assertNotNull(doc); + assertFalse(doc.getFieldNames().isEmpty(), "test if doc is not empty"); + assertTrue( doc.getField("foo").getValues().contains("Original Echo class added 'bar'"), + "test if doc still has new field with arbitrary value running with new indexer"); + assertTrue(doc.getField("bogusSite").getValues().contains("https://www.updatedNutchPluginJunitTest.com"), + "test if updated POJO created new field with arbitrary value"); + assertTrue(doc.getField("description").getValues().contains("-3.14"), + "test updated POJO set new value in existing field"); + assertTrue(doc.getField("summary").getValues().contains("1025.0"), + "test POJO with both constructor styles supports old calls"); } } diff --git a/src/plugin/index-basic/src/test/org/apache/nutch/indexer/basic/TestBasicIndexingFilter.java b/src/plugin/index-basic/src/test/org/apache/nutch/indexer/basic/TestBasicIndexingFilter.java index 3684c9907b..2157f28809 100644 --- a/src/plugin/index-basic/src/test/org/apache/nutch/indexer/basic/TestBasicIndexingFilter.java +++ b/src/plugin/index-basic/src/test/org/apache/nutch/indexer/basic/TestBasicIndexingFilter.java @@ -27,11 +27,12 @@ import org.apache.nutch.parse.ParseImpl; import org.apache.nutch.parse.ParseStatus; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; import java.util.Date; +import static org.junit.jupiter.api.Assertions.*; + /** * JUnit test case which tests 1. that basic searchable fields are added to a * document 2. that domain is added as per {@code indexer.add.domain} in @@ -54,7 +55,7 @@ public void testBasicIndexingFilter() throws Exception { BasicIndexingFilter filter = new BasicIndexingFilter(); filter.setConf(conf); - Assert.assertNotNull(filter); + assertNotNull(filter); NutchDocument doc = new NutchDocument(); @@ -77,22 +78,21 @@ public void testBasicIndexingFilter() throws Exception { crawlDatum, inlinks); } catch (Exception e) { e.printStackTrace(); - Assert.fail(e.getMessage()); + fail(e.getMessage()); } - Assert.assertNotNull(doc); - Assert.assertEquals("test title, expect \"The Foo Pa\"", "The Foo Pa", doc - .getField("title").getValues().get(0)); - Assert.assertEquals("test domain, expect \"apache.org\"", "apache.org", doc - .getField("domain").getValues().get(0)); - Assert.assertEquals("test host, expect \"nutch.apache.org\"", - "nutch.apache.org", doc.getField("host").getValues().get(0)); - Assert.assertEquals( - "test url, expect \"http://nutch.apache.org/index.html\"", - "http://nutch.apache.org/index.html", doc.getField("url").getValues() - .get(0)); - Assert.assertEquals("test content", "this is a sample foo", - doc.getField("content").getValues().get(0)); - Assert.assertEquals("test fetch time", new Date(100L), - doc.getField("tstamp").getValues().get(0)); + assertNotNull(doc); + assertEquals("The Foo Pa", doc.getField("title").getValues().get(0), + "test title, expect \"The Foo Pa\""); + assertEquals("apache.org", doc.getField("domain").getValues().get(0), + "test domain, expect \"apache.org\""); + assertEquals("nutch.apache.org", doc.getField("host").getValues().get(0), + "test host, expect \"nutch.apache.org\""); + assertEquals("http://nutch.apache.org/index.html", + doc.getField("url").getValues().get(0), + "test url, expect \"http://nutch.apache.org/index.html\""); + assertEquals(doc.getField("content").getValues().get(0), + "this is a sample foo", "test content"); + assertEquals(new Date(100L), doc.getField("tstamp").getValues().get(0), + "test fetch time"); } } diff --git a/src/plugin/index-jexl-filter/src/test/org/apache/nutch/indexer/jexl/TestJexlIndexingFilter.java b/src/plugin/index-jexl-filter/src/test/org/apache/nutch/indexer/jexl/TestJexlIndexingFilter.java index f3cc655683..e8ad163401 100644 --- a/src/plugin/index-jexl-filter/src/test/org/apache/nutch/indexer/jexl/TestJexlIndexingFilter.java +++ b/src/plugin/index-jexl-filter/src/test/org/apache/nutch/indexer/jexl/TestJexlIndexingFilter.java @@ -27,14 +27,19 @@ import org.apache.nutch.parse.ParseImpl; import org.apache.nutch.parse.ParseStatus; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Rule; -import org.junit.Test; -import org.junit.rules.ExpectedException; +import org.junit.jupiter.api.Test; + +import java.lang.invoke.MethodHandles; + +import static org.junit.jupiter.api.Assertions.*; + +import org.slf4j.Logger; +import org.slf4j.LoggerFactory; public class TestJexlIndexingFilter { - @Rule - public ExpectedException thrown = ExpectedException.none(); + + private static final Logger LOG = LoggerFactory + .getLogger(MethodHandles.lookup().lookupClass()); @Test public void testAllowMatchingDocument() throws Exception { @@ -43,7 +48,7 @@ public void testAllowMatchingDocument() throws Exception { JexlIndexingFilter filter = new JexlIndexingFilter(); filter.setConf(conf); - Assert.assertNotNull(filter); + assertNotNull(filter); NutchDocument doc = new NutchDocument(); @@ -66,8 +71,8 @@ public void testAllowMatchingDocument() throws Exception { NutchDocument result = filter.filter(doc, parse, new Text("http://nutch.apache.org/index.html"), crawlDatum, inlinks); - Assert.assertNotNull(result); - Assert.assertEquals(doc, result); + assertNotNull(result); + assertEquals(doc, result); } @Test @@ -77,7 +82,7 @@ public void testBlockNotMatchingDocuments() throws Exception { JexlIndexingFilter filter = new JexlIndexingFilter(); filter.setConf(conf); - Assert.assertNotNull(filter); + assertNotNull(filter); NutchDocument doc = new NutchDocument(); @@ -100,16 +105,21 @@ public void testBlockNotMatchingDocuments() throws Exception { NutchDocument result = filter.filter(doc, parse, new Text("http://nutch.apache.org/index.html"), crawlDatum, inlinks); - Assert.assertNull(result); + assertNull(result); } @Test public void testMissingConfiguration() throws Exception { Configuration conf = NutchConfiguration.create(); - JexlIndexingFilter filter = new JexlIndexingFilter(); - thrown.expect(RuntimeException.class); - filter.setConf(conf); + Exception exception = assertThrows(RuntimeException.class, () -> { + filter.setConf(conf); + }); + String expectedMessage = "The property index.jexl.filter must have a " + + "value when index-jexl-filter is used. You can use 'true' or 'false' " + + "to index all/none"; + String actualMessage = exception.getMessage(); + assertTrue(actualMessage.contains(expectedMessage)); } @Test @@ -118,7 +128,11 @@ public void testInvalidExpression() throws Exception { conf.set("index.jexl.filter", "doc.lang=<>:='en'"); JexlIndexingFilter filter = new JexlIndexingFilter(); - thrown.expect(RuntimeException.class); - filter.setConf(conf); + Exception exception = assertThrows(RuntimeException.class, () -> { + filter.setConf(conf); + }); + String expectedMessage = "Failed parsing JEXL from index.jexl.filter"; + String actualMessage = exception.getMessage(); + assertTrue(actualMessage.contains(expectedMessage)); } } diff --git a/src/plugin/index-links/src/test/org/apache/nutch/indexer/links/TestLinksIndexingFilter.java b/src/plugin/index-links/src/test/org/apache/nutch/indexer/links/TestLinksIndexingFilter.java index 0f8b660b75..a362079732 100644 --- a/src/plugin/index-links/src/test/org/apache/nutch/indexer/links/TestLinksIndexingFilter.java +++ b/src/plugin/index-links/src/test/org/apache/nutch/indexer/links/TestLinksIndexingFilter.java @@ -31,19 +31,20 @@ import org.apache.nutch.parse.ParseStatus; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Before; -import org.junit.Test; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; import java.net.URL; +import static org.junit.jupiter.api.Assertions.assertEquals; + public class TestLinksIndexingFilter { Configuration conf = NutchConfiguration.create(); LinksIndexingFilter filter = new LinksIndexingFilter(); Metadata metadata = new Metadata(); - @Before + @BeforeEach public void setUp() throws Exception { metadata.add(Response.CONTENT_TYPE, "text/html"); } @@ -79,10 +80,10 @@ public void testFilterOutlinks() throws Exception { new ParseData(new ParseStatus(), "title", outlinks, metadata)), new Text("http://www.example.com/"), new CrawlDatum(), new Inlinks()); - Assert.assertEquals(1, doc.getField("outlinks").getValues().size()); + assertEquals(1, doc.getField("outlinks").getValues().size()); - Assert.assertEquals("Filter outlinks, allow only those from a different host", - outlinks[0].getToUrl(), doc.getFieldValue("outlinks")); + assertEquals(outlinks[0].getToUrl(), doc.getFieldValue("outlinks"), + "Filter outlinks, allow only those from a different host"); } @Test @@ -98,10 +99,10 @@ public void testFilterInlinks() throws Exception { new ParseData(new ParseStatus(), "title", new Outlink[0], metadata)), new Text("http://www.example.com/"), new CrawlDatum(), inlinks); - Assert.assertEquals(1, doc.getField("inlinks").getValues().size()); + assertEquals(1, doc.getField("inlinks").getValues().size()); - Assert.assertEquals("Filter inlinks, allow only those from a different host", - "http://www.test.com", doc.getFieldValue("inlinks")); + assertEquals("http://www.test.com", doc.getFieldValue("inlinks"), + "Filter inlinks, allow only those from a different host"); } @Test @@ -114,8 +115,8 @@ public void testNoFilterOutlinks() throws Exception { new ParseData(new ParseStatus(), "title", outlinks, metadata)), new Text("http://www.example.com/"), new CrawlDatum(), new Inlinks()); - Assert.assertEquals("All outlinks must be indexed even those from the same host", - outlinks.length, doc.getField("outlinks").getValues().size()); + assertEquals(outlinks.length, doc.getField("outlinks").getValues().size(), + "All outlinks must be indexed even those from the same host"); } @Test @@ -131,8 +132,8 @@ public void testNoFilterInlinks() throws Exception { new ParseData(new ParseStatus(), "title", new Outlink[0], metadata)), new Text("http://www.example.com/"), new CrawlDatum(), inlinks); - Assert.assertEquals("All inlinks must be indexed even those from the same host", - inlinks.size(), doc.getField("inlinks").getValues().size()); + assertEquals(inlinks.size(), doc.getField("inlinks").getValues().size(), + "All inlinks must be indexed even those from the same host"); } @Test @@ -156,16 +157,16 @@ public void testIndexOnlyHostPart() throws Exception { NutchField docOutlinks = doc.getField("outlinks"); - Assert.assertEquals("Only the host portion of the outlink URL must be indexed", - new URL("http://www.test.com").getHost(), - docOutlinks.getValues().get(0)); + assertEquals(new URL("http://www.test.com").getHost(), + docOutlinks.getValues().get(0), + "Only the host portion of the outlink URL must be indexed"); - Assert.assertEquals( - "The inlinks coming from the same host must count only once", 1, - doc.getField("inlinks").getValues().size()); + assertEquals(1, doc.getField("inlinks").getValues().size(), + "The inlinks coming from the same host must count only once"); - Assert.assertEquals("Only the host portion of the inlinks URL must be indexed", - new URL("http://www.test.com").getHost(), doc.getFieldValue("inlinks")); + assertEquals(new URL("http://www.test.com").getHost(), + doc.getFieldValue("inlinks"), + "Only the host portion of the inlinks URL must be indexed"); } @Test @@ -182,12 +183,11 @@ public void testIndexHostsOnlyAndFilterOutlinks() throws Exception { new ParseData(new ParseStatus(), "title", outlinks, metadata)), new Text("http://www.example.com/"), new CrawlDatum(), new Inlinks()); - Assert.assertEquals(1, doc.getField("outlinks").getValues().size()); + assertEquals(1, doc.getField("outlinks").getValues().size()); - Assert.assertEquals( - "Index only the host portion of the outlinks after filtering", - new URL("http://www.test.com").getHost(), - doc.getFieldValue("outlinks")); + assertEquals(new URL("http://www.test.com").getHost(), + doc.getFieldValue("outlinks"), + "Index only the host portion of the outlinks after filtering"); } @Test @@ -206,12 +206,11 @@ public void testIndexHostsOnlyAndFilterInlinks() throws Exception { new ParseData(new ParseStatus(), "title", new Outlink[0], metadata)), new Text("http://www.example.com/"), new CrawlDatum(), inlinks); - Assert.assertEquals(1, doc.getField("inlinks").getValues().size()); + assertEquals(1, doc.getField("inlinks").getValues().size()); - Assert.assertEquals( - "Index only the host portion of the inlinks after filtering", - new URL("http://www.test.com").getHost(), - doc.getFieldValue("inlinks")); + assertEquals(new URL("http://www.test.com").getHost(), + doc.getFieldValue("inlinks"), + "Index only the host portion of the inlinks after filtering"); } } diff --git a/src/plugin/index-more/src/test/org/apache/nutch/indexer/more/TestMoreIndexingFilter.java b/src/plugin/index-more/src/test/org/apache/nutch/indexer/more/TestMoreIndexingFilter.java index 5a22cf6de0..4ff496c5ae 100644 --- a/src/plugin/index-more/src/test/org/apache/nutch/indexer/more/TestMoreIndexingFilter.java +++ b/src/plugin/index-more/src/test/org/apache/nutch/indexer/more/TestMoreIndexingFilter.java @@ -32,8 +32,9 @@ import org.apache.nutch.parse.ParseImpl; import org.apache.nutch.parse.ParseStatus; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.*; public class TestMoreIndexingFilter { @@ -59,7 +60,7 @@ public void testNoParts() { conf.setBoolean("moreIndexingFilter.indexMimeTypeParts", false); MoreIndexingFilter filter = new MoreIndexingFilter(); filter.setConf(conf); - Assert.assertNotNull(filter); + assertNotNull(filter); NutchDocument doc = new NutchDocument(); ParseImpl parse = new ParseImpl("foo bar", new ParseData()); @@ -68,12 +69,12 @@ public void testNoParts() { new CrawlDatum(), new Inlinks()); } catch (Exception e) { e.printStackTrace(); - Assert.fail(e.getMessage()); + fail(e.getMessage()); } - Assert.assertNotNull(doc); - Assert.assertTrue(doc.getFieldNames().contains("type")); - Assert.assertEquals(1, doc.getField("type").getValues().size()); - Assert.assertEquals("text/html", doc.getFieldValue("type")); + assertNotNull(doc); + assertTrue(doc.getFieldNames().contains("type")); + assertEquals(1, doc.getField("type").getValues().size()); + assertEquals("text/html", doc.getFieldValue("type")); } @Test @@ -92,21 +93,21 @@ public void testContentDispositionTitle() throws IndexingException { NutchDocument doc = new NutchDocument(); doc = filter.filter(doc, parseImpl, url, new CrawlDatum(), new Inlinks()); - Assert.assertEquals("content-disposition not detected", "filename.ext", - doc.getFieldValue("title")); + assertEquals("filename.ext", doc.getFieldValue("title"), + "content-disposition not detected"); /* NUTCH-1140: do not add second title to avoid a multi-valued title field */ doc = new NutchDocument(); doc.add("title", "title"); doc = filter.filter(doc, parseImpl, url, new CrawlDatum(), new Inlinks()); - Assert.assertEquals("do not add second title by content-disposition", - "title", doc.getFieldValue("title")); + assertEquals("title", doc.getFieldValue("title"), + "do not add second title by content-disposition"); } private void assertParts(String[] parts, int count, String... expected) { - Assert.assertEquals(count, parts.length); + assertEquals(count, parts.length); for (int i = 0; i < expected.length; i++) { - Assert.assertEquals(expected[i], parts[i]); + assertEquals(expected[i], parts[i]); } } @@ -120,8 +121,8 @@ private void assertContentType(Configuration conf, String source, "text", new ParseData(new ParseStatus(), "title", new Outlink[0], metadata)), new Text("http://www.example.com/"), new CrawlDatum(), new Inlinks()); - Assert.assertEquals("mime type not detected", expected, - doc.getFieldValue("type")); + assertEquals(expected, doc.getFieldValue("type"), + "mime type not detected"); } @Test @@ -148,8 +149,8 @@ public void testDates() throws IndexingException { doc = filter.filter(doc, parseImpl, url, fetchDatum, new Inlinks()); - Assert.assertEquals("last fetch date not extracted", - new Date(dateEpocheSeconds * 1000), doc.getFieldValue("date")); + assertEquals(new Date(dateEpocheSeconds * 1000), + doc.getFieldValue("date"), "last fetch date not extracted"); // set last-modified time (7 days before fetch time) Date lastModifiedDate = new Date( @@ -158,7 +159,7 @@ public void testDates() throws IndexingException { parseImpl.getData().getParseMeta().set(Metadata.LAST_MODIFIED, lastModifiedDateStr); doc = filter.filter(doc, parseImpl, url, fetchDatum, new Inlinks()); - Assert.assertEquals("last-modified date not extracted", lastModifiedDate, - doc.getFieldValue("lastModified")); + assertEquals(lastModifiedDate, doc.getFieldValue("lastModified"), + "last-modified date not extracted"); } } diff --git a/src/plugin/index-replace/src/test/org/apache/nutch/indexer/replace/TestIndexReplace.java b/src/plugin/index-replace/src/test/org/apache/nutch/indexer/replace/TestIndexReplace.java index fcd8cb5c84..0e48cee0ec 100644 --- a/src/plugin/index-replace/src/test/org/apache/nutch/indexer/replace/TestIndexReplace.java +++ b/src/plugin/index-replace/src/test/org/apache/nutch/indexer/replace/TestIndexReplace.java @@ -29,8 +29,9 @@ import org.apache.nutch.protocol.Protocol; import org.apache.nutch.protocol.ProtocolFactory; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.*; /** * JUnit tests for the index-replace plugin. @@ -65,15 +66,15 @@ public NutchDocument parseAndFilterFile(String fileName, Configuration conf) { BasicIndexingFilter basicIndexer = new BasicIndexingFilter(); basicIndexer.setConf(conf); - Assert.assertNotNull(basicIndexer); + assertNotNull(basicIndexer); MetadataIndexer metaIndexer = new MetadataIndexer(); metaIndexer.setConf(conf); - Assert.assertNotNull(basicIndexer); + assertNotNull(basicIndexer); ReplaceIndexer replaceIndexer = new ReplaceIndexer(); replaceIndexer.setConf(conf); - Assert.assertNotNull(replaceIndexer); + assertNotNull(replaceIndexer); try { String urlString = "file:" + sampleDir + fileSeparator + fileName; @@ -91,7 +92,7 @@ public NutchDocument parseAndFilterFile(String fileName, Configuration conf) { doc = replaceIndexer.filter(doc, parse, text, crawlDatum, inlinks); } catch (Exception e) { e.printStackTrace(); - Assert.fail(e.toString()); + fail(e.toString()); } return doc; @@ -121,14 +122,14 @@ public void testPropertyParse() { try { rp.setConf(conf); } catch (RuntimeException ohno) { - Assert.fail("Unable to parse a valid index.replace.regexp property! " + fail("Unable to parse a valid index.replace.regexp property! " + ohno.getMessage()); } Configuration parsedConf = rp.getConf(); // Does the getter equal the setter? Too easy! - Assert.assertEquals(indexReplaceProperty, + assertEquals(indexReplaceProperty, parsedConf.get(INDEX_REPLACE_PROPERTY)); } @@ -160,11 +161,10 @@ public void testGlobalReplacement() { // Run the document through the parser and index filters. NutchDocument doc = parseAndFilterFile(sampleFile, conf); - Assert.assertEquals(expectedDescription, + assertEquals(expectedDescription, doc.getFieldValue("metatag.description")); - Assert - .assertEquals(expectedKeywords, doc.getFieldValue("metatag.keywords")); - Assert.assertEquals(expectedAuthor, doc.getFieldValue("metatag.author")); + assertEquals(expectedKeywords, doc.getFieldValue("metatag.keywords")); + assertEquals(expectedAuthor, doc.getFieldValue("metatag.author")); } /** @@ -199,11 +199,10 @@ public void testInvalidPatterns() { NutchDocument doc = parseAndFilterFile(sampleFile, conf); // Assert that our metatags have not changed. - Assert.assertEquals(expectedDescription, + assertEquals(expectedDescription, doc.getFieldValue("metatag.description")); - Assert - .assertEquals(expectedKeywords, doc.getFieldValue("metatag.keywords")); - Assert.assertEquals(expectedAuthor, doc.getFieldValue("metatag.author")); + assertEquals(expectedKeywords, doc.getFieldValue("metatag.keywords")); + assertEquals(expectedAuthor, doc.getFieldValue("metatag.author")); } @@ -234,11 +233,10 @@ public void testUrlMatchesPattern() { NutchDocument doc = parseAndFilterFile(sampleFile, conf); // Assert that our metatags have changed. - Assert.assertEquals(expectedDescription, + assertEquals(expectedDescription, doc.getFieldValue("metatag.description")); - Assert - .assertEquals(expectedKeywords, doc.getFieldValue("metatag.keywords")); - Assert.assertEquals(expectedAuthor, doc.getFieldValue("metatag.author")); + assertEquals(expectedKeywords, doc.getFieldValue("metatag.keywords")); + assertEquals(expectedAuthor, doc.getFieldValue("metatag.author")); } @@ -271,11 +269,10 @@ public void testUrlNotMatchesPattern() { NutchDocument doc = parseAndFilterFile(sampleFile, conf); // Assert that our metatags have not changed. - Assert.assertEquals(expectedDescription, + assertEquals(expectedDescription, doc.getFieldValue("metatag.description")); - Assert - .assertEquals(expectedKeywords, doc.getFieldValue("metatag.keywords")); - Assert.assertEquals(expectedAuthor, doc.getFieldValue("metatag.author")); + assertEquals(expectedKeywords, doc.getFieldValue("metatag.keywords")); + assertEquals(expectedAuthor, doc.getFieldValue("metatag.author")); } @@ -310,11 +307,10 @@ public void testGlobalAndUrlMatchesPattern() { NutchDocument doc = parseAndFilterFile(sampleFile, conf); // Assert that our metatags have changed. - Assert.assertEquals(expectedDescription, + assertEquals(expectedDescription, doc.getFieldValue("metatag.description")); - Assert - .assertEquals(expectedKeywords, doc.getFieldValue("metatag.keywords")); - Assert.assertEquals(expectedAuthor, doc.getFieldValue("metatag.author")); + assertEquals(expectedKeywords, doc.getFieldValue("metatag.keywords")); + assertEquals(expectedAuthor, doc.getFieldValue("metatag.author")); } @@ -349,11 +345,10 @@ public void testGlobalAndUrlNotMatchesPattern() { NutchDocument doc = parseAndFilterFile(sampleFile, conf); // Assert that description has changed and the others have not changed. - Assert.assertEquals(expectedDescription, + assertEquals(expectedDescription, doc.getFieldValue("metatag.description")); - Assert - .assertEquals(expectedKeywords, doc.getFieldValue("metatag.keywords")); - Assert.assertEquals(expectedAuthor, doc.getFieldValue("metatag.author")); + assertEquals(expectedKeywords, doc.getFieldValue("metatag.keywords")); + assertEquals(expectedAuthor, doc.getFieldValue("metatag.author")); } /** @@ -386,7 +381,7 @@ public void testReplacementsRunInSpecifedOrder() { NutchDocument doc = parseAndFilterFile(sampleFile, conf); // Check that the value produced by the last replacement has worked. - Assert.assertEquals(expectedDescription, + assertEquals(expectedDescription, doc.getFieldValue("metatag.description")); } @@ -417,7 +412,7 @@ public void testReplacementsWithFlags() { // Check that the value produced by the case-insensitive replacement has // worked. - Assert.assertEquals(expectedDescription, + assertEquals(expectedDescription, doc.getFieldValue("metatag.description")); } @@ -446,10 +441,10 @@ public void testReplacementsDifferentTarget() { NutchDocument doc = parseAndFilterFile(sampleFile, conf); // Check that the input field has not been modified - Assert.assertEquals(expectedDescription, + assertEquals(expectedDescription, doc.getFieldValue("metatag.description")); // Check that the output field has created - Assert.assertEquals(expectedTargetDescription, + assertEquals(expectedTargetDescription, doc.getFieldValue("new")); } } diff --git a/src/plugin/index-static/src/test/org/apache/nutch/indexer/staticfield/TestStaticFieldIndexerTest.java b/src/plugin/index-static/src/test/org/apache/nutch/indexer/staticfield/TestStaticFieldIndexerTest.java index 42cd46deac..6d190b3874 100644 --- a/src/plugin/index-static/src/test/org/apache/nutch/indexer/staticfield/TestStaticFieldIndexerTest.java +++ b/src/plugin/index-static/src/test/org/apache/nutch/indexer/staticfield/TestStaticFieldIndexerTest.java @@ -23,9 +23,10 @@ import org.apache.nutch.indexer.NutchDocument; import org.apache.nutch.parse.ParseImpl; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Before; -import org.junit.Test; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.*; /** * JUnit test case which tests 1. that static data fields are added to a @@ -46,7 +47,7 @@ public class TestStaticFieldIndexerTest { Text url; StaticFieldIndexer filter; - @Before + @BeforeEach public void setUp() throws Exception { conf = NutchConfiguration.create(); parse = new ParseImpl(); @@ -64,7 +65,7 @@ public void setUp() throws Exception { @Test public void testEmptyIndexStatic() throws Exception { - Assert.assertNotNull(filter); + assertNotNull(filter); filter.setConf(conf); NutchDocument doc = new NutchDocument(); @@ -73,12 +74,12 @@ public void testEmptyIndexStatic() throws Exception { filter.filter(doc, parse, url, crawlDatum, inlinks); } catch (Exception e) { e.printStackTrace(); - Assert.fail(e.getMessage()); + fail(e.getMessage()); } - Assert.assertNotNull(doc); - Assert.assertTrue("tests if no field is set for empty index.static", doc - .getFieldNames().isEmpty()); + assertNotNull(doc); + assertTrue(doc.getFieldNames().isEmpty(), + "tests if no field is set for empty index.static"); } /** @@ -91,7 +92,7 @@ public void testNormalScenario() throws Exception { conf.set("index.static", "field1:val1, field2 : val2 val3 , field3, field4 :val4 , "); - Assert.assertNotNull(filter); + assertNotNull(filter); filter.setConf(conf); NutchDocument doc = new NutchDocument(); @@ -100,20 +101,20 @@ public void testNormalScenario() throws Exception { filter.filter(doc, parse, url, crawlDatum, inlinks); } catch (Exception e) { e.printStackTrace(); - Assert.fail(e.getMessage()); + fail(e.getMessage()); } - Assert.assertNotNull(doc); - Assert.assertFalse("test if doc is not empty", doc.getFieldNames() - .isEmpty()); - Assert.assertEquals("test if doc has 3 fields", 3, doc.getFieldNames() - .size()); - Assert.assertTrue("test if doc has field1", doc.getField("field1") - .getValues().contains("val1")); - Assert.assertTrue("test if doc has field2", doc.getField("field2") - .getValues().contains("val2")); - Assert.assertTrue("test if doc has field4", doc.getField("field4") - .getValues().contains("val4")); + assertNotNull(doc); + assertFalse(doc.getFieldNames().isEmpty(), + "test if doc is not empty"); + assertEquals(3, doc.getFieldNames().size(), + "test if doc has 3 fields"); + assertTrue(doc.getField("field1").getValues().contains("val1"), + "test if doc has field1"); + assertTrue(doc.getField("field2").getValues().contains("val2"), + "test if doc has field2"); + assertTrue(doc.getField("field4").getValues().contains("val4"), + "test if doc has field4"); } /** @@ -129,7 +130,7 @@ public void testCustomDelimiters() throws Exception { conf.set("index.static.valuesep", "|"); conf.set("index.static", "field1=val1>field2 = val2|val3 >field3>field4 =val4 > "); - Assert.assertNotNull(filter); + assertNotNull(filter); filter.setConf(conf); NutchDocument doc = new NutchDocument(); @@ -138,20 +139,20 @@ public void testCustomDelimiters() throws Exception { filter.filter(doc, parse, url, crawlDatum, inlinks); } catch (Exception e) { e.printStackTrace(); - Assert.fail(e.getMessage()); + fail(e.getMessage()); } - Assert.assertNotNull(doc); - Assert.assertFalse("test if doc is not empty", doc.getFieldNames() - .isEmpty()); - Assert.assertEquals("test if doc has 3 fields", 3, doc.getFieldNames() - .size()); - Assert.assertTrue("test if doc has field1", doc.getField("field1") - .getValues().contains("val1")); - Assert.assertTrue("test if doc has field2", doc.getField("field2") - .getValues().contains("val2")); - Assert.assertTrue("test if doc has field4", doc.getField("field4") - .getValues().contains("val4")); + assertNotNull(doc); + assertFalse(doc.getFieldNames().isEmpty(), + "test if doc is not empty"); + assertEquals(3, doc.getFieldNames().size(), + "test if doc has 3 fields"); + assertTrue(doc.getField("field1").getValues().contains("val1"), + "test if doc has field1"); + assertTrue(doc.getField("field2").getValues().contains("val2"), + "test if doc has field2"); + assertTrue(doc.getField("field4").getValues().contains("val4"), + "test if doc has field4"); } /** @@ -167,7 +168,7 @@ public void testCustomMulticharacterDelimiters() throws Exception { conf.set("index.static.valuesep", "***"); conf.set("index.static", "field1\t\tval1\n\n" + "field2\t\tval2***val3\n\n" + "field3\n\n" + "field4\t\tval4\n\n\n\n"); - Assert.assertNotNull(filter); + assertNotNull(filter); filter.setConf(conf); NutchDocument doc = new NutchDocument(); @@ -176,19 +177,19 @@ public void testCustomMulticharacterDelimiters() throws Exception { filter.filter(doc, parse, url, crawlDatum, inlinks); } catch (Exception e) { e.printStackTrace(); - Assert.fail(e.getMessage()); + fail(e.getMessage()); } - Assert.assertNotNull(doc); - Assert.assertFalse("test if doc is not empty", doc.getFieldNames() - .isEmpty()); - Assert.assertEquals("test if doc has 3 fields", 3, doc.getFieldNames() - .size()); - Assert.assertTrue("test if doc has field1", doc.getField("field1") - .getValues().contains("val1")); - Assert.assertTrue("test if doc has field2", doc.getField("field2") - .getValues().contains("val2")); - Assert.assertTrue("test if doc has field4", doc.getField("field4") - .getValues().contains("val4")); + assertNotNull(doc); + assertFalse(doc.getFieldNames().isEmpty(), + "test if doc is not empty"); + assertEquals(3, doc.getFieldNames().size(), + "test if doc has 3 fields"); + assertTrue(doc.getField("field1").getValues().contains("val1"), + "test if doc has field1"); + assertTrue(doc.getField("field2").getValues().contains("val2"), + "test if doc has field2"); + assertTrue(doc.getField("field4").getValues().contains("val4"), + "test if doc has field4"); } } diff --git a/src/plugin/indexer-csv/src/test/org/apache/nutch/indexwriter/csv/TestCSVIndexWriter.java b/src/plugin/indexer-csv/src/test/org/apache/nutch/indexwriter/csv/TestCSVIndexWriter.java index 5714cc235b..bb9f212293 100644 --- a/src/plugin/indexer-csv/src/test/org/apache/nutch/indexwriter/csv/TestCSVIndexWriter.java +++ b/src/plugin/indexer-csv/src/test/org/apache/nutch/indexwriter/csv/TestCSVIndexWriter.java @@ -16,9 +16,6 @@ */ package org.apache.nutch.indexwriter.csv; -import static org.junit.Assert.assertEquals; -import static org.junit.Assert.assertTrue; - import java.io.ByteArrayOutputStream; import java.io.IOException; import java.io.UnsupportedEncodingException; @@ -32,10 +29,13 @@ import org.apache.nutch.indexer.IndexWriterParams; import org.apache.nutch.indexer.NutchDocument; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Test; +import org.junit.jupiter.api.Test; import org.slf4j.Logger; import org.slf4j.LoggerFactory; +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertTrue; + /** * Test CSVIndexWriter. Focus is on CSV-specific potential issues, mainly quoting and escaping. */ @@ -126,8 +126,8 @@ public void testCSVdefault() throws IOException { "Apache Nutch is an open source web-search software project. ..." }; String csv = getCSV(new String[0], fields); for (int i = 0; i < fields.length; i += 2) { - assertTrue("Testing field " + i + " (" + fields[i] + ")", - csv.contains(fields[i + 1])); + assertTrue(csv.contains(fields[i + 1]), + "Testing field " + i + " (" + fields[i] + ")"); } } @@ -136,8 +136,8 @@ public void testCSVquoteFieldSeparators() throws IOException { String[] params = { CSVConstants.CSV_FIELDS, "test,test2" }; String[] fields = { "test", "a,b", "test2", "c,d" }; String csv = getCSV(params, fields); - assertEquals("If field contains a fields separator, it must be quoted", - "\"a,b\",\"c,d\"", csv.trim()); + assertEquals("\"a,b\",\"c,d\"", csv.trim(), + "If field contains a fields separator, it must be quoted"); } @Test @@ -145,8 +145,8 @@ public void testCSVquoteRecordSeparators() throws IOException { String[] params = { CSVConstants.CSV_FIELDS, "test" }; String[] fields = { "test", "a\nb" }; String csv = getCSV(params, fields); - assertEquals("If field contains a fields separator, it must be quoted", - "\"a\nb\"", csv.trim()); + assertEquals("\"a\nb\"", csv.trim(), + "If field contains a fields separator, it must be quoted"); } @Test @@ -154,8 +154,8 @@ public void testCSVescapeQuotes() throws IOException { String[] params = { CSVConstants.CSV_FIELDS, "test" }; String[] fields = { "test", "a,b:\"quote\",c" }; String csv = getCSV(params, fields); - assertEquals("Quotes inside a quoted field must be escaped", - "\"a,b:\"\"quote\"\",c\"", csv.trim()); + assertEquals("\"a,b:\"\"quote\"\",c\"", csv.trim(), + "Quotes inside a quoted field must be escaped"); } @Test @@ -163,8 +163,8 @@ public void testCSVescapeLeadingQuotes() throws IOException { String[] params = { CSVConstants.CSV_FIELDS, "test" }; String[] fields = { "test", "\"quote\"" }; String csv = getCSV(params, fields); - assertEquals("Leading quotes inside a quoted field must be escaped", - "\"\"\"quote\"\"\"", csv.trim()); + assertEquals("\"\"\"quote\"\"\"", csv.trim(), + "Leading quotes inside a quoted field must be escaped"); } @Test @@ -173,7 +173,8 @@ public void testCSVclipMaxLength() throws IOException { CSVConstants.CSV_MAXFIELDLENGTH, "8" }; String[] fields = { "test", "0123456789" }; String csv = getCSV(params, fields); - assertEquals("Field clipped to max. length = 8", "01234567", csv.trim()); + assertEquals("01234567", csv.trim(), + "Field clipped to max. length = 8"); } @Test @@ -182,8 +183,8 @@ public void testCSVclipMaxLengthQuote() throws IOException { CSVConstants.CSV_MAXFIELDLENGTH, "7" }; String[] fields = { "test", "1,\"2\",3,\"4\"" }; String csv = getCSV(params, fields); - assertEquals("Field clipped to max. length = 7", "\"1,\"\"2\"\",3\"", - csv.trim()); + assertEquals( "\"1,\"\"2\"\",3\"", csv.trim(), + "Field clipped to max. length = 7"); } @Test @@ -193,8 +194,8 @@ public void testCSVmultiValueFields() throws IOException { CSVConstants.CSV_QUOTECHARACTER, "" }; String[] fields = { "test", "abc", "test", "def" }; String csv = getCSV(params, fields); - assertEquals("Values of multi-value fields are concatenated by |", - "abc|def", csv.trim()); + assertEquals("abc|def", csv.trim(), + "Values of multi-value fields are concatenated by |"); } @Test @@ -211,7 +212,7 @@ public void testCSVEncoding() throws IOException { CSVConstants.CSV_CHARSET, charset }; String[] fields = { "test", test }; String csv = getCSV(params, fields); - assertEquals("wrong charset conversion", test, csv.trim()); + assertEquals( test, csv.trim(), "wrong charset conversion"); } } @@ -225,8 +226,8 @@ public void testCSVEncodingSeparator() throws IOException { }; String[] fields = { "test", "abc", "test", "def" }; String csv = getCSV(params, fields); - assertEquals("Values of multi-value fields are concatenated by ¦", - "abc\u00a6def", csv.trim()); + assertEquals("abc\u00a6def", csv.trim(), + "Values of multi-value fields are concatenated by ¦"); } @Test @@ -247,8 +248,8 @@ public void testCSVtabSeparated() throws IOException { docs[1].add("3", "C"); String csv = getCSV(params, docs); String[] records = csv.trim().split("\\r\\n"); - assertEquals("tab-separated output", "a|b\ta\"2\"b\tc,d", records[0]); - assertEquals("tab-separated output", "A\tB\tC", records[1]); + assertEquals("a|b\ta\"2\"b\tc,d", records[0], "tab-separated output"); + assertEquals( "A\tB\tC", records[1], "tab-separated output"); } @Test @@ -259,7 +260,7 @@ public void testCSVdateField() throws IOException { docs[0] = new NutchDocument(); docs[0].add("date", new Date(0)); // 1970-01-01 String csv = getCSV(params, docs); - assertTrue("date conversion", csv.contains("1970")); + assertTrue(csv.contains("1970"), "date conversion"); } } diff --git a/src/plugin/language-identifier/src/test/org/apache/nutch/analysis/lang/TestHTMLLanguageParser.java b/src/plugin/language-identifier/src/test/org/apache/nutch/analysis/lang/TestHTMLLanguageParser.java index 97be1408e5..e664e221b9 100644 --- a/src/plugin/language-identifier/src/test/org/apache/nutch/analysis/lang/TestHTMLLanguageParser.java +++ b/src/plugin/language-identifier/src/test/org/apache/nutch/analysis/lang/TestHTMLLanguageParser.java @@ -26,8 +26,10 @@ import org.apache.nutch.parse.ParseUtil; import org.apache.nutch.protocol.Content; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.fail; public class TestHTMLLanguageParser { @@ -55,12 +57,12 @@ public void testMetaHTMLParsing() { for (int t = 0; t < docs.length; t++) { Content content = getContent(docs[t]); Parse parse = parser.parse(content).get(content.getUrl()); - Assert.assertEquals(metalanguages[t], (String) parse.getData() + assertEquals(metalanguages[t], (String) parse.getData() .getParseMeta().get(Metadata.LANGUAGE)); } } catch (Exception e) { e.printStackTrace(System.out); - Assert.fail(e.toString()); + fail(e.toString()); } } @@ -89,7 +91,7 @@ public void testParseLanguage() { { "torp, stuga, uthyres, bed & breakfast", null } }; for (int i = 0; i < 44; i++) { - Assert.assertEquals(tests[i][1], + assertEquals(tests[i][1], HTMLLanguageParser.LanguageParser.parseLanguage(tests[i][0])); } } @@ -124,7 +126,7 @@ public void testLanguageIndentifier() throws IOException { testLine = testLine.trim(); if (testLine.length() > 256) { lang = identifier.identifyLanguage(testLine); - Assert.assertEquals(tokens[1], lang); + assertEquals(tokens[1], lang); } } testFile.close(); @@ -136,7 +138,7 @@ public void testLanguageIndentifier() throws IOException { lang = identifier.identifyLanguage(content); System.out.println(lang); total += System.currentTimeMillis() - start; - Assert.assertEquals(tokens[1], lang); + assertEquals(tokens[1], lang); } } in.close(); diff --git a/src/plugin/lib-http/src/test/org/apache/nutch/protocol/http/api/TestRobotRulesParser.java b/src/plugin/lib-http/src/test/org/apache/nutch/protocol/http/api/TestRobotRulesParser.java index 202d2d08b9..20049d2fc6 100644 --- a/src/plugin/lib-http/src/test/org/apache/nutch/protocol/http/api/TestRobotRulesParser.java +++ b/src/plugin/lib-http/src/test/org/apache/nutch/protocol/http/api/TestRobotRulesParser.java @@ -18,11 +18,13 @@ import java.util.Set; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; import crawlercommons.robots.BaseRobotRules; +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertTrue; + /** * JUnit test case which tests *
    @@ -100,10 +102,9 @@ private void testRulesOnPaths(String agent, String[] paths, boolean[] results) { for (int counter = 0; counter < paths.length; counter++) { boolean res = rules.isAllowed(paths[counter]); - Assert.assertTrue( - "testing on agent (" + agent + "), and " + "path " + paths[counter] - + " got " + res + ", expected " + results[counter], - res == results[counter]); + assertEquals(res, results[counter], + "testing on agent (" + agent + "), and " + "path " + + paths[counter] + " got " + res + ", expected " + results[counter]); } } @@ -137,21 +138,21 @@ public void testCrawlDelay() { // returned by the parser rules = parser.parseRules("testCrawlDelay", ROBOTS_STRING.getBytes(), CONTENT_TYPE, Set.of(SINGLE_AGENT1.toLowerCase())); - Assert.assertTrue("testing crawl delay for agent " + SINGLE_AGENT1 + " : ", - (rules.getCrawlDelay() == 10000)); + assertTrue((rules.getCrawlDelay() == 10000), + "testing crawl delay for agent " + SINGLE_AGENT1 + " : "); // for SINGLE_AGENT2, the crawl delay of 20 seconds, i.e. 20000 msec must be // returned by the parser rules = parser.parseRules("testCrawlDelay", ROBOTS_STRING.getBytes(), CONTENT_TYPE, Set.of(SINGLE_AGENT2.toLowerCase())); - Assert.assertTrue("testing crawl delay for agent " + SINGLE_AGENT2 + " : ", - (rules.getCrawlDelay() == 20000)); + assertTrue((rules.getCrawlDelay() == 20000), + "testing crawl delay for agent " + SINGLE_AGENT2 + " : "); // for UNKNOWN_AGENT, the default crawl delay must be returned. rules = parser.parseRules("testCrawlDelay", ROBOTS_STRING.getBytes(), CONTENT_TYPE, Set.of(UNKNOWN_AGENT.toLowerCase())); - Assert.assertTrue("testing crawl delay for agent " + UNKNOWN_AGENT + " : ", - (rules.getCrawlDelay() == Long.MIN_VALUE)); + assertTrue((rules.getCrawlDelay() == Long.MIN_VALUE), + "testing crawl delay for agent " + UNKNOWN_AGENT + " : "); } /** @@ -186,20 +187,20 @@ public void testCrawlDelayDeprecatedAPIMethod() { // returned by the parser rules = parser.parseRules("testCrawlDelay", ROBOTS_STRING.getBytes(), CONTENT_TYPE, SINGLE_AGENT1); - Assert.assertTrue("testing crawl delay for agent " + SINGLE_AGENT1 + " : ", - (rules.getCrawlDelay() == 10000)); + assertTrue((rules.getCrawlDelay() == 10000), + "testing crawl delay for agent " + SINGLE_AGENT1 + " : "); // for SINGLE_AGENT2, the crawl delay of 20 seconds, i.e. 20000 msec must be // returned by the parser rules = parser.parseRules("testCrawlDelay", ROBOTS_STRING.getBytes(), CONTENT_TYPE, SINGLE_AGENT2); - Assert.assertTrue("testing crawl delay for agent " + SINGLE_AGENT2 + " : ", - (rules.getCrawlDelay() == 20000)); + assertTrue((rules.getCrawlDelay() == 20000), + "testing crawl delay for agent " + SINGLE_AGENT2 + " : "); // for UNKNOWN_AGENT, the default crawl delay must be returned. rules = parser.parseRules("testCrawlDelay", ROBOTS_STRING.getBytes(), CONTENT_TYPE, UNKNOWN_AGENT); - Assert.assertTrue("testing crawl delay for agent " + UNKNOWN_AGENT + " : ", - (rules.getCrawlDelay() == Long.MIN_VALUE)); + assertTrue((rules.getCrawlDelay() == Long.MIN_VALUE), + "testing crawl delay for agent " + UNKNOWN_AGENT + " : "); } } diff --git a/src/plugin/lib-regex-filter/src/test/org/apache/nutch/urlfilter/api/RegexURLFilterBaseTest.java b/src/plugin/lib-regex-filter/src/test/org/apache/nutch/urlfilter/api/RegexURLFilterBaseTest.java index 080b2e5870..3f20f28ef1 100644 --- a/src/plugin/lib-regex-filter/src/test/org/apache/nutch/urlfilter/api/RegexURLFilterBaseTest.java +++ b/src/plugin/lib-regex-filter/src/test/org/apache/nutch/urlfilter/api/RegexURLFilterBaseTest.java @@ -26,12 +26,13 @@ import java.util.concurrent.TimeUnit; import org.apache.commons.lang3.time.StopWatch; -import org.junit.Assert; import org.slf4j.Logger; import org.slf4j.LoggerFactory; import org.apache.nutch.net.URLFilter; +import static org.junit.jupiter.api.Assertions.*; + /** * JUnit based test of class RegexURLFilterBase. * @@ -53,7 +54,7 @@ protected void bench(int loops, String file) { bench(loops, new FileReader(SAMPLES + SEPARATOR + file + ".rules"), new FileReader(SAMPLES + SEPARATOR + file + ".urls")); } catch (Exception e) { - Assert.fail(e.toString()); + fail(e.toString()); } } @@ -67,7 +68,7 @@ protected void bench(int loops, Reader rules, Reader urls) { test(filter, expected); } } catch (Exception e) { - Assert.fail(e.toString()); + fail(e.toString()); } stopWatch.stop(); LOG.info("bench time {} loops {} ms", loops, stopWatch.getTime(TimeUnit.MILLISECONDS)); @@ -78,7 +79,7 @@ protected void bench(int loops, String rulesFile, String urlsFile) { bench(loops, new FileReader(SAMPLES + SEPARATOR + rulesFile), new FileReader(SAMPLES + SEPARATOR + urlsFile)); } catch (Exception e) { - Assert.fail(e.toString()); + fail(e.toString()); } } @@ -87,7 +88,7 @@ protected void test(String rulesFile, String urlsFile) { test(new FileReader(SAMPLES + SEPARATOR + rulesFile), new FileReader(SAMPLES + SEPARATOR + urlsFile)); } catch (Exception e) { - Assert.fail(e.toString()); + fail(e.toString()); } } @@ -96,7 +97,7 @@ protected void test(String file) { test(new FileReader(SAMPLES + SEPARATOR + file + ".rules"), new FileReader(SAMPLES + SEPARATOR + file + ".urls")); } catch (Exception e) { - Assert.fail(e.toString()); + fail(e.toString()); } } @@ -104,7 +105,7 @@ protected void test(Reader rules, Reader urls) { try { test(getURLFilter(rules), readURLFile(urls)); } catch (Exception e) { - Assert.fail(e.toString()); + fail(e.toString()); } } @@ -112,9 +113,9 @@ protected void test(URLFilter filter, FilteredURL[] expected) { for (int i = 0; i < expected.length; i++) { String result = filter.filter(expected[i].url); if (result != null) { - Assert.assertTrue(expected[i].url, expected[i].sign); + assertTrue(expected[i].sign, expected[i].url); } else { - Assert.assertFalse(expected[i].url, expected[i].sign); + assertFalse(expected[i].sign, expected[i].url); } } } diff --git a/src/plugin/mimetype-filter/src/test/org/apache/nutch/indexer/filter/MimeTypeIndexingFilterTest.java b/src/plugin/mimetype-filter/src/test/org/apache/nutch/indexer/filter/MimeTypeIndexingFilterTest.java index 28b9f401f7..6f2be4cf80 100644 --- a/src/plugin/mimetype-filter/src/test/org/apache/nutch/indexer/filter/MimeTypeIndexingFilterTest.java +++ b/src/plugin/mimetype-filter/src/test/org/apache/nutch/indexer/filter/MimeTypeIndexingFilterTest.java @@ -29,9 +29,10 @@ import org.apache.nutch.parse.ParseStatus; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Before; -import org.junit.Test; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.*; /** * JUnit based tests of class @@ -45,7 +46,7 @@ public class MimeTypeIndexingFilterTest { private String[] MIME_TYPES = { "text/html", "image/png", "application/pdf" }; private ParseImpl[] parses = new ParseImpl[MIME_TYPES.length]; - @Before + @BeforeEach public void setUp() throws Exception { for (int i = 0; i < MIME_TYPES.length; i++) { Metadata metadata = new Metadata(); @@ -61,9 +62,10 @@ public void setUp() throws Exception { @Test public void testMissingConfigFile() throws Exception { String file = conf.get(MimeTypeIndexingFilter.MIMEFILTER_REGEX_FILE, ""); - Assert.assertEquals(String - .format("Property %s must not be present in the the configuration file", - MimeTypeIndexingFilter.MIMEFILTER_REGEX_FILE), "", file); + assertEquals("", file, + String + .format("Property %s must not be present in the the configuration file", + MimeTypeIndexingFilter.MIMEFILTER_REGEX_FILE)); filter.setConf(conf); @@ -72,7 +74,7 @@ public void testMissingConfigFile() throws Exception { NutchDocument doc = filter.filter(new NutchDocument(), parses[i], new Text("http://www.example.com/"), new CrawlDatum(), new Inlinks()); - Assert.assertNotNull("All documents must be allowed by default", doc); + assertNotNull(doc, "All documents must be allowed by default"); } } @@ -86,9 +88,9 @@ public void testAllowOnlyImages() throws Exception { new Text("http://www.example.com/"), new CrawlDatum(), new Inlinks()); if (MIME_TYPES[i].contains("image")) { - Assert.assertNotNull("Allow only images", doc); + assertNotNull(doc, "Allow only images"); } else { - Assert.assertNull("Block everything else", doc); + assertNull(doc, "Block everything else"); } } } @@ -103,9 +105,9 @@ public void testBlockHTML() throws Exception { new Text("http://www.example.com/"), new CrawlDatum(), new Inlinks()); if (MIME_TYPES[i].contains("html")) { - Assert.assertNull("Block only HTML documents", doc); + assertNull(doc, "Block only HTML documents"); } else { - Assert.assertNotNull("Allow everything else", doc); + assertNotNull(doc, "Allow everything else"); } } } diff --git a/src/plugin/parse-ext/src/test/org/apache/nutch/parse/ext/TestExtParser.java b/src/plugin/parse-ext/src/test/org/apache/nutch/parse/ext/TestExtParser.java index 782a1529b9..c8698e815f 100644 --- a/src/plugin/parse-ext/src/test/org/apache/nutch/parse/ext/TestExtParser.java +++ b/src/plugin/parse-ext/src/test/org/apache/nutch/parse/ext/TestExtParser.java @@ -27,15 +27,17 @@ import org.apache.nutch.util.NutchConfiguration; import org.apache.hadoop.io.Text; import org.apache.nutch.crawl.CrawlDatum; -import org.junit.After; -import org.junit.Assert; -import org.junit.Before; -import org.junit.Test; +import org.junit.jupiter.api.AfterEach; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; import java.io.File; import java.io.FileOutputStream; import java.io.IOException; +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertTrue; + /** * Unit tests for ExtParser. First creates a temp file with fixed content, then * fetch and parse it using external command 'cat' and 'md5sum' alternately for @@ -56,7 +58,7 @@ public class TestExtParser { // echo -n "nutch rocks nutch rocks nutch rocks" | md5sum private String expectedMD5sum = "df46711a1a48caafc98b1c3b83aa1526"; - @Before + @BeforeEach protected void setUp() throws ProtocolException, IOException { // prepare a temp file with expectedText as its content // This system property is defined in ./src/plugin/build-plugin.xml @@ -85,7 +87,7 @@ protected void setUp() throws ProtocolException, IOException { protocol = null; } - @After + @AfterEach protected void tearDown() { // clean content content = null; @@ -115,14 +117,14 @@ public void testIt() throws ParseException { content.setContentType(contentType); parse = new ParseUtil(conf).parseByExtensionId("parse-ext", content).get( content.getUrl()); - Assert.assertEquals(expectedText, parse.getText()); + assertEquals(expectedText, parse.getText()); // check external parser that does 'md5sum' contentType = "application/vnd.nutch.example.md5sum"; content.setContentType(contentType); parse = new ParseUtil(conf).parseByExtensionId("parse-ext", content).get( content.getUrl()); - Assert.assertTrue(parse.getText().startsWith(expectedMD5sum)); + assertTrue(parse.getText().startsWith(expectedMD5sum)); } } diff --git a/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestDOMContentUtils.java b/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestDOMContentUtils.java index d50e9052de..bcebd98049 100644 --- a/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestDOMContentUtils.java +++ b/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestDOMContentUtils.java @@ -27,13 +27,15 @@ import java.util.StringTokenizer; import org.cyberneko.html.parsers.*; -import org.junit.Assert; -import org.junit.Before; -import org.junit.Test; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; import org.xml.sax.*; import org.w3c.dom.*; import org.apache.html.dom.*; +import static org.junit.jupiter.api.Assertions.assertTrue; +import static org.junit.jupiter.api.Assertions.fail; + /** * Unit tests for DOMContentUtils. */ @@ -179,7 +181,7 @@ public class TestDOMContentUtils { private static Configuration conf; private static DOMContentUtils utils = null; - @Before + @BeforeEach public void setup() { conf = NutchConfiguration.create(); conf.setBoolean("parser.html.form.use_action", true); @@ -200,7 +202,7 @@ public void setup() { node); testBaseHrefURLs[i] = new URL(testBaseHrefs[i]); } catch (Exception e) { - Assert.assertTrue("caught exception: " + e, false); + fail("caught exception: " + e); } testDOMs[i] = node; } @@ -273,11 +275,10 @@ public void testGetText() { StringBuffer sb = new StringBuffer(); utils.getText(sb, testDOMs[i]); String text = sb.toString(); - Assert.assertTrue( + assertTrue(equalsIgnoreWhitespace(answerText[i], text), "expecting text: " + answerText[i] + System.getProperty("line.separator") - + System.getProperty("line.separator") + "got text: " + text, - equalsIgnoreWhitespace(answerText[i], text)); + + System.getProperty("line.separator") + "got text: " + text); } } @@ -289,11 +290,10 @@ public void testGetTitle() { StringBuffer sb = new StringBuffer(); utils.getTitle(sb, testDOMs[i]); String text = sb.toString(); - Assert.assertTrue( + assertTrue(equalsIgnoreWhitespace(answerTitle[i], text), "expecting text: " + answerText[i] + System.getProperty("line.separator") - + System.getProperty("line.separator") + "got text: " + text, - equalsIgnoreWhitespace(answerTitle[i], text)); + + System.getProperty("line.separator") + "got text: " + text); } } @@ -332,26 +332,25 @@ private static final String outlinksString(Outlink[] o) { private static final void compareOutlinks(Outlink[] o1, Outlink[] o2) { if (o1.length != o2.length) { - Assert.assertTrue( + fail( "got wrong number of outlinks (expecting " + o1.length + ", got " - + o2.length + ")" + System.getProperty("line.separator") - + "answer: " + System.getProperty("line.separator") - + outlinksString(o1) + System.getProperty("line.separator") - + "got: " + System.getProperty("line.separator") - + outlinksString(o2) + System.getProperty("line.separator"), - false); + + o2.length + ")" + System.getProperty( + "line.separator") + "answer: " + System.getProperty( + "line.separator") + outlinksString(o1) + System.getProperty( + "line.separator") + "got: " + System.getProperty( + "line.separator") + outlinksString(o2) + System.getProperty( + "line.separator")); } for (int i = 0; i < o1.length; i++) { if (!o1[i].equals(o2[i])) { - Assert.assertTrue( - "got wrong outlinks at position " + i - + System.getProperty("line.separator") + "answer: " - + System.getProperty("line.separator") + "'" + o1[i].getToUrl() - + "', anchor: '" + o1[i].getAnchor() + "'" - + System.getProperty("line.separator") + "got: " - + System.getProperty("line.separator") + "'" + o2[i].getToUrl() - + "', anchor: '" + o2[i].getAnchor() + "'", false); + fail("got wrong outlinks at position " + i + System.getProperty( + "line.separator") + "answer: " + System.getProperty( + "line.separator") + "'" + o1[i].getToUrl() + "', anchor: '" + + o1[i].getAnchor() + "'" + System.getProperty( + "line.separator") + "got: " + System.getProperty( + "line.separator") + "'" + o2[i].getToUrl() + "', anchor: '" + + o2[i].getAnchor() + "'"); } } diff --git a/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestHtmlParser.java b/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestHtmlParser.java index 6e13a69ea6..47beff5a0f 100644 --- a/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestHtmlParser.java +++ b/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestHtmlParser.java @@ -27,11 +27,12 @@ import org.apache.nutch.parse.Parser; import org.apache.nutch.protocol.Content; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; import org.slf4j.Logger; import org.slf4j.LoggerFactory; +import static org.junit.jupiter.api.Assertions.*; + public class TestHtmlParser { private static final Logger LOG = LoggerFactory @@ -117,15 +118,15 @@ public void testEncodingDetection() { LOG.info("title:\t{}", title); LOG.info("keywords:\t{}", keywords); LOG.info("text:\t{}", text); - Assert.assertEquals("Title not extracted properly (" + name + ")", - encodingTestKeywords, title); + assertEquals(encodingTestKeywords, title, + "Title not extracted properly (" + name + ")"); for (String keyword : encodingTestKeywords.split(",\\s*")) { - Assert.assertTrue(keyword + " not found in text (" + name + ")", - text.contains(keyword)); + assertTrue(text.contains(keyword), + keyword + " not found in text (" + name + ")"); } - Assert.assertNotNull("No keywords extracted", keywords); - Assert.assertEquals("Keywords not extracted properly (" + name + ")", - encodingTestKeywords, keywords); + assertNotNull(keywords, "No keywords extracted"); + assertEquals(encodingTestKeywords, keywords, + "Keywords not extracted properly (" + name + ")"); } } @@ -137,8 +138,8 @@ public void testResolveBaseUrl() { Parse parse = parse(contentBytes); LOG.info(parse.getData().toString()); Outlink[] outlinks = parse.getData().getOutlinks(); - Assert.assertEquals(1, outlinks.length); - Assert.assertEquals("http://www.example.com/index.html", + assertEquals(1, outlinks.length); + assertEquals("http://www.example.com/index.html", outlinks[0].getToUrl()); } diff --git a/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestRobotsMetaProcessor.java b/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestRobotsMetaProcessor.java index e75635090c..b977b82abe 100644 --- a/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestRobotsMetaProcessor.java +++ b/src/plugin/parse-html/src/test/org/apache/nutch/parse/html/TestRobotsMetaProcessor.java @@ -22,12 +22,13 @@ import java.net.URL; import org.cyberneko.html.parsers.*; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; import org.xml.sax.*; import org.w3c.dom.*; import org.apache.html.dom.*; +import static org.junit.jupiter.api.Assertions.*; + /** Unit tests for HTMLMetaProcessor. */ public class TestRobotsMetaProcessor { @@ -117,7 +118,7 @@ public void testRobotsMetaProcessor() { { new URL("http://www.nutch.org"), new URL("http://www.nutch.org/base/") } }; } catch (Exception e) { - Assert.assertTrue("couldn't make test URLs!", false); + fail("couldn't make test URLs!"); } for (int i = 0; i < tests.length; i++) { @@ -134,19 +135,18 @@ public void testRobotsMetaProcessor() { HTMLMetaTags robotsMeta = new HTMLMetaTags(); HTMLMetaProcessor.getMetaTags(robotsMeta, node, currURLsAndAnswers[i][0]); - Assert.assertTrue("got index wrong on test " + i, - robotsMeta.getNoIndex() == answers[i][0]); - Assert.assertTrue("got follow wrong on test " + i, - robotsMeta.getNoFollow() == answers[i][1]); - Assert.assertTrue("got cache wrong on test " + i, - robotsMeta.getNoCache() == answers[i][2]); - Assert - .assertTrue( - "got base href wrong on test " + i + " (got " - + robotsMeta.getBaseHref() + ")", - ((robotsMeta.getBaseHref() == null) && (currURLsAndAnswers[i][1] == null)) + assertEquals(robotsMeta.getNoIndex(), answers[i][0], + "got index wrong on test " + i); + assertEquals(robotsMeta.getNoFollow(), answers[i][1], + "got follow wrong on test " + i); + assertEquals(robotsMeta.getNoCache(), answers[i][2], + "got cache wrong on test " + i); + assertTrue(((robotsMeta.getBaseHref() == null) && + (currURLsAndAnswers[i][1] == null)) || ((robotsMeta.getBaseHref() != null) && robotsMeta - .getBaseHref().equals(currURLsAndAnswers[i][1]))); + .getBaseHref().equals(currURLsAndAnswers[i][1])), + "got base href wrong on test " + i + " (got " + + robotsMeta.getBaseHref() + ")"); } } diff --git a/src/plugin/parse-js/src/test/org/apache/nutch/parse/js/TestJSParseFilter.java b/src/plugin/parse-js/src/test/org/apache/nutch/parse/js/TestJSParseFilter.java index 01bee13213..b0ee519142 100644 --- a/src/plugin/parse-js/src/test/org/apache/nutch/parse/js/TestJSParseFilter.java +++ b/src/plugin/parse-js/src/test/org/apache/nutch/parse/js/TestJSParseFilter.java @@ -37,8 +37,8 @@ import org.apache.nutch.protocol.ProtocolException; import org.apache.nutch.protocol.ProtocolFactory; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Before; -import org.junit.Test; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; import org.slf4j.Logger; import org.slf4j.LoggerFactory; @@ -66,7 +66,7 @@ public class TestJSParseFilter { private Configuration conf; - @Before + @BeforeEach public void setUp() { conf = NutchConfiguration.create(); conf.set("file.content.limit", "-1"); diff --git a/src/plugin/parse-metatags/src/test/org/apache/nutch/parse/metatags/TestMetatagParser.java b/src/plugin/parse-metatags/src/test/org/apache/nutch/parse/metatags/TestMetatagParser.java index 5702c10b00..9a3eb96c01 100644 --- a/src/plugin/parse-metatags/src/test/org/apache/nutch/parse/metatags/TestMetatagParser.java +++ b/src/plugin/parse-metatags/src/test/org/apache/nutch/parse/metatags/TestMetatagParser.java @@ -31,11 +31,12 @@ import org.apache.nutch.protocol.Protocol; import org.apache.nutch.protocol.ProtocolFactory; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; import org.slf4j.Logger; import org.slf4j.LoggerFactory; +import static org.junit.jupiter.api.Assertions.*; + public class TestMetatagParser { private String fileSeparator = System.getProperty("file.separator"); @@ -59,7 +60,7 @@ public Metadata parseMeta(String fileName, Configuration conf) { metadata = parse.getData().getParseMeta(); } catch (Exception e) { e.printStackTrace(); - Assert.fail(e.toString()); + fail(e.toString()); } return metadata; } @@ -72,8 +73,8 @@ public void testIt() { // check that we get the same values Metadata parseMeta = parseMeta(sampleFile, conf); - Assert.assertEquals(description, parseMeta.get("metatag.description")); - Assert.assertEquals(keywords, parseMeta.get("metatag.keywords")); + assertEquals(description, parseMeta.get("metatag.description")); + assertEquals(keywords, parseMeta.get("metatag.keywords")); } @Test @@ -93,7 +94,7 @@ public void testMultiValueMetatags() { } String[] expectedValues1 = { "Doug Cutting", "Michael Cafarella" }; for (String val : expectedValues1) { - Assert.assertTrue(failMessage + val, valueSet.contains(val)); + assertTrue(valueSet.contains(val), failMessage + val); } valueSet.clear(); @@ -103,7 +104,7 @@ public void testMultiValueMetatags() { String[] expectedValues2 = { "robot d'indexation", "web crawler", "Webcrawler" }; for (String val : expectedValues2) { - Assert.assertTrue(failMessage + val, valueSet.contains(val)); + assertTrue(valueSet.contains(val), failMessage + val); } } @@ -123,9 +124,8 @@ public void testDuplicatedMetatags() { LOG.info("metatags ({}): {}", parsePlugin, Arrays.toString(parseMeta.getValues("metatag.keywords"))); - Assert.assertEquals( - "Test document contains a single value of , metatag.keywords should be also single-valued", - 1, parseMeta.getValues("metatag.keywords").length); + assertEquals(1, parseMeta.getValues("metatag.keywords").length, + "Test document contains a single value of , metatag.keywords should be also single-valued"); } } } diff --git a/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestDOMContentUtils.java b/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestDOMContentUtils.java index b449e1e4db..423fa870d7 100644 --- a/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestDOMContentUtils.java +++ b/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestDOMContentUtils.java @@ -27,11 +27,13 @@ import org.apache.nutch.parse.Outlink; import org.apache.nutch.protocol.Content; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Before; -import org.junit.Test; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; import org.w3c.dom.DocumentFragment; +import static org.junit.jupiter.api.Assertions.assertTrue; +import static org.junit.jupiter.api.Assertions.fail; + /** * Unit tests for DOMContentUtils. */ @@ -175,7 +177,7 @@ public class TestDOMContentUtils { private static Configuration conf; private static DOMContentUtils utils = null; - @Before + @BeforeEach public void setup() throws Exception { conf = NutchConfiguration.create(); utils = new DOMContentUtils(conf); @@ -196,7 +198,7 @@ public void setup() throws Exception { parser.getParse(content, doc, root); testDOMs[i] = root; } catch (Exception e) { - Assert.assertTrue("caught exception: " + e, false); + fail("caught exception: " + e); } } answerOutlinks = new Outlink[][] { @@ -260,11 +262,10 @@ public void testGetText() throws Exception { StringBuffer sb = new StringBuffer(); utils.getText(sb, testDOMs[i]); String text = sb.toString(); - Assert.assertTrue( + assertTrue(equalsIgnoreWhitespace(answerText[i], text), "expecting text: " + answerText[i] + System.getProperty("line.separator") - + System.getProperty("line.separator") + "got text: " + text, - equalsIgnoreWhitespace(answerText[i], text)); + + System.getProperty("line.separator") + "got text: " + text); } } @@ -276,11 +277,10 @@ public void testGetTitle() throws Exception { StringBuffer sb = new StringBuffer(); utils.getTitle(sb, testDOMs[i]); String text = sb.toString(); - Assert.assertTrue( + assertTrue(equalsIgnoreWhitespace(answerTitle[i], text), "expecting text: " + answerText[i] + System.getProperty("line.separator") - + System.getProperty("line.separator") + "got text: " + text, - equalsIgnoreWhitespace(answerTitle[i], text)); + + System.getProperty("line.separator") + "got text: " + text); } } @@ -319,26 +319,25 @@ private static final String outlinksString(Outlink[] o) { private static final void compareOutlinks(Outlink[] o1, Outlink[] o2) { if (o1.length != o2.length) { - Assert.assertTrue( - "got wrong number of outlinks (expecting " + o1.length + ", got " - + o2.length + ")" + System.getProperty("line.separator") - + "answer: " + System.getProperty("line.separator") - + outlinksString(o1) + System.getProperty("line.separator") - + "got: " + System.getProperty("line.separator") - + outlinksString(o2) + System.getProperty("line.separator"), - false); + fail( + "got wrong number of outlinks (expecting " + o1.length + + ", got " + o2.length + ")" + System.getProperty( + "line.separator") + "answer: " + System.getProperty( + "line.separator") + outlinksString(o1) + System.getProperty( + "line.separator") + "got: " + System.getProperty( + "line.separator") + outlinksString(o2) + System.getProperty( + "line.separator")); } for (int i = 0; i < o1.length; i++) { if (!o1[i].equals(o2[i])) { - Assert.assertTrue( - "got wrong outlinks at position " + i - + System.getProperty("line.separator") + "answer: " - + System.getProperty("line.separator") + "'" + o1[i].getToUrl() - + "', anchor: '" + o1[i].getAnchor() + "'" - + System.getProperty("line.separator") + "got: " - + System.getProperty("line.separator") + "'" + o2[i].getToUrl() - + "', anchor: '" + o2[i].getAnchor() + "'", false); + fail("got wrong outlinks at position " + i + System.getProperty( + "line.separator") + "answer: " + System.getProperty( + "line.separator") + "'" + o1[i].getToUrl() + "', anchor: '" + + o1[i].getAnchor() + "'" + System.getProperty( + "line.separator") + "got: " + System.getProperty( + "line.separator") + "'" + o2[i].getToUrl() + "', anchor: '" + + o2[i].getAnchor() + "'"); } } } diff --git a/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestEmbeddedDocuments.java b/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestEmbeddedDocuments.java index 8d5b29030a..b66132e24c 100644 --- a/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestEmbeddedDocuments.java +++ b/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestEmbeddedDocuments.java @@ -18,9 +18,10 @@ import org.apache.nutch.parse.ParseException; import org.apache.nutch.protocol.ProtocolException; -import org.junit.Assert; -import org.junit.Before; -import org.junit.Test; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertTrue; /** * Unit tests for MSWordParser. @@ -34,7 +35,7 @@ public class TestEmbeddedDocuments extends TikaParserTest { private String expectedText = "When in the Course of human events"; @Override - @Before + @BeforeEach public void setUp() { super.setUp(); conf.setBoolean("tika.parse.embedded", true); @@ -44,8 +45,7 @@ public void setUp() { public void testIt() throws ProtocolException, ParseException { for (int i = 0; i < sampleFiles.length; i++) { String found = getTextContent(sampleFiles[i]); - Assert.assertTrue("text found : '" + found + "'", - found.contains(expectedText)); + assertTrue(found.contains(expectedText), "text found : '" + found + "'"); } } diff --git a/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestFeedParser.java b/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestFeedParser.java index 94eec535e6..6350984ed5 100644 --- a/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestFeedParser.java +++ b/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestFeedParser.java @@ -16,8 +16,7 @@ */ package org.apache.nutch.parse.tika; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; import org.apache.hadoop.conf.Configuration; import org.apache.hadoop.io.Text; import org.apache.nutch.crawl.CrawlDatum; @@ -32,6 +31,9 @@ import org.apache.nutch.protocol.ProtocolFactory; import org.apache.nutch.util.NutchConfiguration; +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.fail; + /** * Test Suite for the RSS feeds with the {@link TikaParser}. */ @@ -79,8 +81,8 @@ public void testIt() throws ProtocolException, ParseException { Outlink[] theOutlinks = theParseData.getOutlinks(); - Assert.assertTrue("There aren't 2 outlinks read!", - theOutlinks.length == 2); + assertEquals(2, theOutlinks.length, + "There aren't 2 outlinks read!"); // now check to make sure that those are the two outlinks boolean hasLink1 = false, hasLink2 = false; @@ -97,7 +99,7 @@ public void testIt() throws ProtocolException, ParseException { } if (!hasLink1 || !hasLink2) { - Assert.fail("Outlinks read from sample rss file are not correct!"); + fail("Outlinks read from sample rss file are not correct!"); } } } diff --git a/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestHtmlParser.java b/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestHtmlParser.java index e03622246f..17d2444473 100644 --- a/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestHtmlParser.java +++ b/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestHtmlParser.java @@ -27,11 +27,12 @@ import org.apache.nutch.parse.Parser; import org.apache.nutch.protocol.Content; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; import org.slf4j.Logger; import org.slf4j.LoggerFactory; +import static org.junit.jupiter.api.Assertions.*; + public class TestHtmlParser { private static final Logger LOG = LoggerFactory @@ -117,15 +118,15 @@ public void testEncodingDetection() { LOG.info("title:\t{}", title); LOG.info("keywords:\t{}", keywords); LOG.info("text:\t{}", text); - Assert.assertEquals("Title not extracted properly (" + name + ")", - encodingTestKeywords, title); + assertEquals(encodingTestKeywords, title, + "Title not extracted properly (" + name + ")"); for (String keyword : encodingTestKeywords.split(",\\s*")) { - Assert.assertTrue(keyword + " not found in text (" + name + ")", - text.contains(keyword)); + assertTrue(text.contains(keyword), + keyword + " not found in text (" + name + ")"); } - Assert.assertNotNull("No keywords extracted", keywords); - Assert.assertEquals("Keywords not extracted properly (" + name + ")", - encodingTestKeywords, keywords); + assertNotNull(keywords, "No keywords extracted"); + assertEquals(encodingTestKeywords, keywords, + "Keywords not extracted properly (" + name + ")"); } } @@ -137,8 +138,8 @@ public void testResolveBaseUrl() { Parse parse = parse(contentBytes); LOG.info(parse.getData().toString()); Outlink[] outlinks = parse.getData().getOutlinks(); - Assert.assertEquals(1, outlinks.length); - Assert.assertEquals("http://www.example.com/index.html", + assertEquals(1, outlinks.length); + assertEquals("http://www.example.com/index.html", outlinks[0].getToUrl()); } diff --git a/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestImageMetadata.java b/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestImageMetadata.java index 0f1505d0a0..6f694a59e0 100644 --- a/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestImageMetadata.java +++ b/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestImageMetadata.java @@ -27,8 +27,9 @@ import org.apache.nutch.util.NutchConfiguration; import org.apache.hadoop.io.Text; import org.apache.nutch.crawl.CrawlDatum; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertEquals; /** * Test extraction of image metadata @@ -55,8 +56,8 @@ public void testIt() throws ProtocolException, ParseException { parse = new ParseUtil(conf).parseByExtensionId("parse-tika", content) .get(content.getUrl()); - Assert.assertEquals("121", parse.getData().getMeta("width")); - Assert.assertEquals("48", parse.getData().getMeta("height")); + assertEquals("121", parse.getData().getMeta("width")); + assertEquals("48", parse.getData().getMeta("height")); } } diff --git a/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestMSWordParser.java b/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestMSWordParser.java index c5062f6841..a8f5dc93fc 100644 --- a/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestMSWordParser.java +++ b/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestMSWordParser.java @@ -20,8 +20,10 @@ import org.apache.nutch.parse.ParseException; import org.apache.nutch.protocol.ProtocolException; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertFalse; +import static org.junit.jupiter.api.Assertions.assertTrue; /** * Unit tests for MSWordParser. @@ -38,8 +40,7 @@ public class TestMSWordParser extends TikaParserTest { public void testIt() throws ProtocolException, ParseException { for (int i = 0; i < sampleFiles.length; i++) { String found = getTextContent(sampleFiles[i]); - Assert.assertTrue("text found : '" + found + "'", - found.startsWith(expectedText)); + assertTrue(found.startsWith(expectedText), "text found : '" + found + "'"); } } @@ -49,8 +50,8 @@ public void testOpeningDocs() throws ProtocolException, ParseException { for (int i = 0; i < filenames.length; i++) { if (filenames[i].endsWith(".doc") == false) continue; - Assert.assertTrue("can't read content of " + filenames[i], - getTextContent(filenames[i]).length() > 0); + assertFalse(getTextContent(filenames[i]).isEmpty(), + "can't read content of " + filenames[i]); } } } diff --git a/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestOOParser.java b/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestOOParser.java index 41c47e9ba8..90667a64d5 100644 --- a/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestOOParser.java +++ b/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestOOParser.java @@ -21,8 +21,9 @@ import org.apache.nutch.parse.ParseException; import org.apache.nutch.protocol.ProtocolException; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertFalse; /** * Unit tests for OOParser. @@ -53,7 +54,7 @@ public void testIt() throws ProtocolException, ParseException { // simply test for the presence of a text - the ordering of the elements // may differ from what was expected // in the previous tests - Assert.assertTrue(text != null && text.length() > 0); + assertFalse(text.isEmpty()); System.out.println("Found " + sampleFiles[i] + ": " + text); } diff --git a/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestPdfParser.java b/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestPdfParser.java index 784b55c1e7..2671a4fd9e 100644 --- a/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestPdfParser.java +++ b/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestPdfParser.java @@ -18,8 +18,9 @@ import org.apache.nutch.parse.ParseException; import org.apache.nutch.protocol.ProtocolException; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertTrue; /** * Unit tests for PdfParser. @@ -36,7 +37,7 @@ public class TestPdfParser extends TikaParserTest { public void testIt() throws ProtocolException, ParseException { for (int i = 0; i < sampleFiles.length; i++) { int index = getTextContent(sampleFiles[i]).indexOf(expectedText); - Assert.assertTrue(index > 0); + assertTrue(index > 0); } } diff --git a/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestRTFParser.java b/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestRTFParser.java index de0c1c6309..22147cba62 100644 --- a/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestRTFParser.java +++ b/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestRTFParser.java @@ -27,8 +27,10 @@ import org.apache.nutch.protocol.ProtocolException; import org.apache.nutch.protocol.ProtocolFactory; import org.apache.tika.metadata.DublinCore; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertTrue; /** * Unit tests for TestRTFParser. @@ -54,13 +56,13 @@ public void testIt() throws ProtocolException, ParseException { parse = new ParseUtil(conf).parseByExtensionId("parse-tika", content).get( content.getUrl()); String text = parse.getText(); - Assert.assertTrue(text.contains("The quick brown fox jumps over the lazy dog")); + assertTrue(text.contains("The quick brown fox jumps over the lazy dog")); String title = parse.getData().getTitle(); Metadata meta = parse.getData().getParseMeta(); - Assert.assertEquals("test rft document", title); - Assert.assertEquals("tests", meta.get(DublinCore.SUBJECT.getName())); + assertEquals("test rft document", title); + assertEquals("tests", meta.get(DublinCore.SUBJECT.getName())); } } diff --git a/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestRobotsMetaProcessor.java b/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestRobotsMetaProcessor.java index 7591cef2e5..a9eb2adf27 100644 --- a/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestRobotsMetaProcessor.java +++ b/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestRobotsMetaProcessor.java @@ -25,10 +25,11 @@ import org.apache.nutch.parse.Parse; import org.apache.nutch.protocol.Content; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; import org.w3c.dom.DocumentFragment; +import static org.junit.jupiter.api.Assertions.*; + /** Unit tests for HTMLMetaProcessor. */ public class TestRobotsMetaProcessor { @@ -125,7 +126,7 @@ public void testRobotsMetaProcessor() { new URL("http://www.nutch.org/base/") }, { new URL("http://www.nutch.org"), null } }; } catch (Exception e) { - Assert.assertTrue("couldn't make test URLs!", false); + fail("couldn't make test URLs!"); } for (int i = 0; i < tests.length; i++) { @@ -148,30 +149,27 @@ public void testRobotsMetaProcessor() { HTMLMetaTags robotsMeta = new HTMLMetaTags(); HTMLMetaProcessor.getMetaTags(robotsMeta, root, currURLsAndAnswers[i][0]); - Assert.assertEquals("got noindex wrong on test " + i, - answers[i][0], robotsMeta.getNoIndex()); - Assert.assertEquals("got nofollow wrong on test " + i, - answers[i][1], robotsMeta.getNoFollow()); - Assert.assertEquals("got nocache wrong on test " + i, - answers[i][2], robotsMeta.getNoCache()); - Assert - .assertTrue( - "got base href wrong on test " + i + " (got " - + robotsMeta.getBaseHref() + ")", + assertEquals(answers[i][0], robotsMeta.getNoIndex(), + "got noindex wrong on test " + i); + assertEquals(answers[i][1], robotsMeta.getNoFollow(), + "got nofollow wrong on test " + i); + assertEquals(answers[i][2], robotsMeta.getNoCache(), + "got nocache wrong on test " + i); + assertTrue( ((robotsMeta.getBaseHref() == null) && (currURLsAndAnswers[i][1] == null)) || ((robotsMeta.getBaseHref() != null) && robotsMeta - .getBaseHref().equals(currURLsAndAnswers[i][1]))); + .getBaseHref().equals(currURLsAndAnswers[i][1])), + "got base href wrong on test " + i + " (got " + + robotsMeta.getBaseHref() + ")"); if (tests[i].contains("meta-refresh redirect")) { // test for NUTCH-2589 URL metaRefreshUrl = robotsMeta.getRefreshHref(); - Assert.assertNotNull("failed to get meta-refresh redirect", - metaRefreshUrl); - Assert.assertEquals("failed to get meta-refresh redirect", - "http://example.com/", metaRefreshUrl.toString()); - Assert.assertEquals( - "failed to add meta-refresh redirect to parse status", - "http://example.com/", parse.getData().getStatus().getArgs()[0]); + assertNotNull(metaRefreshUrl, "failed to get meta-refresh redirect"); + assertEquals("http://example.com/", metaRefreshUrl.toString(), + "failed to get meta-refresh redirect"); + assertEquals("http://example.com/", parse.getData().getStatus().getArgs()[0], + "failed to add meta-refresh redirect to parse status"); } } } diff --git a/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestXlsxParser.java b/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestXlsxParser.java index 85427dbdc0..1228f8fab9 100644 --- a/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestXlsxParser.java +++ b/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TestXlsxParser.java @@ -16,13 +16,16 @@ */ package org.apache.nutch.parse.tika; -import static org.junit.Assert.*; +import static org.hamcrest.CoreMatchers.is; +import static org.hamcrest.MatcherAssert.assertThat; import java.io.IOException; import org.apache.nutch.parse.ParseException; import org.apache.nutch.protocol.ProtocolException; -import org.junit.Test; +import org.hamcrest.CoreMatchers.*; +import org.hamcrest.MatcherAssert.*; +import org.junit.jupiter.api.Test; public class TestXlsxParser extends TikaParserTest { @@ -32,7 +35,7 @@ public void testIt() throws ProtocolException, ParseException, IOException { String expected = "test.txt This is a test for spreadsheets xlsx"; // text is distributed over columns and rows, need to normalize white space found = found.replaceAll("\\s+", " ").trim(); - assertEquals(found, expected); + assertThat(found, is(expected)); } } diff --git a/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TikaParserTest.java b/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TikaParserTest.java index 781debbae9..9fd468b55e 100644 --- a/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TikaParserTest.java +++ b/src/plugin/parse-tika/src/test/org/apache/nutch/parse/tika/TikaParserTest.java @@ -27,7 +27,7 @@ import org.apache.nutch.protocol.ProtocolException; import org.apache.nutch.protocol.ProtocolFactory; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Before; +import org.junit.jupiter.api.BeforeEach; /** * Base class to extend Tika parser tests from. @@ -45,7 +45,7 @@ public class TikaParserTest { protected Configuration conf; - @Before + @BeforeEach public void setUp() { conf = NutchConfiguration.create(); conf.set("file.content.limit", "-1"); diff --git a/src/plugin/parse-zip/src/test/org/apache/nutch/parse/zip/TestZipParser.java b/src/plugin/parse-zip/src/test/org/apache/nutch/parse/zip/TestZipParser.java index 767099ef6a..1877cf6cc6 100644 --- a/src/plugin/parse-zip/src/test/org/apache/nutch/parse/zip/TestZipParser.java +++ b/src/plugin/parse-zip/src/test/org/apache/nutch/parse/zip/TestZipParser.java @@ -27,8 +27,9 @@ import org.apache.nutch.util.NutchConfiguration; import org.apache.hadoop.io.Text; import org.apache.nutch.crawl.CrawlDatum; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertTrue; /** * Based on Unit tests for MSWordParser by John Xing @@ -63,10 +64,9 @@ public void testIt() throws ProtocolException, ParseException { new CrawlDatum()).getContent(); parse = new ParseUtil(conf).parseByExtensionId("parse-zip", content).get( content.getUrl()); - Assert.assertTrue( + assertTrue(parse.getText().startsWith(expectedText), "Extracted text does not start with <" + expectedText + ">: <" - + parse.getText() + ">", - parse.getText().startsWith(expectedText)); + + parse.getText() + ">"); } } diff --git a/src/plugin/parsefilter-regex/src/test/org/apache/nutch/parsefilter/regex/TestRegexParseFilter.java b/src/plugin/parsefilter-regex/src/test/org/apache/nutch/parsefilter/regex/TestRegexParseFilter.java index 64fa7f6666..44f6cb1148 100644 --- a/src/plugin/parsefilter-regex/src/test/org/apache/nutch/parsefilter/regex/TestRegexParseFilter.java +++ b/src/plugin/parsefilter-regex/src/test/org/apache/nutch/parsefilter/regex/TestRegexParseFilter.java @@ -24,14 +24,17 @@ import org.apache.nutch.parse.ParseResult; import org.apache.nutch.protocol.Content; import org.apache.nutch.util.NutchConfiguration; -import junit.framework.TestCase; +import org.junit.jupiter.api.Test; -public class TestRegexParseFilter extends TestCase { +import static org.junit.jupiter.api.Assertions.assertEquals; + +class TestRegexParseFilter { private final static String SEPARATOR = System.getProperty("file.separator"); private final static String SAMPLES = System.getProperty("test.data", "."); - public void testPositiveFilter() throws Exception { + @Test + void testPositiveFilter() throws Exception { Configuration conf = NutchConfiguration.create(); String file = SAMPLES + SEPARATOR + "regex-parsefilter.txt"; @@ -52,8 +55,9 @@ public void testPositiveFilter() throws Exception { assertEquals("true", meta.get("first")); assertEquals("true", meta.get("second")); } - - public void testNegativeFilter() throws Exception { + + @Test + void testNegativeFilter() throws Exception { Configuration conf = NutchConfiguration.create(); String file = SAMPLES + SEPARATOR + "regex-parsefilter.txt"; diff --git a/src/plugin/protocol-file/src/test/org/apache/nutch/protocol/file/TestProtocolFile.java b/src/plugin/protocol-file/src/test/org/apache/nutch/protocol/file/TestProtocolFile.java index ffee2ba4dd..0139151fc6 100644 --- a/src/plugin/protocol-file/src/test/org/apache/nutch/protocol/file/TestProtocolFile.java +++ b/src/plugin/protocol-file/src/test/org/apache/nutch/protocol/file/TestProtocolFile.java @@ -29,9 +29,11 @@ import org.apache.nutch.protocol.ProtocolOutput; import org.apache.nutch.protocol.ProtocolStatus; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Before; -import org.junit.Test; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertNotNull; /** * @author mattmann @@ -57,7 +59,7 @@ public class TestProtocolFile { private Configuration conf; - @Before + @BeforeEach public void setUp() { conf = NutchConfiguration.create(); } @@ -77,20 +79,20 @@ public void testSetContentType() throws ProtocolException { */ public void setContentType(String testTextFile) throws ProtocolException { String urlString = "file:" + sampleDir + fileSeparator + testTextFile; - Assert.assertNotNull(urlString); + assertNotNull(urlString); Protocol protocol = new ProtocolFactory(conf).getProtocol(urlString); ProtocolOutput output = protocol.getProtocolOutput(new Text(urlString), datum); - Assert.assertNotNull(output); - Assert.assertEquals("Status code: [" + output.getStatus().getCode() - + "], not equal to: [" + ProtocolStatus.SUCCESS + "]: args: [" - + output.getStatus().getArgs() + "]", ProtocolStatus.SUCCESS, output - .getStatus().getCode()); - Assert.assertNotNull(output.getContent()); - Assert.assertNotNull(output.getContent().getContentType()); - Assert.assertEquals(expectedMimeType, output.getContent().getContentType()); - Assert.assertNotNull(output.getContent().getMetadata()); - Assert.assertEquals(expectedMimeType, output.getContent().getMetadata() + assertNotNull(output); + assertEquals(ProtocolStatus.SUCCESS, output.getStatus().getCode(), + "Status code: [" + output.getStatus().getCode() + + "], not equal to: [" + ProtocolStatus.SUCCESS + "]: args: [" + + output.getStatus().getArgs() + "]"); + assertNotNull(output.getContent()); + assertNotNull(output.getContent().getContentType()); + assertEquals(expectedMimeType, output.getContent().getContentType()); + assertNotNull(output.getContent().getMetadata()); + assertEquals(expectedMimeType, output.getContent().getMetadata() .get(Response.CONTENT_TYPE)); } diff --git a/src/plugin/protocol-http/src/test/org/apache/nutch/protocol/http/TestBadServerResponses.java b/src/plugin/protocol-http/src/test/org/apache/nutch/protocol/http/TestBadServerResponses.java index 13c25587cb..02e381e466 100644 --- a/src/plugin/protocol-http/src/test/org/apache/nutch/protocol/http/TestBadServerResponses.java +++ b/src/plugin/protocol-http/src/test/org/apache/nutch/protocol/http/TestBadServerResponses.java @@ -16,19 +16,18 @@ */ package org.apache.nutch.protocol.http; -import static org.junit.Assert.assertEquals; -import static org.junit.Assert.assertNotNull; -import static org.junit.Assert.assertTrue; - -import java.lang.invoke.MethodHandles; -import java.nio.charset.StandardCharsets; - import org.apache.nutch.protocol.AbstractHttpProtocolPluginTest; import org.apache.nutch.protocol.ProtocolOutput; -import org.junit.Test; +import org.junit.jupiter.api.Test; import org.slf4j.Logger; import org.slf4j.LoggerFactory; +import java.lang.invoke.MethodHandles; +import java.nio.charset.StandardCharsets; + +import static org.hamcrest.CoreMatchers.*; +import static org.hamcrest.MatcherAssert.assertThat; + /** * Test cases for protocol-http - robustness regarding bad server responses: * malformed HTTP header lines, etc. See, NUTCH-2549. @@ -100,9 +99,10 @@ public void testIgnoreErrorInRedirectPayload() throws Exception { launchServer("HTTP/1.1 302 Found\r\nLocation: http://example.com/\r\n" + "Transfer-Encoding: chunked\r\n\r\nNot a valid chunk."); ProtocolOutput fetched = fetchPage("/", 302); - assertNotNull("No redirect Location.", getHeader(fetched, "Location")); - assertEquals("Wrong redirect Location.", "http://example.com/", - getHeader(fetched, "Location")); + assertThat("No redirect Location.", getHeader(fetched, "Location"), + notNullValue()); + assertThat("Wrong redirect Location.", getHeader(fetched, "Location"), + is("http://example.com/")); } /** @@ -113,8 +113,9 @@ public void testNoStatusLine() throws Exception { String text = "This is a text containing non-ASCII characters: \u00e4\u00f6\u00fc\u00df"; launchServer(text); ProtocolOutput fetched = fetchPage("/", 200); - assertEquals("Wrong text returned for response with no status line.", text, - new String(fetched.getContent().getContent(), StandardCharsets.UTF_8)); + assertThat("Wrong text returned for response with no status line.", + new String(fetched.getContent().getContent(), StandardCharsets.UTF_8), + is(text)); server.close(); text = "\n\n\n" + "Testing no HTTP header èéâ\n" @@ -123,8 +124,9 @@ public void testNoStatusLine() throws Exception { + "\u00e4\u00f6\u00fc\u00df\n getResponses(String headerValue) { @@ -126,7 +124,7 @@ public void testGetHeader() throws Exception { // testHeader(200, "MYCUSTOMHEADER", value, "MyCustomHeader"); } - @Ignore("Only for benchmarking") + @Disabled("Only for benchmarking") @Test public void testMetadataBenchmark() throws MalformedURLException, ProtocolException, IOException, InterruptedException { diff --git a/src/plugin/protocol-httpclient/src/test/org/apache/nutch/protocol/httpclient/TestProtocolHttpClient.java b/src/plugin/protocol-httpclient/src/test/org/apache/nutch/protocol/httpclient/TestProtocolHttpClient.java index f7277bd9ed..ddc0e24e43 100644 --- a/src/plugin/protocol-httpclient/src/test/org/apache/nutch/protocol/httpclient/TestProtocolHttpClient.java +++ b/src/plugin/protocol-httpclient/src/test/org/apache/nutch/protocol/httpclient/TestProtocolHttpClient.java @@ -30,7 +30,7 @@ import org.apache.nutch.protocol.AbstractHttpProtocolPluginTest; -import org.junit.Test; +import org.junit.jupiter.api.Test; import org.slf4j.Logger; import org.slf4j.LoggerFactory; diff --git a/src/plugin/protocol-okhttp/src/test/org/apache/nutch/protocol/okhttp/TestBadServerResponses.java b/src/plugin/protocol-okhttp/src/test/org/apache/nutch/protocol/okhttp/TestBadServerResponses.java index 7c5d0f15c7..bea9a1ba5e 100644 --- a/src/plugin/protocol-okhttp/src/test/org/apache/nutch/protocol/okhttp/TestBadServerResponses.java +++ b/src/plugin/protocol-okhttp/src/test/org/apache/nutch/protocol/okhttp/TestBadServerResponses.java @@ -16,9 +16,9 @@ */ package org.apache.nutch.protocol.okhttp; -import static org.junit.Assert.assertEquals; -import static org.junit.Assert.assertNotNull; -import static org.junit.Assert.assertTrue; +import static org.hamcrest.CoreMatchers.*; +import static org.hamcrest.MatcherAssert.*; +import static org.junit.jupiter.api.Assertions.assertTrue; import java.io.ByteArrayOutputStream; import java.lang.invoke.MethodHandles; @@ -28,8 +28,8 @@ import org.apache.nutch.net.protocols.Response; import org.apache.nutch.protocol.AbstractHttpProtocolPluginTest; import org.apache.nutch.protocol.ProtocolOutput; -import org.junit.Ignore; -import org.junit.Test; +import org.junit.jupiter.api.Disabled; +import org.junit.jupiter.api.Test; import org.slf4j.Logger; import org.slf4j.LoggerFactory; @@ -79,7 +79,7 @@ public void testContentLengthNotANumber() throws Exception { /** * NUTCH-2559 protocol-http cannot handle colons after the HTTP status code */ - @Ignore("Fails with okhttp 3.10.0") + @Disabled("Fails with okhttp 3.10.0") @Test public void testHeaderWithColon() throws Exception { launchServer("HTTP/1.1 200: OK\r\n" + simpleContent); @@ -100,28 +100,30 @@ public void testHeaderSpellChecking() throws Exception { * NUTCH-2557 protocol-http fails to follow redirections when an HTTP response * body is invalid */ - @Ignore("Fails with okhttp 3.10.0") + @Disabled("Fails with okhttp 3.10.0") @Test public void testIgnoreErrorInRedirectPayload() throws Exception { launchServer("HTTP/1.1 302 Found\r\nLocation: http://example.com/\r\n" + "Transfer-Encoding: chunked\r\n\r\nNot a valid chunk."); ProtocolOutput fetched = fetchPage("/", 302); - assertNotNull("No redirect Location.", getHeader(fetched, "Location")); - assertEquals("Wrong redirect Location.", "http://example.com/", - getHeader(fetched, "Location")); + assertThat("No redirect Location.", getHeader(fetched, "Location"), + notNullValue()); + assertThat("Wrong redirect Location.", getHeader(fetched, "Location"), + is("http://example.com/")); } /** * NUTCH-2558 protocol-http cannot handle a missing HTTP status line */ - @Ignore("Fails with okhttp 3.10.0") + @Disabled("Fails with okhttp 3.10.0") @Test public void testNoStatusLine() throws Exception { String text = "This is a text containing non-ASCII characters: \u00e4\u00f6\u00fc\u00df"; launchServer(text); ProtocolOutput fetched = fetchPage("/", 200); - assertEquals("Wrong text returned for response with no status line.", text, - new String(fetched.getContent().getContent(), StandardCharsets.UTF_8)); + assertThat("Wrong text returned for response with no status line.", + new String(fetched.getContent().getContent(), StandardCharsets.UTF_8), + is(text)); server.close(); text = "\n\n\n" + "Testing no HTTP header èéâ\n" @@ -130,15 +132,16 @@ public void testNoStatusLine() throws Exception { + "\u00e4\u00f6\u00fc\u00df\n 65536) { - assertNotNull("Content truncation not marked", - fetched.getContent().getMetadata().get(Response.TRUNCATED_CONTENT)); - assertEquals("Content truncation not marked", - Response.TruncatedContentReason.LENGTH.toString().toLowerCase(), - fetched.getContent().getMetadata().get(Response.TRUNCATED_CONTENT_REASON)); + assertThat("Content truncation not marked", + fetched.getContent().getMetadata().get(Response.TRUNCATED_CONTENT), + notNullValue()); + assertThat("Content truncation not marked", + fetched.getContent().getMetadata().get(Response.TRUNCATED_CONTENT_REASON), + is(Response.TruncatedContentReason.LENGTH.toString().toLowerCase())); } server.close(); // need to close server before next loop iteration } @@ -273,14 +278,16 @@ public void testTruncationMarkingGzip() throws Exception { launchServer("/", response.toByteArray()); ProtocolOutput fetched = fetchPage("/", 200); - assertEquals("Content not truncated according to http.content.limit", - Math.min(kB * 1024, 65536), fetched.getContent().getContent().length); + assertThat("Content not truncated according to http.content.limit", + fetched.getContent().getContent().length, + is(Math.min(kB * 1024, 65536))); if (kB * 1024 > 65536) { - assertNotNull("Content truncation not marked", - fetched.getContent().getMetadata().get(Response.TRUNCATED_CONTENT)); - assertEquals("Content truncation not marked", - Response.TruncatedContentReason.LENGTH.toString().toLowerCase(), - fetched.getContent().getMetadata().get(Response.TRUNCATED_CONTENT_REASON)); + assertThat("Content truncation not marked", + fetched.getContent().getMetadata().get(Response.TRUNCATED_CONTENT), + notNullValue()); + assertThat("Content truncation not marked", + fetched.getContent().getMetadata().get(Response.TRUNCATED_CONTENT_REASON), + is(Response.TruncatedContentReason.LENGTH.toString().toLowerCase())); } server.close(); // need to close server before next loop iteration } @@ -299,10 +306,12 @@ public void testPartialContentTruncated() throws Exception { launchServer( responseHeader + "Content-Length: 50000\r\n\r\n" + testContent); ProtocolOutput fetched = fetchPage("/", 200); - assertEquals("Content not saved as truncated", testContent, - new String(fetched.getContent().getContent(), StandardCharsets.UTF_8)); - assertNotNull("Content truncation not marked", - fetched.getContent().getMetadata().get(Response.TRUNCATED_CONTENT)); + assertThat("Content not saved as truncated", + new String(fetched.getContent().getContent(), StandardCharsets.UTF_8), + is(testContent)); + assertThat("Content truncation not marked", + fetched.getContent().getMetadata().get(Response.TRUNCATED_CONTENT), + notNullValue()); } @Test @@ -325,8 +334,8 @@ public void testNoContentLimit() throws Exception { } launchServer(response.toString()); ProtocolOutput fetched = fetchPage("/", 200); - assertEquals("Content truncated although http.content.limit == -1", - (kB * 1024), fetched.getContent().getContent().length); + assertThat("Content truncated although http.content.limit == -1", + fetched.getContent().getContent().length, is((kB * 1024))); } /** @@ -339,9 +348,8 @@ public void testHttpStatusNoMessage() throws Exception { String statusLineNoMessage = "HTTP/1.1 200 \r\n"; launchServer(statusLineNoMessage + simpleContent); ProtocolOutput fetched = fetchPage("/", 200); - assertTrue( - "Invalid HTTP status line (see NUTCH-2763, missing whitespace between status code and message)", - getHeaders(fetched).startsWith(statusLineNoMessage)); + assertTrue(getHeaders(fetched).startsWith(statusLineNoMessage), + "Invalid HTTP status line (see NUTCH-2763, missing whitespace between status code and message)"); } } diff --git a/src/plugin/protocol-okhttp/src/test/org/apache/nutch/protocol/okhttp/TestIPAddressFiltering.java b/src/plugin/protocol-okhttp/src/test/org/apache/nutch/protocol/okhttp/TestIPAddressFiltering.java index dbd1b846d7..850b088186 100644 --- a/src/plugin/protocol-okhttp/src/test/org/apache/nutch/protocol/okhttp/TestIPAddressFiltering.java +++ b/src/plugin/protocol-okhttp/src/test/org/apache/nutch/protocol/okhttp/TestIPAddressFiltering.java @@ -17,16 +17,14 @@ package org.apache.nutch.protocol.okhttp; import static java.nio.charset.StandardCharsets.UTF_8; -import static org.junit.Assert.assertFalse; -import static org.junit.Assert.assertTrue; -import static org.junit.Assert.fail; +import static org.junit.jupiter.api.Assertions.*; import java.net.InetAddress; import java.util.function.Function; import org.apache.hadoop.conf.Configuration; import org.apache.nutch.protocol.AbstractHttpProtocolPluginTest; -import org.junit.Test; +import org.junit.jupiter.api.Test; import com.google.common.net.InetAddresses; @@ -53,13 +51,13 @@ public InetAddress parseIP(String ip) { public void testCIDRcontains(String cidr, String ip) { CIDR c = new CIDR(cidr); InetAddress i = parseIP(ip); - assertTrue(i + " should be in " + c, c.contains(i)); + assertTrue(c.contains(i), i + " should be in " + c); } public void testCIDRnotContains(String cidr, String ip) { CIDR c = new CIDR(cidr); InetAddress i = parseIP(ip); - assertFalse(i + " should not be in " + c, c.contains(i)); + assertFalse(c.contains(i), i + " should not be in " + c); } /** Tests for {@link CIDR} */ @@ -93,12 +91,12 @@ public void testCIDRs() { public void testFilter(Configuration conf, String[] included, String[] excluded) { IPFilterRules ipFilterRules = new IPFilterRules(conf); for (String address : included) { - assertTrue("Address " + address + " should be included", - ipFilterRules.accept(parseIP(address))); + assertTrue(ipFilterRules.accept(parseIP(address)), + "Address " + address + " should be included"); } for (String address : excluded) { - assertFalse("Address " + address + " should be excluded", - ipFilterRules.accept(parseIP(address))); + assertFalse(ipFilterRules.accept(parseIP(address)), + "Address " + address + " should be excluded"); } } @@ -141,7 +139,7 @@ public void testPredefinedAddressRange(String ipAddress, String type) { default: fail("Unknown IP address type " + type); } - assertTrue(ipAddress + " is not recognized as " + type + " address", pred.apply(addr)); + assertTrue(pred.apply(addr), ipAddress + " is not recognized as " + type + " address"); } catch (IllegalArgumentException e) { fail("Not a valid IP address string: " + ipAddress); } diff --git a/src/plugin/protocol-okhttp/src/test/org/apache/nutch/protocol/okhttp/TestProtocolOkHttp.java b/src/plugin/protocol-okhttp/src/test/org/apache/nutch/protocol/okhttp/TestProtocolOkHttp.java index e740ed288f..a9b7267b3d 100644 --- a/src/plugin/protocol-okhttp/src/test/org/apache/nutch/protocol/okhttp/TestProtocolOkHttp.java +++ b/src/plugin/protocol-okhttp/src/test/org/apache/nutch/protocol/okhttp/TestProtocolOkHttp.java @@ -22,7 +22,7 @@ import java.util.TreeMap; import org.apache.nutch.protocol.AbstractHttpProtocolPluginTest; -import org.junit.Test; +import org.junit.jupiter.api.Test; /** * Test cases for protocol-okhttp diff --git a/src/plugin/protocol-okhttp/src/test/org/apache/nutch/protocol/okhttp/TestResponse.java b/src/plugin/protocol-okhttp/src/test/org/apache/nutch/protocol/okhttp/TestResponse.java index 5cff6ddbc2..f0944ff4c5 100644 --- a/src/plugin/protocol-okhttp/src/test/org/apache/nutch/protocol/okhttp/TestResponse.java +++ b/src/plugin/protocol-okhttp/src/test/org/apache/nutch/protocol/okhttp/TestResponse.java @@ -17,8 +17,7 @@ package org.apache.nutch.protocol.okhttp; import static java.nio.charset.StandardCharsets.UTF_8; - -import static org.junit.Assert.assertEquals; +import static org.junit.jupiter.api.Assertions.assertEquals; import java.io.IOException; import java.lang.invoke.MethodHandles; @@ -31,9 +30,9 @@ import org.apache.nutch.net.protocols.Response; import org.apache.nutch.protocol.AbstractHttpProtocolPluginTest; import org.apache.nutch.protocol.ProtocolException; -import org.junit.Before; -import org.junit.Ignore; -import org.junit.Test; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Disabled; +import org.junit.jupiter.api.Test; import org.slf4j.Logger; import org.slf4j.LoggerFactory; import org.apache.hadoop.conf.Configuration; @@ -53,7 +52,7 @@ protected String getPluginClassName() { } @Override - @Before + @BeforeEach public void setUp() throws Exception { conf = new Configuration(); conf.addResource("nutch-default.xml"); @@ -82,10 +81,9 @@ protected void headerTest(int statusCode, String headerName, String value, Strin OkHttpResponse response = getResponse(statusCode, headerName); LOG.info("Response headers:"); LOG.info(response.getHeaders().get(Response.RESPONSE_HEADERS)); - assertEquals( + assertEquals(value, response.getHeader(lookupName), "No or unexpected value of header \"" + headerName - + "\" returned when retrieving header \"" + lookupName + "\"", - value, response.getHeader(lookupName)); + + "\" returned when retrieving header \"" + lookupName + "\""); } protected Map getResponses(String headerValue) { @@ -127,7 +125,7 @@ public void testGetHeader() throws Exception { // testHeader(200, "MYCUSTOMHEADER", value, "MyCustomHeader"); } - @Ignore("Only for benchmarking") + @Disabled("Only for benchmarking") @Test public void testMetadataBenchmark() throws MalformedURLException, ProtocolException, IOException, InterruptedException { diff --git a/src/plugin/scoring-metadata/src/test/org/apache/nutch/scoring/metadata/TestMetadataScoringFilter.java b/src/plugin/scoring-metadata/src/test/org/apache/nutch/scoring/metadata/TestMetadataScoringFilter.java index 0112239586..e0ad7e68b0 100644 --- a/src/plugin/scoring-metadata/src/test/org/apache/nutch/scoring/metadata/TestMetadataScoringFilter.java +++ b/src/plugin/scoring-metadata/src/test/org/apache/nutch/scoring/metadata/TestMetadataScoringFilter.java @@ -23,11 +23,12 @@ import org.apache.nutch.protocol.Content; import org.apache.nutch.scoring.ScoringFilterException; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; import java.util.HashMap; +import static org.junit.jupiter.api.Assertions.assertEquals; + public class TestMetadataScoringFilter { @@ -60,8 +61,8 @@ public void distributeScoreToOutlinks() throws ScoringFilterException { Text parent = (Text) outlink.getMetaData().get(new Text(PARENT)); Text depth = (Text) outlink.getMetaData().get(new Text(DEPTH)); - Assert.assertEquals(parentMD,parent.toString()); - Assert.assertEquals(depthMD,depth.toString()); + assertEquals(parentMD,parent.toString()); + assertEquals(depthMD,depth.toString()); } } @@ -87,8 +88,8 @@ public void passScoreBeforeParsing() { metadataScoringFilter.passScoreBeforeParsing(from,crawlDatum,content); - Assert.assertEquals(parentMD,content.getMetadata().get(PARENT)); - Assert.assertEquals(depthMD,content.getMetadata().get(DEPTH)); + assertEquals(parentMD,content.getMetadata().get(PARENT)); + assertEquals(depthMD,content.getMetadata().get(DEPTH)); } @Test @@ -118,7 +119,7 @@ public void passScoreAfterParsing() { metadataScoringFilter.passScoreAfterParsing(from,content,parse); - Assert.assertEquals(parentMD,parse.getData().getMeta(PARENT)); - Assert.assertEquals(depthMD,parse.getData().getMeta(DEPTH)); + assertEquals(parentMD,parse.getData().getMeta(PARENT)); + assertEquals(depthMD,parse.getData().getMeta(DEPTH)); } } diff --git a/src/plugin/scoring-orphan/src/test/org/apache/nutch/scoring/orphan/TestOrphanScoringFilter.java b/src/plugin/scoring-orphan/src/test/org/apache/nutch/scoring/orphan/TestOrphanScoringFilter.java index f49a996078..568c6521d3 100644 --- a/src/plugin/scoring-orphan/src/test/org/apache/nutch/scoring/orphan/TestOrphanScoringFilter.java +++ b/src/plugin/scoring-orphan/src/test/org/apache/nutch/scoring/orphan/TestOrphanScoringFilter.java @@ -16,9 +16,6 @@ */ package org.apache.nutch.scoring.orphan; -import static org.junit.Assert.assertEquals; -import static org.junit.Assert.assertTrue; - import java.util.ArrayList; import java.util.List; @@ -28,7 +25,10 @@ import org.apache.nutch.crawl.CrawlDatum; import org.apache.nutch.scoring.ScoringFilter; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertTrue; public class TestOrphanScoringFilter { @@ -72,10 +72,9 @@ public void testOrphanScoringFilter() throws Exception { filter.updateDbScore(url, null, datum, emptyListOfInlinks); int thirdOrphanTime = getTime(datum); assertEquals(thirdOrphanTime, secondOrphanTime); - assertEquals( + assertEquals(CrawlDatum.STATUS_DB_NOTMODIFIED, datum.getStatus(), "Expected status db_notmodified but got " - + CrawlDatum.getStatusName(datum.getStatus()), - CrawlDatum.STATUS_DB_NOTMODIFIED, datum.getStatus()); + + CrawlDatum.getStatusName(datum.getStatus())); // Wait a little bit try { @@ -86,10 +85,9 @@ public void testOrphanScoringFilter() throws Exception { // Act as if no more inlinks, time will not increase, status is still the // same filter.updateDbScore(url, null, datum, emptyListOfInlinks); - assertEquals( + assertEquals(CrawlDatum.STATUS_DB_NOTMODIFIED, datum.getStatus(), "Expected status db_notmodified but got " - + CrawlDatum.getStatusName(datum.getStatus()), - CrawlDatum.STATUS_DB_NOTMODIFIED, datum.getStatus()); + + CrawlDatum.getStatusName(datum.getStatus())); // Wait until scoring.orphan.mark.gone.after try { @@ -101,10 +99,9 @@ public void testOrphanScoringFilter() throws Exception { filter.updateDbScore(url, null, datum, emptyListOfInlinks); int fourthOrphanTime = getTime(datum); assertEquals(fourthOrphanTime, thirdOrphanTime); - assertEquals( + assertEquals(CrawlDatum.STATUS_DB_GONE, datum.getStatus(), "Expected status db_gone but got " - + CrawlDatum.getStatusName(datum.getStatus()), - CrawlDatum.STATUS_DB_GONE, datum.getStatus()); + + CrawlDatum.getStatusName(datum.getStatus())); // Wait until scoring.orphan.mark.orphan.after try { @@ -114,10 +111,9 @@ public void testOrphanScoringFilter() throws Exception { // Again, but now markgoneafter has expired and record should be DB_ORPHAN filter.updateDbScore(url, null, datum, emptyListOfInlinks); - assertEquals( + assertEquals(CrawlDatum.STATUS_DB_ORPHAN, datum.getStatus(), "Expected status db_orphan but got " - + CrawlDatum.getStatusName(datum.getStatus()), - CrawlDatum.STATUS_DB_ORPHAN, datum.getStatus()); + + CrawlDatum.getStatusName(datum.getStatus())); } protected int getTime(CrawlDatum datum) { diff --git a/src/plugin/subcollection/src/test/org/apache/nutch/collection/TestSubcollection.java b/src/plugin/subcollection/src/test/org/apache/nutch/collection/TestSubcollection.java index a2d2772769..5b148034f8 100644 --- a/src/plugin/subcollection/src/test/org/apache/nutch/collection/TestSubcollection.java +++ b/src/plugin/subcollection/src/test/org/apache/nutch/collection/TestSubcollection.java @@ -21,8 +21,10 @@ import java.util.Collection; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertNull; public class TestSubcollection { @@ -38,16 +40,15 @@ public void testFilter() throws Exception { sc.setBlackList("jpg\nwww.apache.org/zecret/"); // matches whitelist - Assert.assertEquals("http://www.apache.org/index.html", + assertEquals("http://www.apache.org/index.html", sc.filter("http://www.apache.org/index.html")); // matches blacklist - Assert.assertEquals(null, - sc.filter("http://www.apache.org/zecret/index.html")); - Assert.assertEquals(null, sc.filter("http://www.apache.org/img/image.jpg")); + assertNull(sc.filter("http://www.apache.org/zecret/index.html")); + assertNull(sc.filter("http://www.apache.org/img/image.jpg")); // no match - Assert.assertEquals(null, sc.filter("http://www.google.com/")); + assertNull(sc.filter("http://www.google.com/")); } @Test @@ -77,36 +78,36 @@ public void testInput() { Collection c = cm.getAll(); // test that size matches - Assert.assertEquals(1, c.size()); + assertEquals(1, c.size()); Subcollection collection = (Subcollection) c.toArray()[0]; // test collection id - Assert.assertEquals("nutch", collection.getId()); + assertEquals("nutch", collection.getId()); // test collection name - Assert.assertEquals("nutch collection", collection.getName()); + assertEquals("nutch collection", collection.getName()); // test whitelist - Assert.assertEquals(2, collection.whiteList.size()); + assertEquals(2, collection.whiteList.size()); String wlUrl = (String) collection.whiteList.get(0); - Assert.assertEquals("http://lucene.apache.org/nutch/", wlUrl); + assertEquals("http://lucene.apache.org/nutch/", wlUrl); wlUrl = (String) collection.whiteList.get(1); - Assert.assertEquals("http://wiki.apache.org/nutch/", wlUrl); + assertEquals("http://wiki.apache.org/nutch/", wlUrl); // matches whitelist - Assert.assertEquals("http://lucene.apache.org/nutch/", + assertEquals("http://lucene.apache.org/nutch/", collection.filter("http://lucene.apache.org/nutch/")); // test blacklist - Assert.assertEquals(1, collection.blackList.size()); + assertEquals(1, collection.blackList.size()); String blUrl = (String) collection.blackList.get(0); - Assert.assertEquals("http://www.xxx.yyy", blUrl); + assertEquals("http://www.xxx.yyy", blUrl); // no match - Assert.assertEquals(null, collection.filter("http://www.google.com/")); + assertNull(collection.filter("http://www.google.com/")); } } diff --git a/src/plugin/urlfilter-automaton/src/test/org/apache/nutch/urlfilter/automaton/TestAutomatonURLFilter.java b/src/plugin/urlfilter-automaton/src/test/org/apache/nutch/urlfilter/automaton/TestAutomatonURLFilter.java index bf2d6b2362..143d65fdc1 100644 --- a/src/plugin/urlfilter-automaton/src/test/org/apache/nutch/urlfilter/automaton/TestAutomatonURLFilter.java +++ b/src/plugin/urlfilter-automaton/src/test/org/apache/nutch/urlfilter/automaton/TestAutomatonURLFilter.java @@ -23,8 +23,9 @@ import org.apache.nutch.net.*; // Nutch imports import org.apache.nutch.urlfilter.api.RegexURLFilterBaseTest; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.fail; /** * JUnit based test of class AutomatonURLFilter. @@ -38,7 +39,7 @@ protected URLFilter getURLFilter(Reader rules) { try { return new AutomatonURLFilter(rules); } catch (IOException e) { - Assert.fail(e.toString()); + fail(e.toString()); return null; } } diff --git a/src/plugin/urlfilter-domain/src/test/org/apache/nutch/urlfilter/domain/TestDomainURLFilter.java b/src/plugin/urlfilter-domain/src/test/org/apache/nutch/urlfilter/domain/TestDomainURLFilter.java index 7878aa198e..aa829bfbad 100644 --- a/src/plugin/urlfilter-domain/src/test/org/apache/nutch/urlfilter/domain/TestDomainURLFilter.java +++ b/src/plugin/urlfilter-domain/src/test/org/apache/nutch/urlfilter/domain/TestDomainURLFilter.java @@ -18,8 +18,10 @@ import org.apache.hadoop.conf.Configuration; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertNotNull; +import static org.junit.jupiter.api.Assertions.assertNull; public class TestDomainURLFilter { @@ -34,16 +36,16 @@ public void testFilter() throws Exception { conf.set("urlfilter.domain.file", domainFile); DomainURLFilter domainFilter = new DomainURLFilter(); domainFilter.setConf(conf); - Assert.assertNotNull(domainFilter.filter("http://lucene.apache.org")); - Assert.assertNotNull(domainFilter.filter("http://hadoop.apache.org")); - Assert.assertNotNull(domainFilter.filter("http://www.apache.org")); - Assert.assertNull(domainFilter.filter("http://www.google.com")); - Assert.assertNull(domainFilter.filter("http://mail.yahoo.com")); - Assert.assertNotNull(domainFilter.filter("http://www.foobar.net")); - Assert.assertNotNull(domainFilter.filter("http://www.foobas.net")); - Assert.assertNotNull(domainFilter.filter("http://www.yahoo.com")); - Assert.assertNotNull(domainFilter.filter("http://www.foobar.be")); - Assert.assertNull(domainFilter.filter("http://www.adobe.com")); + assertNotNull(domainFilter.filter("http://lucene.apache.org")); + assertNotNull(domainFilter.filter("http://hadoop.apache.org")); + assertNotNull(domainFilter.filter("http://www.apache.org")); + assertNull(domainFilter.filter("http://www.google.com")); + assertNull(domainFilter.filter("http://mail.yahoo.com")); + assertNotNull(domainFilter.filter("http://www.foobar.net")); + assertNotNull(domainFilter.filter("http://www.foobas.net")); + assertNotNull(domainFilter.filter("http://www.yahoo.com")); + assertNotNull(domainFilter.filter("http://www.foobar.be")); + assertNull(domainFilter.filter("http://www.adobe.com")); } @Test @@ -54,16 +56,16 @@ public void testNoFilter() throws Exception { conf.set("urlfilter.domain.file", domainFile); DomainURLFilter domainFilter = new DomainURLFilter(); domainFilter.setConf(conf); - Assert.assertNotNull(domainFilter.filter("http://lucene.apache.org")); - Assert.assertNotNull(domainFilter.filter("http://hadoop.apache.org")); - Assert.assertNotNull(domainFilter.filter("http://www.apache.org")); - Assert.assertNotNull(domainFilter.filter("http://www.google.com")); - Assert.assertNotNull(domainFilter.filter("http://mail.yahoo.com")); - Assert.assertNotNull(domainFilter.filter("http://www.foobar.net")); - Assert.assertNotNull(domainFilter.filter("http://www.foobas.net")); - Assert.assertNotNull(domainFilter.filter("http://www.yahoo.com")); - Assert.assertNotNull(domainFilter.filter("http://www.foobar.be")); - Assert.assertNotNull(domainFilter.filter("http://www.adobe.com")); + assertNotNull(domainFilter.filter("http://lucene.apache.org")); + assertNotNull(domainFilter.filter("http://hadoop.apache.org")); + assertNotNull(domainFilter.filter("http://www.apache.org")); + assertNotNull(domainFilter.filter("http://www.google.com")); + assertNotNull(domainFilter.filter("http://mail.yahoo.com")); + assertNotNull(domainFilter.filter("http://www.foobar.net")); + assertNotNull(domainFilter.filter("http://www.foobas.net")); + assertNotNull(domainFilter.filter("http://www.yahoo.com")); + assertNotNull(domainFilter.filter("http://www.foobar.be")); + assertNotNull(domainFilter.filter("http://www.adobe.com")); } } diff --git a/src/plugin/urlfilter-domaindenylist/src/test/org/apache/nutch/urlfilter/domaindenylist/TestDomainDenylistURLFilter.java b/src/plugin/urlfilter-domaindenylist/src/test/org/apache/nutch/urlfilter/domaindenylist/TestDomainDenylistURLFilter.java index 0dde234377..f60315715a 100644 --- a/src/plugin/urlfilter-domaindenylist/src/test/org/apache/nutch/urlfilter/domaindenylist/TestDomainDenylistURLFilter.java +++ b/src/plugin/urlfilter-domaindenylist/src/test/org/apache/nutch/urlfilter/domaindenylist/TestDomainDenylistURLFilter.java @@ -16,11 +16,13 @@ */ package org.apache.nutch.urlfilter.domaindenylist; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; import org.apache.hadoop.conf.Configuration; import org.apache.nutch.util.NutchConfiguration; +import static org.junit.jupiter.api.Assertions.assertNotNull; +import static org.junit.jupiter.api.Assertions.assertNull; + public class TestDomainDenylistURLFilter { private final static String SEPARATOR = System.getProperty("file.separator"); @@ -34,16 +36,16 @@ public void testFilter() throws Exception { conf.set("urlfilter.domaindenylist.file", domainDenylistFile); DomainDenylistURLFilter domainDenylistFilter = new DomainDenylistURLFilter(); domainDenylistFilter.setConf(conf); - Assert.assertNull(domainDenylistFilter.filter("http://lucene.apache.org")); - Assert.assertNull(domainDenylistFilter.filter("http://hadoop.apache.org")); - Assert.assertNull(domainDenylistFilter.filter("http://www.apache.org")); - Assert.assertNotNull(domainDenylistFilter.filter("http://www.google.com")); - Assert.assertNotNull(domainDenylistFilter.filter("http://mail.yahoo.com")); - Assert.assertNull(domainDenylistFilter.filter("http://www.foobar.net")); - Assert.assertNull(domainDenylistFilter.filter("http://www.foobas.net")); - Assert.assertNull(domainDenylistFilter.filter("http://www.yahoo.com")); - Assert.assertNull(domainDenylistFilter.filter("http://www.foobar.be")); - Assert.assertNotNull(domainDenylistFilter.filter("http://www.adobe.com")); + assertNull(domainDenylistFilter.filter("http://lucene.apache.org")); + assertNull(domainDenylistFilter.filter("http://hadoop.apache.org")); + assertNull(domainDenylistFilter.filter("http://www.apache.org")); + assertNotNull(domainDenylistFilter.filter("http://www.google.com")); + assertNotNull(domainDenylistFilter.filter("http://mail.yahoo.com")); + assertNull(domainDenylistFilter.filter("http://www.foobar.net")); + assertNull(domainDenylistFilter.filter("http://www.foobas.net")); + assertNull(domainDenylistFilter.filter("http://www.yahoo.com")); + assertNull(domainDenylistFilter.filter("http://www.foobar.be")); + assertNotNull(domainDenylistFilter.filter("http://www.adobe.com")); } } diff --git a/src/plugin/urlfilter-fast/src/test/org/apache/nutch/urlfilter/fast/TestFastURLFilter.java b/src/plugin/urlfilter-fast/src/test/org/apache/nutch/urlfilter/fast/TestFastURLFilter.java index 2c9ceea6f4..82b13330ec 100644 --- a/src/plugin/urlfilter-fast/src/test/org/apache/nutch/urlfilter/fast/TestFastURLFilter.java +++ b/src/plugin/urlfilter-fast/src/test/org/apache/nutch/urlfilter/fast/TestFastURLFilter.java @@ -23,8 +23,9 @@ import org.apache.hadoop.conf.Configuration; import org.apache.nutch.net.URLFilter; import org.apache.nutch.urlfilter.api.RegexURLFilterBaseTest; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.*; public class TestFastURLFilter extends RegexURLFilterBaseTest { @@ -33,7 +34,7 @@ protected URLFilter getURLFilter(Reader rules) { try { return new FastURLFilter(rules); } catch (IOException e) { - Assert.fail(e.toString()); + fail(e.toString()); return null; } } @@ -65,14 +66,14 @@ public void lengthQueryAndPath() throws Exception { for (int i = 0; i < 50; i++) { url.append(i); } - Assert.assertEquals(null, filter.filter(url.toString())); + assertNull(filter.filter(url.toString())); url = new StringBuilder("http://nutch.apache.org/path?"); for (int i = 0; i < 50; i++) { url.append(i); } - Assert.assertEquals(null, filter.filter(url.toString())); + assertNull(filter.filter(url.toString())); } @Test @@ -86,6 +87,6 @@ public void overalLengthTest() throws Exception { for (int i = 0; i < 500; i++) { url.append(i); } - Assert.assertEquals(null, filter.filter(url.toString())); + assertNull(filter.filter(url.toString())); } } diff --git a/src/plugin/urlfilter-prefix/src/test/org/apache/nutch/urlfilter/prefix/TestPrefixURLFilter.java b/src/plugin/urlfilter-prefix/src/test/org/apache/nutch/urlfilter/prefix/TestPrefixURLFilter.java index 0079fa7fd1..ebf86d9411 100644 --- a/src/plugin/urlfilter-prefix/src/test/org/apache/nutch/urlfilter/prefix/TestPrefixURLFilter.java +++ b/src/plugin/urlfilter-prefix/src/test/org/apache/nutch/urlfilter/prefix/TestPrefixURLFilter.java @@ -16,13 +16,12 @@ */ package org.apache.nutch.urlfilter.prefix; -import junit.framework.Test; -import junit.framework.TestCase; -import junit.framework.TestSuite; -import junit.textui.TestRunner; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; import java.io.IOException; +import static org.junit.jupiter.api.Assertions.assertSame; /** * JUnit test for PrefixURLFilter. @@ -30,7 +29,7 @@ * @author Talat Uyarer * @author Cihad Guzel */ -public class TestPrefixURLFilter extends TestCase { +public class TestPrefixURLFilter { private static final String prefixes = "# this is a comment\n" + "\n" + @@ -59,22 +58,15 @@ public class TestPrefixURLFilter extends TestCase { private PrefixURLFilter filter = null; - public static Test suite() { - return new TestSuite(TestPrefixURLFilter.class); - } - - public static void main(String[] args) { - TestRunner.run(suite()); - } - - @Override + @BeforeEach public void setUp() throws IOException { filter = new PrefixURLFilter(prefixes); } + @Test public void testModeAccept() { for (int i = 0; i < urls.length; i++) { - assertTrue(urlsModeAccept[i] == filter.filter(urls[i])); + assertSame(urlsModeAccept[i], filter.filter(urls[i])); } } } diff --git a/src/plugin/urlfilter-regex/src/test/org/apache/nutch/urlfilter/regex/TestRegexURLFilter.java b/src/plugin/urlfilter-regex/src/test/org/apache/nutch/urlfilter/regex/TestRegexURLFilter.java index bd0503eb4c..cc79c003f5 100644 --- a/src/plugin/urlfilter-regex/src/test/org/apache/nutch/urlfilter/regex/TestRegexURLFilter.java +++ b/src/plugin/urlfilter-regex/src/test/org/apache/nutch/urlfilter/regex/TestRegexURLFilter.java @@ -23,8 +23,9 @@ import org.apache.nutch.net.*; // Nutch imports import org.apache.nutch.urlfilter.api.RegexURLFilterBaseTest; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.fail; /** * JUnit based test of class RegexURLFilter. @@ -38,7 +39,7 @@ protected URLFilter getURLFilter(Reader rules) { try { return new RegexURLFilter(rules); } catch (IOException e) { - Assert.fail(e.toString()); + fail(e.toString()); return null; } } diff --git a/src/plugin/urlfilter-suffix/src/test/org/apache/nutch/urlfilter/suffix/TestSuffixURLFilter.java b/src/plugin/urlfilter-suffix/src/test/org/apache/nutch/urlfilter/suffix/TestSuffixURLFilter.java index eecb2b2d7e..bb9e62ae02 100644 --- a/src/plugin/urlfilter-suffix/src/test/org/apache/nutch/urlfilter/suffix/TestSuffixURLFilter.java +++ b/src/plugin/urlfilter-suffix/src/test/org/apache/nutch/urlfilter/suffix/TestSuffixURLFilter.java @@ -19,9 +19,11 @@ import java.io.IOException; import java.io.StringReader; -import org.junit.Assert; -import org.junit.Before; -import org.junit.Test; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertSame; +import static org.junit.jupiter.api.Assertions.assertTrue; /** * JUnit test for SuffixURLFilter. @@ -59,7 +61,7 @@ public class TestSuffixURLFilter { private SuffixURLFilter filter = null; - @Before + @BeforeEach public void setUp() throws IOException { filter = new SuffixURLFilter(new StringReader(suffixes)); } @@ -69,7 +71,7 @@ public void testModeAccept() { filter.setIgnoreCase(false); filter.setModeAccept(true); for (int i = 0; i < urls.length; i++) { - Assert.assertTrue(urlsModeAccept[i] == filter.filter(urls[i])); + assertSame(urlsModeAccept[i], filter.filter(urls[i])); } } @@ -78,7 +80,7 @@ public void testModeReject() { filter.setIgnoreCase(false); filter.setModeAccept(false); for (int i = 0; i < urls.length; i++) { - Assert.assertTrue(urlsModeReject[i] == filter.filter(urls[i])); + assertSame(urlsModeReject[i], filter.filter(urls[i])); } } @@ -87,7 +89,7 @@ public void testModeAcceptIgnoreCase() { filter.setIgnoreCase(true); filter.setModeAccept(true); for (int i = 0; i < urls.length; i++) { - Assert.assertTrue(urlsModeAcceptIgnoreCase[i] == filter.filter(urls[i])); + assertSame(urlsModeAcceptIgnoreCase[i], filter.filter(urls[i])); } } @@ -96,7 +98,7 @@ public void testModeRejectIgnoreCase() { filter.setIgnoreCase(true); filter.setModeAccept(false); for (int i = 0; i < urls.length; i++) { - Assert.assertTrue(urlsModeRejectIgnoreCase[i] == filter.filter(urls[i])); + assertSame(urlsModeRejectIgnoreCase[i], filter.filter(urls[i])); } } @@ -105,8 +107,7 @@ public void testModeAcceptAndNonPathFilter() { filter.setModeAccept(true); filter.setFilterFromPath(false); for (int i = 0; i < urls.length; i++) { - Assert.assertTrue(urlsModeAcceptAndNonPathFilter[i] == filter - .filter(urls[i])); + assertSame(urlsModeAcceptAndNonPathFilter[i], filter.filter(urls[i])); } } @@ -115,8 +116,7 @@ public void testModeAcceptAndPathFilter() { filter.setModeAccept(true); filter.setFilterFromPath(true); for (int i = 0; i < urls.length; i++) { - Assert.assertTrue(urlsModeAcceptAndPathFilter[i] == filter - .filter(urls[i])); + assertSame(urlsModeAcceptAndPathFilter[i], filter.filter(urls[i])); } } diff --git a/src/plugin/urlfilter-validator/src/test/org/apache/nutch/urlfilter/validator/TestUrlValidator.java b/src/plugin/urlfilter-validator/src/test/org/apache/nutch/urlfilter/validator/TestUrlValidator.java index a52285bded..d815486de6 100644 --- a/src/plugin/urlfilter-validator/src/test/org/apache/nutch/urlfilter/validator/TestUrlValidator.java +++ b/src/plugin/urlfilter-validator/src/test/org/apache/nutch/urlfilter/validator/TestUrlValidator.java @@ -16,8 +16,10 @@ */ package org.apache.nutch.urlfilter.validator; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertNotNull; +import static org.junit.jupiter.api.Assertions.assertNull; /** * JUnit test case which tests 1. that valid urls are not filtered while invalid @@ -38,41 +40,37 @@ public class TestUrlValidator { @Test public void testFilter() { UrlValidator url_validator = new UrlValidator(); - Assert.assertNotNull(url_validator); + assertNotNull(url_validator); - Assert.assertNull("Filtering on a null object should return null", - url_validator.filter(null)); - Assert.assertNull("Invalid url: example.com/file[/].html", - url_validator.filter("example.com/file[/].html")); - Assert.assertNull("Invalid url: http://www.example.com/space here.html", - url_validator.filter("http://www.example.com/space here.html")); - Assert.assertNull("Invalid url: /main.html", - url_validator.filter("/main.html")); - Assert.assertNull("Invalid url: www.example.com/main.html", - url_validator.filter("www.example.com/main.html")); - Assert.assertNull("Invalid url: ftp:www.example.com/main.html", - url_validator.filter("ftp:www.example.com/main.html")); - Assert.assertNull( - "Inalid url: http://999.000.456.32/nutch/trunk/README.txt", - url_validator.filter("http://999.000.456.32/nutch/trunk/README.txt")); - Assert.assertNull("Invalid url: http://www.example.com/ma|in\\toc.html", - url_validator.filter(" http://www.example.com/ma|in\\toc.html")); + assertNull(url_validator.filter(null), + "Filtering on a null object should return null"); + assertNull(url_validator.filter("example.com/file[/].html"), + "Invalid url: example.com/file[/].html"); + assertNull(url_validator.filter("http://www.example.com/space here.html"), + "Invalid url: http://www.example.com/space here.html"); + assertNull(url_validator.filter("/main.html"), + "Invalid url: /main.html"); + assertNull(url_validator.filter("www.example.com/main.html"), + "Invalid url: www.example.com/main.html"); + assertNull(url_validator.filter("ftp:www.example.com/main.html"), + "Invalid url: ftp:www.example.com/main.html"); + assertNull( + url_validator.filter("http://999.000.456.32/nutch/trunk/README.txt"), + "Inalid url: http://999.000.456.32/nutch/trunk/README.txt"); + assertNull(url_validator.filter(" http://www.example.com/ma|in\\toc.html"), + "Invalid url: http://www.example.com/ma|in\\toc.html"); - Assert.assertNotNull( - "Valid url: https://issues.apache.org/jira/NUTCH-1127", - url_validator.filter("https://issues.apache.org/jira/NUTCH-1127")); - Assert - .assertNotNull( - "Valid url: http://domain.tld/function.cgi?url=http://fonzi.com/&name=Fonzi&mood=happy&coat=leather", - url_validator - .filter("http://domain.tld/function.cgi?url=http://fonzi.com/&name=Fonzi&mood=happy&coat=leather")); - Assert - .assertNotNull( - "Valid url: http://validator.w3.org/feed/check.cgi?url=http%3A%2F%2Ffeeds.feedburner.com%2Fperishablepress", - url_validator - .filter("http://validator.w3.org/feed/check.cgi?url=http%3A%2F%2Ffeeds.feedburner.com%2Fperishablepress")); - Assert.assertNotNull("Valid url: ftp://alfa.bravo.pi/foo/bar/plan.pdf", - url_validator.filter("ftp://alfa.bravo.pi/mike/check/plan.pdf")); + assertNotNull( + url_validator.filter("https://issues.apache.org/jira/NUTCH-1127"), + "Valid url: https://issues.apache.org/jira/NUTCH-1127"); + assertNotNull(url_validator + .filter("http://domain.tld/function.cgi?url=http://fonzi.com/&name=Fonzi&mood=happy&coat=leather"), + "Valid url: http://domain.tld/function.cgi?url=http://fonzi.com/&name=Fonzi&mood=happy&coat=leather"); + assertNotNull(url_validator + .filter("http://validator.w3.org/feed/check.cgi?url=http%3A%2F%2Ffeeds.feedburner.com%2Fperishablepress"), + "Valid url: http://validator.w3.org/feed/check.cgi?url=http%3A%2F%2Ffeeds.feedburner.com%2Fperishablepress"); + assertNotNull(url_validator.filter("ftp://alfa.bravo.pi/mike/check/plan.pdf"), + "Valid url: ftp://alfa.bravo.pi/foo/bar/plan.pdf"); } } diff --git a/src/plugin/urlnormalizer-ajax/src/test/org/apache/nutch/net/urlnormalizer/ajax/TestAjaxURLNormalizer.java b/src/plugin/urlnormalizer-ajax/src/test/org/apache/nutch/net/urlnormalizer/ajax/TestAjaxURLNormalizer.java index 5a13879707..310761e374 100644 --- a/src/plugin/urlnormalizer-ajax/src/test/org/apache/nutch/net/urlnormalizer/ajax/TestAjaxURLNormalizer.java +++ b/src/plugin/urlnormalizer-ajax/src/test/org/apache/nutch/net/urlnormalizer/ajax/TestAjaxURLNormalizer.java @@ -19,21 +19,22 @@ import org.apache.hadoop.conf.Configuration; import org.apache.nutch.net.URLNormalizers; import org.apache.nutch.util.NutchConfiguration; +import org.junit.jupiter.api.Test; -import junit.framework.TestCase; +import static org.junit.jupiter.api.Assertions.assertEquals; /** Unit tests for AjaxURLNormalizer. */ -public class TestAjaxURLNormalizer extends TestCase { +public class TestAjaxURLNormalizer { private AjaxURLNormalizer normalizer; private Configuration conf; - public TestAjaxURLNormalizer(String name) { - super(name); + public TestAjaxURLNormalizer() { normalizer = new AjaxURLNormalizer(); conf = NutchConfiguration.create(); normalizer.setConf(conf); } + @Test public void testNormalizer() throws Exception { // check if AJAX URL's are normalized to an _escaped_frament_ form normalizeTest("http://example.org/#!k=v", "http://example.org/?_escaped_fragment_=k=v"); @@ -44,7 +45,8 @@ public void testNormalizer() throws Exception { // Check with query string and multiple fragment params normalizeTest("http://example.org/path.html?queryparam=queryvalue#!key1=value1&key2=value2", "http://example.org/path.html?queryparam=queryvalue&_escaped_fragment_=key1=value1%26key2=value2"); } - + + @Test public void testNormalizerWhenIndexing() throws Exception { // check if it works the other way around normalizeTest("http://example.org/?_escaped_fragment_=key=value", "http://example.org/#!key=value", URLNormalizers.SCOPE_INDEXER); @@ -53,7 +55,8 @@ public void testNormalizerWhenIndexing() throws Exception { } private void normalizeTest(String weird, String normal) throws Exception { - assertEquals(normal, normalizer.normalize(weird, URLNormalizers.SCOPE_DEFAULT)); + assertEquals(normal, + normalizer.normalize(weird, URLNormalizers.SCOPE_DEFAULT)); } private void normalizeTest(String weird, String normal, String scope) throws Exception { @@ -61,6 +64,6 @@ private void normalizeTest(String weird, String normal, String scope) throws Exc } public static void main(String[] args) throws Exception { - new TestAjaxURLNormalizer("test").testNormalizer(); + new TestAjaxURLNormalizer().testNormalizer(); } } diff --git a/src/plugin/urlnormalizer-basic/src/test/org/apache/nutch/net/urlnormalizer/basic/TestBasicURLNormalizer.java b/src/plugin/urlnormalizer-basic/src/test/org/apache/nutch/net/urlnormalizer/basic/TestBasicURLNormalizer.java index d48e585101..a6bad41f2e 100644 --- a/src/plugin/urlnormalizer-basic/src/test/org/apache/nutch/net/urlnormalizer/basic/TestBasicURLNormalizer.java +++ b/src/plugin/urlnormalizer-basic/src/test/org/apache/nutch/net/urlnormalizer/basic/TestBasicURLNormalizer.java @@ -23,8 +23,10 @@ import org.apache.hadoop.conf.Configuration; import org.apache.nutch.net.URLNormalizers; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.fail; /** Unit tests for BasicURLNormalizer. */ public class TestBasicURLNormalizer { @@ -295,12 +297,12 @@ private void normalizeTest(String weird, String normal) throws Exception { private void normalizeTest(BasicURLNormalizer normalizer, String weird, String normal) throws Exception { - Assert.assertEquals("normalizing: " + weird, normal, - normalizer.normalize(weird, URLNormalizers.SCOPE_DEFAULT)); + assertEquals(normal, normalizer.normalize(weird, URLNormalizers.SCOPE_DEFAULT), + "normalizing: " + weird); try { (new URL(normal)).toURI(); } catch (MalformedURLException | URISyntaxException e) { - Assert.fail("Output of normalization fails to validate as URL or URI: " + fail("Output of normalization fails to validate as URL or URI: " + e.getMessage()); } } @@ -318,7 +320,7 @@ private void normalizeTestAssertThrowsMalformedURLException( // ok, expected return; } - Assert.fail("Expected MalformedURLException was not thrown on " + weird + fail("Expected MalformedURLException was not thrown on " + weird + " (normalized: " + normalized + ")"); } diff --git a/src/plugin/urlnormalizer-host/src/test/org/apache/nutch/net/urlnormalizer/host/TestHostURLNormalizer.java b/src/plugin/urlnormalizer-host/src/test/org/apache/nutch/net/urlnormalizer/host/TestHostURLNormalizer.java index 68cb50ab51..03df349d76 100644 --- a/src/plugin/urlnormalizer-host/src/test/org/apache/nutch/net/urlnormalizer/host/TestHostURLNormalizer.java +++ b/src/plugin/urlnormalizer-host/src/test/org/apache/nutch/net/urlnormalizer/host/TestHostURLNormalizer.java @@ -19,8 +19,9 @@ import org.apache.hadoop.conf.Configuration; import org.apache.nutch.net.URLNormalizers; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertEquals; public class TestHostURLNormalizer { @@ -37,22 +38,22 @@ public void testHostURLNormalizer() throws Exception { normalizer.setConf(conf); // Force www. sub domain when hitting link without sub domain - Assert.assertEquals("http://www.example.org/page.html", + assertEquals("http://www.example.org/page.html", normalizer.normalize("http://example.org/page.html", URLNormalizers.SCOPE_DEFAULT)); // Force no sub domain to www. URL's - Assert.assertEquals("http://example.net/path/to/something.html", normalizer + assertEquals("http://example.net/path/to/something.html", normalizer .normalize("http://www.example.net/path/to/something.html", URLNormalizers.SCOPE_DEFAULT)); // Force all sub domains to www. - Assert.assertEquals("http://example.com/?does=it&still=work", normalizer + assertEquals("http://example.com/?does=it&still=work", normalizer .normalize("http://example.com/?does=it&still=work", URLNormalizers.SCOPE_DEFAULT)); - Assert.assertEquals("http://example.com/buh", normalizer.normalize( + assertEquals("http://example.com/buh", normalizer.normalize( "http://http.www.example.com/buh", URLNormalizers.SCOPE_DEFAULT)); - Assert.assertEquals("http://example.com/blaat", normalizer.normalize( + assertEquals("http://example.com/blaat", normalizer.normalize( "http://whatever.example.com/blaat", URLNormalizers.SCOPE_DEFAULT)); } } diff --git a/src/plugin/urlnormalizer-pass/src/test/org/apache/nutch/net/urlnormalizer/pass/TestPassURLNormalizer.java b/src/plugin/urlnormalizer-pass/src/test/org/apache/nutch/net/urlnormalizer/pass/TestPassURLNormalizer.java index f470c627e4..83367a14e6 100644 --- a/src/plugin/urlnormalizer-pass/src/test/org/apache/nutch/net/urlnormalizer/pass/TestPassURLNormalizer.java +++ b/src/plugin/urlnormalizer-pass/src/test/org/apache/nutch/net/urlnormalizer/pass/TestPassURLNormalizer.java @@ -21,8 +21,10 @@ import org.apache.hadoop.conf.Configuration; import org.apache.nutch.net.URLNormalizers; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.fail; public class TestPassURLNormalizer { @@ -37,9 +39,9 @@ public void testPassURLNormalizer() { try { result = normalizer.normalize(url, URLNormalizers.SCOPE_DEFAULT); } catch (MalformedURLException mue) { - Assert.fail(mue.toString()); + fail(mue.toString()); } - Assert.assertEquals(url, result); + assertEquals(url, result); } } diff --git a/src/plugin/urlnormalizer-protocol/src/test/org/apache/nutch/net/urlnormalizer/protocol/TestProtocolURLNormalizer.java b/src/plugin/urlnormalizer-protocol/src/test/org/apache/nutch/net/urlnormalizer/protocol/TestProtocolURLNormalizer.java index 977525073a..91e1fb70b9 100644 --- a/src/plugin/urlnormalizer-protocol/src/test/org/apache/nutch/net/urlnormalizer/protocol/TestProtocolURLNormalizer.java +++ b/src/plugin/urlnormalizer-protocol/src/test/org/apache/nutch/net/urlnormalizer/protocol/TestProtocolURLNormalizer.java @@ -20,14 +20,17 @@ import org.apache.nutch.net.URLNormalizers; import org.apache.nutch.util.NutchConfiguration; -import junit.framework.TestCase; +import org.junit.jupiter.api.Test; -public class TestProtocolURLNormalizer extends TestCase { +import static org.junit.jupiter.api.Assertions.assertEquals; + +class TestProtocolURLNormalizer { private final static String SEPARATOR = System.getProperty("file.separator"); private final static String SAMPLES = System.getProperty("test.data", "."); - public void testProtocolURLNormalizer() throws Exception { + @Test + void testProtocolURLNormalizer() throws Exception { Configuration conf = NutchConfiguration.create(); String protocolsFile = SAMPLES + SEPARATOR + "protocols.txt"; diff --git a/src/plugin/urlnormalizer-querystring/src/test/org/apache/nutch/net/urlnormalizer/querystring/TestQuerystringURLNormalizer.java b/src/plugin/urlnormalizer-querystring/src/test/org/apache/nutch/net/urlnormalizer/querystring/TestQuerystringURLNormalizer.java index e9a02cd2cc..0029f5c71f 100644 --- a/src/plugin/urlnormalizer-querystring/src/test/org/apache/nutch/net/urlnormalizer/querystring/TestQuerystringURLNormalizer.java +++ b/src/plugin/urlnormalizer-querystring/src/test/org/apache/nutch/net/urlnormalizer/querystring/TestQuerystringURLNormalizer.java @@ -20,11 +20,14 @@ import org.apache.nutch.net.URLNormalizers; import org.apache.nutch.util.NutchConfiguration; -import junit.framework.TestCase; +import org.junit.jupiter.api.Test; -public class TestQuerystringURLNormalizer extends TestCase { +import static org.junit.jupiter.api.Assertions.assertEquals; - public void testQuerystringURLNormalizer() throws Exception { +class TestQuerystringURLNormalizer { + + @Test + void testQuerystringURLNormalizer() throws Exception { Configuration conf = NutchConfiguration.create(); QuerystringURLNormalizer normalizer = new QuerystringURLNormalizer(); diff --git a/src/plugin/urlnormalizer-regex/src/test/org/apache/nutch/net/urlnormalizer/regex/TestRegexURLNormalizer.java b/src/plugin/urlnormalizer-regex/src/test/org/apache/nutch/net/urlnormalizer/regex/TestRegexURLNormalizer.java index 250ad2527c..ed4b99abfe 100644 --- a/src/plugin/urlnormalizer-regex/src/test/org/apache/nutch/net/urlnormalizer/regex/TestRegexURLNormalizer.java +++ b/src/plugin/urlnormalizer-regex/src/test/org/apache/nutch/net/urlnormalizer/regex/TestRegexURLNormalizer.java @@ -27,8 +27,7 @@ import java.util.*; import java.util.concurrent.TimeUnit; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; import org.slf4j.Logger; import org.slf4j.LoggerFactory; import org.apache.commons.lang3.time.StopWatch; @@ -36,6 +35,9 @@ import org.apache.nutch.net.URLNormalizers; import org.apache.nutch.util.NutchConfiguration; +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.fail; + /** Unit tests for RegexUrlNormalizer. */ public class TestRegexURLNormalizer { private static final Logger LOG = LoggerFactory @@ -101,7 +103,7 @@ private void normalizeTest(NormalizedURL[] urls, String scope) String expected = urls[i].expectedURL; LOG.info("scope: {} url: {} | normalized: {} | expected: {}", scope, url, normalized, expected); - Assert.assertEquals(urls[i].expectedURL, normalized); + assertEquals(urls[i].expectedURL, normalized); } } @@ -116,7 +118,7 @@ private void bench(int loops, String scope) { normalizeTest(expected, scope); } } catch (Exception e) { - Assert.fail(e.toString()); + fail(e.toString()); } stopWatch.stop(); LOG.info("bench time ({}) {}ms", loops, diff --git a/src/plugin/urlnormalizer-slash/src/test/org/apache/nutch/net/urlnormalizer/slash/TestSlashURLNormalizer.java b/src/plugin/urlnormalizer-slash/src/test/org/apache/nutch/net/urlnormalizer/slash/TestSlashURLNormalizer.java index 54af2bf12c..6e2e5d115d 100644 --- a/src/plugin/urlnormalizer-slash/src/test/org/apache/nutch/net/urlnormalizer/slash/TestSlashURLNormalizer.java +++ b/src/plugin/urlnormalizer-slash/src/test/org/apache/nutch/net/urlnormalizer/slash/TestSlashURLNormalizer.java @@ -19,15 +19,17 @@ import org.apache.hadoop.conf.Configuration; import org.apache.nutch.net.URLNormalizers; import org.apache.nutch.util.NutchConfiguration; +import org.junit.jupiter.api.Test; -import junit.framework.TestCase; +import static org.junit.jupiter.api.Assertions.assertEquals; -public class TestSlashURLNormalizer extends TestCase { +class TestSlashURLNormalizer { private final static String SEPARATOR = System.getProperty("file.separator"); private final static String SAMPLES = System.getProperty("test.data", "."); - public void testSlashURLNormalizer() throws Exception { + @Test + void testSlashURLNormalizer() throws Exception { Configuration conf = NutchConfiguration.create(); String slashesFile = SAMPLES + SEPARATOR + "slashes.txt"; @@ -36,37 +38,53 @@ public void testSlashURLNormalizer() throws Exception { normalizer.setConf(conf); // No change - assertEquals("http://example.org/", normalizer.normalize("http://example.org/", URLNormalizers.SCOPE_DEFAULT)); - assertEquals("http://example.net/", normalizer.normalize("http://example.net/", URLNormalizers.SCOPE_DEFAULT)); + assertEquals("http://example.org/", + normalizer.normalize("http://example.org/", URLNormalizers.SCOPE_DEFAULT)); + assertEquals("http://example.net/", + normalizer.normalize("http://example.net/", URLNormalizers.SCOPE_DEFAULT)); // Don't touch base URL's - assertEquals("http://example.org", normalizer.normalize("http://example.org", URLNormalizers.SCOPE_DEFAULT)); - assertEquals("http://example.net", normalizer.normalize("http://example.net", URLNormalizers.SCOPE_DEFAULT)); - assertEquals("http://example.org/", normalizer.normalize("http://example.org/", URLNormalizers.SCOPE_DEFAULT)); - assertEquals("http://example.net/", normalizer.normalize("http://example.net/", URLNormalizers.SCOPE_DEFAULT)); + assertEquals("http://example.org", + normalizer.normalize("http://example.org", URLNormalizers.SCOPE_DEFAULT)); + assertEquals("http://example.net", + normalizer.normalize("http://example.net", URLNormalizers.SCOPE_DEFAULT)); + assertEquals("http://example.org/", + normalizer.normalize("http://example.org/", URLNormalizers.SCOPE_DEFAULT)); + assertEquals("http://example.net/", + normalizer.normalize("http://example.net/", URLNormalizers.SCOPE_DEFAULT)); // Change - assertEquals("http://www.example.org/page/", normalizer.normalize("http://www.example.org/page", URLNormalizers.SCOPE_DEFAULT)); + assertEquals("http://www.example.org/page/", + normalizer.normalize("http://www.example.org/page", URLNormalizers.SCOPE_DEFAULT)); assertEquals("http://www.example.net/path/to/something", normalizer.normalize("http://www.example.net/path/to/something/", URLNormalizers.SCOPE_DEFAULT)); // No change - assertEquals("http://example.org/buh/", normalizer.normalize("http://example.org/buh/", URLNormalizers.SCOPE_DEFAULT)); - assertEquals("http://example.net/blaat", normalizer.normalize("http://example.net/blaat", URLNormalizers.SCOPE_DEFAULT)); + assertEquals("http://example.org/buh/", + normalizer.normalize("http://example.org/buh/", URLNormalizers.SCOPE_DEFAULT)); + assertEquals("http://example.net/blaat", + normalizer.normalize("http://example.net/blaat", URLNormalizers.SCOPE_DEFAULT)); // No change - assertEquals("http://example.nl/buh/", normalizer.normalize("http://example.nl/buh/", URLNormalizers.SCOPE_DEFAULT)); - assertEquals("http://example.de/blaat", normalizer.normalize("http://example.de/blaat", URLNormalizers.SCOPE_DEFAULT)); + assertEquals("http://example.nl/buh/", + normalizer.normalize("http://example.nl/buh/", URLNormalizers.SCOPE_DEFAULT)); + assertEquals("http://example.de/blaat", + normalizer.normalize("http://example.de/blaat", URLNormalizers.SCOPE_DEFAULT)); // Change assertEquals("http://www.example.org/page/?a=b&c=d", normalizer.normalize("http://www.example.org/page?a=b&c=d", URLNormalizers.SCOPE_DEFAULT)); - assertEquals("http://www.example.net/path/to/something?a=b&c=d", normalizer.normalize("http://www.example.net/path/to/something/?a=b&c=d", URLNormalizers.SCOPE_DEFAULT)); + assertEquals("http://www.example.net/path/to/something?a=b&c=d", + normalizer.normalize("http://www.example.net/path/to/something/?a=b&c=d", URLNormalizers.SCOPE_DEFAULT)); // No change - assertEquals("http://www.example.org/noise.mp3", normalizer.normalize("http://www.example.org/noise.mp3", URLNormalizers.SCOPE_DEFAULT)); - assertEquals("http://www.example.org/page.html", normalizer.normalize("http://www.example.org/page.html", URLNormalizers.SCOPE_DEFAULT)); - assertEquals("http://www.example.org/page.shtml", normalizer.normalize("http://www.example.org/page.shtml", URLNormalizers.SCOPE_DEFAULT)); + assertEquals("http://www.example.org/noise.mp3", + normalizer.normalize("http://www.example.org/noise.mp3", URLNormalizers.SCOPE_DEFAULT)); + assertEquals("http://www.example.org/page.html", + normalizer.normalize("http://www.example.org/page.html", URLNormalizers.SCOPE_DEFAULT)); + assertEquals("http://www.example.org/page.shtml", + normalizer.normalize("http://www.example.org/page.shtml", URLNormalizers.SCOPE_DEFAULT)); // Change - assertEquals("http://www.example.org/this.is.not.an_extension/", normalizer.normalize("http://www.example.org/this.is.not.an_extension", URLNormalizers.SCOPE_DEFAULT)); + assertEquals("http://www.example.org/this.is.not.an_extension/", + normalizer.normalize("http://www.example.org/this.is.not.an_extension", URLNormalizers.SCOPE_DEFAULT)); } } diff --git a/src/test/org/apache/nutch/crawl/ContinuousCrawlTestUtil.java b/src/test/org/apache/nutch/crawl/ContinuousCrawlTestUtil.java index c3dab2cee8..f0971c2e16 100644 --- a/src/test/org/apache/nutch/crawl/ContinuousCrawlTestUtil.java +++ b/src/test/org/apache/nutch/crawl/ContinuousCrawlTestUtil.java @@ -16,15 +16,6 @@ */ package org.apache.nutch.crawl; -import java.io.IOException; -import java.lang.invoke.MethodHandles; -import java.util.ArrayList; -import java.util.Arrays; -import java.util.Date; -import java.util.List; - -import junit.framework.TestCase; - import org.apache.hadoop.conf.Configuration; import org.apache.hadoop.io.Text; import org.apache.hadoop.mapreduce.Reducer; @@ -33,11 +24,20 @@ import org.slf4j.Logger; import org.slf4j.LoggerFactory; +import java.io.IOException; +import java.lang.invoke.MethodHandles; +import java.util.ArrayList; +import java.util.Arrays; +import java.util.Date; +import java.util.List; + +import static org.junit.jupiter.api.Assertions.*; + /** * Emulate a continuous crawl for one URL. * */ -public class ContinuousCrawlTestUtil extends TestCase { +public class ContinuousCrawlTestUtil { private static final Logger LOG = LoggerFactory .getLogger(MethodHandles.lookup().lookupClass()); @@ -239,9 +239,9 @@ protected boolean run(int maxErrors) throws IOException { values.add(fetchDatum); values.addAll(parse(fetchDatum)); List res = updateDb.update(values); - assertNotNull("null returned", res); - assertFalse("no CrawlDatum", 0 == res.size()); - assertEquals("more than one CrawlDatum", 1, res.size()); + assertNotNull(res, "null returned"); + assertNotEquals(0, res.size(), "no CrawlDatum"); + assertEquals(1, res.size(), "more than one CrawlDatum"); if (!check(res.get(0))) { LOG.info("previously in CrawlDb: {}", copyDbDatum); LOG.info("after shouldFetch(): {}", afterShouldFetch); diff --git a/src/test/org/apache/nutch/crawl/TODOTestCrawlDbStates.java b/src/test/org/apache/nutch/crawl/TODOTestCrawlDbStates.java index 09863656d9..dfad393512 100644 --- a/src/test/org/apache/nutch/crawl/TODOTestCrawlDbStates.java +++ b/src/test/org/apache/nutch/crawl/TODOTestCrawlDbStates.java @@ -16,22 +16,19 @@ */ package org.apache.nutch.crawl; -import java.io.IOException; -import java.lang.invoke.MethodHandles; - -import static org.apache.nutch.crawl.CrawlDatum.*; - import org.apache.hadoop.conf.Configuration; -import org.apache.nutch.util.TimingUtil; - import org.apache.hadoop.mapreduce.Reducer.Context; - -import static org.junit.Assert.*; - -import org.junit.Test; +import org.apache.nutch.util.TimingUtil; +import org.junit.jupiter.api.Test; import org.slf4j.Logger; import org.slf4j.LoggerFactory; +import java.io.IOException; +import java.lang.invoke.MethodHandles; + +import static org.apache.nutch.crawl.CrawlDatum.*; +import static org.junit.jupiter.api.Assertions.fail; + public class TODOTestCrawlDbStates extends TestCrawlDbStates { private static final Logger LOG = LoggerFactory diff --git a/src/test/org/apache/nutch/crawl/TestAdaptiveFetchSchedule.java b/src/test/org/apache/nutch/crawl/TestAdaptiveFetchSchedule.java index f83c8d9fbe..377d49ec81 100644 --- a/src/test/org/apache/nutch/crawl/TestAdaptiveFetchSchedule.java +++ b/src/test/org/apache/nutch/crawl/TestAdaptiveFetchSchedule.java @@ -16,19 +16,20 @@ */ package org.apache.nutch.crawl; -import junit.framework.TestCase; - import org.apache.hadoop.conf.Configuration; import org.apache.hadoop.io.Text; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Before; -import org.junit.Test; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; + +import static org.hamcrest.CoreMatchers.is; +import static org.hamcrest.MatcherAssert.assertThat; /** * Test cases for AdaptiveFetchSchedule. * */ -public class TestAdaptiveFetchSchedule extends TestCase { +public class TestAdaptiveFetchSchedule { private float inc_rate; private float dec_rate; @@ -36,10 +37,8 @@ public class TestAdaptiveFetchSchedule extends TestCase { private long curTime, lastModified; private int changed, interval, calculateInterval; - @Override - @Before + @BeforeEach public void setUp() throws Exception { - super.setUp(); conf = NutchConfiguration.create(); inc_rate = conf.getFloat("db.fetch.schedule.adaptive.inc_rate", 0.2f); dec_rate = conf.getFloat("db.fetch.schedule.adaptive.dec_rate", 0.2f); @@ -106,15 +105,15 @@ public CrawlDatum prepareCrawlDatum() { private void validateFetchInterval(int changed, int getInterval) { if (changed == FetchSchedule.STATUS_UNKNOWN) { - assertEquals(getInterval, interval); + assertThat(interval, is(getInterval)); } else if (changed == FetchSchedule.STATUS_MODIFIED) { calculateInterval = (int) (interval - (interval * dec_rate)); - assertEquals(getInterval, calculateInterval); + assertThat(calculateInterval, is(getInterval)); } else if (changed == FetchSchedule.STATUS_NOTMODIFIED) { calculateInterval = (int) (interval + (interval * inc_rate)); - assertEquals(getInterval, calculateInterval); + assertThat(calculateInterval, is(getInterval)); } } diff --git a/src/test/org/apache/nutch/crawl/TestCrawlDbDeduplication.java b/src/test/org/apache/nutch/crawl/TestCrawlDbDeduplication.java index 25850a9365..646749db5e 100644 --- a/src/test/org/apache/nutch/crawl/TestCrawlDbDeduplication.java +++ b/src/test/org/apache/nutch/crawl/TestCrawlDbDeduplication.java @@ -16,10 +16,6 @@ */ package org.apache.nutch.crawl; -import java.io.File; -import java.io.IOException; -import java.lang.invoke.MethodHandles; - import org.apache.commons.io.FileUtils; import org.apache.hadoop.conf.Configuration; import org.apache.hadoop.fs.FileStatus; @@ -28,13 +24,18 @@ import org.apache.hadoop.io.Text; import org.apache.hadoop.util.ToolRunner; import org.apache.nutch.util.NutchConfiguration; -import org.junit.After; -import org.junit.Assert; -import org.junit.Before; -import org.junit.Test; +import org.junit.jupiter.api.AfterEach; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; import org.slf4j.Logger; import org.slf4j.LoggerFactory; +import java.io.File; +import java.io.IOException; +import java.lang.invoke.MethodHandles; + +import static org.junit.jupiter.api.Assertions.*; + public class TestCrawlDbDeduplication { private static final Logger LOG = LoggerFactory .getLogger(MethodHandles.lookup().lookupClass()); @@ -45,7 +46,7 @@ public class TestCrawlDbDeduplication { Path testCrawlDb; CrawlDbReader reader; - @Before + @BeforeEach public void setUp() throws Exception { conf = NutchConfiguration.create(); fs = FileSystem.get(conf); @@ -62,7 +63,7 @@ public void setUp() throws Exception { reader = new CrawlDbReader(); } - @After + @AfterEach public void tearDown() { try { if (fs.exists(testCrawlDb)) @@ -82,7 +83,7 @@ public void testDeduplication() throws Exception { args[1] = "-compareOrder"; args[2] = "fetchTime,urlLength,score"; int result = ToolRunner.run(conf, new DeduplicationJob(), args); - Assert.assertEquals("DeduplicationJob did not succeed", 0, result); + assertEquals(0, result, "DeduplicationJob did not succeed"); String url1 = "http://nutch.apache.org/"; String url2 = "https://nutch.apache.org/"; // url1 has been fetched earlier, so it should "survive" as "db_fetched": @@ -97,7 +98,7 @@ public void testDeduplicationHttpsOverHttp() throws Exception { args[1] = "-compareOrder"; args[2] = "httpsOverHttp,fetchTime,urlLength,score"; int result = ToolRunner.run(conf, new DeduplicationJob(), args); - Assert.assertEquals("DeduplicationJob did not succeed", 0, result); + assertEquals(0, result, "DeduplicationJob did not succeed"); String url1 = "http://nutch.apache.org/"; String url2 = "https://nutch.apache.org/"; // url2 is https://, so it should "survive" as "db_fetched": @@ -107,10 +108,10 @@ public void testDeduplicationHttpsOverHttp() throws Exception { private void checkStatus(String url, byte status) throws IOException { CrawlDatum datum = reader.get(testCrawlDb.toString(), url, conf); - Assert.assertNotNull("No CrawlDatum found in CrawlDb for " + url, datum); - Assert.assertEquals( - "Expected status for " + url + ": " + CrawlDatum.getStatusName(status), - status, datum.getStatus()); + assertNotNull(datum, "No CrawlDatum found in CrawlDb for " + url); + assertEquals( + status, datum.getStatus(), + "Expected status for " + url + ": " + CrawlDatum.getStatusName(status)); } static class TestDedupReducer extends DeduplicationJob.DedupReducer { @@ -142,22 +143,22 @@ public String getDuplicateURL(String compareOrder, String url1, String url2) { public void testCompareURLs() { // test same protocol, same length: no decision possible String url0 = "https://example.com/"; - Assert.assertNull(getDuplicateURL("httpsOverHttp,urlLength", url0, url0)); + assertNull(getDuplicateURL("httpsOverHttp,urlLength", url0, url0)); String url1 = "http://nutch.apache.org/"; String url2 = "https://nutch.apache.org/"; // test httpsOverHttp - Assert.assertEquals(url1, getDuplicateURL("httpsOverHttp", url1, url2)); + assertEquals(url1, getDuplicateURL("httpsOverHttp", url1, url2)); // test urlLength - Assert.assertEquals(url2, getDuplicateURL("urlLength", url1, url2)); + assertEquals(url2, getDuplicateURL("urlLength", url1, url2)); // test urlLength with percent-encoded URLs // "b%C3%BCcher" (unescaped "bücher") is shorter than "buecher" String url3 = "https://example.com/b%C3%BCcher"; String url4 = "https://example.com/buecher"; - Assert.assertEquals(url4, getDuplicateURL("urlLength", url3, url4)); + assertEquals(url4, getDuplicateURL("urlLength", url3, url4)); // test NUTCH-2935: should not throw error on invalid percent-encoding String url5 = "https://example.com/%YR"; String url6 = "https://example.com/%YR%YR"; - Assert.assertEquals(url6, getDuplicateURL("urlLength", url5, url6)); + assertEquals(url6, getDuplicateURL("urlLength", url5, url6)); } } diff --git a/src/test/org/apache/nutch/crawl/TestCrawlDbFilter.java b/src/test/org/apache/nutch/crawl/TestCrawlDbFilter.java index 812d4a6a8f..d18b3139f8 100644 --- a/src/test/org/apache/nutch/crawl/TestCrawlDbFilter.java +++ b/src/test/org/apache/nutch/crawl/TestCrawlDbFilter.java @@ -16,9 +16,6 @@ */ package org.apache.nutch.crawl; -import java.io.IOException; -import java.util.ArrayList; - import org.apache.hadoop.conf.Configuration; import org.apache.hadoop.fs.FileSystem; import org.apache.hadoop.fs.Path; @@ -26,15 +23,19 @@ import org.apache.hadoop.io.SequenceFile.Reader.Option; import org.apache.hadoop.io.Text; import org.apache.hadoop.mapreduce.Job; -import org.apache.hadoop.mapreduce.lib.input.SequenceFileInputFormat; -import org.apache.hadoop.mapreduce.lib.output.MapFileOutputFormat; import org.apache.hadoop.mapreduce.lib.input.FileInputFormat; +import org.apache.hadoop.mapreduce.lib.input.SequenceFileInputFormat; import org.apache.hadoop.mapreduce.lib.output.FileOutputFormat; +import org.apache.hadoop.mapreduce.lib.output.MapFileOutputFormat; import org.apache.nutch.crawl.CrawlDBTestUtil.URLCrawlDatum; -import org.junit.After; -import org.junit.Assert; -import org.junit.Before; -import org.junit.Test; +import org.junit.jupiter.api.AfterEach; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; + +import java.io.IOException; +import java.util.ArrayList; + +import static org.junit.jupiter.api.Assertions.assertEquals; /** * CrawlDbFiltering test which tests for correct, error free url normalization @@ -50,14 +51,14 @@ public class TestCrawlDbFilter { final static Path testdir = new Path("build/test/crawldbfilter-test"); FileSystem fs; - @Before + @BeforeEach public void setUp() throws Exception { conf = CrawlDBTestUtil.createContext().getConfiguration(); fs = FileSystem.get(conf); fs.delete(testdir, true); } - @After + @AfterEach public void tearDown() { delete(testdir); } @@ -114,7 +115,7 @@ public void testUrl404Purging() throws Exception { ArrayList l = readContents(fetchlist); // verify we got right amount of records - Assert.assertEquals(2, l.size()); + assertEquals(2, l.size()); } /** diff --git a/src/test/org/apache/nutch/crawl/TestCrawlDbMerger.java b/src/test/org/apache/nutch/crawl/TestCrawlDbMerger.java index 7fb83f601b..48892e49ea 100644 --- a/src/test/org/apache/nutch/crawl/TestCrawlDbMerger.java +++ b/src/test/org/apache/nutch/crawl/TestCrawlDbMerger.java @@ -16,25 +16,27 @@ */ package org.apache.nutch.crawl; -import java.lang.invoke.MethodHandles; -import java.util.HashMap; -import java.util.Iterator; -import java.util.TreeSet; - -import org.slf4j.Logger; -import org.slf4j.LoggerFactory; import org.apache.hadoop.conf.Configuration; import org.apache.hadoop.fs.FileSystem; import org.apache.hadoop.fs.Path; import org.apache.hadoop.io.MapFile; +import org.apache.hadoop.io.MapFile.Writer.Option; import org.apache.hadoop.io.SequenceFile; import org.apache.hadoop.io.Text; -import org.apache.hadoop.io.MapFile.Writer.Option; import org.apache.nutch.util.NutchConfiguration; -import org.junit.After; -import org.junit.Assert; -import org.junit.Before; -import org.junit.Test; +import org.junit.jupiter.api.AfterEach; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; +import org.slf4j.Logger; +import org.slf4j.LoggerFactory; + +import java.lang.invoke.MethodHandles; +import java.util.HashMap; +import java.util.Iterator; +import java.util.TreeSet; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertNotNull; public class TestCrawlDbMerger { private static final Logger LOG = LoggerFactory @@ -55,7 +57,7 @@ public class TestCrawlDbMerger { Path testDir; CrawlDbReader reader; - @Before + @BeforeEach public void setUp() throws Exception { init1.add(url10); init1.add(url11); @@ -85,7 +87,7 @@ public void setUp() throws Exception { fs.mkdirs(testDir); } - @After + @AfterEach public void tearDown() { try { if (fs.exists(testDir)) @@ -133,8 +135,8 @@ public void testMerge() throws Exception { System.out.println(" cd " + cd); System.out.println(" res " + res); // may not be null - Assert.assertNotNull(res); - Assert.assertTrue(cd.equals(res)); + assertNotNull(res); + assertEquals(cd, res); } reader.close(); fs.delete(testDir, true); diff --git a/src/test/org/apache/nutch/crawl/TestCrawlDbStates.java b/src/test/org/apache/nutch/crawl/TestCrawlDbStates.java index c6aac88f96..bae7b29fd8 100644 --- a/src/test/org/apache/nutch/crawl/TestCrawlDbStates.java +++ b/src/test/org/apache/nutch/crawl/TestCrawlDbStates.java @@ -16,6 +16,17 @@ */ package org.apache.nutch.crawl; +import org.apache.hadoop.conf.Configuration; +import org.apache.hadoop.io.Text; +import org.apache.hadoop.mapreduce.Reducer; +import org.apache.hadoop.mapreduce.Reducer.Context; +import org.apache.hadoop.util.StringUtils; +import org.apache.nutch.scoring.ScoringFilterException; +import org.apache.nutch.scoring.ScoringFilters; +import org.junit.jupiter.api.Test; +import org.slf4j.Logger; +import org.slf4j.LoggerFactory; + import java.io.IOException; import java.lang.invoke.MethodHandles; import java.util.ArrayList; @@ -23,24 +34,8 @@ import java.util.Iterator; import java.util.List; -import org.apache.hadoop.conf.Configuration; -import org.apache.hadoop.util.StringUtils; - import static org.apache.nutch.crawl.CrawlDatum.*; - -import org.apache.nutch.scoring.ScoringFilterException; -import org.apache.nutch.scoring.ScoringFilters; - -import org.apache.hadoop.mapreduce.Reducer; -import org.apache.hadoop.mapreduce.Reducer.Context; -import org.apache.hadoop.io.Text; - -import static org.junit.Assert.*; - -import org.junit.Test; - -import org.slf4j.Logger; -import org.slf4j.LoggerFactory; +import static org.junit.jupiter.api.Assertions.fail; /** * Test transitions of {@link CrawlDatum} states during an update of diff --git a/src/test/org/apache/nutch/crawl/TestGenerator.java b/src/test/org/apache/nutch/crawl/TestGenerator.java index 1003e40c54..cf4fa49558 100644 --- a/src/test/org/apache/nutch/crawl/TestGenerator.java +++ b/src/test/org/apache/nutch/crawl/TestGenerator.java @@ -16,22 +16,24 @@ */ package org.apache.nutch.crawl; -import java.io.IOException; -import java.util.ArrayList; -import java.util.Collections; -import java.util.Comparator; - import org.apache.hadoop.conf.Configuration; import org.apache.hadoop.fs.FileSystem; import org.apache.hadoop.fs.Path; import org.apache.hadoop.io.SequenceFile; -import org.apache.hadoop.io.Text; import org.apache.hadoop.io.SequenceFile.Reader.Option; +import org.apache.hadoop.io.Text; import org.apache.nutch.crawl.CrawlDBTestUtil.URLCrawlDatum; -import org.junit.After; -import org.junit.Assert; -import org.junit.Before; -import org.junit.Test; +import org.junit.jupiter.api.AfterEach; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; + +import java.io.IOException; +import java.util.ArrayList; +import java.util.Collections; +import java.util.Comparator; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertNull; /** * Basic generator test. 1. Insert entries in crawldb 2. Generates entries to @@ -51,14 +53,14 @@ public class TestGenerator { final static Path testdir = new Path("build/test/generator-test"); - @Before + @BeforeEach public void setUp() throws Exception { conf = CrawlDBTestUtil.createContext().getConfiguration(); fs = FileSystem.get(conf); fs.delete(testdir, true); } - @After + @AfterEach public void tearDown() { delete(testdir); } @@ -99,11 +101,11 @@ public void testGenerateHighest() throws Exception { Collections.sort(l, new ScoreComparator()); // verify we got right amount of records - Assert.assertEquals(NUM_RESULTS, l.size()); + assertEquals(NUM_RESULTS, l.size()); // verify we have the highest scoring urls - Assert.assertEquals("http://aaa/100", (l.get(0).url.toString())); - Assert.assertEquals("http://aaa/099", (l.get(1).url.toString())); + assertEquals("http://aaa/100", (l.get(0).url.toString())); + assertEquals("http://aaa/099", (l.get(1).url.toString())); } private String pad(int i) { @@ -159,8 +161,8 @@ public void testGenerateHostLimit() throws Exception { // verify we got right amount of records int expectedFetchListSize = Math.min(maxPerHost, list.size()); - Assert.assertEquals("Failed to apply generate.max.count by host", - expectedFetchListSize, fetchList.size()); + assertEquals(expectedFetchListSize, fetchList.size(), + "Failed to apply generate.max.count by host"); maxPerHost = 2; myConfiguration = new Configuration(conf); @@ -175,8 +177,8 @@ public void testGenerateHostLimit() throws Exception { // verify we got right amount of records expectedFetchListSize = Math.min(maxPerHost, list.size()); - Assert.assertEquals("Failed to apply generate.max.count by host", - expectedFetchListSize, fetchList.size()); + assertEquals(expectedFetchListSize, fetchList.size(), + "Failed to apply generate.max.count by host"); maxPerHost = 3; myConfiguration = new Configuration(conf); @@ -191,8 +193,8 @@ public void testGenerateHostLimit() throws Exception { // verify we got right amount of records expectedFetchListSize = Math.min(maxPerHost, list.size()); - Assert.assertEquals("Failed to apply generate.max.count by host", - expectedFetchListSize, fetchList.size()); + assertEquals(expectedFetchListSize, fetchList.size(), + "Failed to apply generate.max.count by host"); } /** @@ -227,8 +229,8 @@ public void testGenerateDomainLimit() throws Exception { // verify we got right amount of records int expectedFetchListSize = Math.min(maxPerDomain, list.size()); - Assert.assertEquals("Failed to apply generate.max.count by domain", - expectedFetchListSize, fetchList.size()); + assertEquals(expectedFetchListSize, fetchList.size(), + "Failed to apply generate.max.count by domain"); maxPerDomain = 2; myConfiguration = new Configuration(myConfiguration); @@ -243,8 +245,8 @@ public void testGenerateDomainLimit() throws Exception { // verify we got right amount of records expectedFetchListSize = Math.min(maxPerDomain, list.size()); - Assert.assertEquals("Failed to apply generate.max.count by domain", - expectedFetchListSize, fetchList.size()); + assertEquals(expectedFetchListSize, fetchList.size(), + "Failed to apply generate.max.count by domain"); maxPerDomain = 3; myConfiguration = new Configuration(myConfiguration); @@ -259,8 +261,8 @@ public void testGenerateDomainLimit() throws Exception { // verify we got right amount of records expectedFetchListSize = Math.min(maxPerDomain, list.size()); - Assert.assertEquals("Failed to apply generate.max.count by domain", - expectedFetchListSize, fetchList.size()); + assertEquals(expectedFetchListSize, fetchList.size(), + "Failed to apply generate.max.count by domain"); } /** @@ -286,7 +288,7 @@ public void testFilter() throws IOException, Exception { Path generatedSegment = generateFetchlist(Integer.MAX_VALUE, myConfiguration, true); - Assert.assertNull("should be null (0 entries)", generatedSegment); + assertNull(generatedSegment, "should be null (0 entries)"); generatedSegment = generateFetchlist(Integer.MAX_VALUE, myConfiguration, false); @@ -297,7 +299,7 @@ public void testFilter() throws IOException, Exception { ArrayList fetchList = readContents(fetchlistPath); // verify nothing got filtered - Assert.assertEquals(list.size(), fetchList.size()); + assertEquals(list.size(), fetchList.size()); } @@ -333,8 +335,8 @@ public void testURLNoHost() throws IOException, Exception { ArrayList fetchList = readContents(fetchlistPath); - Assert.assertEquals("Size of fetch list does not fit", - numValidURLs, fetchList.size()); + assertEquals(numValidURLs, fetchList.size(), + "Size of fetch list does not fit"); myConfiguration.set(Generator.GENERATOR_COUNT_MODE, Generator.GENERATOR_COUNT_VALUE_DOMAIN); @@ -347,8 +349,8 @@ public void testURLNoHost() throws IOException, Exception { fetchList = readContents(fetchlistPath); - Assert.assertEquals("Size of fetch list does not fit", - numValidURLs, fetchList.size()); + assertEquals(numValidURLs, fetchList.size(), + "Size of fetch list does not fit"); } /** diff --git a/src/test/org/apache/nutch/crawl/TestInjector.java b/src/test/org/apache/nutch/crawl/TestInjector.java index c6db4ba77a..74ccc44d0d 100644 --- a/src/test/org/apache/nutch/crawl/TestInjector.java +++ b/src/test/org/apache/nutch/crawl/TestInjector.java @@ -16,23 +16,21 @@ */ package org.apache.nutch.crawl; -import java.io.IOException; -import java.util.ArrayList; -import java.util.Collections; -import java.util.HashMap; -import java.util.List; -import java.util.Map; - import org.apache.hadoop.conf.Configuration; import org.apache.hadoop.fs.FileSystem; import org.apache.hadoop.fs.Path; import org.apache.hadoop.io.SequenceFile; -import org.apache.hadoop.io.Text; import org.apache.hadoop.io.SequenceFile.Reader.Option; -import org.junit.After; -import org.junit.Assert; -import org.junit.Before; -import org.junit.Test; +import org.apache.hadoop.io.Text; +import org.junit.jupiter.api.AfterEach; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; + +import java.io.IOException; +import java.util.*; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertTrue; /** * Basic injector test: 1. Creates a text file with urls 2. Injects them into @@ -48,7 +46,7 @@ public class TestInjector { Path crawldbPath; Path urlPath; - @Before + @BeforeEach public void setUp() throws Exception { conf = CrawlDBTestUtil.createContext().getConfiguration(); urlPath = new Path(testdir, "urls"); @@ -60,7 +58,7 @@ public void setUp() throws Exception { fs.delete(crawldbPath, true); } - @After + @AfterEach public void tearDown() throws IOException { fs.delete(testdir, true); } @@ -88,10 +86,10 @@ public void testInject() Collections.sort(read); Collections.sort(urls); - Assert.assertEquals(urls.size(), read.size()); + assertEquals(urls.size(), read.size()); - Assert.assertTrue(read.containsAll(urls)); - Assert.assertTrue(urls.containsAll(read)); + assertTrue(read.containsAll(urls)); + assertTrue(urls.containsAll(read)); // inject more urls ArrayList urls2 = new ArrayList(); @@ -114,10 +112,10 @@ public void testInject() Collections.sort(urls); // We should have 100 less records because we've overwritten - Assert.assertEquals(urls.size() - 100, read.size()); + assertEquals(urls.size() - 100, read.size()); - Assert.assertTrue(read.containsAll(urls)); - Assert.assertTrue(urls.containsAll(read)); + assertTrue(read.containsAll(urls)); + assertTrue(urls.containsAll(read)); // Check if we correctly preserved MD Map records = readCrawldbRecords(); @@ -129,11 +127,11 @@ public void testInject() for (String url : urls) { if (url.indexOf("http://zzz") == 0) { // Check for fetch interval - Assert.assertTrue(records.get(url).getFetchInterval() == 171717); + assertEquals(171717, records.get(url).getFetchInterval()); // Check for default score - Assert.assertTrue(records.get(url).getScore() != 1.0); + assertTrue(records.get(url).getScore() != 1.0); // Check for MD key=value - Assert.assertEquals(writableValue, + assertEquals(writableValue, records.get(url).getMetaData().get(writableKey)); } } diff --git a/src/test/org/apache/nutch/crawl/TestLinkDbMerger.java b/src/test/org/apache/nutch/crawl/TestLinkDbMerger.java index 8e634c5b9c..837090909d 100644 --- a/src/test/org/apache/nutch/crawl/TestLinkDbMerger.java +++ b/src/test/org/apache/nutch/crawl/TestLinkDbMerger.java @@ -16,27 +16,29 @@ */ package org.apache.nutch.crawl; -import java.lang.invoke.MethodHandles; -import java.util.ArrayList; -import java.util.HashMap; -import java.util.Iterator; -import java.util.TreeMap; - import org.apache.hadoop.conf.Configuration; import org.apache.hadoop.fs.FileSystem; import org.apache.hadoop.fs.Path; import org.apache.hadoop.io.MapFile; +import org.apache.hadoop.io.MapFile.Writer.Option; import org.apache.hadoop.io.SequenceFile; import org.apache.hadoop.io.Text; -import org.apache.hadoop.io.MapFile.Writer.Option; import org.apache.nutch.util.NutchConfiguration; -import org.junit.After; -import org.junit.Assert; -import org.junit.Before; -import org.junit.Test; +import org.junit.jupiter.api.AfterEach; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; import org.slf4j.Logger; import org.slf4j.LoggerFactory; +import java.lang.invoke.MethodHandles; +import java.util.ArrayList; +import java.util.HashMap; +import java.util.Iterator; +import java.util.TreeMap; + +import static org.junit.jupiter.api.Assertions.assertNotNull; +import static org.junit.jupiter.api.Assertions.assertTrue; + public class TestLinkDbMerger { private static final Logger LOG = LoggerFactory .getLogger(MethodHandles.lookup().lookupClass()); @@ -68,7 +70,7 @@ public class TestLinkDbMerger { FileSystem fs; LinkDbReader reader; - @Before + @BeforeEach public void setUp() throws Exception { init1.put(url10, urls10); init1.put(url11, urls11); @@ -85,7 +87,7 @@ public void setUp() throws Exception { fs.mkdirs(testDir); } - @After + @AfterEach public void tearDown() { try { if (fs.exists(testDir)) @@ -120,7 +122,7 @@ public void testMerge() throws Exception { String[] vals = expected.get(url); Inlinks inlinks = reader.getInlinks(new Text(url)); // may not be null - Assert.assertNotNull(inlinks); + assertNotNull(inlinks); ArrayList links = new ArrayList(); Iterator it2 = inlinks.iterator(); while (it2.hasNext()) { @@ -129,7 +131,7 @@ public void testMerge() throws Exception { } for (int i = 0; i < vals.length; i++) { LOG.debug(" -> {}", vals[i]); - Assert.assertTrue(links.contains(vals[i])); + assertTrue(links.contains(vals[i])); } } reader.close(); diff --git a/src/test/org/apache/nutch/crawl/TestSignatureFactory.java b/src/test/org/apache/nutch/crawl/TestSignatureFactory.java index db82d7acb0..986fcf47d8 100644 --- a/src/test/org/apache/nutch/crawl/TestSignatureFactory.java +++ b/src/test/org/apache/nutch/crawl/TestSignatureFactory.java @@ -18,8 +18,10 @@ import org.apache.hadoop.conf.Configuration; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertNotNull; public class TestSignatureFactory { @@ -28,8 +30,8 @@ public void testGetSignature() { Configuration conf = NutchConfiguration.create(); Signature signature1 = SignatureFactory.getSignature(conf); Signature signature2 = SignatureFactory.getSignature(conf); - Assert.assertNotNull(signature1); - Assert.assertNotNull(signature2); - Assert.assertEquals(signature1, signature2); + assertNotNull(signature1); + assertNotNull(signature2); + assertEquals(signature1, signature2); } } diff --git a/src/test/org/apache/nutch/crawl/TestTextProfileSignature.java b/src/test/org/apache/nutch/crawl/TestTextProfileSignature.java index adf4b5e571..1d415b42fb 100644 --- a/src/test/org/apache/nutch/crawl/TestTextProfileSignature.java +++ b/src/test/org/apache/nutch/crawl/TestTextProfileSignature.java @@ -16,10 +16,6 @@ */ package org.apache.nutch.crawl; -import java.util.Arrays; -import java.util.Collections; -import java.util.List; - import org.apache.hadoop.conf.Configuration; import org.apache.nutch.metadata.Metadata; import org.apache.nutch.parse.Outlink; @@ -29,8 +25,14 @@ import org.apache.nutch.protocol.Content; import org.apache.nutch.util.NutchConfiguration; import org.apache.nutch.util.StringUtil; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import java.util.Arrays; +import java.util.Collections; +import java.util.List; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertNotNull; public class TestTextProfileSignature { @@ -44,14 +46,14 @@ public void testGetSignature() { new Outlink[0], new Metadata()); byte[] signature1 = textProf.calculate(new Content(), new ParseImpl(text, pd)); - Assert.assertNotNull(signature1); + assertNotNull(signature1); List words = Arrays.asList(text.split("\\s")); Collections.shuffle(words); String text2 = String.join(" ", words); byte[] signature2 = textProf.calculate(new Content(), new ParseImpl(text2, pd)); - Assert.assertNotNull(signature2); - Assert.assertEquals(StringUtil.toHexString(signature1), + assertNotNull(signature2); + assertEquals(StringUtil.toHexString(signature1), StringUtil.toHexString(signature2)); } } diff --git a/src/test/org/apache/nutch/fetcher/TestFetcher.java b/src/test/org/apache/nutch/fetcher/TestFetcher.java index 19e4104b27..f25cab5459 100644 --- a/src/test/org/apache/nutch/fetcher/TestFetcher.java +++ b/src/test/org/apache/nutch/fetcher/TestFetcher.java @@ -16,10 +16,6 @@ */ package org.apache.nutch.fetcher; -import java.io.IOException; -import java.util.ArrayList; -import java.util.Collections; - import org.apache.hadoop.conf.Configuration; import org.apache.hadoop.fs.FileSystem; import org.apache.hadoop.fs.Path; @@ -32,11 +28,17 @@ import org.apache.nutch.metadata.Nutch; import org.apache.nutch.parse.ParseData; import org.apache.nutch.protocol.Content; -import org.junit.After; -import org.junit.Assert; -import org.junit.Before; -import org.junit.Test; import org.eclipse.jetty.server.Server; +import org.junit.jupiter.api.AfterEach; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; + +import java.io.IOException; +import java.util.ArrayList; +import java.util.Collections; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertTrue; /** * Basic fetcher test 1. generate seedlist 2. inject 3. generate 3. fetch 4. @@ -53,7 +55,7 @@ public class TestFetcher { Path urlPath; Server server; - @Before + @BeforeEach public void setUp() throws Exception { conf = CrawlDBTestUtil.createContext().getConfiguration(); fs = FileSystem.get(conf); @@ -67,7 +69,7 @@ public void setUp() throws Exception { server.start(); } - @After + @AfterEach public void tearDown() throws Exception { server.stop(); for (int i = 0; i < 5; i++) { @@ -116,7 +118,7 @@ public void testFetch() throws IOException, ClassNotFoundException, InterruptedE // verify politeness, time taken should be more than (num_of_pages +1)*delay int minimumTime = (int) ((urls.size() + 1) * 1000 * conf.getFloat( "fetcher.server.delay", 5)); - Assert.assertTrue(time > minimumTime); + assertTrue(time > minimumTime); // verify content Path content = new Path(new Path(generatedSegment[0], Content.DIR_NAME), @@ -142,11 +144,11 @@ public void testFetch() throws IOException, ClassNotFoundException, InterruptedE Collections.sort(handledurls); // verify that enough pages were handled - Assert.assertEquals(urls.size(), handledurls.size()); + assertEquals(urls.size(), handledurls.size()); // verify that correct pages were handled - Assert.assertTrue(handledurls.containsAll(urls)); - Assert.assertTrue(urls.containsAll(handledurls)); + assertTrue(handledurls.containsAll(urls)); + assertTrue(urls.containsAll(handledurls)); handledurls.clear(); @@ -174,10 +176,10 @@ public void testFetch() throws IOException, ClassNotFoundException, InterruptedE Collections.sort(handledurls); - Assert.assertEquals(urls.size(), handledurls.size()); + assertEquals(urls.size(), handledurls.size()); - Assert.assertTrue(handledurls.containsAll(urls)); - Assert.assertTrue(urls.containsAll(handledurls)); + assertTrue(handledurls.containsAll(urls)); + assertTrue(urls.containsAll(handledurls)); } private void addUrl(ArrayList urls, String page) { @@ -201,7 +203,7 @@ public void testAgentNameCheck() { } catch (Exception e) { } - Assert.assertTrue(failedNoAgentName); + assertTrue(failedNoAgentName); } } diff --git a/src/test/org/apache/nutch/indexer/TestIndexerMapReduce.java b/src/test/org/apache/nutch/indexer/TestIndexerMapReduce.java index 95a5b41151..c275b1941a 100644 --- a/src/test/org/apache/nutch/indexer/TestIndexerMapReduce.java +++ b/src/test/org/apache/nutch/indexer/TestIndexerMapReduce.java @@ -16,9 +16,10 @@ */ package org.apache.nutch.indexer; -import java.lang.invoke.MethodHandles; - import org.apache.commons.codec.binary.Base64; +import org.apache.hadoop.conf.Configuration; +import org.apache.hadoop.io.Text; +import org.apache.hadoop.mapreduce.Reducer; import org.apache.hadoop.mrunit.mapreduce.ReduceDriver; import org.apache.hadoop.mrunit.types.Pair; import org.apache.hadoop.util.StringUtils; @@ -32,22 +33,19 @@ import org.apache.nutch.parse.ParseText; import org.apache.nutch.protocol.Content; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Test; +import org.junit.jupiter.api.Test; import org.slf4j.Logger; import org.slf4j.LoggerFactory; -import org.apache.hadoop.io.Text; -import org.apache.hadoop.mapreduce.Reducer; - -import static org.junit.Assert.*; - import java.io.IOException; +import java.lang.invoke.MethodHandles; import java.nio.charset.Charset; import java.nio.charset.StandardCharsets; import java.util.ArrayList; import java.util.List; -import org.apache.hadoop.conf.Configuration; +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertNotNull; /** Test {@link IndexerMapReduce} */ public class TestIndexerMapReduce { @@ -128,7 +126,7 @@ public void testBinaryContentBase64() { NutchDocument doc = runIndexer(crawlDatumDbFetched, crawlDatumFetchSuccess, parseText, parseData, content); - assertNotNull("No NutchDocument indexed", doc); + assertNotNull(doc, "No NutchDocument indexed"); String binaryContentBase64 = (String) doc.getField("binaryContent") .getValues().get(0); @@ -136,9 +134,8 @@ public void testBinaryContentBase64() { String binaryContent = new String( Base64.decodeBase64(binaryContentBase64), charset); LOG.info("binary content (decoded): {}", binaryContent); - assertEquals( - "Binary content (" + charset + ") not correctly saved as base64", - htmlDoc, binaryContent); + assertEquals(htmlDoc, binaryContent, + "Binary content (" + charset + ") not correctly saved as base64"); } } diff --git a/src/test/org/apache/nutch/indexer/TestIndexingFilters.java b/src/test/org/apache/nutch/indexer/TestIndexingFilters.java index 4d5849fdc3..f4a1b2cf7d 100644 --- a/src/test/org/apache/nutch/indexer/TestIndexingFilters.java +++ b/src/test/org/apache/nutch/indexer/TestIndexingFilters.java @@ -26,8 +26,10 @@ import org.apache.nutch.parse.ParseImpl; import org.apache.nutch.parse.ParseStatus; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertNull; public class TestIndexingFilters { @@ -67,7 +69,7 @@ public void testNutchDocumentNullIndexingFilter() throws IndexingException { new Metadata())), new Text("http://www.example.com/"), new CrawlDatum(), new Inlinks()); - Assert.assertNull(doc); + assertNull(doc); } /** @@ -103,7 +105,7 @@ public void testFilterCacheIndexingFilter() throws IndexingException { NutchDocument fdoc2 = filters2.filter(new NutchDocument(), new ParseImpl( "text", new ParseData(new ParseStatus(), "title", new Outlink[0], md)), new Text("http://www.example.com/"), new CrawlDatum(), new Inlinks()); - Assert.assertEquals(fdoc1.getFieldNames().size(), fdoc2.getFieldNames() + assertEquals(fdoc1.getFieldNames().size(), fdoc2.getFieldNames() .size()); } diff --git a/src/test/org/apache/nutch/metadata/TestMetadata.java b/src/test/org/apache/nutch/metadata/TestMetadata.java index 0804e3e351..9c7791c727 100644 --- a/src/test/org/apache/nutch/metadata/TestMetadata.java +++ b/src/test/org/apache/nutch/metadata/TestMetadata.java @@ -23,8 +23,9 @@ import java.io.IOException; import java.util.Properties; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.*; /** * JUnit based tests of class {@link org.apache.nutch.metadata.Metadata}. @@ -48,14 +49,14 @@ public void testWriteNonNull() { met.add(CONTENTTYPE, "text/bogus2"); met = writeRead(met); - Assert.assertNotNull(met); - Assert.assertEquals(met.size(), 1); + assertNotNull(met); + assertEquals(1, met.size()); boolean hasBogus = false, hasBogus2 = false; String[] values = met.getValues(CONTENTTYPE); - Assert.assertNotNull(values); - Assert.assertEquals(values.length, 2); + assertNotNull(values); + assertEquals(2, values.length); for (int i = 0; i < values.length; i++) { if (values[i].equals("text/bogus")) { @@ -67,7 +68,7 @@ public void testWriteNonNull() { } } - Assert.assertTrue(hasBogus && hasBogus2); + assertTrue(hasBogus && hasBogus2); } /** Test for the add(String, String) method. */ @@ -77,27 +78,27 @@ public void testAdd() { Metadata meta = new Metadata(); values = meta.getValues(CONTENTTYPE); - Assert.assertEquals(0, values.length); + assertEquals(0, values.length); meta.add(CONTENTTYPE, "value1"); values = meta.getValues(CONTENTTYPE); - Assert.assertEquals(1, values.length); - Assert.assertEquals("value1", values[0]); + assertEquals(1, values.length); + assertEquals("value1", values[0]); meta.add(CONTENTTYPE, "value2"); values = meta.getValues(CONTENTTYPE); - Assert.assertEquals(2, values.length); - Assert.assertEquals("value1", values[0]); - Assert.assertEquals("value2", values[1]); + assertEquals(2, values.length); + assertEquals("value1", values[0]); + assertEquals("value2", values[1]); // NOTE : For now, the same value can be added many times. // Should it be changed? meta.add(CONTENTTYPE, "value1"); values = meta.getValues(CONTENTTYPE); - Assert.assertEquals(3, values.length); - Assert.assertEquals("value1", values[0]); - Assert.assertEquals("value2", values[1]); - Assert.assertEquals("value1", values[2]); + assertEquals(3, values.length); + assertEquals("value1", values[0]); + assertEquals("value2", values[1]); + assertEquals("value1", values[2]); } /** Test for the set(String, String) method. */ @@ -107,24 +108,24 @@ public void testSet() { Metadata meta = new Metadata(); values = meta.getValues(CONTENTTYPE); - Assert.assertEquals(0, values.length); + assertEquals(0, values.length); meta.set(CONTENTTYPE, "value1"); values = meta.getValues(CONTENTTYPE); - Assert.assertEquals(1, values.length); - Assert.assertEquals("value1", values[0]); + assertEquals(1, values.length); + assertEquals("value1", values[0]); meta.set(CONTENTTYPE, "value2"); values = meta.getValues(CONTENTTYPE); - Assert.assertEquals(1, values.length); - Assert.assertEquals("value2", values[0]); + assertEquals(1, values.length); + assertEquals("value2", values[0]); meta.set(CONTENTTYPE, "new value 1"); meta.add("contenttype", "new value 2"); values = meta.getValues(CONTENTTYPE); - Assert.assertEquals(2, values.length); - Assert.assertEquals("new value 1", values[0]); - Assert.assertEquals("new value 2", values[1]); + assertEquals(2, values.length); + assertEquals("new value 1", values[0]); + assertEquals("new value 2", values[1]); } /** Test for setAll(Properties) method. */ @@ -135,46 +136,46 @@ public void testSetProperties() { Properties props = new Properties(); meta.setAll(props); - Assert.assertEquals(0, meta.size()); + assertEquals(0, meta.size()); props.setProperty("name-one", "value1.1"); meta.setAll(props); - Assert.assertEquals(1, meta.size()); + assertEquals(1, meta.size()); values = meta.getValues("name-one"); - Assert.assertEquals(1, values.length); - Assert.assertEquals("value1.1", values[0]); + assertEquals(1, values.length); + assertEquals("value1.1", values[0]); props.setProperty("name-two", "value2.1"); meta.setAll(props); - Assert.assertEquals(2, meta.size()); + assertEquals(2, meta.size()); values = meta.getValues("name-one"); - Assert.assertEquals(1, values.length); - Assert.assertEquals("value1.1", values[0]); + assertEquals(1, values.length); + assertEquals("value1.1", values[0]); values = meta.getValues("name-two"); - Assert.assertEquals(1, values.length); - Assert.assertEquals("value2.1", values[0]); + assertEquals(1, values.length); + assertEquals("value2.1", values[0]); } /** Test for get(String) method. */ @Test public void testGet() { Metadata meta = new Metadata(); - Assert.assertNull(meta.get("a-name")); + assertNull(meta.get("a-name")); meta.add("a-name", "value-1"); - Assert.assertEquals("value-1", meta.get("a-name")); + assertEquals("value-1", meta.get("a-name")); meta.add("a-name", "value-2"); - Assert.assertEquals("value-1", meta.get("a-name")); + assertEquals("value-1", meta.get("a-name")); } /** Test for isMultiValued() method. */ @Test public void testIsMultiValued() { Metadata meta = new Metadata(); - Assert.assertFalse(meta.isMultiValued("key")); + assertFalse(meta.isMultiValued("key")); meta.add("key", "value1"); - Assert.assertFalse(meta.isMultiValued("key")); + assertFalse(meta.isMultiValued("key")); meta.add("key", "value2"); - Assert.assertTrue(meta.isMultiValued("key")); + assertTrue(meta.isMultiValued("key")); } /** Test for names method. */ @@ -183,15 +184,15 @@ public void testNames() { String[] names = null; Metadata meta = new Metadata(); names = meta.names(); - Assert.assertEquals(0, names.length); + assertEquals(0, names.length); meta.add("name-one", "value"); names = meta.names(); - Assert.assertEquals(1, names.length); - Assert.assertEquals("name-one", names[0]); + assertEquals(1, names.length); + assertEquals("name-one", names[0]); meta.add("name-two", "value"); names = meta.names(); - Assert.assertEquals(2, names.length); + assertEquals(2, names.length); } /** Test for remove(String) method. */ @@ -199,21 +200,21 @@ public void testNames() { public void testRemove() { Metadata meta = new Metadata(); meta.remove("name-one"); - Assert.assertEquals(0, meta.size()); + assertEquals(0, meta.size()); meta.add("name-one", "value-1.1"); meta.add("name-one", "value-1.2"); meta.add("name-two", "value-2.2"); - Assert.assertEquals(2, meta.size()); - Assert.assertNotNull(meta.get("name-one")); - Assert.assertNotNull(meta.get("name-two")); + assertEquals(2, meta.size()); + assertNotNull(meta.get("name-one")); + assertNotNull(meta.get("name-two")); meta.remove("name-one"); - Assert.assertEquals(1, meta.size()); - Assert.assertNull(meta.get("name-one")); - Assert.assertNotNull(meta.get("name-two")); + assertEquals(1, meta.size()); + assertNull(meta.get("name-one")); + assertNotNull(meta.get("name-two")); meta.remove("name-two"); - Assert.assertEquals(0, meta.size()); - Assert.assertNull(meta.get("name-one")); - Assert.assertNull(meta.get("name-two")); + assertEquals(0, meta.size()); + assertNull(meta.get("name-one")); + assertNull(meta.get("name-two")); } /** Test for equals(Object) method. */ @@ -221,25 +222,25 @@ public void testRemove() { public void testObject() { Metadata meta1 = new Metadata(); Metadata meta2 = new Metadata(); - Assert.assertFalse(meta1.equals(null)); - Assert.assertFalse(meta1.equals("String")); - Assert.assertTrue(meta1.equals(meta2)); + assertNotNull(meta1); + assertNotEquals("String", meta1); + assertEquals(meta1, meta2); meta1.add("name-one", "value-1.1"); - Assert.assertFalse(meta1.equals(meta2)); + assertNotEquals(meta1, meta2); meta2.add("name-one", "value-1.1"); - Assert.assertTrue(meta1.equals(meta2)); + assertEquals(meta1, meta2); meta1.add("name-one", "value-1.2"); - Assert.assertFalse(meta1.equals(meta2)); + assertNotEquals(meta1, meta2); meta2.add("name-one", "value-1.2"); - Assert.assertTrue(meta1.equals(meta2)); + assertEquals(meta1, meta2); meta1.add("name-two", "value-2.1"); - Assert.assertFalse(meta1.equals(meta2)); + assertNotEquals(meta1, meta2); meta2.add("name-two", "value-2.1"); - Assert.assertTrue(meta1.equals(meta2)); + assertEquals(meta1, meta2); meta1.add("name-two", "value-2.2"); - Assert.assertFalse(meta1.equals(meta2)); + assertNotEquals(meta1, meta2); meta2.add("name-two", "value-2.x"); - Assert.assertFalse(meta1.equals(meta2)); + assertNotEquals(meta1, meta2); } /** Test for Writable implementation. */ @@ -248,21 +249,21 @@ public void testWritable() { Metadata result = null; Metadata meta = new Metadata(); result = writeRead(meta); - Assert.assertEquals(0, result.size()); + assertEquals(0, result.size()); meta.add("name-one", "value-1.1"); result = writeRead(meta); - Assert.assertEquals(1, result.size()); - Assert.assertEquals(1, result.getValues("name-one").length); - Assert.assertEquals("value-1.1", result.get("name-one")); + assertEquals(1, result.size()); + assertEquals(1, result.getValues("name-one").length); + assertEquals("value-1.1", result.get("name-one")); meta.add("name-two", "value-2.1"); meta.add("name-two", "value-2.2"); result = writeRead(meta); - Assert.assertEquals(2, result.size()); - Assert.assertEquals(1, result.getValues("name-one").length); - Assert.assertEquals("value-1.1", result.getValues("name-one")[0]); - Assert.assertEquals(2, result.getValues("name-two").length); - Assert.assertEquals("value-2.1", result.getValues("name-two")[0]); - Assert.assertEquals("value-2.2", result.getValues("name-two")[1]); + assertEquals(2, result.size()); + assertEquals(1, result.getValues("name-one").length); + assertEquals("value-1.1", result.getValues("name-one")[0]); + assertEquals(2, result.getValues("name-two").length); + assertEquals("value-2.1", result.getValues("name-two")[0]); + assertEquals("value-2.2", result.getValues("name-two")[1]); } private Metadata writeRead(Metadata meta) { @@ -273,7 +274,7 @@ private Metadata writeRead(Metadata meta) { readed.readFields(new DataInputStream(new ByteArrayInputStream(out .toByteArray()))); } catch (IOException ioe) { - Assert.fail(ioe.toString()); + fail(ioe.toString()); } return readed; } diff --git a/src/test/org/apache/nutch/metadata/TestSpellCheckedMetadata.java b/src/test/org/apache/nutch/metadata/TestSpellCheckedMetadata.java index 9041736aef..b33d514477 100644 --- a/src/test/org/apache/nutch/metadata/TestSpellCheckedMetadata.java +++ b/src/test/org/apache/nutch/metadata/TestSpellCheckedMetadata.java @@ -24,8 +24,9 @@ import java.util.HashMap; import java.util.Properties; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.*; /** * JUnit based tests of class @@ -41,17 +42,17 @@ public class TestSpellCheckedMetadata { /** Test for the getNormalizedName(String) method. */ @Test public void testGetNormalizedName() { - Assert.assertEquals("Content-Type", + assertEquals("Content-Type", SpellCheckedMetadata.getNormalizedName("Content-Type")); - Assert.assertEquals("Content-Type", + assertEquals("Content-Type", SpellCheckedMetadata.getNormalizedName("ContentType")); - Assert.assertEquals("Content-Type", + assertEquals("Content-Type", SpellCheckedMetadata.getNormalizedName("Content-type")); - Assert.assertEquals("Content-Type", + assertEquals("Content-Type", SpellCheckedMetadata.getNormalizedName("contenttype")); - Assert.assertEquals("Content-Type", + assertEquals("Content-Type", SpellCheckedMetadata.getNormalizedName("contentype")); - Assert.assertEquals("Content-Type", + assertEquals("Content-Type", SpellCheckedMetadata.getNormalizedName("contntype")); } @@ -62,27 +63,27 @@ public void testAdd() { SpellCheckedMetadata meta = new SpellCheckedMetadata(); values = meta.getValues("contentype"); - Assert.assertEquals(0, values.length); + assertEquals(0, values.length); meta.add("contentype", "value1"); values = meta.getValues("contentype"); - Assert.assertEquals(1, values.length); - Assert.assertEquals("value1", values[0]); + assertEquals(1, values.length); + assertEquals("value1", values[0]); meta.add("Content-Type", "value2"); values = meta.getValues("contentype"); - Assert.assertEquals(2, values.length); - Assert.assertEquals("value1", values[0]); - Assert.assertEquals("value2", values[1]); + assertEquals(2, values.length); + assertEquals("value1", values[0]); + assertEquals("value2", values[1]); // NOTE : For now, the same value can be added many times. // Should it be changed? meta.add("ContentType", "value1"); values = meta.getValues("Content-Type"); - Assert.assertEquals(3, values.length); - Assert.assertEquals("value1", values[0]); - Assert.assertEquals("value2", values[1]); - Assert.assertEquals("value1", values[2]); + assertEquals(3, values.length); + assertEquals("value1", values[0]); + assertEquals("value2", values[1]); + assertEquals("value1", values[2]); } /** Test for the set(String, String) method. */ @@ -92,24 +93,24 @@ public void testSet() { SpellCheckedMetadata meta = new SpellCheckedMetadata(); values = meta.getValues("contentype"); - Assert.assertEquals(0, values.length); + assertEquals(0, values.length); meta.set("contentype", "value1"); values = meta.getValues("contentype"); - Assert.assertEquals(1, values.length); - Assert.assertEquals("value1", values[0]); + assertEquals(1, values.length); + assertEquals("value1", values[0]); meta.set("Content-Type", "value2"); values = meta.getValues("contentype"); - Assert.assertEquals(1, values.length); - Assert.assertEquals("value2", values[0]); + assertEquals(1, values.length); + assertEquals("value2", values[0]); meta.set("contenttype", "new value 1"); meta.add("contenttype", "new value 2"); values = meta.getValues("contentype"); - Assert.assertEquals(2, values.length); - Assert.assertEquals("new value 1", values[0]); - Assert.assertEquals("new value 2", values[1]); + assertEquals(2, values.length); + assertEquals("new value 1", values[0]); + assertEquals("new value 2", values[1]); } /** Test for the set(String, String) method. */ @@ -119,24 +120,24 @@ public void testSetCaseInsensitive() { SpellCheckedMetadata meta = new SpellCheckedMetadata(); values = meta.getValues("name-one"); - Assert.assertEquals(0, values.length); + assertEquals(0, values.length); meta.set("name-one", "value1"); values = meta.getValues("name-one"); - Assert.assertEquals(1, values.length); - Assert.assertEquals("value1", values[0]); + assertEquals(1, values.length); + assertEquals("value1", values[0]); meta.set("naMe-OnE", "value2"); values = meta.getValues("name-one"); - Assert.assertEquals(1, values.length); - Assert.assertEquals("value2", values[0]); + assertEquals(1, values.length); + assertEquals("value2", values[0]); meta.set("nAme-One", "new value 1"); meta.add("NamE-oNe", "new value 2"); values = meta.getValues("namE-OnE"); - Assert.assertEquals(2, values.length); - Assert.assertEquals("new value 1", values[0]); - Assert.assertEquals("new value 2", values[1]); + assertEquals(2, values.length); + assertEquals("new value 1", values[0]); + assertEquals("new value 2", values[1]); } /** Test for setAll(Properties) method. */ @@ -147,60 +148,60 @@ public void testSetProperties() { Properties props = new Properties(); meta.setAll(props); - Assert.assertEquals(0, meta.size()); + assertEquals(0, meta.size()); props.setProperty("name-one", "value1.1"); meta.setAll(props); - Assert.assertEquals(1, meta.size()); + assertEquals(1, meta.size()); values = meta.getValues("name-one"); - Assert.assertEquals(1, values.length); - Assert.assertEquals("value1.1", values[0]); + assertEquals(1, values.length); + assertEquals("value1.1", values[0]); props.setProperty("name-two", "value2.1"); meta.setAll(props); - Assert.assertEquals(2, meta.size()); + assertEquals(2, meta.size()); values = meta.getValues("name-one"); - Assert.assertEquals(1, values.length); - Assert.assertEquals("value1.1", values[0]); + assertEquals(1, values.length); + assertEquals("value1.1", values[0]); values = meta.getValues("name-two"); - Assert.assertEquals(1, values.length); - Assert.assertEquals("value2.1", values[0]); + assertEquals(1, values.length); + assertEquals("value2.1", values[0]); } /** Test for get(String) method. */ @Test public void testGet() { SpellCheckedMetadata meta = new SpellCheckedMetadata(); - Assert.assertNull(meta.get("a-name")); + assertNull(meta.get("a-name")); meta.add("a-name", "value-1"); - Assert.assertEquals("value-1", meta.get("a-name")); + assertEquals("value-1", meta.get("a-name")); meta.add("a-name", "value-2"); - Assert.assertEquals("value-1", meta.get("a-name")); + assertEquals("value-1", meta.get("a-name")); } /** Test for get(String) method. */ @Test public void testGetCaseInsensitive() { SpellCheckedMetadata meta = new SpellCheckedMetadata(); - Assert.assertNull(meta.get("a-name")); + assertNull(meta.get("a-name")); meta.add("a-name", "value-1"); - Assert.assertEquals("value-1", meta.get("a-name")); + assertEquals("value-1", meta.get("a-name")); - Assert.assertNotNull(meta.get("a-NamE")); - Assert.assertEquals("value-1", meta.get("a-NamE")); + assertNotNull(meta.get("a-NamE")); + assertEquals("value-1", meta.get("a-NamE")); } /** Test for isMultiValued() method. */ @Test public void testIsMultiValued() { SpellCheckedMetadata meta = new SpellCheckedMetadata(); - Assert.assertFalse(meta.isMultiValued("key")); + assertFalse(meta.isMultiValued("key")); meta.add("key", "value1"); - Assert.assertFalse(meta.isMultiValued("key")); + assertFalse(meta.isMultiValued("key")); meta.add("key", "value2"); - Assert.assertTrue(meta.isMultiValued("key")); + assertTrue(meta.isMultiValued("key")); } /** Test for names method. */ @@ -209,15 +210,15 @@ public void testNames() { String[] names = null; SpellCheckedMetadata meta = new SpellCheckedMetadata(); names = meta.names(); - Assert.assertEquals(0, names.length); + assertEquals(0, names.length); meta.add("name-one", "value"); names = meta.names(); - Assert.assertEquals(1, names.length); - Assert.assertEquals("name-one", names[0]); + assertEquals(1, names.length); + assertEquals("name-one", names[0]); meta.add("name-two", "value"); names = meta.names(); - Assert.assertEquals(2, names.length); + assertEquals(2, names.length); } /** Test for remove(String) method. */ @@ -225,21 +226,21 @@ public void testNames() { public void testRemove() { SpellCheckedMetadata meta = new SpellCheckedMetadata(); meta.remove("name-one"); - Assert.assertEquals(0, meta.size()); + assertEquals(0, meta.size()); meta.add("name-one", "value-1.1"); meta.add("name-one", "value-1.2"); meta.add("name-two", "value-2.2"); - Assert.assertEquals(2, meta.size()); - Assert.assertNotNull(meta.get("name-one")); - Assert.assertNotNull(meta.get("name-two")); + assertEquals(2, meta.size()); + assertNotNull(meta.get("name-one")); + assertNotNull(meta.get("name-two")); meta.remove("name-one"); - Assert.assertEquals(1, meta.size()); - Assert.assertNull(meta.get("name-one")); - Assert.assertNotNull(meta.get("name-two")); + assertEquals(1, meta.size()); + assertNull(meta.get("name-one")); + assertNotNull(meta.get("name-two")); meta.remove("name-two"); - Assert.assertEquals(0, meta.size()); - Assert.assertNull(meta.get("name-one")); - Assert.assertNull(meta.get("name-two")); + assertEquals(0, meta.size()); + assertNull(meta.get("name-one")); + assertNull(meta.get("name-two")); } /** Test for equals(Object) method. */ @@ -247,25 +248,25 @@ public void testRemove() { public void testObject() { SpellCheckedMetadata meta1 = new SpellCheckedMetadata(); SpellCheckedMetadata meta2 = new SpellCheckedMetadata(); - Assert.assertFalse(meta1.equals(null)); - Assert.assertFalse(meta1.equals("String")); - Assert.assertTrue(meta1.equals(meta2)); + assertNotEquals(null, meta1); + assertNotEquals("String", meta1); + assertEquals(meta1, meta2); meta1.add("name-one", "value-1.1"); - Assert.assertFalse(meta1.equals(meta2)); + assertNotEquals(meta1, meta2); meta2.add("name-one", "value-1.1"); - Assert.assertTrue(meta1.equals(meta2)); + assertEquals(meta1, meta2); meta1.add("name-one", "value-1.2"); - Assert.assertFalse(meta1.equals(meta2)); + assertNotEquals(meta1, meta2); meta2.add("name-one", "value-1.2"); - Assert.assertTrue(meta1.equals(meta2)); + assertEquals(meta1, meta2); meta1.add("name-two", "value-2.1"); - Assert.assertFalse(meta1.equals(meta2)); + assertNotEquals(meta1, meta2); meta2.add("name-two", "value-2.1"); - Assert.assertTrue(meta1.equals(meta2)); + assertEquals(meta1, meta2); meta1.add("name-two", "value-2.2"); - Assert.assertFalse(meta1.equals(meta2)); + assertNotEquals(meta1, meta2); meta2.add("name-two", "value-2.x"); - Assert.assertFalse(meta1.equals(meta2)); + assertNotEquals(meta1, meta2); } /** Test for Writable implementation. */ @@ -274,23 +275,23 @@ public void testWritable() { SpellCheckedMetadata result = null; SpellCheckedMetadata meta = new SpellCheckedMetadata(); result = writeRead(meta); - Assert.assertEquals(0, result.size()); + assertEquals(0, result.size()); meta.add("name-one", "value-1.1"); result = writeRead(meta); meta.add("Contenttype", "text/html"); - Assert.assertEquals(1, result.size()); - Assert.assertEquals(1, result.getValues("name-one").length); - Assert.assertEquals("value-1.1", result.get("name-one")); + assertEquals(1, result.size()); + assertEquals(1, result.getValues("name-one").length); + assertEquals("value-1.1", result.get("name-one")); meta.add("name-two", "value-2.1"); meta.add("name-two", "value-2.2"); result = writeRead(meta); - Assert.assertEquals(3, result.size()); - Assert.assertEquals(1, result.getValues("name-one").length); - Assert.assertEquals("value-1.1", result.getValues("name-one")[0]); - Assert.assertEquals(2, result.getValues("name-two").length); - Assert.assertEquals("value-2.1", result.getValues("name-two")[0]); - Assert.assertEquals("value-2.2", result.getValues("name-two")[1]); - Assert.assertEquals("text/html", result.get(Metadata.CONTENT_TYPE)); + assertEquals(3, result.size()); + assertEquals(1, result.getValues("name-one").length); + assertEquals("value-1.1", result.getValues("name-one")[0]); + assertEquals(2, result.getValues("name-two").length); + assertEquals("value-2.1", result.getValues("name-two")[0]); + assertEquals("value-2.2", result.getValues("name-two")[1]); + assertEquals("text/html", result.get(Metadata.CONTENT_TYPE)); } /** Test for Writable implementation. */ @@ -302,30 +303,30 @@ public void testWritableBackwardCompatibility() { CaseSensitiveSpellCheckedMetadata meta = new CaseSensitiveSpellCheckedMetadata(); result = writeRead(meta); - Assert.assertEquals(0, result.size()); + assertEquals(0, result.size()); meta.add("name-One", "value-1.1"); // Check that the original case is kept for old Metadata class - Assert.assertEquals(0, result.getValues("naMe-one").length); + assertEquals(0, result.getValues("naMe-one").length); // Check that the values written by old instances can be // read by new instances of SpellCheckedMetadata result = writeRead(meta); - Assert.assertEquals(1, result.size()); - Assert.assertEquals(1, result.getValues("naMe-one").length); - Assert.assertEquals("value-1.1", result.get("nAme-oNe")); + assertEquals(1, result.size()); + assertEquals(1, result.getValues("naMe-one").length); + assertEquals("value-1.1", result.get("nAme-oNe")); meta.add("Contenttype", "text/html"); meta.add("name-Two", "value-2.1"); meta.add("namE-two", "value-2.2"); result = writeRead(meta); - Assert.assertEquals(3, result.size()); - Assert.assertEquals(1, result.getValues("name-onE").length); - Assert.assertEquals("value-1.1", result.getValues("namE-one")[0]); - Assert.assertEquals(2, result.getValues("name-two").length); - Assert.assertEquals("value-2.1", result.getValues("nAme-tWo")[0]); - Assert.assertEquals("value-2.2", result.getValues("namE-Two")[1]); - Assert.assertEquals("text/html", result.get(Metadata.CONTENT_TYPE)); + assertEquals(3, result.size()); + assertEquals(1, result.getValues("name-onE").length); + assertEquals("value-1.1", result.getValues("namE-one")[0]); + assertEquals(2, result.getValues("name-two").length); + assertEquals("value-2.1", result.getValues("nAme-tWo")[0]); + assertEquals("value-2.2", result.getValues("namE-Two")[1]); + assertEquals("text/html", result.get(Metadata.CONTENT_TYPE)); } /** @@ -363,7 +364,7 @@ private SpellCheckedMetadata writeRead(SpellCheckedMetadata meta) { readed.readFields(new DataInputStream(new ByteArrayInputStream(out .toByteArray()))); } catch (IOException ioe) { - Assert.fail(ioe.toString()); + fail(ioe.toString()); } return readed; } diff --git a/src/test/org/apache/nutch/net/TestURLFilters.java b/src/test/org/apache/nutch/net/TestURLFilters.java index c43941a94d..00c9b14894 100644 --- a/src/test/org/apache/nutch/net/TestURLFilters.java +++ b/src/test/org/apache/nutch/net/TestURLFilters.java @@ -18,7 +18,7 @@ import org.apache.hadoop.conf.Configuration; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Test; +import org.junit.jupiter.api.Test; public class TestURLFilters { diff --git a/src/test/org/apache/nutch/net/TestURLNormalizers.java b/src/test/org/apache/nutch/net/TestURLNormalizers.java index 6fdbb9d884..3427109d1e 100644 --- a/src/test/org/apache/nutch/net/TestURLNormalizers.java +++ b/src/test/org/apache/nutch/net/TestURLNormalizers.java @@ -20,8 +20,9 @@ import org.apache.hadoop.conf.Configuration; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.*; public class TestURLNormalizers { @@ -35,12 +36,12 @@ public void testURLNormalizers() { URLNormalizers normalizers = new URLNormalizers(conf, URLNormalizers.SCOPE_DEFAULT); - Assert.assertNotNull(normalizers); + assertNotNull(normalizers); try { normalizers.normalize("http://www.example.com/", URLNormalizers.SCOPE_DEFAULT); } catch (MalformedURLException mue) { - Assert.fail(mue.toString()); + fail(mue.toString()); } // NUTCH-1011 - Get rid of superfluous slashes @@ -48,10 +49,10 @@ public void testURLNormalizers() { String normalizedSlashes = normalizers.normalize( "http://www.example.com//path/to//somewhere.html", URLNormalizers.SCOPE_DEFAULT); - Assert.assertEquals(normalizedSlashes, - "http://www.example.com/path/to/somewhere.html"); + assertEquals("http://www.example.com/path/to/somewhere.html", + normalizedSlashes); } catch (MalformedURLException mue) { - Assert.fail(mue.toString()); + fail(mue.toString()); } // HostNormalizer NUTCH-1319 @@ -59,10 +60,10 @@ public void testURLNormalizers() { String normalizedHost = normalizers.normalize( "http://www.example.org//path/to//somewhere.html", URLNormalizers.SCOPE_DEFAULT); - Assert.assertEquals(normalizedHost, - "http://www.example.org/path/to/somewhere.html"); + assertEquals("http://www.example.org/path/to/somewhere.html", + normalizedHost); } catch (MalformedURLException mue) { - Assert.fail(mue.toString()); + fail(mue.toString()); } // check the order @@ -76,8 +77,7 @@ public void testURLNormalizers() { pos2 = i; } if (pos1 != -1 && pos2 != -1) { - Assert.assertTrue("RegexURLNormalizer before BasicURLNormalizer", - pos1 < pos2); + assertTrue(pos1 < pos2, "RegexURLNormalizer before BasicURLNormalizer"); } } } diff --git a/src/test/org/apache/nutch/net/protocols/TestHttpDateFormat.java b/src/test/org/apache/nutch/net/protocols/TestHttpDateFormat.java index 94f30c3e66..818d35b303 100644 --- a/src/test/org/apache/nutch/net/protocols/TestHttpDateFormat.java +++ b/src/test/org/apache/nutch/net/protocols/TestHttpDateFormat.java @@ -19,8 +19,13 @@ import java.text.ParseException; import java.util.Date; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; +import static org.junit.jupiter.api.Assertions.assertThrows; +import static org.junit.jupiter.api.Assertions.assertTrue; +import static org.junit.jupiter.api.Assertions.assertEquals; + +import static org.hamcrest.CoreMatchers.is; +import static org.hamcrest.MatcherAssert.assertThat; public class TestHttpDateFormat { @@ -38,18 +43,25 @@ public class TestHttpDateFormat { @Test public void testHttpDateFormat() throws ParseException { - Assert.assertEquals(dateMillis, HttpDateFormat.toLong(dateString)); - Assert.assertEquals(dateString, HttpDateFormat.toString(dateMillis)); - Assert.assertEquals(new Date(dateMillis), HttpDateFormat.toDate(dateString)); + assertThat(HttpDateFormat.toLong(dateString), is(dateMillis)); + assertThat(HttpDateFormat.toString(dateMillis), is(dateString)); + assertThat(HttpDateFormat.toDate(dateString), is(new Date(dateMillis))); String ds2 = "Sun, 6 Nov 1994 08:49:37 GMT"; - Assert.assertEquals(dateMillis, HttpDateFormat.toLong(ds2)); + assertThat(HttpDateFormat.toLong(ds2), is(dateMillis)); } - @Test(expected = ParseException.class) + @Test public void testHttpDateFormatException() throws ParseException { String ds = "this is not a valid date"; - HttpDateFormat.toLong(ds); + Exception exception = assertThrows(ParseException.class, () -> { + HttpDateFormat.toLong(ds); + }); + String expectedMessage = + "Text 'this is not a valid date' could not be parsed at index 0"; + String actualMessage = exception.getMessage(); + assertTrue(actualMessage.contains(expectedMessage)); + } /** @@ -60,6 +72,6 @@ public void testHttpDateFormatException() throws ParseException { public void testHttpDateFormatTimeZone() throws ParseException { String dateStringPDT = "Mon, 21 Oct 2019 03:18:16 PDT"; HttpDateFormat.toLong(dateStringPDT); // must not affect internal time zone - Assert.assertEquals(dateString, HttpDateFormat.toString(dateMillis)); + assertEquals(dateString, HttpDateFormat.toString(dateMillis)); } } diff --git a/src/test/org/apache/nutch/parse/TestOutlinkExtractor.java b/src/test/org/apache/nutch/parse/TestOutlinkExtractor.java index 218e4fc460..4b4c4e9dd9 100644 --- a/src/test/org/apache/nutch/parse/TestOutlinkExtractor.java +++ b/src/test/org/apache/nutch/parse/TestOutlinkExtractor.java @@ -18,8 +18,9 @@ import org.apache.hadoop.conf.Configuration; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.*; /** * TestCase to check regExp extraction of URLs. @@ -37,12 +38,12 @@ public void testGetNoOutlinks() { Outlink[] outlinks = null; outlinks = OutlinkExtractor.getOutlinks(null, conf); - Assert.assertNotNull(outlinks); - Assert.assertEquals(0, outlinks.length); + assertNotNull(outlinks); + assertEquals(0, outlinks.length); outlinks = OutlinkExtractor.getOutlinks("", conf); - Assert.assertNotNull(outlinks); - Assert.assertEquals(0, outlinks.length); + assertNotNull(outlinks); + assertEquals(0, outlinks.length); } @Test @@ -54,13 +55,13 @@ public void testGetOutlinksHttp() { + "A longer URL could be http://www.sybit.com/solutions/portals.html", conf); - Assert.assertTrue("Url not found!", outlinks.length == 3); - Assert.assertEquals("Wrong URL", "http://www.nutch.org/index.html", - outlinks[0].getToUrl()); - Assert.assertEquals("Wrong URL", "http://www.google.de", - outlinks[1].getToUrl()); - Assert.assertEquals("Wrong URL", - "http://www.sybit.com/solutions/portals.html", outlinks[2].getToUrl()); + assertEquals(3, outlinks.length, "Url not found!"); + assertEquals("http://www.nutch.org/index.html", + outlinks[0].getToUrl(), "Wrong URL"); + assertEquals( "http://www.google.de", + outlinks[1].getToUrl(), "Wrong URL"); + assertEquals("http://www.sybit.com/solutions/portals.html", + outlinks[2].getToUrl(), "Wrong URL"); } @Test @@ -72,13 +73,13 @@ public void testGetOutlinksHttp2() { + "A longer URL could be http://www.sybit.com/solutions/portals.html", "http://www.sybit.de", conf); - Assert.assertTrue("Url not found!", outlinks.length == 3); - Assert.assertEquals("Wrong URL", "http://www.nutch.org/index.html", - outlinks[0].getToUrl()); - Assert.assertEquals("Wrong URL", "http://www.google.de", - outlinks[1].getToUrl()); - Assert.assertEquals("Wrong URL", - "http://www.sybit.com/solutions/portals.html", outlinks[2].getToUrl()); + assertEquals(3, outlinks.length, "Url not found!"); + assertEquals("http://www.nutch.org/index.html", + outlinks[0].getToUrl(), "Wrong URL"); + assertEquals("http://www.google.de", outlinks[1].getToUrl(), + "Wrong URL"); + assertEquals("http://www.sybit.com/solutions/portals.html", + outlinks[2].getToUrl(), "Wrong URL"); } @Test @@ -87,10 +88,10 @@ public void testGetOutlinksFtp() { "Test with ftp://www.nutch.org is it found? " + "What about www.google.com at ftp://www.google.de", conf); - Assert.assertTrue("Url not found!", outlinks.length > 1); - Assert.assertEquals("Wrong URL", "ftp://www.nutch.org", - outlinks[0].getToUrl()); - Assert.assertEquals("Wrong URL", "ftp://www.google.de", - outlinks[1].getToUrl()); + assertTrue(outlinks.length > 1, "Url not found!"); + assertEquals( "ftp://www.nutch.org", outlinks[0].getToUrl(), + "Wrong URL"); + assertEquals( "ftp://www.google.de", outlinks[1].getToUrl(), + "Wrong URL"); } } diff --git a/src/test/org/apache/nutch/parse/TestOutlinks.java b/src/test/org/apache/nutch/parse/TestOutlinks.java index 78c051e51b..e0e46a2003 100644 --- a/src/test/org/apache/nutch/parse/TestOutlinks.java +++ b/src/test/org/apache/nutch/parse/TestOutlinks.java @@ -16,12 +16,12 @@ */ package org.apache.nutch.parse; -import org.junit.Test; +import org.junit.jupiter.api.Test; import java.util.HashSet; import java.util.Set; -import static org.junit.Assert.*; +import static org.junit.jupiter.api.Assertions.assertEquals; public class TestOutlinks { @@ -33,7 +33,7 @@ public void testAddSameObject() throws Exception { set.add(o); set.add(o); - assertEquals("Adding the same Outlink twice", 1, set.size()); + assertEquals(1, set.size(), "Adding the same Outlink twice"); } @Test @@ -43,11 +43,11 @@ public void testAddOtherObjectWithSameData() throws Exception { Outlink o = new Outlink("http://www.example.com", "Example"); Outlink o1 = new Outlink("http://www.example.com", "Example"); - assertTrue("The two Outlink objects are the same", o.equals(o1)); + assertEquals(o, o1, "The two Outlink objects are the same"); set.add(o); set.add(o1); - assertEquals("The set should contain only 1 Outlink", 1, set.size()); + assertEquals(1, set.size(), "The set should contain only 1 Outlink"); } } diff --git a/src/test/org/apache/nutch/parse/TestParseData.java b/src/test/org/apache/nutch/parse/TestParseData.java index 0dbbf78cd7..c4f12cc3c8 100644 --- a/src/test/org/apache/nutch/parse/TestParseData.java +++ b/src/test/org/apache/nutch/parse/TestParseData.java @@ -18,8 +18,9 @@ import org.apache.nutch.util.WritableTestUtils; import org.apache.nutch.metadata.Metadata; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertEquals; /** Unit tests for ParseData. */ @@ -52,6 +53,6 @@ public void testMaxOutlinks() throws Exception { ParseData original = new ParseData(ParseStatus.STATUS_SUCCESS, "Max Outlinks Title", outlinks, new Metadata()); ParseData data = (ParseData) WritableTestUtils.writeRead(original, null); - Assert.assertEquals(outlinks.length, data.getOutlinks().length); + assertEquals(outlinks.length, data.getOutlinks().length); } } diff --git a/src/test/org/apache/nutch/parse/TestParseSegment.java b/src/test/org/apache/nutch/parse/TestParseSegment.java index dd7f4f9202..d989c7a0b2 100644 --- a/src/test/org/apache/nutch/parse/TestParseSegment.java +++ b/src/test/org/apache/nutch/parse/TestParseSegment.java @@ -16,15 +16,14 @@ */ package org.apache.nutch.parse; -import static junit.framework.TestCase.assertFalse; -import static junit.framework.TestCase.assertTrue; - import java.nio.charset.StandardCharsets; import org.apache.nutch.metadata.Metadata; import org.apache.nutch.net.protocols.Response; import org.apache.nutch.protocol.Content; -import org.junit.Test; +import org.junit.jupiter.api.Test; +import static org.junit.jupiter.api.Assertions.assertFalse; +import static org.junit.jupiter.api.Assertions.assertTrue; public class TestParseSegment { private static byte[] BYTES = "the quick brown fox".getBytes(StandardCharsets.UTF_8); diff --git a/src/test/org/apache/nutch/parse/TestParseText.java b/src/test/org/apache/nutch/parse/TestParseText.java index 3873632992..56418a1fc8 100644 --- a/src/test/org/apache/nutch/parse/TestParseText.java +++ b/src/test/org/apache/nutch/parse/TestParseText.java @@ -17,7 +17,7 @@ package org.apache.nutch.parse; import org.apache.nutch.util.WritableTestUtils; -import org.junit.Test; +import org.junit.jupiter.api.Test; /** Unit tests for ParseText. */ diff --git a/src/test/org/apache/nutch/parse/TestParserFactory.java b/src/test/org/apache/nutch/parse/TestParserFactory.java index c996ef76af..228b41972e 100644 --- a/src/test/org/apache/nutch/parse/TestParserFactory.java +++ b/src/test/org/apache/nutch/parse/TestParserFactory.java @@ -20,9 +20,11 @@ import org.apache.nutch.plugin.Extension; import org.apache.hadoop.conf.Configuration; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Before; -import org.junit.Test; +import org.junit.jupiter.api.Test; +import org.junit.jupiter.api.BeforeEach; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertNotNull; /** * Unit test for new parse plugin selection. @@ -36,7 +38,7 @@ public class TestParserFactory { private ParserFactory parserFactory; /** Inits the Test Case with the test parse-plugin file */ - @Before + @BeforeEach public void setUp() throws Exception { conf = NutchConfiguration.create(); conf.set("plugin.includes", ".*"); @@ -49,55 +51,54 @@ public void setUp() throws Exception { @Test public void testGetExtensions() throws Exception { Extension ext = parserFactory.getExtensions("text/html").get(0); - Assert.assertEquals("parse-tika", ext.getDescriptor().getPluginId()); + assertEquals("parse-tika", ext.getDescriptor().getPluginId()); ext = parserFactory.getExtensions("text/html; charset=ISO-8859-1").get(0); - Assert.assertEquals("parse-tika", ext.getDescriptor().getPluginId()); + assertEquals("parse-tika", ext.getDescriptor().getPluginId()); ext = parserFactory.getExtensions("foo/bar").get(0); - Assert.assertEquals("parse-tika", ext.getDescriptor().getPluginId()); + assertEquals("parse-tika", ext.getDescriptor().getPluginId()); } /** Unit test to check getParsers method */ @Test public void testGetParsers() throws Exception { Parser[] parsers = parserFactory.getParsers("text/html", "http://foo.com"); - Assert.assertNotNull(parsers); - Assert.assertEquals(1, parsers.length); - Assert.assertEquals("org.apache.nutch.parse.tika.TikaParser", parsers[0] + assertNotNull(parsers); + assertEquals(1, parsers.length); + assertEquals("org.apache.nutch.parse.tika.TikaParser", parsers[0] .getClass().getName()); parsers = parserFactory.getParsers("text/html; charset=ISO-8859-1", "http://foo.com"); - Assert.assertNotNull(parsers); - Assert.assertEquals(1, parsers.length); - Assert.assertEquals("org.apache.nutch.parse.tika.TikaParser", parsers[0] + assertNotNull(parsers); + assertEquals(1, parsers.length); + assertEquals("org.apache.nutch.parse.tika.TikaParser", parsers[0] .getClass().getName()); parsers = parserFactory.getParsers("application/x-javascript", "http://foo.com"); - Assert.assertNotNull(parsers); - Assert.assertEquals(1, parsers.length); - Assert.assertEquals("org.apache.nutch.parse.js.JSParseFilter", parsers[0] + assertNotNull(parsers); + assertEquals(1, parsers.length); + assertEquals("org.apache.nutch.parse.js.JSParseFilter", parsers[0] .getClass().getName()); parsers = parserFactory.getParsers("text/plain", "http://foo.com"); - Assert.assertNotNull(parsers); - Assert.assertEquals(1, parsers.length); - Assert.assertEquals("org.apache.nutch.parse.tika.TikaParser", parsers[0] + assertNotNull(parsers); + assertEquals(1, parsers.length); + assertEquals("org.apache.nutch.parse.tika.TikaParser", parsers[0] .getClass().getName()); Parser parser1 = parserFactory.getParsers("text/plain", "http://foo.com")[0]; Parser parser2 = parserFactory.getParsers("*", "http://foo.com")[0]; - Assert.assertEquals("Different instances!", parser1.hashCode(), - parser2.hashCode()); + assertEquals(parser1.hashCode(), parser2.hashCode(), "Different instances!"); // test and make sure that the rss parser is loaded even though its // plugin.xml // doesn't claim to support text/rss, only application/rss+xml parsers = parserFactory.getParsers("text/rss", "http://foo.com"); - Assert.assertNotNull(parsers); - Assert.assertEquals(1, parsers.length); - Assert.assertEquals("org.apache.nutch.parse.tika.TikaParser", parsers[0] + assertNotNull(parsers); + assertEquals(1, parsers.length); + assertEquals("org.apache.nutch.parse.tika.TikaParser", parsers[0] .getClass().getName()); } diff --git a/src/test/org/apache/nutch/plugin/TestPluginSystem.java b/src/test/org/apache/nutch/plugin/TestPluginSystem.java index 7c1362aa56..049c49adf2 100644 --- a/src/test/org/apache/nutch/plugin/TestPluginSystem.java +++ b/src/test/org/apache/nutch/plugin/TestPluginSystem.java @@ -28,10 +28,12 @@ import org.apache.hadoop.conf.Configuration; import org.apache.hadoop.mapreduce.Job; import org.apache.nutch.util.NutchConfiguration; -import org.junit.After; -import org.junit.Assert; -import org.junit.Before; -import org.junit.Test; + +import org.junit.jupiter.api.AfterEach; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.*; /** * Unit tests for the plugin system @@ -43,7 +45,7 @@ public class TestPluginSystem { private Configuration conf; private PluginRepository repository; - @Before + @BeforeEach public void setUp() throws Exception { this.conf = NutchConfiguration.create(); conf.set("plugin.includes", ".*"); @@ -54,12 +56,7 @@ public void setUp() throws Exception { this.repository = PluginRepository.get(conf); } - /* - * (non-Javadoc) - * - * @see junit.framework.TestCase#tearDown() - */ - @After + @AfterEach public void tearDown() throws Exception { for (int i = 0; i < fFolders.size(); i++) { File folder = fFolders.get(i); @@ -77,7 +74,7 @@ public void testPluginConfiguration() { if (!file.exists()) { file.mkdir(); } - Assert.assertTrue(file.exists()); + assertTrue(file.exists()); } /** @@ -86,14 +83,14 @@ public void testPluginConfiguration() { public void testLoadPlugins() { PluginDescriptor[] descriptors = repository.getPluginDescriptors(); int k = descriptors.length; - Assert.assertTrue(fPluginCount <= k); + assertTrue(fPluginCount <= k); for (int i = 0; i < descriptors.length; i++) { PluginDescriptor descriptor = descriptors[i]; if (!descriptor.getPluginId().startsWith("getPluginFolder()")) { continue; } - Assert.assertEquals(1, descriptor.getExportedLibUrls().length); - Assert.assertEquals(1, descriptor.getNotExportedLibUrls().length); + assertEquals(1, descriptor.getExportedLibUrls().length); + assertEquals(1, descriptor.getNotExportedLibUrls().length); } } @@ -104,7 +101,7 @@ public void testRepositoryCache() throws IOException { Job job = Job.getInstance(config); config = job.getConfiguration(); PluginRepository repo1 = PluginRepository.get(config); - Assert.assertTrue(repo == repo1); + assertSame(repo, repo1); // now construct a config without UUID config = new Configuration(); config.addResource("nutch-default.xml"); @@ -113,7 +110,7 @@ public void testRepositoryCache() throws IOException { job = Job.getInstance(config); config = job.getConfiguration(); repo1 = PluginRepository.get(config); - Assert.assertTrue(repo1 != repo); + assertNotSame(repo1, repo); } /** @@ -123,14 +120,14 @@ public void testRepositoryCache() throws IOException { public void testGetExtensionAndAttributes() { String xpId = " sdsdsd"; ExtensionPoint extensionPoint = repository.getExtensionPoint(xpId); - Assert.assertEquals(extensionPoint, null); + assertNull(extensionPoint); Extension[] extension1 = repository.getExtensionPoint(getGetExtensionId()) .getExtensions(); - Assert.assertEquals(extension1.length, fPluginCount); + assertEquals(extension1.length, fPluginCount); for (int i = 0; i < extension1.length; i++) { Extension extension2 = extension1[i]; String string = extension2.getAttribute(getGetConfigElementName()); - Assert.assertEquals(string, getParameterValue()); + assertEquals(string, getParameterValue()); } } @@ -141,15 +138,15 @@ public void testGetExtensionAndAttributes() { public void testGetExtensionInstances() throws PluginRuntimeException { Extension[] extensions = repository.getExtensionPoint(getGetExtensionId()) .getExtensions(); - Assert.assertEquals(extensions.length, fPluginCount); + assertEquals(extensions.length, fPluginCount); for (int i = 0; i < extensions.length; i++) { Extension extension = extensions[i]; Object object = extension.getExtensionInstance(); if (!(object instanceof HelloWorldExtension)) - Assert.fail(" object is not a instance of HelloWorldExtension"); + fail(" object is not a instance of HelloWorldExtension"); ((ITestExtension) object).testGetExtension("Bla "); String string = ((ITestExtension) object).testGetExtension("Hello"); - Assert.assertEquals("Hello World", string); + assertEquals("Hello World", string); } } @@ -162,7 +159,7 @@ public void testGetClassLoader() { PluginDescriptor[] descriptors = repository.getPluginDescriptors(); for (int i = 0; i < descriptors.length; i++) { PluginDescriptor descriptor = descriptors[i]; - Assert.assertNotNull(descriptor.getClassLoader()); + assertNotNull(descriptor.getClassLoader()); } } @@ -178,9 +175,9 @@ public void testGetResources() throws IOException { continue; } String value = descriptor.getResourceString("key", Locale.UK); - Assert.assertEquals("value", value); + assertEquals("value", value); value = descriptor.getResourceString("key", Locale.TRADITIONAL_CHINESE); - Assert.assertEquals("value", value); + assertEquals("value", value); } } @@ -191,7 +188,7 @@ public void testGetResources() throws IOException { private String getPluginFolder() { String[] strings = conf.getStrings("plugin.folders"); if (strings == null || strings.length == 0) - Assert.fail("no plugin directory setuped.."); + fail("no plugin directory setuped.."); String name = strings[0]; return new PluginManifestParser(conf, this.repository) diff --git a/src/test/org/apache/nutch/protocol/AbstractHttpProtocolPluginTest.java b/src/test/org/apache/nutch/protocol/AbstractHttpProtocolPluginTest.java index be26d9147e..17b87ebe22 100644 --- a/src/test/org/apache/nutch/protocol/AbstractHttpProtocolPluginTest.java +++ b/src/test/org/apache/nutch/protocol/AbstractHttpProtocolPluginTest.java @@ -41,8 +41,8 @@ import org.apache.nutch.crawl.CrawlDatum; import org.apache.nutch.metadata.Nutch; import org.apache.nutch.net.protocols.Response; -import org.junit.After; -import org.junit.Before; +import org.junit.jupiter.api.AfterEach; +import org.junit.jupiter.api.BeforeEach; import org.slf4j.Logger; import org.slf4j.LoggerFactory; @@ -94,7 +94,7 @@ public abstract class AbstractHttpProtocolPluginTest { protected abstract String getPluginClassName(); - @Before + @BeforeEach public void setUp() throws Exception { conf = new Configuration(); conf.addResource("nutch-default.xml"); @@ -110,7 +110,7 @@ public void setUp() throws Exception { http.setConf(conf); } - @After + @AfterEach public void tearDown() throws Exception { if (server != null) { server.close(); diff --git a/src/test/org/apache/nutch/protocol/TestContent.java b/src/test/org/apache/nutch/protocol/TestContent.java index e6a2a0e85e..fa5c7d17da 100644 --- a/src/test/org/apache/nutch/protocol/TestContent.java +++ b/src/test/org/apache/nutch/protocol/TestContent.java @@ -22,8 +22,12 @@ import org.apache.nutch.util.NutchConfiguration; import org.apache.nutch.util.WritableTestUtils; import org.apache.tika.mime.MimeTypes; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import java.nio.charset.StandardCharsets; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertNotNull; /** Unit tests for Content. */ @@ -46,9 +50,9 @@ public void testContent() throws Exception { metaData, conf); WritableTestUtils.testWritable(r); - Assert.assertEquals("text/html", r.getMetadata().get("Content-Type")); - Assert.assertEquals("text/html", r.getMetadata().get("content-type")); - Assert.assertEquals("text/html", r.getMetadata().get("CONTENTYPE")); + assertEquals("text/html", r.getMetadata().get("Content-Type")); + assertEquals("text/html", r.getMetadata().get("content-type")); + assertEquals("text/html", r.getMetadata().get("CONTENTYPE")); } /** Unit tests for getContentType(String, String, byte[]) method. */ @@ -58,36 +62,36 @@ public void testGetContentType() throws Exception { Metadata p = new Metadata(); c = new Content("http://www.foo.com/", "http://www.foo.com/", - "".getBytes("UTF8"), "text/html; charset=UTF-8", p, conf); - Assert.assertEquals("text/html", c.getContentType()); + "".getBytes(StandardCharsets.UTF_8), "text/html; charset=UTF-8", p, conf); + assertEquals("text/html", c.getContentType()); c = new Content("http://www.foo.com/foo.html", "http://www.foo.com/", - "".getBytes("UTF8"), "", p, conf); - Assert.assertEquals("text/html", c.getContentType()); + "".getBytes(StandardCharsets.UTF_8), "", p, conf); + assertEquals("text/html", c.getContentType()); c = new Content("http://www.foo.com/foo.html", "http://www.foo.com/", - "".getBytes("UTF8"), null, p, conf); - Assert.assertEquals("text/html", c.getContentType()); + "".getBytes(StandardCharsets.UTF_8), null, p, conf); + assertEquals("text/html", c.getContentType()); c = new Content("http://www.foo.com/", "http://www.foo.com/", - "".getBytes("UTF8"), "", p, conf); - Assert.assertEquals("text/html", c.getContentType()); + "".getBytes(StandardCharsets.UTF_8), "", p, conf); + assertEquals("text/html", c.getContentType()); c = new Content("http://www.foo.com/foo.html", "http://www.foo.com/", - "".getBytes("UTF8"), "text/plain", p, conf); - Assert.assertEquals("text/html", c.getContentType()); + "".getBytes(StandardCharsets.UTF_8), "text/plain", p, conf); + assertEquals("text/html", c.getContentType()); c = new Content("http://www.foo.com/foo.png", "http://www.foo.com/", - "".getBytes("UTF8"), "text/plain", p, conf); - Assert.assertEquals("text/html", c.getContentType()); + "".getBytes(StandardCharsets.UTF_8), "text/plain", p, conf); + assertEquals("text/html", c.getContentType()); c = new Content("http://www.foo.com/", "http://www.foo.com/", - "".getBytes("UTF8"), "", p, conf); - Assert.assertEquals(MimeTypes.OCTET_STREAM, c.getContentType()); + "".getBytes(StandardCharsets.UTF_8), "", p, conf); + assertEquals(MimeTypes.OCTET_STREAM, c.getContentType()); c = new Content("http://www.foo.com/", "http://www.foo.com/", - "".getBytes("UTF8"), null, p, conf); - Assert.assertNotNull(c.getContentType()); + "".getBytes(StandardCharsets.UTF_8), null, p, conf); + assertNotNull(c.getContentType()); } } diff --git a/src/test/org/apache/nutch/protocol/TestProtocolFactory.java b/src/test/org/apache/nutch/protocol/TestProtocolFactory.java index 2266b084e9..b6639f304d 100644 --- a/src/test/org/apache/nutch/protocol/TestProtocolFactory.java +++ b/src/test/org/apache/nutch/protocol/TestProtocolFactory.java @@ -18,16 +18,17 @@ import org.apache.hadoop.conf.Configuration; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Before; -import org.junit.Test; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.*; public class TestProtocolFactory { Configuration conf; ProtocolFactory factory; - @Before + @BeforeEach public void setUp() throws Exception { conf = NutchConfiguration.create(); conf.set("plugin.includes", ".*"); @@ -41,11 +42,11 @@ public void testGetProtocol() { // non existing protocol try { factory.getProtocol("xyzxyz://somehost"); - Assert.fail("Must throw ProtocolNotFound"); + fail("Must throw ProtocolNotFound"); } catch (ProtocolNotFound e) { // all is ok } catch (Exception ex) { - Assert.fail("Must not throw any other exception"); + fail("Must not throw any other exception"); } Protocol httpProtocol = null; @@ -53,26 +54,26 @@ public void testGetProtocol() { // existing protocol try { httpProtocol = factory.getProtocol("http://somehost"); - Assert.assertNotNull(httpProtocol); + assertNotNull(httpProtocol); } catch (Exception ex) { - Assert.fail("Must not throw any other exception"); + fail("Must not throw any other exception"); } // test same object instance try { - Assert.assertTrue(httpProtocol == factory.getProtocol("http://somehost")); + assertSame(httpProtocol, factory.getProtocol("http://somehost")); } catch (ProtocolNotFound e) { - Assert.fail("Must not throw any exception"); + fail("Must not throw any exception"); } } @Test public void testContains() { - Assert.assertTrue(factory.contains("http", "http")); - Assert.assertTrue(factory.contains("http", "http,ftp")); - Assert.assertTrue(factory.contains("http", " http , ftp")); - Assert.assertTrue(factory.contains("smb", "ftp,smb,http")); - Assert.assertFalse(factory.contains("smb", "smbb")); + assertTrue(factory.contains("http", "http")); + assertTrue(factory.contains("http", "http,ftp")); + assertTrue(factory.contains("http", " http , ftp")); + assertTrue(factory.contains("smb", "ftp,smb,http")); + assertFalse(factory.contains("smb", "smbb")); } } diff --git a/src/test/org/apache/nutch/segment/TestSegmentMerger.java b/src/test/org/apache/nutch/segment/TestSegmentMerger.java index e705188738..0df88a2de6 100644 --- a/src/test/org/apache/nutch/segment/TestSegmentMerger.java +++ b/src/test/org/apache/nutch/segment/TestSegmentMerger.java @@ -29,12 +29,14 @@ import org.apache.hadoop.mapreduce.lib.output.MapFileOutputFormat; import org.apache.nutch.parse.ParseText; import org.apache.nutch.util.NutchConfiguration; -import org.junit.After; -import org.junit.Assert; -import org.junit.Assume; -import org.junit.Before; -import org.junit.BeforeClass; -import org.junit.Test; +import org.junit.jupiter.api.AfterEach; +import org.junit.jupiter.api.BeforeAll; +import org.junit.jupiter.api.Test; +import org.junit.jupiter.api.BeforeEach; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertTrue; +import static org.junit.jupiter.api.Assumptions.assumeTrue; public class TestSegmentMerger { Configuration conf; @@ -45,12 +47,12 @@ public class TestSegmentMerger { Path out; int countSeg1, countSeg2; - @BeforeClass + @BeforeAll public static void checkConditions() throws Exception { - Assume.assumeTrue(Boolean.getBoolean("test.include.slow")); + assumeTrue(Boolean.getBoolean("test.include.slow")); } - @Before + @BeforeEach public void setUp() throws Exception { conf = NutchConfiguration.create(); fs = FileSystem.get(conf); @@ -98,7 +100,7 @@ public void setUp() throws Exception { System.err.println(" - done: " + countSeg2 + " records."); } - @After + @AfterEach public void tearDown() throws Exception { fs.delete(testDir, true); } @@ -110,7 +112,7 @@ public void testLargeMerge() throws Exception { // verify output FileStatus[] stats = fs.listStatus(out); // there should be just one path - Assert.assertEquals(1, stats.length); + assertEquals(1, stats.length); Path outSeg = stats[0].getPath(); Text k = new Text(); ParseText v = new ParseText(); @@ -123,16 +125,16 @@ public void testLargeMerge() throws Exception { String vs = v.getText(); if (ks.startsWith("seg1-")) { cnt1++; - Assert.assertTrue(vs.startsWith("seg1 ")); + assertTrue(vs.startsWith("seg1 ")); } else if (ks.startsWith("seg2-")) { cnt2++; - Assert.assertTrue(vs.startsWith("seg2 ")); + assertTrue(vs.startsWith("seg2 ")); } } r.close(); } - Assert.assertEquals(countSeg1, cnt1); - Assert.assertEquals(countSeg2, cnt2); + assertEquals(countSeg1, cnt1); + assertEquals(countSeg2, cnt2); } } diff --git a/src/test/org/apache/nutch/segment/TestSegmentMergerCrawlDatums.java b/src/test/org/apache/nutch/segment/TestSegmentMergerCrawlDatums.java index 8cc4744a26..cd0db1e50a 100644 --- a/src/test/org/apache/nutch/segment/TestSegmentMergerCrawlDatums.java +++ b/src/test/org/apache/nutch/segment/TestSegmentMergerCrawlDatums.java @@ -31,14 +31,15 @@ import org.apache.hadoop.mapreduce.lib.output.MapFileOutputFormat; import org.apache.nutch.crawl.CrawlDatum; import org.apache.nutch.util.NutchConfiguration; -import org.junit.Assert; -import org.junit.Assume; -import org.junit.Before; -import org.junit.BeforeClass; -import org.junit.Test; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.BeforeAll; +import org.junit.jupiter.api.Test; import org.slf4j.Logger; import org.slf4j.LoggerFactory; +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assumptions.assumeTrue; + /** * New SegmentMerger unit test focusing on several crappy issues with the * segment merger. The general problem is disappearing records and incorrect @@ -63,12 +64,12 @@ public class TestSegmentMergerCrawlDatums { private static final Logger LOG = LoggerFactory .getLogger(MethodHandles.lookup().lookupClass()); - @BeforeClass + @BeforeAll public static void checkConditions() throws Exception { - Assume.assumeTrue(Boolean.getBoolean("test.include.slow")); + assumeTrue(Boolean.getBoolean("test.include.slow")); } - @Before + @BeforeEach public void setUp() throws Exception { conf = NutchConfiguration.create(); fs = FileSystem.get(conf); @@ -80,7 +81,7 @@ public void setUp() throws Exception { */ @Test public void testSingleRandomSequence() throws Exception { - Assert.assertEquals( + assertEquals( Byte.valueOf(CrawlDatum.STATUS_FETCH_SUCCESS), Byte.valueOf(executeSequence(CrawlDatum.STATUS_FETCH_GONE, CrawlDatum.STATUS_FETCH_SUCCESS, 256, false))); @@ -118,7 +119,7 @@ public void testMostlyRedirects() throws Exception { segment3, segment4, segment5, segment6, segment7, segment8 }); Byte status = Byte.valueOf(status = checkMergedSegment(testDir, mergedSegment)); - Assert.assertEquals(Byte.valueOf(CrawlDatum.STATUS_FETCH_SUCCESS), status); + assertEquals(Byte.valueOf(CrawlDatum.STATUS_FETCH_SUCCESS), status); } /** @@ -139,12 +140,11 @@ public void testRandomizedSequences() throws Exception { byte resultStatus = executeSequence(randomStatus, expectedStatus, rounds, withRedirects); - Assert.assertEquals( + assertEquals(expectedStatus, resultStatus, "Expected status = " + CrawlDatum.getStatusName(expectedStatus) + ", but got " + CrawlDatum.getStatusName(resultStatus) + " when merging " + rounds + " segments" - + (withRedirects ? " with redirects" : ""), expectedStatus, - resultStatus); + + (withRedirects ? " with redirects" : "")); } } @@ -153,7 +153,7 @@ public void testRandomizedSequences() throws Exception { */ @Test public void testRandomTestSequenceWithRedirects() throws Exception { - Assert.assertEquals( + assertEquals( Byte.valueOf(CrawlDatum.STATUS_FETCH_SUCCESS), Byte.valueOf(executeSequence(CrawlDatum.STATUS_FETCH_GONE, CrawlDatum.STATUS_FETCH_SUCCESS, 128, true))); @@ -181,7 +181,7 @@ public void testFixedSequence() throws Exception { segment3 }); Byte status = Byte.valueOf(status = checkMergedSegment(testDir, mergedSegment)); - Assert.assertEquals(Byte.valueOf(CrawlDatum.STATUS_FETCH_SUCCESS), status); + assertEquals(Byte.valueOf(CrawlDatum.STATUS_FETCH_SUCCESS), status); } /** @@ -201,7 +201,7 @@ public void testRedirFetchInOneSegment() throws Exception { Path mergedSegment = merge(testDir, new Path[] { segment }); Byte status = Byte.valueOf(status = checkMergedSegment(testDir, mergedSegment)); - Assert.assertEquals(Byte.valueOf(CrawlDatum.STATUS_FETCH_SUCCESS), status); + assertEquals(Byte.valueOf(CrawlDatum.STATUS_FETCH_SUCCESS), status); } /** @@ -223,7 +223,7 @@ public void testEndsWithRedirect() throws Exception { Path mergedSegment = merge(testDir, new Path[] { segment1, segment2 }); Byte status = Byte.valueOf(status = checkMergedSegment(testDir, mergedSegment)); - Assert.assertEquals(Byte.valueOf(CrawlDatum.STATUS_FETCH_SUCCESS), status); + assertEquals(Byte.valueOf(CrawlDatum.STATUS_FETCH_SUCCESS), status); } /** @@ -349,7 +349,7 @@ protected Path merge(Path testDir, Path[] segments) throws Exception { merger.merge(out, segments, false, false, -1); FileStatus[] stats = fs.listStatus(out); - Assert.assertEquals(1, stats.length); + assertEquals(1, stats.length); return stats[0].getPath(); } diff --git a/src/test/org/apache/nutch/service/TestNutchServer.java b/src/test/org/apache/nutch/service/TestNutchServer.java index 811285faf5..11397a9b12 100644 --- a/src/test/org/apache/nutch/service/TestNutchServer.java +++ b/src/test/org/apache/nutch/service/TestNutchServer.java @@ -20,7 +20,7 @@ import javax.ws.rs.core.Response; import org.apache.cxf.jaxrs.client.WebClient; -import org.junit.Test; +import org.junit.jupiter.api.Test; import org.slf4j.Logger; import org.slf4j.LoggerFactory; diff --git a/src/test/org/apache/nutch/tools/TestCommonCrawlDataDumper.java b/src/test/org/apache/nutch/tools/TestCommonCrawlDataDumper.java index 547007aea3..fee72b65a5 100644 --- a/src/test/org/apache/nutch/tools/TestCommonCrawlDataDumper.java +++ b/src/test/org/apache/nutch/tools/TestCommonCrawlDataDumper.java @@ -16,15 +16,15 @@ */ package org.apache.nutch.tools; -import static org.junit.Assert.assertTrue; - import java.io.File; import java.nio.file.Files; import java.util.Collection; import org.apache.commons.io.FileUtils; import org.apache.commons.io.filefilter.FileFilterUtils; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertTrue; /** * @@ -100,8 +100,8 @@ public void testDump() throws Exception { FileFilterUtils.directoryFileFilter()); for (String expectedFileName : crawledFiles) { - assertTrue("Missed file " + expectedFileName + " in dump", - hasFile(expectedFileName, tempFiles)); + assertTrue(hasFile(expectedFileName, tempFiles), + "Missed file " + expectedFileName + " in dump"); } } diff --git a/src/test/org/apache/nutch/util/DumpFileUtilTest.java b/src/test/org/apache/nutch/util/DumpFileUtilTest.java index 249d9783f6..546ea7e6bc 100644 --- a/src/test/org/apache/nutch/util/DumpFileUtilTest.java +++ b/src/test/org/apache/nutch/util/DumpFileUtilTest.java @@ -16,9 +16,11 @@ */ package org.apache.nutch.util; -import org.junit.Test; +import org.junit.jupiter.api.Test; -import static org.junit.Assert.*; +import static org.hamcrest.CoreMatchers.is; +import static org.hamcrest.CoreMatchers.nullValue; +import static org.hamcrest.MatcherAssert.assertThat; public class DumpFileUtilTest { @@ -28,7 +30,7 @@ public void testGetUrlMD5() throws Exception { String result = DumpFileUtil.getUrlMD5(testUrl); - assertEquals("991e599262e04ea2ec76b6c5aed499a7", result); + assertThat(result, is("991e599262e04ea2ec76b6c5aed499a7")); } @Test @@ -37,12 +39,12 @@ public void testCreateTwoLevelsDirectory() throws Exception { String basePath = "/tmp"; String fullDir = DumpFileUtil.createTwoLevelsDirectory(basePath, DumpFileUtil.getUrlMD5(testUrl)); - assertEquals("/tmp/96/ea", fullDir); + assertThat(fullDir, is("/tmp/96/ea")); String basePath2 = "/this/path/is/not/existed/just/for/testing"; String fullDir2 = DumpFileUtil.createTwoLevelsDirectory(basePath2, DumpFileUtil.getUrlMD5(testUrl)); - assertNull(fullDir2); + assertThat(fullDir2, nullValue()); } @Test @@ -52,16 +54,17 @@ public void testCreateFileName() throws Exception { String extension = "html"; String fullDir = DumpFileUtil.createFileName(DumpFileUtil.getUrlMD5(testUrl), baseName, extension); - assertEquals("991e599262e04ea2ec76b6c5aed499a7_test.html", fullDir); + assertThat(fullDir, is("991e599262e04ea2ec76b6c5aed499a7_test.html")); String tooLongBaseName = "testtesttesttesttesttesttesttesttesttesttesttesttesttesttesttesttesttest"; String fullDir2 = DumpFileUtil.createFileName(DumpFileUtil.getUrlMD5(testUrl), tooLongBaseName, extension); - assertEquals("991e599262e04ea2ec76b6c5aed499a7_testtesttesttesttesttesttesttest.html", fullDir2); + assertThat(fullDir2, + is("991e599262e04ea2ec76b6c5aed499a7_testtesttesttesttesttesttesttest.html")); String tooLongExtension = "testtesttesttesttesttesttesttesttesttesttesttesttesttesttesttesttesttest"; String fullDir3 = DumpFileUtil.createFileName(DumpFileUtil.getUrlMD5(testUrl), baseName, tooLongExtension); - assertEquals("991e599262e04ea2ec76b6c5aed499a7_test.testt", fullDir3); + assertThat(fullDir3, is("991e599262e04ea2ec76b6c5aed499a7_test.testt")); } } diff --git a/src/test/org/apache/nutch/util/TestEncodingDetector.java b/src/test/org/apache/nutch/util/TestEncodingDetector.java index 8697a62f08..153b026b05 100644 --- a/src/test/org/apache/nutch/util/TestEncodingDetector.java +++ b/src/test/org/apache/nutch/util/TestEncodingDetector.java @@ -22,8 +22,9 @@ import org.apache.nutch.metadata.Metadata; import org.apache.nutch.net.protocols.Response; import org.apache.nutch.protocol.Content; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertEquals; public class TestEncodingDetector { private static Configuration conf = NutchConfiguration.create(); @@ -54,7 +55,7 @@ public void testGuessing() { detector.autoDetectClues(content, true); encoding = detector.guessEncoding(content, "windows-1252"); // no information is available, so it should return default encoding - Assert.assertEquals("windows-1252", encoding.toLowerCase()); + assertEquals("windows-1252", encoding.toLowerCase()); metadata.clear(); metadata.set(Response.CONTENT_TYPE, "text/plain; charset=UTF-16"); @@ -63,7 +64,7 @@ public void testGuessing() { detector = new EncodingDetector(conf); detector.autoDetectClues(content, true); encoding = detector.guessEncoding(content, "windows-1252"); - Assert.assertEquals("utf-16", encoding.toLowerCase()); + assertEquals("utf-16", encoding.toLowerCase()); metadata.clear(); content = new Content("http://www.example.com", "http://www.example.com/", @@ -72,7 +73,7 @@ public void testGuessing() { detector.autoDetectClues(content, true); detector.addClue("windows-1254", "sniffed"); encoding = detector.guessEncoding(content, "windows-1252"); - Assert.assertEquals("windows-1254", encoding.toLowerCase()); + assertEquals("windows-1254", encoding.toLowerCase()); // enable autodetection conf.setInt(EncodingDetector.MIN_CONFIDENCE_KEY, 50); @@ -84,7 +85,7 @@ public void testGuessing() { detector.autoDetectClues(content, true); detector.addClue("utf-32", "sniffed"); encoding = detector.guessEncoding(content, "windows-1252"); - Assert.assertEquals("utf-8", encoding.toLowerCase()); + assertEquals("utf-8", encoding.toLowerCase()); } } diff --git a/src/test/org/apache/nutch/util/TestGZIPUtils.java b/src/test/org/apache/nutch/util/TestGZIPUtils.java index fcda06c120..fb405c5ad4 100644 --- a/src/test/org/apache/nutch/util/TestGZIPUtils.java +++ b/src/test/org/apache/nutch/util/TestGZIPUtils.java @@ -18,8 +18,9 @@ import java.io.IOException; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.*; /** Unit tests for GZIPUtils methods. */ public class TestGZIPUtils { @@ -154,37 +155,37 @@ public void testLimit() { public void testZipUnzip(byte[] origBytes) { byte[] compressedBytes = GZIPUtils.zip(origBytes); - Assert.assertTrue("compressed array is not smaller!", - compressedBytes.length < origBytes.length); + assertTrue(compressedBytes.length < origBytes.length, + "compressed array is not smaller!"); byte[] uncompressedBytes = null; try { uncompressedBytes = GZIPUtils.unzip(compressedBytes); } catch (IOException e) { e.printStackTrace(); - Assert.assertTrue("caught exception '" + e + "' during unzip()", false); + fail("caught exception '" + e + "' during unzip()"); } - Assert.assertTrue("uncompressedBytes is wrong size", - uncompressedBytes.length == origBytes.length); + assertEquals(uncompressedBytes.length, origBytes.length, + "uncompressedBytes is wrong size"); for (int i = 0; i < origBytes.length; i++) if (origBytes[i] != uncompressedBytes[i]) - Assert.assertTrue("uncompressedBytes does not match origBytes", false); + fail("uncompressedBytes does not match origBytes"); } public void testZipUnzipBestEffort(byte[] origBytes) { byte[] compressedBytes = GZIPUtils.zip(origBytes); - Assert.assertTrue("compressed array is not smaller!", - compressedBytes.length < origBytes.length); + assertTrue(compressedBytes.length < origBytes.length, + "compressed array is not smaller!"); byte[] uncompressedBytes = GZIPUtils.unzipBestEffort(compressedBytes); - Assert.assertTrue("uncompressedBytes is wrong size", - uncompressedBytes.length == origBytes.length); + assertEquals(uncompressedBytes.length, origBytes.length, + "uncompressedBytes is wrong size"); for (int i = 0; i < origBytes.length; i++) if (origBytes[i] != uncompressedBytes[i]) - Assert.assertTrue("uncompressedBytes does not match origBytes", false); + fail("uncompressedBytes does not match origBytes"); } public void testTruncation(byte[] origBytes) { @@ -210,9 +211,8 @@ public void testTruncation(byte[] origBytes) { for (int j = 0; j < trunc.length; j++) if (trunc[j] != origBytes[j]) - Assert.assertTrue("truncated/uncompressed array differs at pos " - + j + " (compressed data had been truncated to len " + i + ")", - false); + fail("truncated/uncompressed array differs at pos " + j + + " (compressed data had been truncated to len " + i + ")"); } } } @@ -220,20 +220,19 @@ public void testTruncation(byte[] origBytes) { public void testLimit(byte[] origBytes) { byte[] compressedBytes = GZIPUtils.zip(origBytes); - Assert.assertTrue("compressed array is not smaller!", - compressedBytes.length < origBytes.length); + assertTrue(compressedBytes.length < origBytes.length, + "compressed array is not smaller!"); for (int i = 0; i < origBytes.length; i++) { byte[] uncompressedBytes = GZIPUtils.unzipBestEffort(compressedBytes, i); - Assert.assertTrue("uncompressedBytes is wrong size", - uncompressedBytes.length == i); + assertEquals(uncompressedBytes.length, i, + "uncompressedBytes is wrong size"); for (int j = 0; j < i; j++) if (origBytes[j] != uncompressedBytes[j]) - Assert - .assertTrue("uncompressedBytes does not match origBytes", false); + fail("uncompressedBytes does not match origBytes"); } } diff --git a/src/test/org/apache/nutch/util/TestMimeUtil.java b/src/test/org/apache/nutch/util/TestMimeUtil.java index 6ebe766698..c9b09226a3 100644 --- a/src/test/org/apache/nutch/util/TestMimeUtil.java +++ b/src/test/org/apache/nutch/util/TestMimeUtil.java @@ -24,9 +24,10 @@ import com.google.common.io.Files; -import junit.framework.TestCase; +import org.junit.jupiter.api.Test; +import static org.junit.jupiter.api.Assertions.assertEquals; -public class TestMimeUtil extends TestCase { +public class TestMimeUtil { public static String urlPrefix = "http://localhost/"; @@ -96,15 +97,17 @@ private String getMimeType(String url, byte[] bytes, String contentType, } /** use HTTP Content-Type, URL pattern, and MIME magic */ + @Test public void testWithMimeMagic() { for (String[] testPage : textBasedFormats) { String mimeType = getMimeType(urlPrefix, testPage[3].getBytes(defaultCharset), testPage[2], true); - assertEquals("", testPage[0], mimeType); + assertEquals(testPage[0], mimeType); } } /** use only HTTP Content-Type (if given) and URL pattern */ + @Test public void testWithoutMimeMagic() { for (String[] testPage : textBasedFormats) { if (testPage.length > 4 && "requires-mime-magic".equals(testPage[4])) { @@ -112,26 +115,28 @@ public void testWithoutMimeMagic() { } String mimeType = getMimeType(urlPrefix + testPage[1], testPage[3].getBytes(defaultCharset), testPage[2], false); - assertEquals("", testPage[0], mimeType); + assertEquals(testPage[0], mimeType); } } /** use only MIME magic (detection from content bytes) */ + @Test public void testOnlyMimeMagic() { for (String[] testPage : textBasedFormats) { String mimeType = getMimeType(urlPrefix, testPage[3].getBytes(defaultCharset), "", true); - assertEquals("", testPage[0], mimeType); + assertEquals(testPage[0], mimeType); } } /** test binary file formats (real files) */ + @Test public void testBinaryFiles() throws IOException { for (String[] testPage : binaryFiles) { File dataFile = new File(sampleDir, testPage[1]); String mimeType = getMimeType(urlPrefix + testPage[1], dataFile, testPage[2], false); - assertEquals("", testPage[0], mimeType); + assertEquals(testPage[0], mimeType); } } diff --git a/src/test/org/apache/nutch/util/TestNodeWalker.java b/src/test/org/apache/nutch/util/TestNodeWalker.java index 066bf1aade..892adaa1ba 100644 --- a/src/test/org/apache/nutch/util/TestNodeWalker.java +++ b/src/test/org/apache/nutch/util/TestNodeWalker.java @@ -19,12 +19,14 @@ import java.io.ByteArrayInputStream; import org.apache.xerces.parsers.DOMParser; -import org.junit.Assert; -import org.junit.Before; -import org.junit.Test; +import org.junit.jupiter.api.BeforeEach; +import org.junit.jupiter.api.Test; import org.w3c.dom.Node; import org.xml.sax.InputSource; +import static org.junit.jupiter.api.Assertions.assertFalse; +import static org.junit.jupiter.api.Assertions.assertTrue; + /** Unit tests for NodeWalker methods. */ public class TestNodeWalker { @@ -40,7 +42,7 @@ public class TestNodeWalker { private final static String[] ULCONTENT = new String[4]; - @Before + @BeforeEach public void setUp() throws Exception { ULCONTENT[0] = "crawl several billion pages per month"; ULCONTENT[1] = "maintain an index of these pages"; @@ -74,8 +76,8 @@ public void testSkipChildren() { sb.append(text); } } - Assert.assertTrue("UL Content can NOT be found in the node", - findSomeUlContent(sb.toString())); + assertTrue(findSomeUlContent(sb.toString()), + "UL Content can NOT be found in the node"); StringBuffer sbSkip = new StringBuffer(); NodeWalker walkerSkip = new NodeWalker(parser.getDocument()); @@ -92,8 +94,8 @@ public void testSkipChildren() { sbSkip.append(text); } } - Assert.assertFalse("UL Content can be found in the node", - findSomeUlContent(sbSkip.toString())); + assertFalse(findSomeUlContent(sbSkip.toString()), + "UL Content can be found in the node"); } public boolean findSomeUlContent(String str) { diff --git a/src/test/org/apache/nutch/util/TestPrefixStringMatcher.java b/src/test/org/apache/nutch/util/TestPrefixStringMatcher.java index f86070dd0b..82fca11b26 100644 --- a/src/test/org/apache/nutch/util/TestPrefixStringMatcher.java +++ b/src/test/org/apache/nutch/util/TestPrefixStringMatcher.java @@ -16,8 +16,10 @@ */ package org.apache.nutch.util; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertTrue; /** Unit tests for PrefixStringMatcher. */ public class TestPrefixStringMatcher { @@ -90,18 +92,16 @@ public void testPrefixMatcher() { numInputsTested++; - Assert.assertTrue("'" + input + "' should " + (matches ? "" : "not ") - + "match!", matches == prematcher.matches(input)); + assertEquals(matches, prematcher.matches(input), + "'" + input + "' should " + (matches ? "" : "not ") + "match!"); if (matches) { - Assert.assertTrue(shortestMatch == prematcher.shortestMatch(input) - .length()); - Assert.assertTrue(input.substring(0, shortestMatch).equals( - prematcher.shortestMatch(input))); - - Assert.assertTrue(longestMatch == prematcher.longestMatch(input) - .length()); - Assert.assertTrue(input.substring(0, longestMatch).equals( - prematcher.longestMatch(input))); + assertEquals(shortestMatch, prematcher.shortestMatch(input).length()); + assertEquals(input.substring(0, shortestMatch), + prematcher.shortestMatch(input)); + + assertEquals(longestMatch, prematcher.longestMatch(input).length()); + assertEquals(input.substring(0, longestMatch), + prematcher.longestMatch(input)); } } diff --git a/src/test/org/apache/nutch/util/TestStringUtil.java b/src/test/org/apache/nutch/util/TestStringUtil.java index 9b82912116..b42c45e917 100644 --- a/src/test/org/apache/nutch/util/TestStringUtil.java +++ b/src/test/org/apache/nutch/util/TestStringUtil.java @@ -18,8 +18,9 @@ import java.util.regex.Pattern; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.*; /** Unit tests for StringUtil methods. */ public class TestStringUtil { @@ -28,16 +29,16 @@ public void testRightPad() { String s = "my string"; String ps = StringUtil.rightPad(s, 0); - Assert.assertTrue(s.equals(ps)); + assertEquals(s, ps); ps = StringUtil.rightPad(s, 9); - Assert.assertTrue(s.equals(ps)); + assertEquals(s, ps); ps = StringUtil.rightPad(s, 10); - Assert.assertTrue((s + " ").equals(ps)); + assertEquals((s + " "), ps); ps = StringUtil.rightPad(s, 15); - Assert.assertTrue((s + " ").equals(ps)); + assertEquals((s + " "), ps); } @Test @@ -45,38 +46,38 @@ public void testLeftPad() { String s = "my string"; String ps = StringUtil.leftPad(s, 0); - Assert.assertTrue(s.equals(ps)); + assertEquals(s, ps); ps = StringUtil.leftPad(s, 9); - Assert.assertTrue(s.equals(ps)); + assertEquals(s, ps); ps = StringUtil.leftPad(s, 10); - Assert.assertTrue((" " + s).equals(ps)); + assertEquals((" " + s), ps); ps = StringUtil.leftPad(s, 15); - Assert.assertTrue((" " + s).equals(ps)); + assertEquals((" " + s), ps); } @Test public void testMaskPasswords() { String secret = "password"; String masked = StringUtil.mask(secret); - Assert.assertNotEquals(secret, masked); - Assert.assertEquals(secret.length(), masked.length()); + assertNotEquals(secret, masked); + assertEquals(secret.length(), masked.length()); char mask = 'X'; masked = StringUtil.mask(secret, mask); - Assert.assertNotEquals(secret, masked); - Assert.assertEquals(secret.length(), masked.length()); - masked.chars().forEach((c) -> Assert.assertEquals(mask, c)); + assertNotEquals(secret, masked); + assertEquals(secret.length(), masked.length()); + masked.chars().forEach((c) -> assertEquals(mask, c)); String strWithSecret = "amqp://username:password@example.org:5672/virtualHost"; Pattern maskPasswordPattern = Pattern.compile("^amqp://[^:]+:([^@]+)@"); masked = StringUtil.mask(strWithSecret, maskPasswordPattern, mask); - Assert.assertNotEquals(strWithSecret, masked); - Assert.assertEquals(strWithSecret.length(), masked.length()); - Assert.assertFalse(masked.contains(secret)); - Assert.assertTrue(masked.contains(StringUtil.mask(secret, mask))); + assertNotEquals(strWithSecret, masked); + assertEquals(strWithSecret.length(), masked.length()); + assertFalse(masked.contains(secret)); + assertTrue(masked.contains(StringUtil.mask(secret, mask))); } } diff --git a/src/test/org/apache/nutch/util/TestSuffixStringMatcher.java b/src/test/org/apache/nutch/util/TestSuffixStringMatcher.java index 104907cc59..6396afd98e 100644 --- a/src/test/org/apache/nutch/util/TestSuffixStringMatcher.java +++ b/src/test/org/apache/nutch/util/TestSuffixStringMatcher.java @@ -16,8 +16,9 @@ */ package org.apache.nutch.util; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertEquals; /** Unit tests for SuffixStringMatcher. */ public class TestSuffixStringMatcher { @@ -90,18 +91,16 @@ public void testSuffixMatcher() { numInputsTested++; - Assert.assertTrue("'" + input + "' should " + (matches ? "" : "not ") - + "match!", matches == sufmatcher.matches(input)); + assertEquals(matches, sufmatcher.matches(input), + "'" + input + "' should " + (matches ? "" : "not ") + "match!"); if (matches) { - Assert.assertTrue(shortestMatch == sufmatcher.shortestMatch(input) - .length()); - Assert.assertTrue(input.substring(input.length() - shortestMatch) - .equals(sufmatcher.shortestMatch(input))); - - Assert.assertTrue(longestMatch == sufmatcher.longestMatch(input) - .length()); - Assert.assertTrue(input.substring(input.length() - longestMatch) - .equals(sufmatcher.longestMatch(input))); + assertEquals(shortestMatch, sufmatcher.shortestMatch(input).length()); + assertEquals(input.substring(input.length() - shortestMatch), + sufmatcher.shortestMatch(input)); + + assertEquals(longestMatch, sufmatcher.longestMatch(input).length()); + assertEquals(input.substring(input.length() - longestMatch), + sufmatcher.longestMatch(input)); } } } diff --git a/src/test/org/apache/nutch/util/TestTableUtil.java b/src/test/org/apache/nutch/util/TestTableUtil.java index d31acdb13d..2a9e5df3be 100644 --- a/src/test/org/apache/nutch/util/TestTableUtil.java +++ b/src/test/org/apache/nutch/util/TestTableUtil.java @@ -16,8 +16,9 @@ */ package org.apache.nutch.util; -import org.junit.Test; -import static org.junit.Assert.*; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertEquals; public class TestTableUtil { diff --git a/src/test/org/apache/nutch/util/TestURLUtil.java b/src/test/org/apache/nutch/util/TestURLUtil.java index f8a0a88766..59e486d696 100644 --- a/src/test/org/apache/nutch/util/TestURLUtil.java +++ b/src/test/org/apache/nutch/util/TestURLUtil.java @@ -18,8 +18,9 @@ import java.net.URL; -import org.junit.Assert; -import org.junit.Test; +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.*; /** Test class for URLUtil */ public class TestURLUtil { @@ -30,60 +31,60 @@ public void testGetDomainName() throws Exception { URL url = null; url = new URL("http://lucene.apache.org/nutch"); - Assert.assertEquals("apache.org", URLUtil.getDomainName(url)); + assertEquals("apache.org", URLUtil.getDomainName(url)); // hostname with trailing dot url = new URL("https://lucene.apache.org./nutch"); - Assert.assertEquals("apache.org", URLUtil.getDomainName(url)); + assertEquals("apache.org", URLUtil.getDomainName(url)); url = new URL("http://en.wikipedia.org/wiki/Java_coffee"); - Assert.assertEquals("wikipedia.org", URLUtil.getDomainName(url)); + assertEquals("wikipedia.org", URLUtil.getDomainName(url)); url = new URL("http://140.211.11.130/foundation/contributing.html"); - Assert.assertEquals("140.211.11.130", URLUtil.getDomainName(url)); + assertEquals("140.211.11.130", URLUtil.getDomainName(url)); url = new URL("http://www.example.co.uk:8080/index.html"); - Assert.assertEquals("example.co.uk", URLUtil.getDomainName(url)); + assertEquals("example.co.uk", URLUtil.getDomainName(url)); url = new URL("http://com"); - Assert.assertEquals("com", URLUtil.getDomainName(url)); + assertEquals("com", URLUtil.getDomainName(url)); url = new URL("http://www.example.co.uk.com"); - Assert.assertEquals("uk.com", URLUtil.getDomainName(url)); + assertEquals("uk.com", URLUtil.getDomainName(url)); // "nn" is not a public suffix url = new URL("http://example.com.nn"); - Assert.assertEquals("example.com.nn", URLUtil.getDomainName(url)); + assertEquals("example.com.nn", URLUtil.getDomainName(url)); url = new URL("http://"); - Assert.assertEquals("", URLUtil.getDomainName(url)); + assertEquals("", URLUtil.getDomainName(url)); /* * "xyz" is an ICANN suffix since 2014, see * https://www.iana.org/domains/root/db/xyz.html */ url = new URL("http://www.edu.tr.xyz"); - Assert.assertEquals("tr.xyz", URLUtil.getDomainName(url)); + assertEquals("tr.xyz", URLUtil.getDomainName(url)); url = new URL("http://www.example.c.se"); - Assert.assertEquals("example.c.se", URLUtil.getDomainName(url)); + assertEquals("example.c.se", URLUtil.getDomainName(url)); // plc.co.im is listed as a domain suffix url = new URL("http://www.example.plc.co.im"); - Assert.assertEquals("example.plc.co.im", URLUtil.getDomainName(url)); + assertEquals("example.plc.co.im", URLUtil.getDomainName(url)); // 2000.hu is listed as a domain suffix url = new URL("http://www.example.2000.hu"); - Assert.assertEquals("example.2000.hu", URLUtil.getDomainName(url)); + assertEquals("example.2000.hu", URLUtil.getDomainName(url)); // test non-ascii url = new URL("http://www.example.商業.tw"); - Assert.assertEquals("example.商業.tw", URLUtil.getDomainName(url)); + assertEquals("example.商業.tw", URLUtil.getDomainName(url)); // test URL without host/authority url = new URL("file:/path/index.html"); - Assert.assertNotNull(URLUtil.getDomainName(url)); - Assert.assertEquals("", URLUtil.getDomainName(url)); + assertNotNull(URLUtil.getDomainName(url)); + assertEquals("", URLUtil.getDomainName(url)); } @Test @@ -91,58 +92,58 @@ public void testGetDomainSuffix() throws Exception { URL url = null; url = new URL("http://lucene.apache.org/nutch"); - Assert.assertEquals("org", URLUtil.getDomainSuffix(url)); + assertEquals("org", URLUtil.getDomainSuffix(url)); // hostname with trailing dot url = new URL("https://lucene.apache.org./nutch"); - Assert.assertEquals("org", URLUtil.getDomainSuffix(url)); + assertEquals("org", URLUtil.getDomainSuffix(url)); url = new URL("http://140.211.11.130/foundation/contributing.html"); - Assert.assertNull(URLUtil.getDomainSuffix(url)); + assertNull(URLUtil.getDomainSuffix(url)); url = new URL("http://www.example.co.uk:8080/index.html"); - Assert.assertEquals("co.uk", URLUtil.getDomainSuffix(url)); + assertEquals("co.uk", URLUtil.getDomainSuffix(url)); url = new URL("http://com"); - Assert.assertEquals("com", URLUtil.getDomainSuffix(url)); + assertEquals("com", URLUtil.getDomainSuffix(url)); url = new URL("http://www.example.co.uk.com"); - Assert.assertEquals("com", URLUtil.getDomainSuffix(url)); + assertEquals("com", URLUtil.getDomainSuffix(url)); // "nn" is not a public suffix url = new URL("http://example.com.nn"); - Assert.assertNull(URLUtil.getDomainSuffix(url)); + assertNull(URLUtil.getDomainSuffix(url)); url = new URL("http://"); - Assert.assertNull(URLUtil.getDomainSuffix(url)); + assertNull(URLUtil.getDomainSuffix(url)); /* * "xyz" is an ICANN suffix since 2014, see * https://www.iana.org/domains/root/db/xyz.html */ url = new URL("http://www.edu.tr.xyz"); - Assert.assertEquals("xyz", URLUtil.getDomainSuffix(url)); + assertEquals("xyz", URLUtil.getDomainSuffix(url)); url = new URL("http://subdomain.example.edu.tr"); - Assert.assertEquals("edu.tr", URLUtil.getDomainSuffix(url)); + assertEquals("edu.tr", URLUtil.getDomainSuffix(url)); url = new URL("http://subdomain.example.presse.fr"); - Assert.assertEquals("fr", URLUtil.getDomainSuffix(url)); + assertEquals("fr", URLUtil.getDomainSuffix(url)); url = new URL("http://subdomain.example.presse.tr"); - Assert.assertEquals("tr", URLUtil.getDomainSuffix(url)); + assertEquals("tr", URLUtil.getDomainSuffix(url)); // plc.co.im is listed as a domain suffix url = new URL("http://www.example.plc.co.im"); - Assert.assertEquals("plc.co.im", URLUtil.getDomainSuffix(url)); + assertEquals("plc.co.im", URLUtil.getDomainSuffix(url)); // 2000.hu is listed as a domain suffix url = new URL("http://www.example.2000.hu"); - Assert.assertEquals("2000.hu", URLUtil.getDomainSuffix(url)); + assertEquals("2000.hu", URLUtil.getDomainSuffix(url)); // test non-ascii url = new URL("http://www.example.商業.tw"); - Assert.assertEquals("xn--czrw28b.tw", URLUtil.getDomainSuffix(url)); + assertEquals("xn--czrw28b.tw", URLUtil.getDomainSuffix(url)); } @Test @@ -150,27 +151,27 @@ public void testGetTopLevelDomain() throws Exception { URL url = null; url = new URL("http://lucene.apache.org/nutch"); - Assert.assertEquals("org", URLUtil.getTopLevelDomainName(url)); + assertEquals("org", URLUtil.getTopLevelDomainName(url)); // hostname with trailing dot url = new URL("https://lucene.apache.org./nutch"); - Assert.assertEquals("org", URLUtil.getTopLevelDomainName(url)); + assertEquals("org", URLUtil.getTopLevelDomainName(url)); url = new URL("http://140.211.11.130/foundation/contributing.html"); - Assert.assertNull(URLUtil.getTopLevelDomainName(url)); + assertNull(URLUtil.getTopLevelDomainName(url)); url = new URL("http://www.example.co.uk:8080/index.html"); - Assert.assertEquals("uk", URLUtil.getTopLevelDomainName(url)); + assertEquals("uk", URLUtil.getTopLevelDomainName(url)); // "nn" is not a public suffix url = new URL("http://example.com.nn"); - Assert.assertNull(URLUtil.getTopLevelDomainName(url)); + assertNull(URLUtil.getTopLevelDomainName(url)); url = new URL("http://"); - Assert.assertNull(URLUtil.getTopLevelDomainName(url)); + assertNull(URLUtil.getTopLevelDomainName(url)); url = new URL("http://nic.삼성/"); - Assert.assertEquals("xn--cg4bki", URLUtil.getTopLevelDomainName(url)); + assertEquals("xn--cg4bki", URLUtil.getTopLevelDomainName(url)); } @Test @@ -180,28 +181,28 @@ public void testGetHostSegments() throws Exception { url = new URL("http://subdomain.example.edu.tr"); segments = URLUtil.getHostSegments(url); - Assert.assertEquals("subdomain", segments[0]); - Assert.assertEquals("example", segments[1]); - Assert.assertEquals("edu", segments[2]); - Assert.assertEquals("tr", segments[3]); + assertEquals("subdomain", segments[0]); + assertEquals("example", segments[1]); + assertEquals("edu", segments[2]); + assertEquals("tr", segments[3]); url = new URL("http://"); segments = URLUtil.getHostSegments(url); - Assert.assertEquals(1, segments.length); - Assert.assertEquals("", segments[0]); + assertEquals(1, segments.length); + assertEquals("", segments[0]); url = new URL("http://140.211.11.130/foundation/contributing.html"); segments = URLUtil.getHostSegments(url); - Assert.assertEquals(1, segments.length); - Assert.assertEquals("140.211.11.130", segments[0]); + assertEquals(1, segments.length); + assertEquals("140.211.11.130", segments[0]); // test non-ascii url = new URL("http://www.example.商業.tw"); segments = URLUtil.getHostSegments(url); - Assert.assertEquals("www", segments[0]); - Assert.assertEquals("example", segments[1]); - Assert.assertEquals("商業", segments[2]); - Assert.assertEquals("tw", segments[3]); + assertEquals("www", segments[0]); + assertEquals("example", segments[1]); + assertEquals("商業", segments[2]); + assertEquals("tw", segments[3]); } @@ -218,40 +219,40 @@ public void testChooseRepr() throws Exception { // 1) different domain them keep dest, temp or perm // a.com -> b.com* - Assert.assertEquals(bDotCom, URLUtil.chooseRepr(aDotCom, bDotCom, true)); - Assert.assertEquals(bDotCom, URLUtil.chooseRepr(aDotCom, bDotCom, false)); + assertEquals(bDotCom, URLUtil.chooseRepr(aDotCom, bDotCom, true)); + assertEquals(bDotCom, URLUtil.chooseRepr(aDotCom, bDotCom, false)); // 2) permanent and root, keep src // *a.com -> a.com?y=1 || *a.com -> a.com/xyz/index.html - Assert.assertEquals(aDotCom, URLUtil.chooseRepr(aDotCom, aQStr, false)); - Assert.assertEquals(aDotCom, URLUtil.chooseRepr(aDotCom, aPath, false)); + assertEquals(aDotCom, URLUtil.chooseRepr(aDotCom, aQStr, false)); + assertEquals(aDotCom, URLUtil.chooseRepr(aDotCom, aPath, false)); // 3) permanent and not root and dest root, keep dest // a.com/xyz/index.html -> a.com* - Assert.assertEquals(aDotCom, URLUtil.chooseRepr(aPath, aDotCom, false)); + assertEquals(aDotCom, URLUtil.chooseRepr(aPath, aDotCom, false)); // 4) permanent and neither root keep dest // a.com/xyz/index.html -> a.com/abc/page.html* - Assert.assertEquals(aPath2, URLUtil.chooseRepr(aPath, aPath2, false)); + assertEquals(aPath2, URLUtil.chooseRepr(aPath, aPath2, false)); // 5) temp and root and dest not root keep src // *a.com -> a.com/xyz/index.html - Assert.assertEquals(aDotCom, URLUtil.chooseRepr(aDotCom, aPath, true)); + assertEquals(aDotCom, URLUtil.chooseRepr(aDotCom, aPath, true)); // 6) temp and not root and dest root keep dest // a.com/xyz/index.html -> a.com* - Assert.assertEquals(aDotCom, URLUtil.chooseRepr(aPath, aDotCom, true)); + assertEquals(aDotCom, URLUtil.chooseRepr(aPath, aDotCom, true)); // 7) temp and neither root, keep shortest, if hosts equal by path else by // hosts // a.com/xyz/index.html -> a.com/abc/page.html* // *www.a.com/xyz/index.html -> www.news.a.com/xyz/index.html - Assert.assertEquals(aPath2, URLUtil.chooseRepr(aPath, aPath2, true)); - Assert.assertEquals(aPath, URLUtil.chooseRepr(aPath, aPath3, true)); + assertEquals(aPath2, URLUtil.chooseRepr(aPath, aPath2, true)); + assertEquals(aPath, URLUtil.chooseRepr(aPath, aPath3, true)); // 8) temp and both root keep shortest sub domain // *www.a.com -> www.news.a.com - Assert.assertEquals(aDotCom, URLUtil.chooseRepr(aDotCom, aSubDotCom, true)); + assertEquals(aDotCom, URLUtil.chooseRepr(aDotCom, aSubDotCom, true)); } // from RFC3986 section 5.4.1 @@ -274,31 +275,29 @@ public void testChooseRepr() throws Exception { public void testResolveURL() throws Exception { // test NUTCH-436 URL u436 = new URL("http://a/b/c/d;p?q#f"); - Assert.assertEquals("http://a/b/c/d;p?q#f", u436.toString()); + assertEquals("http://a/b/c/d;p?q#f", u436.toString()); URL abs = URLUtil.resolveURL(u436, "?y"); - Assert.assertEquals("http://a/b/c/d;p?y", abs.toString()); + assertEquals("http://a/b/c/d;p?y", abs.toString()); // test NUTCH-566 URL u566 = new URL("http://www.fleurie.org/entreprise.asp"); abs = URLUtil.resolveURL(u566, "?id_entrep=111"); - Assert.assertEquals("http://www.fleurie.org/entreprise.asp?id_entrep=111", + assertEquals("http://www.fleurie.org/entreprise.asp?id_entrep=111", abs.toString()); URL base = new URL(baseString); - Assert.assertEquals("base url parsing", baseString, base.toString()); + assertEquals(baseString, base.toString(), "base url parsing"); for (int i = 0; i < targets.length; i++) { URL u = URLUtil.resolveURL(base, targets[i][0]); - Assert.assertEquals(targets[i][1], targets[i][1], u.toString()); + assertEquals(targets[i][1], targets[i][1], u.toString()); } } @Test public void testToUNICODE() throws Exception { - Assert.assertEquals("http://www.çevir.com", + assertEquals("http://www.çevir.com", URLUtil.toUNICODE("http://www.xn--evir-zoa.com")); - Assert.assertEquals("http://uni-tübingen.de/", + assertEquals("http://uni-tübingen.de/", URLUtil.toUNICODE("http://xn--uni-tbingen-xhb.de/")); - Assert - .assertEquals( - "http://www.medizin.uni-tübingen.de:8080/search.php?q=abc#p1", + assertEquals("http://www.medizin.uni-tübingen.de:8080/search.php?q=abc#p1", URLUtil .toUNICODE("http://www.medizin.xn--uni-tbingen-xhb.de:8080/search.php?q=abc#p1")); @@ -306,13 +305,11 @@ public void testToUNICODE() throws Exception { @Test public void testToASCII() throws Exception { - Assert.assertEquals("http://www.xn--evir-zoa.com", + assertEquals("http://www.xn--evir-zoa.com", URLUtil.toASCII("http://www.çevir.com")); - Assert.assertEquals("http://xn--uni-tbingen-xhb.de/", + assertEquals("http://xn--uni-tbingen-xhb.de/", URLUtil.toASCII("http://uni-tübingen.de/")); - Assert - .assertEquals( - "http://www.medizin.xn--uni-tbingen-xhb.de:8080/search.php?q=abc#p1", + assertEquals("http://www.medizin.xn--uni-tbingen-xhb.de:8080/search.php?q=abc#p1", URLUtil .toASCII("http://www.medizin.uni-tübingen.de:8080/search.php?q=abc#p1")); } @@ -320,9 +317,9 @@ public void testToASCII() throws Exception { @Test public void testFileProtocol() throws Exception { // keep one single slash NUTCH-1483 - Assert.assertEquals("file:/path/file.html", + assertEquals("file:/path/file.html", URLUtil.toASCII("file:/path/file.html")); - Assert.assertEquals("file:/path/file.html", + assertEquals("file:/path/file.html", URLUtil.toUNICODE("file:/path/file.html")); } diff --git a/src/test/org/apache/nutch/util/WritableTestUtils.java b/src/test/org/apache/nutch/util/WritableTestUtils.java index 3da2226d15..d4429dbf36 100644 --- a/src/test/org/apache/nutch/util/WritableTestUtils.java +++ b/src/test/org/apache/nutch/util/WritableTestUtils.java @@ -18,7 +18,8 @@ import org.apache.hadoop.io.*; import org.apache.hadoop.conf.*; -import org.junit.Assert; + +import static org.junit.jupiter.api.Assertions.assertEquals; public class WritableTestUtils { @@ -30,7 +31,7 @@ public static void testWritable(Writable before) throws Exception { /** Utility method for testing writables. */ public static void testWritable(Writable before, Configuration conf) throws Exception { - Assert.assertEquals(before, writeRead(before, conf)); + assertEquals(before, writeRead(before, conf)); } /** Utility method for testing writables. */