diff --git a/.gitignore b/.gitignore
index 965000f400..611df9b3c1 100644
--- a/.gitignore
+++ b/.gitignore
@@ -1,3 +1,4 @@
+.claude
*.iml
.idea
target
diff --git a/opennlp-api/src/main/java/opennlp/tools/util/StringUtil.java b/opennlp-api/src/main/java/opennlp/tools/util/StringUtil.java
index 98cca59891..1842556720 100644
--- a/opennlp-api/src/main/java/opennlp/tools/util/StringUtil.java
+++ b/opennlp-api/src/main/java/opennlp/tools/util/StringUtil.java
@@ -268,6 +268,27 @@ public static boolean isEmpty(CharSequence theString) {
return theString.length() == 0;
}
+ /**
+ * Determines whether a {@link CharSequence} is blank: empty, or made up entirely of
+ * code points that {@link #isWhitespace(int)} accepts. Unlike
+ * {@link String#isBlank()}, this follows the toolkit's whitespace definition, which
+ * includes the no-break spaces the JDK predicate leaves out, so a value spelled
+ * entirely from them cannot pass a blank check as content.
+ *
+ * @param theString The {@link CharSequence} to examine. Must not be {@code null}.
+ * @return {@code true} if {@code theString} is empty or all whitespace.
+ */
+ public static boolean isBlank(CharSequence theString) {
+ for (int i = 0; i < theString.length(); ) {
+ final int codePoint = Character.codePointAt(theString, i);
+ if (!isWhitespace(codePoint)) {
+ return false;
+ }
+ i += Character.charCount(codePoint);
+ }
+ return true;
+ }
+
/**
* Get the minimum of three values.
*
diff --git a/opennlp-api/src/main/java/opennlp/tools/wordnet/LexicalKnowledgeBase.java b/opennlp-api/src/main/java/opennlp/tools/wordnet/LexicalKnowledgeBase.java
new file mode 100644
index 0000000000..214ef74db1
--- /dev/null
+++ b/opennlp-api/src/main/java/opennlp/tools/wordnet/LexicalKnowledgeBase.java
@@ -0,0 +1,89 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package opennlp.tools.wordnet;
+
+import java.util.List;
+import java.util.Optional;
+
+/**
+ * Lemma and synset lookup over a loaded lexical-semantic resource in the WordNet family. Synset
+ * identity is opaque and source-qualified (see {@link Synset#id()}). Lookups return their matches
+ * in the source's sense order and never return {@code null}.
+ *
+ *
Lemma matching semantics are the implementation's concern. The reference implementations
+ * match case-insensitively (case folding with the root locale) and treat the underscore some
+ * formats store in multiword lemmas as a space; an implementation with different semantics must
+ * document them. Returned {@link Synset#lemmas() lemmas} preserve the source's written forms,
+ * with spaces in multiword lemmas.
+ *
+ * Implementations must be immutable and thread-safe after loading: one instance is meant to
+ * be shared across an application's threads for concurrent lookups.
+ */
+public interface LexicalKnowledgeBase {
+
+ /**
+ * Finds the synsets containing a lemma with a part of speech, in the source's sense order
+ * (the most salient sense first when the source ranks senses).
+ *
+ * @param lemma The lemma to look up. Must not be {@code null}.
+ * @param pos The part of speech to look it up as. Must not be {@code null}.
+ * @return The matching synsets, never {@code null}; empty when the lexicon does not contain
+ * the lemma with that part of speech.
+ * @throws IllegalArgumentException Thrown if {@code lemma} or {@code pos} is {@code null}.
+ */
+ List lookup(String lemma, WordNetPOS pos);
+
+ /**
+ * Finds a synset by its opaque identifier.
+ *
+ * @param synsetId The synset identifier, as minted by this lexicon. Must not be {@code null}.
+ * @return The synset, or empty when this lexicon has no synset with that identifier.
+ * @throws IllegalArgumentException Thrown if {@code synsetId} is {@code null}.
+ */
+ Optional synset(String synsetId);
+
+ /**
+ * Navigates one typed relation from a synset.
+ *
+ * @param synsetId The source synset identifier. Must not be {@code null}.
+ * @param relation The relation type to follow. Must not be {@code null}.
+ * @return The target synset ids in source order, never {@code null}; empty when the synset is
+ * unknown or has no relation of that type.
+ * @throws IllegalArgumentException Thrown if {@code synsetId} or {@code relation} is
+ * {@code null}.
+ */
+ default List related(String synsetId, WordNetRelation relation) {
+ if (relation == null) {
+ throw new IllegalArgumentException("Relation must not be null");
+ }
+ return synset(synsetId).map(s -> s.related(relation)).orElse(List.of());
+ }
+
+ /**
+ * Tests whether the lexicon contains a lemma with a part of speech. This is the membership
+ * check morphological rules validate their candidates against; implementations may override
+ * it with a cheaper check than {@link #lookup(String, WordNetPOS)}.
+ *
+ * @param lemma The lemma to test. Must not be {@code null}.
+ * @param pos The part of speech to test it as. Must not be {@code null}.
+ * @return {@code true} if the lexicon contains the lemma with that part of speech.
+ * @throws IllegalArgumentException Thrown if {@code lemma} or {@code pos} is {@code null}.
+ */
+ default boolean contains(String lemma, WordNetPOS pos) {
+ return !lookup(lemma, pos).isEmpty();
+ }
+}
diff --git a/opennlp-api/src/main/java/opennlp/tools/wordnet/Synset.java b/opennlp-api/src/main/java/opennlp/tools/wordnet/Synset.java
new file mode 100644
index 0000000000..477fa38d90
--- /dev/null
+++ b/opennlp-api/src/main/java/opennlp/tools/wordnet/Synset.java
@@ -0,0 +1,123 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package opennlp.tools.wordnet;
+
+import java.util.Collections;
+import java.util.EnumMap;
+import java.util.List;
+import java.util.Map;
+
+import opennlp.tools.commons.ThreadSafe;
+
+/**
+ * One synonym set: a single lexicalized concept with its member lemmas, gloss, and typed
+ * relations to other synsets.
+ *
+ * The {@link #id() id} is an opaque, source-qualified string minted by the reader that
+ * produced the synset; consumers must not parse it, only pass it back to
+ * {@link LexicalKnowledgeBase#synset(String)} and compare it for equality. Relations map each
+ * {@link WordNetRelation} present on this synset to the target synset ids in source order.
+ *
+ * Instances are immutable and thread-safe: the list and map components are defensively
+ * copied to immutable views at construction.
+ *
+ * @param id The opaque, source-qualified synset identifier. Must not be {@code null} or
+ * empty.
+ * @param pos The part of speech. Must not be {@code null}.
+ * @param lemmas The member lemmas in source order, human-readable (multiword lemmas use
+ * spaces, not the underscores some formats store). Must not be {@code null} or
+ * empty, and must not contain {@code null} or empty elements.
+ * @param gloss The definition text, possibly empty when the source has none. Must not be
+ * {@code null}.
+ * @param relations The typed relations, each mapping to the target synset ids in source order.
+ * Must not be {@code null}; keys must not be {@code null}; each value must be
+ * a non-empty list of non-{@code null}, non-empty target ids.
+ */
+@ThreadSafe
+public record Synset(
+ String id,
+ WordNetPOS pos,
+ List lemmas,
+ String gloss,
+ Map> relations) {
+
+ /**
+ * Creates a synset.
+ *
+ * @throws IllegalArgumentException Thrown if any component violates its documented constraint.
+ */
+ public Synset {
+ if (id == null || id.isEmpty()) {
+ throw new IllegalArgumentException("Id must not be null or empty");
+ }
+ if (pos == null) {
+ throw new IllegalArgumentException("Pos must not be null");
+ }
+ if (lemmas == null || lemmas.isEmpty()) {
+ throw new IllegalArgumentException("Lemmas must not be null or empty for synset " + id);
+ }
+ for (final String lemma : lemmas) {
+ if (lemma == null || lemma.isEmpty()) {
+ throw new IllegalArgumentException(
+ "Lemmas must not contain a null or empty element for synset " + id);
+ }
+ }
+ if (gloss == null) {
+ throw new IllegalArgumentException("Gloss must not be null for synset " + id);
+ }
+ if (relations == null) {
+ throw new IllegalArgumentException("Relations must not be null for synset " + id);
+ }
+ final Map> copiedRelations =
+ new EnumMap<>(WordNetRelation.class);
+ for (final Map.Entry> relation : relations.entrySet()) {
+ if (relation.getKey() == null) {
+ throw new IllegalArgumentException("Relations must not contain a null key for synset " + id);
+ }
+ final List targets = relation.getValue();
+ if (targets == null || targets.isEmpty()) {
+ throw new IllegalArgumentException("Relation " + relation.getKey()
+ + " must map to a non-empty target list for synset " + id);
+ }
+ for (final String target : targets) {
+ if (target == null || target.isEmpty()) {
+ throw new IllegalArgumentException("Relation " + relation.getKey()
+ + " must not contain a null or empty target id for synset " + id);
+ }
+ }
+ copiedRelations.put(relation.getKey(), List.copyOf(targets));
+ }
+ lemmas = List.copyOf(lemmas);
+ relations = Collections.unmodifiableMap(copiedRelations);
+ }
+
+ /**
+ * Finds the target synset ids of one relation type.
+ *
+ * @param relation The relation type. Must not be {@code null}.
+ * @return The target synset ids in source order, never {@code null}; empty when this synset
+ * has no relation of that type.
+ * @throws IllegalArgumentException Thrown if {@code relation} is {@code null}.
+ */
+ public List related(WordNetRelation relation) {
+ if (relation == null) {
+ throw new IllegalArgumentException("Relation must not be null");
+ }
+ final List targets = relations.get(relation);
+ return targets == null ? List.of() : targets;
+ }
+}
diff --git a/opennlp-api/src/main/java/opennlp/tools/wordnet/WordNetPOS.java b/opennlp-api/src/main/java/opennlp/tools/wordnet/WordNetPOS.java
new file mode 100644
index 0000000000..65f6e691ec
--- /dev/null
+++ b/opennlp-api/src/main/java/opennlp/tools/wordnet/WordNetPOS.java
@@ -0,0 +1,40 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package opennlp.tools.wordnet;
+
+/**
+ * The four parts of speech a wordnet-style lexicon distinguishes.
+ *
+ * The enum carries none of the single-letter codes the on-disk formats use; readers own the
+ * mapping from their format's codes to these values. Adjective satellites normalize to
+ * {@link #ADJECTIVE}, with the cluster structure preserved through
+ * {@link WordNetRelation#SIMILAR_TO}.
+ */
+public enum WordNetPOS {
+
+ /** Nouns. */
+ NOUN,
+
+ /** Verbs. */
+ VERB,
+
+ /** Adjectives, including adjective satellites. */
+ ADJECTIVE,
+
+ /** Adverbs. */
+ ADVERB
+}
diff --git a/opennlp-api/src/main/java/opennlp/tools/wordnet/WordNetRelation.java b/opennlp-api/src/main/java/opennlp/tools/wordnet/WordNetRelation.java
new file mode 100644
index 0000000000..10da3cea34
--- /dev/null
+++ b/opennlp-api/src/main/java/opennlp/tools/wordnet/WordNetRelation.java
@@ -0,0 +1,115 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package opennlp.tools.wordnet;
+
+/**
+ * The typed relations a wordnet-style lexicon draws between {@link Synset synsets}. Readers map
+ * their source format's relation names onto these values.
+ *
+ * Relations that a source format draws between individual word senses (antonymy and
+ * derivation, for example) surface here at the synset level: the synset containing the source
+ * sense carries the relation to the synset containing the target sense.
+ */
+public enum WordNetRelation {
+
+ /** Opposition in meaning, for example between the adjectives for tall and short. */
+ ANTONYM,
+
+ /** The more general concept: a dog is a kind of canid. */
+ HYPERNYM,
+
+ /** The class a named instance belongs to: a specific river is an instance of river. */
+ INSTANCE_HYPERNYM,
+
+ /** The more specific concept: canid has the hyponym dog. */
+ HYPONYM,
+
+ /** A named instance of this class. */
+ INSTANCE_HYPONYM,
+
+ /** The group this synset is a member of. */
+ MEMBER_HOLONYM,
+
+ /** The whole this synset is a substance of. */
+ SUBSTANCE_HOLONYM,
+
+ /** The whole this synset is a part of. */
+ PART_HOLONYM,
+
+ /** A member of this group. */
+ MEMBER_MERONYM,
+
+ /** A substance this synset is made of. */
+ SUBSTANCE_MERONYM,
+
+ /** A part of this synset. */
+ PART_MERONYM,
+
+ /** The attribute a value expresses, or a value of this attribute. */
+ ATTRIBUTE,
+
+ /** A derivationally related form, typically across parts of speech. */
+ DERIVATIONALLY_RELATED,
+
+ /** An action entailed by this verb: snoring entails sleeping. */
+ ENTAILMENT,
+
+ /** The verb that entails this one; the inverse of {@link #ENTAILMENT}. */
+ ENTAILED_BY,
+
+ /** An effect this verb causes. */
+ CAUSE,
+
+ /** The cause of this verb; the inverse of {@link #CAUSE}. */
+ CAUSED_BY,
+
+ /** A related synset worth consulting. */
+ ALSO_SEE,
+
+ /** A verb sense grouped with this one. */
+ VERB_GROUP,
+
+ /** A satellite or head adjective in the same similarity cluster. */
+ SIMILAR_TO,
+
+ /** The verb an adjective is the participle of. */
+ PARTICIPLE,
+
+ /**
+ * The noun an adjective pertains to, or the adjective an adverb derives from. The source
+ * formats use one pointer for both directions of derivation, so this value does too.
+ */
+ PERTAINYM,
+
+ /** The topical domain this synset belongs to. */
+ DOMAIN_TOPIC,
+
+ /** A synset belonging to this topical domain. */
+ MEMBER_OF_DOMAIN_TOPIC,
+
+ /** The regional domain this synset belongs to. */
+ DOMAIN_REGION,
+
+ /** A synset belonging to this regional domain. */
+ MEMBER_OF_DOMAIN_REGION,
+
+ /** The usage domain this synset belongs to, for example slang or archaism. */
+ DOMAIN_USAGE,
+
+ /** A synset belonging to this usage domain. */
+ MEMBER_OF_DOMAIN_USAGE
+}
diff --git a/opennlp-api/src/test/java/opennlp/tools/wordnet/LexicalKnowledgeBaseTest.java b/opennlp-api/src/test/java/opennlp/tools/wordnet/LexicalKnowledgeBaseTest.java
new file mode 100644
index 0000000000..2b698797f1
--- /dev/null
+++ b/opennlp-api/src/test/java/opennlp/tools/wordnet/LexicalKnowledgeBaseTest.java
@@ -0,0 +1,105 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package opennlp.tools.wordnet;
+
+import java.util.List;
+import java.util.Map;
+import java.util.Optional;
+
+import org.junit.jupiter.api.Test;
+
+import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assertions.assertFalse;
+import static org.junit.jupiter.api.Assertions.assertThrows;
+import static org.junit.jupiter.api.Assertions.assertTrue;
+
+/**
+ * Exercises the {@link LexicalKnowledgeBase} default methods against a minimal in-memory
+ * implementation, so the defaults are validated independently of any reader.
+ */
+public class LexicalKnowledgeBaseTest {
+
+ private static final Synset DOG = new Synset("test-1-n", WordNetPOS.NOUN, List.of("dog"),
+ "a domesticated canid", Map.of(WordNetRelation.HYPERNYM, List.of("test-2-n")));
+
+ private static final Synset CANID = new Synset("test-2-n", WordNetPOS.NOUN, List.of("canid"),
+ "a carnivorous mammal", Map.of(WordNetRelation.HYPONYM, List.of("test-1-n")));
+
+ // A deliberately tiny implementation of only the two abstract methods.
+ private static final LexicalKnowledgeBase LEXICON = new LexicalKnowledgeBase() {
+
+ @Override
+ public List lookup(String lemma, WordNetPOS pos) {
+ if (lemma == null) {
+ throw new IllegalArgumentException("Lemma must not be null");
+ }
+ if (pos == null) {
+ throw new IllegalArgumentException("Pos must not be null");
+ }
+ if (pos == WordNetPOS.NOUN && "dog".equals(lemma)) {
+ return List.of(DOG);
+ }
+ return List.of();
+ }
+
+ @Override
+ public Optional synset(String synsetId) {
+ if (synsetId == null) {
+ throw new IllegalArgumentException("SynsetId must not be null");
+ }
+ if (DOG.id().equals(synsetId)) {
+ return Optional.of(DOG);
+ }
+ if (CANID.id().equals(synsetId)) {
+ return Optional.of(CANID);
+ }
+ return Optional.empty();
+ }
+ };
+
+ @Test
+ void testRelatedNavigatesThroughSynset() {
+ assertEquals(List.of("test-2-n"), LEXICON.related("test-1-n", WordNetRelation.HYPERNYM));
+ assertEquals(List.of("test-1-n"), LEXICON.related("test-2-n", WordNetRelation.HYPONYM));
+ }
+
+ @Test
+ void testRelatedIsEmptyForAbsentRelationOrUnknownSynset() {
+ assertTrue(LEXICON.related("test-1-n", WordNetRelation.ANTONYM).isEmpty());
+ assertTrue(LEXICON.related("test-99-n", WordNetRelation.HYPERNYM).isEmpty());
+ }
+
+ @Test
+ void testRelatedRejectsNulls() {
+ assertThrows(IllegalArgumentException.class,
+ () -> LEXICON.related(null, WordNetRelation.HYPERNYM));
+ assertThrows(IllegalArgumentException.class, () -> LEXICON.related("test-1-n", null));
+ }
+
+ @Test
+ void testContainsFollowsLookup() {
+ assertTrue(LEXICON.contains("dog", WordNetPOS.NOUN));
+ assertFalse(LEXICON.contains("dog", WordNetPOS.VERB));
+ assertFalse(LEXICON.contains("cat", WordNetPOS.NOUN));
+ }
+
+ @Test
+ void testContainsRejectsNulls() {
+ assertThrows(IllegalArgumentException.class, () -> LEXICON.contains(null, WordNetPOS.NOUN));
+ assertThrows(IllegalArgumentException.class, () -> LEXICON.contains("dog", null));
+ }
+}
diff --git a/opennlp-api/src/test/java/opennlp/tools/wordnet/SynsetTest.java b/opennlp-api/src/test/java/opennlp/tools/wordnet/SynsetTest.java
new file mode 100644
index 0000000000..e9cbb4c063
--- /dev/null
+++ b/opennlp-api/src/test/java/opennlp/tools/wordnet/SynsetTest.java
@@ -0,0 +1,150 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package opennlp.tools.wordnet;
+
+import java.util.ArrayList;
+import java.util.HashMap;
+import java.util.List;
+import java.util.Map;
+
+import org.junit.jupiter.api.Test;
+
+import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assertions.assertThrows;
+import static org.junit.jupiter.api.Assertions.assertTrue;
+
+public class SynsetTest {
+
+ private static Synset dog() {
+ return new Synset("test-1-n", WordNetPOS.NOUN, List.of("dog", "domestic dog"),
+ "a domesticated canid",
+ Map.of(WordNetRelation.HYPERNYM, List.of("test-2-n")));
+ }
+
+ @Test
+ void testComponents() {
+ final Synset synset = dog();
+ assertEquals("test-1-n", synset.id());
+ assertEquals(WordNetPOS.NOUN, synset.pos());
+ assertEquals(List.of("dog", "domestic dog"), synset.lemmas());
+ assertEquals("a domesticated canid", synset.gloss());
+ assertEquals(Map.of(WordNetRelation.HYPERNYM, List.of("test-2-n")), synset.relations());
+ }
+
+ @Test
+ void testRelatedReturnsTargetsInOrder() {
+ final Synset synset = new Synset("test-1-n", WordNetPOS.NOUN, List.of("dog"), "",
+ Map.of(WordNetRelation.HYPONYM, List.of("test-3-n", "test-2-n")));
+ assertEquals(List.of("test-3-n", "test-2-n"), synset.related(WordNetRelation.HYPONYM));
+ }
+
+ @Test
+ void testRelatedIsEmptyForAbsentRelation() {
+ assertTrue(dog().related(WordNetRelation.ANTONYM).isEmpty());
+ }
+
+ @Test
+ void testRelatedRejectsNull() {
+ assertThrows(IllegalArgumentException.class, () -> dog().related(null));
+ }
+
+ @Test
+ void testEmptyGlossAndNoRelationsAreValid() {
+ final Synset synset = new Synset("test-9-r", WordNetPOS.ADVERB, List.of("well"), "", Map.of());
+ assertEquals("", synset.gloss());
+ assertTrue(synset.relations().isEmpty());
+ }
+
+ @Test
+ void testDefensiveCopies() {
+ final List lemmas = new ArrayList<>(List.of("dog"));
+ final List targets = new ArrayList<>(List.of("test-2-n"));
+ final Map> relations = new HashMap<>();
+ relations.put(WordNetRelation.HYPERNYM, targets);
+ final Synset synset = new Synset("test-1-n", WordNetPOS.NOUN, lemmas, "gloss", relations);
+ lemmas.add("mutated");
+ targets.add("mutated");
+ relations.put(WordNetRelation.ANTONYM, List.of("test-3-n"));
+ assertEquals(List.of("dog"), synset.lemmas());
+ assertEquals(List.of("test-2-n"), synset.related(WordNetRelation.HYPERNYM));
+ assertEquals(1, synset.relations().size());
+ }
+
+ @Test
+ void testReturnedCollectionsAreImmutable() {
+ final Synset synset = dog();
+ assertThrows(UnsupportedOperationException.class, () -> synset.lemmas().add("x"));
+ assertThrows(UnsupportedOperationException.class,
+ () -> synset.relations().put(WordNetRelation.ANTONYM, List.of("x")));
+ assertThrows(UnsupportedOperationException.class,
+ () -> synset.related(WordNetRelation.HYPERNYM).add("x"));
+ }
+
+ @Test
+ void testRejectsNullOrEmptyId() {
+ assertThrows(IllegalArgumentException.class,
+ () -> new Synset(null, WordNetPOS.NOUN, List.of("dog"), "", Map.of()));
+ assertThrows(IllegalArgumentException.class,
+ () -> new Synset("", WordNetPOS.NOUN, List.of("dog"), "", Map.of()));
+ }
+
+ @Test
+ void testRejectsNullPos() {
+ assertThrows(IllegalArgumentException.class,
+ () -> new Synset("test-1-n", null, List.of("dog"), "", Map.of()));
+ }
+
+ @Test
+ void testRejectsNullOrEmptyLemmas() {
+ assertThrows(IllegalArgumentException.class,
+ () -> new Synset("test-1-n", WordNetPOS.NOUN, null, "", Map.of()));
+ assertThrows(IllegalArgumentException.class,
+ () -> new Synset("test-1-n", WordNetPOS.NOUN, List.of(), "", Map.of()));
+ final List withNull = new ArrayList<>();
+ withNull.add(null);
+ assertThrows(IllegalArgumentException.class,
+ () -> new Synset("test-1-n", WordNetPOS.NOUN, withNull, "", Map.of()));
+ assertThrows(IllegalArgumentException.class,
+ () -> new Synset("test-1-n", WordNetPOS.NOUN, List.of(""), "", Map.of()));
+ }
+
+ @Test
+ void testRejectsNullGloss() {
+ assertThrows(IllegalArgumentException.class,
+ () -> new Synset("test-1-n", WordNetPOS.NOUN, List.of("dog"), null, Map.of()));
+ }
+
+ @Test
+ void testRejectsInvalidRelations() {
+ assertThrows(IllegalArgumentException.class,
+ () -> new Synset("test-1-n", WordNetPOS.NOUN, List.of("dog"), "", null));
+ final Map> nullKey = new HashMap<>();
+ nullKey.put(null, List.of("test-2-n"));
+ assertThrows(IllegalArgumentException.class,
+ () -> new Synset("test-1-n", WordNetPOS.NOUN, List.of("dog"), "", nullKey));
+ final Map> nullTargets = new HashMap<>();
+ nullTargets.put(WordNetRelation.HYPERNYM, null);
+ assertThrows(IllegalArgumentException.class,
+ () -> new Synset("test-1-n", WordNetPOS.NOUN, List.of("dog"), "", nullTargets));
+ assertThrows(IllegalArgumentException.class,
+ () -> new Synset("test-1-n", WordNetPOS.NOUN, List.of("dog"), "",
+ Map.of(WordNetRelation.HYPERNYM, List.of())));
+ assertThrows(IllegalArgumentException.class,
+ () -> new Synset("test-1-n", WordNetPOS.NOUN, List.of("dog"), "",
+ Map.of(WordNetRelation.HYPERNYM, List.of(""))));
+ }
+}
diff --git a/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/CFSA2Reader.java b/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/CFSA2Reader.java
new file mode 100644
index 0000000000..f429e129a5
--- /dev/null
+++ b/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/CFSA2Reader.java
@@ -0,0 +1,253 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package opennlp.tools.formats;
+
+import java.io.IOException;
+import java.io.InputStream;
+import java.util.Arrays;
+import java.util.function.Consumer;
+
+/**
+ * Reads a CFSA2 finite-state automaton and enumerates the byte sequences it accepts.
+ *
+ * CFSA2 is the compact automaton format defined by the morfologik project (BSD); the
+ * {@code .dict} files distributed for many languages are CFSA2 automata paired with a plain
+ * {@code .info} metadata file. This is a clean-room reader written from the published format,
+ * with no dependency on that library. It exposes the raw accepted byte sequences; interpreting
+ * them as morphological entries (surface form, separator, encoded base form, separator, tag) is
+ * left to the caller, which also owns the character encoding declared by the dictionary.
+ *
+ * A constructed instance holds only immutable state, so {@link #forEachSequence(Consumer)} may
+ * be called concurrently.
+ */
+public final class CFSA2Reader implements FsaSequenceReader {
+
+ private static final byte VERSION = (byte) 0xc6;
+
+ private static final int FLAG_NUMBERS = 0x0100;
+
+ private static final int BIT_TARGET_NEXT = 0x80;
+ private static final int BIT_LAST_ARC = 0x40;
+ private static final int BIT_FINAL_ARC = 0x20;
+ private static final int LABEL_INDEX_MASK = 0x1f;
+
+ private static final int HEADER_SIZE = 8;
+ private static final int TERMINAL_NODE = 0;
+ private static final int NO_ARC = 0;
+
+ /** Guards against runaway recursion on a malformed automaton. */
+ private static final int MAX_SEQUENCE_LENGTH = 8192;
+
+ private final byte[] arcs;
+ private final byte[] labelMapping;
+ private final boolean hasNumbers;
+ private final int rootNode;
+
+ /**
+ * Initializes the reader over the automaton's arc block.
+ *
+ * @param arcs The arc block, the automaton bytes after the header and label table.
+ * @param labelMapping The label table indexed by an arc's label index.
+ * @param hasNumbers Whether each node is prefixed with a perfect-hash number to skip.
+ */
+ private CFSA2Reader(byte[] arcs, byte[] labelMapping, boolean hasNumbers) {
+ this.arcs = arcs;
+ this.labelMapping = labelMapping;
+ this.hasNumbers = hasNumbers;
+ this.rootNode = destinationNode(firstArc(TERMINAL_NODE));
+ }
+
+ /**
+ * Reads a CFSA2 automaton from a stream.
+ *
+ * @param in The automaton bytes, referenced by an open {@link InputStream}. Must not be
+ * {@code null}.
+ * @return A reader over the automaton.
+ * @throws IllegalArgumentException Thrown if {@code in} is {@code null}.
+ * @throws IOException Thrown on IO errors, or if the stream is not a CFSA2 automaton.
+ */
+ public static CFSA2Reader read(InputStream in) throws IOException {
+ if (in == null) {
+ throw new IllegalArgumentException("in must not be null");
+ }
+ return fromBytes(in.readAllBytes());
+ }
+
+ /**
+ * Reads an automaton from a byte block already held in memory.
+ *
+ * @param bytes The automaton bytes, header included.
+ * @return A reader over the automaton.
+ * @throws IOException Thrown if the block is not a CFSA2 automaton or its header is truncated.
+ */
+ static CFSA2Reader fromBytes(byte[] bytes) throws IOException {
+ FsaSequenceReader.requireFsaHeader(bytes);
+ if (bytes.length < HEADER_SIZE) {
+ throw new IOException("truncated CFSA2 header: fewer than " + HEADER_SIZE + " bytes");
+ }
+ if (bytes[4] != VERSION) {
+ throw new IOException("unsupported FSA version 0x"
+ + Integer.toHexString(bytes[4] & 0xff) + "; only CFSA2 (0xc6) is read");
+ }
+ final int flags = ((bytes[5] & 0xff) << 8) | (bytes[6] & 0xff);
+ final int labelMappingSize = bytes[7] & 0xff;
+ final int arcsStart = HEADER_SIZE + labelMappingSize;
+ if (arcsStart > bytes.length) {
+ throw new IOException("truncated CFSA2 header: label table runs past end of data");
+ }
+ final byte[] labelMapping = Arrays.copyOfRange(bytes, HEADER_SIZE, arcsStart);
+ final byte[] arcs = Arrays.copyOfRange(bytes, arcsStart, bytes.length);
+ return new CFSA2Reader(arcs, labelMapping, (flags & FLAG_NUMBERS) != 0);
+ }
+
+ /**
+ * {@inheritDoc}
+ *
+ * The stored order of a CFSA2 automaton is lexicographic.
+ *
+ * @throws IllegalStateException Thrown if a path exceeds {@value #MAX_SEQUENCE_LENGTH} bytes,
+ * which indicates a malformed automaton.
+ */
+ @Override
+ public void forEachSequence(Consumer action) {
+ if (action == null) {
+ throw new IllegalArgumentException("action must not be null");
+ }
+ enumerate(rootNode, new GrowableByteSequence(), action);
+ }
+
+ /**
+ * Walks the automaton depth first, reporting the path of every arc that ends a word.
+ *
+ * @param node The node whose arcs are walked.
+ * @param path The labels of the arcs walked so far; pushed and popped in place.
+ * @param action The action to run for each accepted sequence.
+ */
+ private void enumerate(int node, GrowableByteSequence path, Consumer action) {
+ if (path.length() > MAX_SEQUENCE_LENGTH) {
+ throw new IllegalStateException(
+ "CFSA2 sequence exceeds " + MAX_SEQUENCE_LENGTH + " bytes; automaton may be malformed");
+ }
+ for (int arc = firstArc(node); arc != NO_ARC; arc = nextArc(arc)) {
+ path.push(arcLabel(arc));
+ if ((arcs[arc] & BIT_FINAL_ARC) != 0) {
+ action.accept(path.toByteArray());
+ }
+ final int destination = destinationNode(arc);
+ if (destination != TERMINAL_NODE) {
+ enumerate(destination, path, action);
+ }
+ path.pop();
+ }
+ }
+
+ /**
+ * Locates the first arc of a node, skipping the node number when the automaton stores one.
+ *
+ * @param node The offset of the node.
+ * @return The offset of the node's first arc.
+ */
+ private int firstArc(int node) {
+ return hasNumbers ? skipVInt(node) : node;
+ }
+
+ /**
+ * Locates the arc following one within the same node.
+ *
+ * @param arc The offset of the current arc.
+ * @return The offset of the next arc, or {@link #NO_ARC} when the current arc is the last.
+ */
+ private int nextArc(int arc) {
+ return (arcs[arc] & BIT_LAST_ARC) != 0 ? NO_ARC : skipArc(arc);
+ }
+
+ /**
+ * Reads the label of an arc, from the label table or from the arc itself.
+ *
+ * @param arc The offset of the arc.
+ * @return The label byte the arc consumes.
+ */
+ private byte arcLabel(int arc) {
+ final int index = arcs[arc] & LABEL_INDEX_MASK;
+ return index > 0 ? labelMapping[index] : arcs[arc + 1];
+ }
+
+ /**
+ * Reads the node an arc leads to, following either its goto address or the next-node bit.
+ *
+ * @param arc The offset of the arc.
+ * @return The offset of the destination node, {@link #TERMINAL_NODE} when the arc ends a word
+ * without continuing.
+ */
+ private int destinationNode(int arc) {
+ if ((arcs[arc] & BIT_TARGET_NEXT) != 0) {
+ int last = arc;
+ while ((arcs[last] & BIT_LAST_ARC) == 0) {
+ last = skipArc(last);
+ }
+ return skipArc(last);
+ }
+ return readVInt(arc + ((arcs[arc] & LABEL_INDEX_MASK) == 0 ? 2 : 1));
+ }
+
+ /**
+ * Skips over one arc, whose width depends on its flags.
+ *
+ * @param offset The offset of the arc.
+ * @return The offset just past the arc.
+ */
+ private int skipArc(int offset) {
+ final int flag = arcs[offset++];
+ if ((flag & LABEL_INDEX_MASK) == 0) {
+ offset++;
+ }
+ if ((flag & BIT_TARGET_NEXT) == 0) {
+ offset = skipVInt(offset);
+ }
+ return offset;
+ }
+
+ /**
+ * Reads a variable-length integer, least significant group first.
+ *
+ * @param offset The offset of the first byte of the integer.
+ * @return The decoded value.
+ */
+ private int readVInt(int offset) {
+ byte b = arcs[offset];
+ int value = b & 0x7f;
+ for (int shift = 7; b < 0; shift += 7) {
+ b = arcs[++offset];
+ value |= (b & 0x7f) << shift;
+ }
+ return value;
+ }
+
+ /**
+ * Skips over a variable-length integer.
+ *
+ * @param offset The offset of the first byte of the integer.
+ * @return The offset just past the integer.
+ */
+ private int skipVInt(int offset) {
+ while (arcs[offset] < 0) {
+ offset++;
+ }
+ return offset + 1;
+ }
+}
diff --git a/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/FSA5Reader.java b/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/FSA5Reader.java
new file mode 100644
index 0000000000..7b7f685c4b
--- /dev/null
+++ b/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/FSA5Reader.java
@@ -0,0 +1,209 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package opennlp.tools.formats;
+
+import java.io.IOException;
+import java.io.InputStream;
+import java.util.Arrays;
+import java.util.function.Consumer;
+
+/**
+ * Reads an FSA5 finite-state automaton and enumerates the byte sequences it accepts.
+ *
+ * FSA5 is the older of the two morfologik automaton formats (the newer being
+ * {@link CFSA2Reader}); its arcs use a fixed-width goto address rather than a variable-length one.
+ * This is a clean-room reader written from the published format, with no dependency on that
+ * library. Interpreting the accepted sequences as morphological entries is left to the caller.
+ *
+ * A constructed instance holds only immutable state, so {@link #forEachSequence(Consumer)} may
+ * be called concurrently.
+ */
+public final class FSA5Reader implements FsaSequenceReader {
+
+ private static final byte VERSION = (byte) 0x05;
+
+ private static final int BIT_FINAL_ARC = 0x01;
+ private static final int BIT_LAST_ARC = 0x02;
+ private static final int BIT_TARGET_NEXT = 0x04;
+
+ /** Offset of the flags/goto field within an arc; the label occupies the byte before it. */
+ private static final int ADDRESS_OFFSET = 1;
+ private static final int HEADER_SIZE = 8;
+ private static final int TERMINAL_NODE = 0;
+ private static final int NO_ARC = 0;
+
+ /** Guards against runaway recursion on a malformed automaton. */
+ private static final int MAX_SEQUENCE_LENGTH = 8192;
+
+ private final byte[] arcs;
+ private final int gotoLength;
+ private final int nodeDataLength;
+ private final int rootNode;
+
+ /**
+ * Initializes the reader over the automaton's arc block.
+ *
+ * @param arcs The arc block, the automaton bytes after the header.
+ * @param gotoLength The width in bytes of an arc's flags and goto address field.
+ * @param nodeDataLength The width in bytes of the optional data preceding a node's arcs.
+ */
+ private FSA5Reader(byte[] arcs, int gotoLength, int nodeDataLength) {
+ this.arcs = arcs;
+ this.gotoLength = gotoLength;
+ this.nodeDataLength = nodeDataLength;
+ // FSA5 keeps a dummy node ahead of the epsilon node: skip the dummy's arc to reach the
+ // epsilon node, then the root is its single arc's destination.
+ final int epsilonNode = skipArc(firstArc(TERMINAL_NODE));
+ this.rootNode = destinationNode(firstArc(epsilonNode));
+ }
+
+ /**
+ * Reads an FSA5 automaton from a stream.
+ *
+ * @param in The automaton bytes, referenced by an open {@link InputStream}. Must not be
+ * {@code null}.
+ * @return A reader over the automaton.
+ * @throws IllegalArgumentException Thrown if {@code in} is {@code null}.
+ * @throws IOException Thrown on IO errors, or if the stream is not an FSA5 automaton.
+ */
+ public static FSA5Reader read(InputStream in) throws IOException {
+ if (in == null) {
+ throw new IllegalArgumentException("in must not be null");
+ }
+ return fromBytes(in.readAllBytes());
+ }
+
+ /**
+ * Reads an automaton from a byte block already held in memory.
+ *
+ * @param bytes The automaton bytes, header included.
+ * @return A reader over the automaton.
+ * @throws IOException Thrown if the block is not an FSA5 automaton or its header is truncated.
+ */
+ static FSA5Reader fromBytes(byte[] bytes) throws IOException {
+ FsaSequenceReader.requireFsaHeader(bytes);
+ if (bytes.length < HEADER_SIZE) {
+ throw new IOException("truncated FSA5 header: fewer than " + HEADER_SIZE + " bytes");
+ }
+ if (bytes[4] != VERSION) {
+ throw new IOException("unsupported FSA version 0x"
+ + Integer.toHexString(bytes[4] & 0xff) + "; only FSA5 (0x05) is read here");
+ }
+ // One header byte packs both widths: the goto length low, the node data length high.
+ final int packedLengths = bytes[7] & 0xff;
+ final int gotoLength = packedLengths & 0x0f;
+ final int nodeDataLength = (packedLengths >>> 4) & 0x0f;
+ if (gotoLength < 1) {
+ throw new IOException("invalid FSA5 goto length: " + gotoLength);
+ }
+ final byte[] arcs = Arrays.copyOfRange(bytes, HEADER_SIZE, bytes.length);
+ return new FSA5Reader(arcs, gotoLength, nodeDataLength);
+ }
+
+ /**
+ * {@inheritDoc}
+ *
+ * The stored order of an FSA5 automaton is lexicographic.
+ *
+ * @throws IllegalStateException Thrown if a path exceeds {@value #MAX_SEQUENCE_LENGTH} bytes,
+ * which indicates a malformed automaton.
+ */
+ @Override
+ public void forEachSequence(Consumer action) {
+ if (action == null) {
+ throw new IllegalArgumentException("action must not be null");
+ }
+ enumerate(rootNode, new GrowableByteSequence(), action);
+ }
+
+ /**
+ * Walks the automaton depth first, reporting the path of every arc that ends a word.
+ *
+ * @param node The node whose arcs are walked.
+ * @param path The labels of the arcs walked so far; pushed and popped in place.
+ * @param action The action to run for each accepted sequence.
+ */
+ private void enumerate(int node, GrowableByteSequence path, Consumer action) {
+ if (path.length() > MAX_SEQUENCE_LENGTH) {
+ throw new IllegalStateException(
+ "FSA5 sequence exceeds " + MAX_SEQUENCE_LENGTH + " bytes; automaton may be malformed");
+ }
+ for (int arc = firstArc(node); arc != NO_ARC; arc = nextArc(arc)) {
+ path.push(arcs[arc]);
+ if ((arcs[arc + ADDRESS_OFFSET] & BIT_FINAL_ARC) != 0) {
+ action.accept(path.toByteArray());
+ }
+ final int destination = destinationNode(arc);
+ if (destination != TERMINAL_NODE) {
+ enumerate(destination, path, action);
+ }
+ path.pop();
+ }
+ }
+
+ /**
+ * Locates the first arc of a node, skipping the node data when the automaton stores any.
+ *
+ * @param node The offset of the node.
+ * @return The offset of the node's first arc.
+ */
+ private int firstArc(int node) {
+ return nodeDataLength + node;
+ }
+
+ /**
+ * Locates the arc following one within the same node.
+ *
+ * @param arc The offset of the current arc.
+ * @return The offset of the next arc, or {@link #NO_ARC} when the current arc is the last.
+ */
+ private int nextArc(int arc) {
+ return (arcs[arc + ADDRESS_OFFSET] & BIT_LAST_ARC) != 0 ? NO_ARC : skipArc(arc);
+ }
+
+ /**
+ * Reads the node an arc leads to, following either its goto address or the next-node bit.
+ *
+ * @param arc The offset of the arc.
+ * @return The offset of the destination node, {@link #TERMINAL_NODE} when the arc ends a word
+ * without continuing.
+ */
+ private int destinationNode(int arc) {
+ if ((arcs[arc + ADDRESS_OFFSET] & BIT_TARGET_NEXT) != 0) {
+ return skipArc(arc);
+ }
+ int value = 0;
+ for (int i = gotoLength - 1; i >= 0; i--) {
+ value = (value << 8) | (arcs[arc + ADDRESS_OFFSET + i] & 0xff);
+ }
+ // The three flag bits share the low end of the goto field.
+ return value >>> 3;
+ }
+
+ /**
+ * Skips over one arc, whose width depends on whether it carries a goto address.
+ *
+ * @param arc The offset of the arc.
+ * @return The offset just past the arc.
+ */
+ private int skipArc(int arc) {
+ return (arcs[arc + ADDRESS_OFFSET] & BIT_TARGET_NEXT) != 0
+ ? arc + ADDRESS_OFFSET + 1
+ : arc + ADDRESS_OFFSET + gotoLength;
+ }
+}
diff --git a/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/FsaSequenceReader.java b/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/FsaSequenceReader.java
new file mode 100644
index 0000000000..6bea9716d3
--- /dev/null
+++ b/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/FsaSequenceReader.java
@@ -0,0 +1,94 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package opennlp.tools.formats;
+
+import java.io.IOException;
+import java.io.InputStream;
+import java.util.function.Consumer;
+
+/**
+ * Enumerates the byte sequences accepted by a finite-state automaton, independent of which
+ * on-disk format encodes it. The two morfologik automaton formats are read: FSA5 (version
+ * {@code 0x05}) and CFSA2 (version {@code 0xc6}).
+ *
+ * Thread safety is implementation specific.
+ */
+public interface FsaSequenceReader {
+
+ /** ASCII {@code \fsa}, the shared magic header of both automaton formats. */
+ String MAGIC = "\\fsa";
+
+ /** Version byte of the FSA5 format. */
+ int VERSION_FSA5 = 0x05;
+
+ /** Version byte of the CFSA2 format. */
+ int VERSION_CFSA2 = 0xc6;
+
+ /**
+ * Passes every accepted byte sequence to {@code action}, in the automaton's stored order. Each
+ * sequence is a fresh array owned by the callee.
+ *
+ * @param action The action to run for each accepted sequence. Must not be {@code null}.
+ * @throws IllegalArgumentException Thrown if {@code action} is {@code null}.
+ */
+ void forEachSequence(Consumer action);
+
+ /**
+ * Checks that a byte block starts with {@link #MAGIC} and carries a version byte.
+ *
+ * @param bytes The automaton bytes. Must not be {@code null}.
+ * @throws IOException Thrown if the block is too short or does not start with {@link #MAGIC}.
+ */
+ static void requireFsaHeader(byte[] bytes) throws IOException {
+ if (bytes.length <= MAGIC.length()) {
+ throw new IOException("not an FSA automaton: bad magic header");
+ }
+ for (int i = 0; i < MAGIC.length(); i++) {
+ if (bytes[i] != (byte) MAGIC.charAt(i)) {
+ throw new IOException("not an FSA automaton: bad magic header");
+ }
+ }
+ }
+
+ /**
+ * Reads an FSA5 or CFSA2 automaton, dispatching on the version byte.
+ *
+ * @param in The automaton bytes, referenced by an open {@link InputStream}. Must not be
+ * {@code null}.
+ * @return A reader over the automaton.
+ * @throws IllegalArgumentException Thrown if {@code in} is {@code null}.
+ * @throws IOException Thrown on IO errors, or if the stream is not a supported FSA automaton.
+ */
+ static FsaSequenceReader read(InputStream in) throws IOException {
+ if (in == null) {
+ throw new IllegalArgumentException("in must not be null");
+ }
+ final byte[] bytes = in.readAllBytes();
+ requireFsaHeader(bytes);
+ final int version = bytes[4] & 0xff;
+ switch (version) {
+ case VERSION_CFSA2:
+ return CFSA2Reader.fromBytes(bytes);
+ case VERSION_FSA5:
+ return FSA5Reader.fromBytes(bytes);
+ default:
+ throw new IOException("unsupported FSA version 0x" + Integer.toHexString(version)
+ + "; only FSA5 (0x05) and CFSA2 (0xc6) are read");
+ }
+ }
+}
diff --git a/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/GrowableByteSequence.java b/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/GrowableByteSequence.java
new file mode 100644
index 0000000000..46a4a63322
--- /dev/null
+++ b/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/GrowableByteSequence.java
@@ -0,0 +1,59 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package opennlp.tools.formats;
+
+import java.util.Arrays;
+
+/**
+ * A growable byte buffer used as the path stack while walking a finite-state automaton. One
+ * instance is reused for a whole traversal: labels are {@link #push(byte)}ed on the way down and
+ * {@link #pop()}ped on the way back up, so only accepted sequences allocate, via
+ * {@link #toByteArray()}. Not thread-safe; each traversal creates its own.
+ */
+final class GrowableByteSequence {
+
+ private byte[] data = new byte[64];
+ private int length;
+
+ /** {@return the number of bytes currently on the stack} */
+ int length() {
+ return length;
+ }
+
+ /**
+ * Appends one byte, growing the buffer when it is full.
+ *
+ * @param value The byte to append.
+ */
+ void push(byte value) {
+ if (length == data.length) {
+ data = Arrays.copyOf(data, data.length << 1);
+ }
+ data[length++] = value;
+ }
+
+ /** Drops the last byte. The caller must not pop more bytes than it pushed. */
+ void pop() {
+ length--;
+ }
+
+ /** {@return a copy of the bytes currently on the stack} */
+ byte[] toByteArray() {
+ return Arrays.copyOf(data, length);
+ }
+}
diff --git a/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/MorfologikDictionaryReader.java b/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/MorfologikDictionaryReader.java
new file mode 100644
index 0000000000..ba8f2c9329
--- /dev/null
+++ b/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/MorfologikDictionaryReader.java
@@ -0,0 +1,370 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package opennlp.tools.formats;
+
+import java.io.ByteArrayInputStream;
+import java.io.IOException;
+import java.io.InputStream;
+import java.io.UncheckedIOException;
+import java.nio.charset.Charset;
+import java.nio.charset.StandardCharsets;
+import java.util.LinkedHashMap;
+import java.util.LinkedHashSet;
+import java.util.Locale;
+import java.util.Map;
+import java.util.Properties;
+
+import opennlp.tools.lemmatizer.DictionaryLemmatizer;
+
+/**
+ * Builds a {@link DictionaryLemmatizer} from a morfologik-format morphological dictionary: an
+ * FSA5 or CFSA2 automaton (read by {@link FsaSequenceReader}) whose accepted byte sequences are
+ * {@code surfaceForm SEP encodedBase SEP tag}, paired with the {@code .info} metadata that
+ * declares the separator byte, the character encoding, and the base-form encoder.
+ *
+ * This is a clean-room reader with no dependency on the morfologik library. The base form
+ * (lemma) is stored relative to the surface form to save space; the four encoders are decoded
+ * here. Each is a run of control bytes offset by {@code 'A'}, followed by literal bytes to
+ * append:
+ *
+ * - {@code NONE}: the encoded bytes are the base form verbatim.
+ * - {@code SUFFIX} ({@code K}): drop {@code K} bytes from the end of the form, then append.
+ * - {@code PREFIX} ({@code P},{@code K}): drop {@code P} from the front and {@code K} from the
+ * end of the form, then append.
+ * - {@code INFIX} ({@code I},{@code L},{@code K}): drop {@code L} bytes at offset {@code I} and
+ * {@code K} from the end of the form, then append.
+ *
+ *
+ * Surface forms are lower-cased on load, because {@link DictionaryLemmatizer} lower-cases the
+ * queried token before lookup. Dictionary data is supplied by the caller and never bundled.
+ */
+public final class MorfologikDictionaryReader {
+
+ /** The base-form encoder declared by a dictionary's {@code fsa.dict.encoder}. */
+ public enum BaseFormEncoding {
+
+ /** The encoded bytes are the base form verbatim. */
+ NONE,
+
+ /** The base form drops trailing bytes of the surface form, then appends the encoded rest. */
+ SUFFIX,
+
+ /** As {@link #SUFFIX}, and leading bytes of the surface form are dropped too. */
+ PREFIX,
+
+ /** As {@link #SUFFIX}, and a run of bytes inside the surface form is dropped too. */
+ INFIX
+ }
+
+ private static final int OFFSET = 'A';
+ private static final String KEY_SEPARATOR = "fsa.dict.separator";
+ private static final String KEY_ENCODING = "fsa.dict.encoding";
+ private static final String KEY_ENCODER = "fsa.dict.encoder";
+
+ private static final String FIELD_SEPARATOR = "\t";
+ private static final String LEMMA_SEPARATOR = "#";
+
+ /** Not instantiable. */
+ private MorfologikDictionaryReader() {
+ }
+
+ /**
+ * Reads a morfologik CFSA2 dictionary into a {@link DictionaryLemmatizer} using an explicit
+ * separator, encoder, and charset.
+ *
+ * @param dictionary The CFSA2 automaton, referenced by an open {@link InputStream}. Must not be
+ * {@code null}.
+ * @param separator The byte separating the form, encoded base, and tag fields.
+ * @param encoding The base-form encoder. Must not be {@code null}.
+ * @param charset The character encoding of the dictionary bytes. Must not be {@code null}.
+ * @return A {@link DictionaryLemmatizer} over the decoded entries.
+ * @throws IllegalArgumentException Thrown if {@code dictionary}, {@code encoding}, or
+ * {@code charset} is {@code null}.
+ * @throws IOException Thrown on IO errors, if the stream is not a CFSA2 automaton, or if an
+ * entry cannot be split into a form and encoded base.
+ */
+ public static DictionaryLemmatizer read(InputStream dictionary, byte separator,
+ BaseFormEncoding encoding, Charset charset) throws IOException {
+ if (dictionary == null) {
+ throw new IllegalArgumentException("dictionary must not be null");
+ }
+ if (encoding == null) {
+ throw new IllegalArgumentException("encoding must not be null");
+ }
+ if (charset == null) {
+ throw new IllegalArgumentException("charset must not be null");
+ }
+
+ final FsaSequenceReader automaton = FsaSequenceReader.read(dictionary);
+ final Map> entries = new LinkedHashMap<>();
+ try {
+ automaton.forEachSequence(
+ sequence -> addEntry(sequence, separator, encoding, charset, entries));
+ } catch (UncheckedIOException e) {
+ throw e.getCause();
+ }
+
+ final StringBuilder adapted = new StringBuilder();
+ for (final Map.Entry> entry : entries.entrySet()) {
+ adapted.append(entry.getKey())
+ .append(FIELD_SEPARATOR)
+ .append(String.join(LEMMA_SEPARATOR, entry.getValue()))
+ .append('\n');
+ }
+ final byte[] bytes = adapted.toString().getBytes(StandardCharsets.UTF_8);
+ return new DictionaryLemmatizer(new ByteArrayInputStream(bytes), StandardCharsets.UTF_8);
+ }
+
+ /**
+ * Reads a morfologik CFSA2 dictionary into a {@link DictionaryLemmatizer}, taking the separator,
+ * charset, and encoder from the dictionary's {@code .info} metadata.
+ *
+ * @param dictionary The CFSA2 automaton, referenced by an open {@link InputStream}. Must not be
+ * {@code null}.
+ * @param info The {@code .info} metadata properties, referenced by an open
+ * {@link InputStream}. Must not be {@code null} and must declare
+ * {@code fsa.dict.separator}, {@code fsa.dict.encoding}, and
+ * {@code fsa.dict.encoder}.
+ * @return A {@link DictionaryLemmatizer} over the decoded entries.
+ * @throws IllegalArgumentException Thrown if an argument is {@code null} or a required metadata
+ * key is missing or invalid.
+ * @throws IOException Thrown on IO errors or invalid dictionary content.
+ */
+ public static DictionaryLemmatizer read(InputStream dictionary, InputStream info)
+ throws IOException {
+ if (info == null) {
+ throw new IllegalArgumentException("info must not be null");
+ }
+ final Properties properties = new Properties();
+ properties.load(info);
+
+ final String separator = required(properties, KEY_SEPARATOR);
+ if (separator.length() != 1) {
+ throw new IllegalArgumentException(KEY_SEPARATOR + " must be a single character");
+ }
+ final Charset charset = Charset.forName(required(properties, KEY_ENCODING));
+ final BaseFormEncoding encoding = BaseFormEncoding.valueOf(
+ required(properties, KEY_ENCODER).toUpperCase(Locale.ROOT));
+ return read(dictionary, (byte) separator.charAt(0), encoding, charset);
+ }
+
+ /**
+ * Reads a metadata value that the dictionary must declare.
+ *
+ * @param properties The parsed {@code .info} metadata.
+ * @param key The metadata key to read.
+ * @return The declared value.
+ * @throws IllegalArgumentException Thrown if the key is not declared.
+ */
+ private static String required(Properties properties, String key) {
+ final String value = properties.getProperty(key);
+ if (value == null) {
+ throw new IllegalArgumentException("missing required metadata key: " + key);
+ }
+ return value;
+ }
+
+ /**
+ * Splits one accepted sequence into form, base form, and tag, and records it.
+ *
+ * @param sequence The accepted byte sequence, the fields joined by {@code separator}.
+ * @param separator The byte separating the fields.
+ * @param encoding The encoder the base form is stored with.
+ * @param charset The character encoding of the dictionary bytes.
+ * @param entries The entries collected so far, keyed by form and tag; updated in place.
+ * @throws UncheckedIOException Thrown if the sequence carries no separator or its base form
+ * cannot be decoded.
+ */
+ private static void addEntry(byte[] sequence, byte separator, BaseFormEncoding encoding,
+ Charset charset, Map> entries) {
+ final int firstSeparator = indexOf(sequence, separator, 0);
+ if (firstSeparator < 0) {
+ throw new UncheckedIOException(new IOException(
+ "morfologik entry has no separator: " + new String(sequence, charset)));
+ }
+ final int secondSeparator = indexOf(sequence, separator, firstSeparator + 1);
+ final int baseEnd = secondSeparator < 0 ? sequence.length : secondSeparator;
+
+ final byte[] form = slice(sequence, 0, firstSeparator);
+ final byte[] encodedBase = slice(sequence, firstSeparator + 1, baseEnd);
+ final String tag = secondSeparator < 0 ? ""
+ : new String(sequence, secondSeparator + 1, sequence.length - secondSeparator - 1, charset);
+
+ final byte[] base;
+ try {
+ base = decodeBaseForm(form, encodedBase, encoding);
+ } catch (IllegalArgumentException e) {
+ throw new UncheckedIOException(new IOException(
+ "malformed morfologik entry: " + new String(sequence, charset), e));
+ }
+
+ final String key = new String(form, charset).toLowerCase() + FIELD_SEPARATOR + tag;
+ entries.computeIfAbsent(key, k -> new LinkedHashSet<>()).add(new String(base, charset));
+ }
+
+ /**
+ * Recovers a base form from a surface form and its encoded representation.
+ *
+ * @param form The surface form bytes.
+ * @param encoded The encoded base bytes: control bytes followed by literal bytes to append.
+ * @param encoding The encoder that produced {@code encoded}.
+ * @return The decoded base form bytes.
+ * @throws IllegalArgumentException Thrown if {@code encoded} is too short for the encoder or the
+ * control bytes address positions outside {@code form}.
+ */
+ static byte[] decodeBaseForm(byte[] form, byte[] encoded, BaseFormEncoding encoding) {
+ switch (encoding) {
+ case NONE:
+ return encoded.clone();
+ case SUFFIX: {
+ require(encoded, 1, encoding);
+ final int keep = bounded(form.length - control(encoded[0]), form);
+ return join(form, 0, keep, encoded, 1);
+ }
+ case PREFIX: {
+ require(encoded, 2, encoding);
+ final int start = bounded(control(encoded[0]), form);
+ final int end = bounded(form.length - control(encoded[1]), form);
+ return join(form, Math.min(start, end), end, encoded, 2);
+ }
+ case INFIX: {
+ require(encoded, 3, encoding);
+ final int cut = bounded(control(encoded[0]), form);
+ final int resume = bounded(cut + control(encoded[1]), form);
+ final int end = bounded(form.length - control(encoded[2]), form);
+ return join3(form, cut, Math.max(resume, cut), Math.max(end, resume), encoded, 3);
+ }
+ default:
+ throw new IllegalArgumentException("unknown encoder: " + encoding);
+ }
+ }
+
+ /**
+ * Reads one control byte as the offset it encodes.
+ *
+ * @param b The control byte.
+ * @return The encoded offset, the byte value less {@code 'A'}.
+ */
+ private static int control(byte b) {
+ return (b & 0xff) - OFFSET;
+ }
+
+ /**
+ * Checks that an encoded base form carries all the control bytes its encoder needs.
+ *
+ * @param encoded The encoded base bytes.
+ * @param prefixBytes The number of control bytes the encoder needs.
+ * @param encoding The encoder, named in the failure message.
+ * @throws IllegalArgumentException Thrown if fewer bytes are present.
+ */
+ private static void require(byte[] encoded, int prefixBytes, BaseFormEncoding encoding) {
+ if (encoded.length < prefixBytes) {
+ throw new IllegalArgumentException(
+ encoding + " encoded base needs at least " + prefixBytes + " control byte(s)");
+ }
+ }
+
+ /**
+ * Checks that a decoded offset addresses a position within a surface form.
+ *
+ * @param index The offset to check.
+ * @param form The surface form bytes.
+ * @return The offset itself.
+ * @throws IllegalArgumentException Thrown if the offset lies outside {@code form}.
+ */
+ private static int bounded(int index, byte[] form) {
+ if (index < 0 || index > form.length) {
+ throw new IllegalArgumentException(
+ "encoded base addresses byte " + index + " outside a form of length " + form.length);
+ }
+ return index;
+ }
+
+ /**
+ * Joins one run of the surface form with the literal tail of an encoded base form.
+ *
+ * @param form The surface form bytes.
+ * @param from The first byte of the run to keep.
+ * @param to The byte after the run to keep.
+ * @param encoded The encoded base bytes.
+ * @param appendFrom The first literal byte of {@code encoded}, past its control bytes.
+ * @return The decoded base form bytes.
+ */
+ private static byte[] join(byte[] form, int from, int to, byte[] encoded, int appendFrom) {
+ final int kept = to - from;
+ final int appended = encoded.length - appendFrom;
+ final byte[] out = new byte[kept + appended];
+ System.arraycopy(form, from, out, 0, kept);
+ System.arraycopy(encoded, appendFrom, out, kept, appended);
+ return out;
+ }
+
+ /**
+ * Joins two runs of the surface form, the second past a dropped infix, with the literal tail of
+ * an encoded base form.
+ *
+ * @param form The surface form bytes.
+ * @param headEnd The byte after the leading run to keep, which starts at zero.
+ * @param tailFrom The first byte of the trailing run to keep.
+ * @param tailEnd The byte after the trailing run to keep.
+ * @param encoded The encoded base bytes.
+ * @param appendFrom The first literal byte of {@code encoded}, past its control bytes.
+ * @return The decoded base form bytes.
+ */
+ private static byte[] join3(byte[] form, int headEnd, int tailFrom, int tailEnd,
+ byte[] encoded, int appendFrom) {
+ final int tail = tailEnd - tailFrom;
+ final int appended = encoded.length - appendFrom;
+ final byte[] out = new byte[headEnd + tail + appended];
+ System.arraycopy(form, 0, out, 0, headEnd);
+ System.arraycopy(form, tailFrom, out, headEnd, tail);
+ System.arraycopy(encoded, appendFrom, out, headEnd + tail, appended);
+ return out;
+ }
+
+ /**
+ * Finds the next occurrence of a byte.
+ *
+ * @param array The bytes to search.
+ * @param value The byte to find.
+ * @param from The index to start at.
+ * @return The index of the first occurrence at or after {@code from}, or {@code -1} when absent.
+ */
+ private static int indexOf(byte[] array, byte value, int from) {
+ for (int i = from; i < array.length; i++) {
+ if (array[i] == value) {
+ return i;
+ }
+ }
+ return -1;
+ }
+
+ /**
+ * Copies a run of bytes.
+ *
+ * @param array The bytes to copy from.
+ * @param from The first byte to copy.
+ * @param to The byte after the last one to copy.
+ * @return The copied run.
+ */
+ private static byte[] slice(byte[] array, int from, int to) {
+ final byte[] out = new byte[to - from];
+ System.arraycopy(array, from, out, 0, to - from);
+ return out;
+ }
+}
diff --git a/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/PoliMorfDictionaryReader.java b/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/PoliMorfDictionaryReader.java
new file mode 100644
index 0000000000..e155bb3408
--- /dev/null
+++ b/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/PoliMorfDictionaryReader.java
@@ -0,0 +1,129 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package opennlp.tools.formats;
+
+import java.io.BufferedReader;
+import java.io.ByteArrayInputStream;
+import java.io.IOException;
+import java.io.InputStream;
+import java.io.InputStreamReader;
+import java.nio.charset.Charset;
+import java.nio.charset.StandardCharsets;
+import java.util.LinkedHashMap;
+import java.util.LinkedHashSet;
+import java.util.Map;
+
+import opennlp.tools.lemmatizer.DictionaryLemmatizer;
+import opennlp.tools.util.StringUtil;
+
+/**
+ * Builds a {@link DictionaryLemmatizer} from a morphological dictionary laid out as one
+ * tab-separated {@code surfaceForm\tlemma\ttag} row per entry.
+ *
+ * This is the layout published by PoliMorf, the successor grammatical dictionary of Polish
+ * (BSD 2-Clause), and emitted by exporting a morfologik dictionary to text; it is otherwise
+ * language-agnostic. {@link DictionaryLemmatizer} expects the opposite column order,
+ * {@code word\tpostag\tlemma}, and a single row per {@code (word, postag)} key with alternative
+ * lemmas joined by {@code #}. This reader performs that adaptation: it re-orders the columns and
+ * merges every lemma seen for the same form and tag into one entry, preserving first-seen order.
+ * The bundled dictionary data itself is never shipped; callers supply it.
+ *
+ * Surface forms are lower-cased on load because {@link DictionaryLemmatizer} lower-cases the
+ * queried token before lookup, so an entry keyed on a mixed-case form would otherwise be
+ * unreachable. Tags are kept verbatim and must match the tags the caller's tagger emits.
+ */
+public final class PoliMorfDictionaryReader {
+
+ private static final String FIELD_SEPARATOR = "\t";
+ private static final String LEMMA_SEPARATOR = "#";
+ private static final int MIN_FIELDS = 3;
+
+ /** Not instantiable. */
+ private PoliMorfDictionaryReader() {
+ }
+
+ /**
+ * Reads a UTF-8 {@code surfaceForm\tlemma\ttag} dictionary into a {@link DictionaryLemmatizer}.
+ *
+ * @param dictionary The dictionary referenced by an open {@link InputStream}. Must not be
+ * {@code null}.
+ * @return A {@link DictionaryLemmatizer} over the adapted entries.
+ * @throws IllegalArgumentException Thrown if {@code dictionary} is {@code null}.
+ * @throws IOException Thrown if IO errors occur while reading, or a non-blank line carries
+ * fewer than three tab-separated fields.
+ */
+ public static DictionaryLemmatizer read(InputStream dictionary) throws IOException {
+ return read(dictionary, StandardCharsets.UTF_8);
+ }
+
+ /**
+ * Reads a {@code surfaceForm\tlemma\ttag} dictionary into a {@link DictionaryLemmatizer}.
+ *
+ * @param dictionary The dictionary referenced by an open {@link InputStream}. Must not be
+ * {@code null}.
+ * @param charset The character encoding of the dictionary. Must not be {@code null}.
+ * @return A {@link DictionaryLemmatizer} over the adapted entries.
+ * @throws IllegalArgumentException Thrown if {@code dictionary} or {@code charset} is
+ * {@code null}.
+ * @throws IOException Thrown if IO errors occur while reading, or a non-blank line carries
+ * fewer than three tab-separated fields.
+ */
+ public static DictionaryLemmatizer read(InputStream dictionary, Charset charset)
+ throws IOException {
+ if (dictionary == null) {
+ throw new IllegalArgumentException("dictionary must not be null");
+ }
+ if (charset == null) {
+ throw new IllegalArgumentException("charset must not be null");
+ }
+
+ // key "form\ttag" -> alternative lemmas, both maps ordered so the output is deterministic.
+ final Map> entries = new LinkedHashMap<>();
+ try (BufferedReader reader = new BufferedReader(new InputStreamReader(dictionary, charset))) {
+ String line;
+ int lineNumber = 0;
+ while ((line = reader.readLine()) != null) {
+ lineNumber++;
+ if (StringUtil.isBlank(line)) {
+ continue;
+ }
+ final String[] fields = line.split(FIELD_SEPARATOR, -1);
+ if (fields.length < MIN_FIELDS) {
+ throw new IOException("PoliMorf line " + lineNumber
+ + " has fewer than " + MIN_FIELDS + " tab-separated fields: " + line);
+ }
+ final String form = fields[0].toLowerCase();
+ final String lemma = fields[1];
+ final String tag = fields[2];
+ entries.computeIfAbsent(form + FIELD_SEPARATOR + tag, key -> new LinkedHashSet<>())
+ .add(lemma);
+ }
+ }
+
+ final StringBuilder adapted = new StringBuilder();
+ for (final Map.Entry> entry : entries.entrySet()) {
+ adapted.append(entry.getKey())
+ .append(FIELD_SEPARATOR)
+ .append(String.join(LEMMA_SEPARATOR, entry.getValue()))
+ .append('\n');
+ }
+
+ final byte[] bytes = adapted.toString().getBytes(StandardCharsets.UTF_8);
+ return new DictionaryLemmatizer(new ByteArrayInputStream(bytes), StandardCharsets.UTF_8);
+ }
+}
diff --git a/opennlp-core/opennlp-formats/src/test/java/opennlp/tools/formats/CFSA2ReaderTest.java b/opennlp-core/opennlp-formats/src/test/java/opennlp/tools/formats/CFSA2ReaderTest.java
new file mode 100644
index 0000000000..558969315f
--- /dev/null
+++ b/opennlp-core/opennlp-formats/src/test/java/opennlp/tools/formats/CFSA2ReaderTest.java
@@ -0,0 +1,75 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package opennlp.tools.formats;
+
+import java.io.ByteArrayInputStream;
+import java.io.IOException;
+import java.nio.charset.StandardCharsets;
+import java.util.ArrayList;
+import java.util.Base64;
+import java.util.List;
+
+import org.junit.jupiter.api.Assertions;
+import org.junit.jupiter.api.Test;
+
+/**
+ * Validates the clean-room {@link CFSA2Reader} against a ground-truth automaton. The fixture below
+ * was generated once by morfologik's {@code CFSA2Serializer} from the words
+ * {@code {cat, cats, do, dog, dogs}} and committed as the expected encoding; the reader must
+ * recover exactly those sequences.
+ */
+public class CFSA2ReaderTest {
+
+ private static final String FIXTURE_BASE64 = "XGZzYcYABwgAdHNvZ2RjYUBeAwYKxePkYgDHYQg=";
+
+ private static List sequences(byte[] cfsa2) throws IOException {
+ final CFSA2Reader reader = CFSA2Reader.read(new ByteArrayInputStream(cfsa2));
+ final List out = new ArrayList<>();
+ reader.forEachSequence(bytes -> out.add(new String(bytes, StandardCharsets.UTF_8)));
+ return out;
+ }
+
+ /** Every accepted sequence is recovered, in the automaton's stored lexicographic order. */
+ @Test
+ void testEnumeratesAllAcceptedSequences() throws IOException {
+ Assertions.assertEquals(List.of("cat", "cats", "do", "dog", "dogs"),
+ sequences(Base64.getDecoder().decode(FIXTURE_BASE64)));
+ }
+
+ /** A stream that is not an FSA automaton fails loudly. */
+ @Test
+ void testRejectsNonFsaMagic() {
+ Assertions.assertThrows(IOException.class, () -> CFSA2Reader.read(
+ new ByteArrayInputStream("not an fsa header".getBytes(StandardCharsets.UTF_8))));
+ }
+
+ /** An FSA of a different version than CFSA2 fails loudly rather than misreading. */
+ @Test
+ void testRejectsUnsupportedVersion() {
+ final byte[] altered = Base64.getDecoder().decode(FIXTURE_BASE64);
+ altered[4] = 0x05;
+ Assertions.assertThrows(IOException.class,
+ () -> CFSA2Reader.read(new ByteArrayInputStream(altered)));
+ }
+
+ /** A null stream is rejected at the boundary. */
+ @Test
+ void testNullStreamRejected() {
+ Assertions.assertThrows(IllegalArgumentException.class, () -> CFSA2Reader.read(null));
+ }
+}
diff --git a/opennlp-core/opennlp-formats/src/test/java/opennlp/tools/formats/FSA5ReaderTest.java b/opennlp-core/opennlp-formats/src/test/java/opennlp/tools/formats/FSA5ReaderTest.java
new file mode 100644
index 0000000000..39423cdb7b
--- /dev/null
+++ b/opennlp-core/opennlp-formats/src/test/java/opennlp/tools/formats/FSA5ReaderTest.java
@@ -0,0 +1,74 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package opennlp.tools.formats;
+
+import java.io.ByteArrayInputStream;
+import java.io.IOException;
+import java.nio.charset.StandardCharsets;
+import java.util.ArrayList;
+import java.util.Base64;
+import java.util.List;
+
+import org.junit.jupiter.api.Assertions;
+import org.junit.jupiter.api.Test;
+
+/**
+ * Validates the clean-room {@link FSA5Reader} against a ground-truth automaton generated once by
+ * morfologik's {@code FSA5Serializer} from the words {@code {cat, cats, do, dog, dogs}}.
+ */
+public class FSA5ReaderTest {
+
+ private static final String FIXTURE_BASE64 = "XGZzYQVfKwEAAF4GY3BkBm8HZwdzA2EGdGM=";
+
+ private static List sequences(FsaSequenceReader reader) {
+ final List out = new ArrayList<>();
+ reader.forEachSequence(bytes -> out.add(new String(bytes, StandardCharsets.UTF_8)));
+ return out;
+ }
+
+ /** Every accepted sequence is recovered, in stored lexicographic order. */
+ @Test
+ void testEnumeratesAllAcceptedSequences() throws IOException {
+ final FSA5Reader reader = FSA5Reader.read(
+ new ByteArrayInputStream(Base64.getDecoder().decode(FIXTURE_BASE64)));
+ Assertions.assertEquals(List.of("cat", "cats", "do", "dog", "dogs"), sequences(reader));
+ }
+
+ /** The format-agnostic dispatcher recognizes and reads FSA5 by its version byte. */
+ @Test
+ void testDispatcherReadsFsa5() throws IOException {
+ final FsaSequenceReader reader = FsaSequenceReader.read(
+ new ByteArrayInputStream(Base64.getDecoder().decode(FIXTURE_BASE64)));
+ Assertions.assertEquals(List.of("cat", "cats", "do", "dog", "dogs"), sequences(reader));
+ }
+
+ /** A different FSA version fails loudly rather than misreading. */
+ @Test
+ void testRejectsUnsupportedVersion() {
+ final byte[] altered = Base64.getDecoder().decode(FIXTURE_BASE64);
+ altered[4] = (byte) 0x99;
+ Assertions.assertThrows(IOException.class,
+ () -> FSA5Reader.read(new ByteArrayInputStream(altered)));
+ }
+
+ /** A null stream is rejected at the boundary. */
+ @Test
+ void testNullStreamRejected() {
+ Assertions.assertThrows(IllegalArgumentException.class, () -> FSA5Reader.read(null));
+ }
+}
diff --git a/opennlp-core/opennlp-formats/src/test/java/opennlp/tools/formats/MorfologikDictionaryReaderTest.java b/opennlp-core/opennlp-formats/src/test/java/opennlp/tools/formats/MorfologikDictionaryReaderTest.java
new file mode 100644
index 0000000000..40eb404f3c
--- /dev/null
+++ b/opennlp-core/opennlp-formats/src/test/java/opennlp/tools/formats/MorfologikDictionaryReaderTest.java
@@ -0,0 +1,144 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package opennlp.tools.formats;
+
+import java.io.ByteArrayInputStream;
+import java.io.IOException;
+import java.io.InputStream;
+import java.nio.charset.StandardCharsets;
+import java.util.Base64;
+
+import org.junit.jupiter.api.Assertions;
+import org.junit.jupiter.api.Test;
+import org.junit.jupiter.params.ParameterizedTest;
+import org.junit.jupiter.params.provider.CsvSource;
+
+import opennlp.tools.formats.MorfologikDictionaryReader.BaseFormEncoding;
+import opennlp.tools.lemmatizer.DictionaryLemmatizer;
+
+/**
+ * Validates the clean-room {@link MorfologikDictionaryReader}. The base-form decode cases and the
+ * end-to-end CFSA2 dictionary below were generated by morfologik's own encoder and serializer, so
+ * the reader is checked against that library's output rather than against restated assumptions.
+ */
+public class MorfologikDictionaryReaderTest {
+
+ // A CFSA2 dictionary of cat+A+NN, cats+B+NNS, dogs+B+NNS, mice+Douse+NN (SUFFIX encoder, '+'
+ // separator, ISO-8859-1), built by morfologik's CFSA2Serializer.
+ private static final String DICTIONARY_BASE64 =
+ "XGZzYcYABxIAdXRtaWdkYVNEQkFvZWNzTitAXgMOHwYVw8TOzdHJzMHPzdHQcADMxc/RytHQ0GgAx8KRTxhLEQ==";
+
+ // The same dictionary serialized in the older FSA5 format, to prove both formats are read.
+ private static final String FSA5_DICTIONARY_BASE64 =
+ "XGZzYQVfKwIAAABeBmPIAWQwAW0GaQZjBmUGKwZEBm8GdQZzBmUGKwZOBk4DAG8G"
+ + "ZwZzBisGQgYrBk4GTgZTAwBhBnQGKxgCc2IBQfoA";
+
+ private static byte[] iso(String s) {
+ return s.getBytes(StandardCharsets.ISO_8859_1);
+ }
+
+ /** Each encoder recovers the base form morfologik encoded. */
+ @ParameterizedTest
+ @CsvSource({
+ "SUFFIX, cats, B, cat",
+ "SUFFIX, mice, Douse, mouse",
+ "SUFFIX, foobar, Gbar, bar",
+ "SUFFIX, cat, A, cat",
+ "PREFIX, cats, AB, cat",
+ "PREFIX, foobar, DA, bar",
+ "PREFIX, unhappy, CA, happy",
+ "PREFIX, mice, ADouse, mouse",
+ "INFIX, cats, AAB, cat",
+ "INFIX, foobar, ADA, bar",
+ "INFIX, unhappy, ACA, happy",
+ "INFIX, mice, AADouse, mouse",
+ "NONE, cats, cat, cat",
+ })
+ void testDecodeBaseForm(String encoder, String form, String encoded, String base) {
+ final byte[] decoded = MorfologikDictionaryReader.decodeBaseForm(
+ iso(form), iso(encoded), BaseFormEncoding.valueOf(encoder));
+ Assertions.assertEquals(base, new String(decoded, StandardCharsets.ISO_8859_1));
+ }
+
+ private static DictionaryLemmatizer dictionary() throws IOException {
+ return MorfologikDictionaryReader.read(
+ new ByteArrayInputStream(Base64.getDecoder().decode(DICTIONARY_BASE64)),
+ (byte) '+', BaseFormEncoding.SUFFIX, StandardCharsets.ISO_8859_1);
+ }
+
+ /** The CFSA2 dictionary lemmatizes each surface form under its tag; misses yield "O". */
+ @Test
+ void testReadsCfsa2DictionaryEndToEnd() throws IOException {
+ final DictionaryLemmatizer lemmatizer = dictionary();
+
+ Assertions.assertArrayEquals(new String[] {"cat"},
+ lemmatizer.lemmatize(new String[] {"cats"}, new String[] {"NNS"}));
+ Assertions.assertArrayEquals(new String[] {"dog"},
+ lemmatizer.lemmatize(new String[] {"dogs"}, new String[] {"NNS"}));
+ Assertions.assertArrayEquals(new String[] {"mouse"},
+ lemmatizer.lemmatize(new String[] {"mice"}, new String[] {"NN"}));
+ Assertions.assertArrayEquals(new String[] {"cat"},
+ lemmatizer.lemmatize(new String[] {"cat"}, new String[] {"NN"}));
+ Assertions.assertArrayEquals(new String[] {"O"},
+ lemmatizer.lemmatize(new String[] {"mice"}, new String[] {"NNS"}));
+ }
+
+ /** The same dictionary in the older FSA5 format is read identically. */
+ @Test
+ void testReadsFsa5DictionaryEndToEnd() throws IOException {
+ final DictionaryLemmatizer lemmatizer = MorfologikDictionaryReader.read(
+ new ByteArrayInputStream(Base64.getDecoder().decode(FSA5_DICTIONARY_BASE64)),
+ (byte) '+', BaseFormEncoding.SUFFIX, StandardCharsets.ISO_8859_1);
+
+ Assertions.assertArrayEquals(new String[] {"mouse"},
+ lemmatizer.lemmatize(new String[] {"mice"}, new String[] {"NN"}));
+ Assertions.assertArrayEquals(new String[] {"cat"},
+ lemmatizer.lemmatize(new String[] {"cats"}, new String[] {"NNS"}));
+ }
+
+ /** The separator, encoding, and encoder are taken from .info metadata when supplied. */
+ @Test
+ void testReadsWithInfoMetadata() throws IOException {
+ final String info = "fsa.dict.separator=+\n"
+ + "fsa.dict.encoding=iso-8859-1\n"
+ + "fsa.dict.encoder=suffix\n";
+ final InputStream dict = new ByteArrayInputStream(Base64.getDecoder().decode(DICTIONARY_BASE64));
+
+ final DictionaryLemmatizer lemmatizer = MorfologikDictionaryReader.read(
+ dict, new ByteArrayInputStream(info.getBytes(StandardCharsets.ISO_8859_1)));
+
+ Assertions.assertArrayEquals(new String[] {"mouse"},
+ lemmatizer.lemmatize(new String[] {"mice"}, new String[] {"NN"}));
+ }
+
+ /** Missing required metadata fails loudly. */
+ @Test
+ void testMissingMetadataKeyRejected() {
+ final String info = "fsa.dict.separator=+\n";
+ final InputStream dict = new ByteArrayInputStream(Base64.getDecoder().decode(DICTIONARY_BASE64));
+ Assertions.assertThrows(IllegalArgumentException.class, () -> MorfologikDictionaryReader.read(
+ dict, new ByteArrayInputStream(info.getBytes(StandardCharsets.ISO_8859_1))));
+ }
+
+ /** A null dictionary stream is rejected at the boundary. */
+ @Test
+ void testNullDictionaryRejected() {
+ Assertions.assertThrows(IllegalArgumentException.class, () -> MorfologikDictionaryReader.read(
+ null, (byte) '+', BaseFormEncoding.SUFFIX, StandardCharsets.ISO_8859_1));
+ }
+}
diff --git a/opennlp-core/opennlp-formats/src/test/java/opennlp/tools/formats/PoliMorfDictionaryReaderUsageExampleTest.java b/opennlp-core/opennlp-formats/src/test/java/opennlp/tools/formats/PoliMorfDictionaryReaderUsageExampleTest.java
new file mode 100644
index 0000000000..e46dfaba5e
--- /dev/null
+++ b/opennlp-core/opennlp-formats/src/test/java/opennlp/tools/formats/PoliMorfDictionaryReaderUsageExampleTest.java
@@ -0,0 +1,113 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package opennlp.tools.formats;
+
+import java.io.ByteArrayInputStream;
+import java.io.IOException;
+import java.io.InputStream;
+import java.nio.charset.StandardCharsets;
+import java.util.List;
+
+import org.junit.jupiter.api.Assertions;
+import org.junit.jupiter.api.Test;
+
+import opennlp.tools.lemmatizer.DictionaryLemmatizer;
+
+/**
+ * Runs the manual's PoliMorf reader examples (docbkx {@code lemmatizer.xml}) verbatim over a
+ * small illustrative dictionary in PoliMorf's {@code surfaceForm\tlemma\ttag} format. It is a
+ * hand-built fixture, not the PoliMorf distribution; every value the chapter states is asserted
+ * here, so a change breaking this test breaks the manual.
+ */
+public class PoliMorfDictionaryReaderUsageExampleTest {
+
+ private static final String[] ROWS = {
+ "pies\tpies\tsubst:sg:nom:m2",
+ "psa\tpies\tsubst:sg:gen:m2",
+ "psy\tpies\tsubst:pl:nom:m2",
+ "kota\tkot\tsubst:sg:gen:m2",
+ // Same (form, tag) listed with two lemmas: the reader merges them into one entry.
+ "formy\tforma\tsubst:pl:nom:f",
+ "formy\tform\tsubst:pl:nom:f",
+ };
+
+ private static InputStream dictionary(String text) {
+ return new ByteArrayInputStream(text.getBytes(StandardCharsets.UTF_8));
+ }
+
+ private static DictionaryLemmatizer fixture() throws IOException {
+ return PoliMorfDictionaryReader.read(dictionary(String.join("\n", ROWS) + "\n"));
+ }
+
+ /** A form and its tag resolve to the base form; a homograph tag or unknown form yields "O". */
+ @Test
+ void testResolvesFormAndTagToLemma() throws IOException {
+ final DictionaryLemmatizer lemmatizer = fixture();
+
+ Assertions.assertArrayEquals(new String[] {"pies"},
+ lemmatizer.lemmatize(new String[] {"psa"}, new String[] {"subst:sg:gen:m2"}));
+ Assertions.assertArrayEquals(new String[] {"kot"},
+ lemmatizer.lemmatize(new String[] {"kota"}, new String[] {"subst:sg:gen:m2"}));
+ // A known form under a tag it never carries is a miss.
+ Assertions.assertArrayEquals(new String[] {"O"},
+ lemmatizer.lemmatize(new String[] {"psa"}, new String[] {"adj:sg:nom:m2:pos"}));
+ // An unknown form is a miss.
+ Assertions.assertArrayEquals(new String[] {"O"},
+ lemmatizer.lemmatize(new String[] {"kanapa"}, new String[] {"subst:sg:nom:f"}));
+ }
+
+ /** Lookup is case-insensitive because forms are lower-cased on load. */
+ @Test
+ void testLookupIsCaseInsensitive() throws IOException {
+ Assertions.assertArrayEquals(new String[] {"pies"},
+ fixture().lemmatize(new String[] {"Psa"}, new String[] {"subst:sg:gen:m2"}));
+ }
+
+ /** Every lemma listed for a form and tag is kept, in first-seen order. */
+ @Test
+ void testAlternativeLemmasAreMerged() throws IOException {
+ final List> lemmas =
+ fixture().lemmatize(List.of("formy"), List.of("subst:pl:nom:f"));
+
+ Assertions.assertEquals(List.of(List.of("forma", "form")), lemmas);
+ }
+
+ /** Blank lines are skipped rather than treated as entries. */
+ @Test
+ void testBlankLinesAreSkipped() throws IOException {
+ final DictionaryLemmatizer lemmatizer = PoliMorfDictionaryReader.read(
+ dictionary("\npsa\tpies\tsubst:sg:gen:m2\n \n"));
+
+ Assertions.assertArrayEquals(new String[] {"pies"},
+ lemmatizer.lemmatize(new String[] {"psa"}, new String[] {"subst:sg:gen:m2"}));
+ }
+
+ /** A non-blank line with fewer than three fields fails loudly. */
+ @Test
+ void testTooFewFieldsThrows() {
+ Assertions.assertThrows(IOException.class,
+ () -> PoliMorfDictionaryReader.read(dictionary("psa\tpies\n")));
+ }
+
+ /** A null dictionary stream is rejected at the boundary. */
+ @Test
+ void testNullDictionaryRejected() {
+ Assertions.assertThrows(IllegalArgumentException.class,
+ () -> PoliMorfDictionaryReader.read(null));
+ }
+}
diff --git a/opennlp-distr/pom.xml b/opennlp-distr/pom.xml
index e9092d8821..39269575e8 100644
--- a/opennlp-distr/pom.xml
+++ b/opennlp-distr/pom.xml
@@ -91,6 +91,10 @@
org.apache.opennlp
opennlp-spellcheck
+
+ org.apache.opennlp
+ opennlp-wordnet
+
diff --git a/opennlp-distr/src/main/assembly/bin.xml b/opennlp-distr/src/main/assembly/bin.xml
index 2db4eafc65..36c556d473 100644
--- a/opennlp-distr/src/main/assembly/bin.xml
+++ b/opennlp-distr/src/main/assembly/bin.xml
@@ -239,6 +239,13 @@
docs/apidocs/opennlp-spellcheck
+
+ ../opennlp-extensions/opennlp-wordnet/target/reports/apidocs
+ 644
+ 755
+ docs/apidocs/opennlp-wordnet
+
+
../opennlp-extensions/opennlp-uima/target/reports/apidocs
644
diff --git a/opennlp-docs/src/docbkx/lemmatizer.xml b/opennlp-docs/src/docbkx/lemmatizer.xml
index 4d33948e1a..09158161ed 100644
--- a/opennlp-docs/src/docbkx/lemmatizer.xml
+++ b/opennlp-docs/src/docbkx/lemmatizer.xml
@@ -316,4 +316,60 @@ Accuracy: 0.9659110277825124]]>
+
+ Dictionary lemmatization from PoliMorf tables
+
+ DictionaryLemmatizer reads a word, postag,
+ lemma table. Many published morphological dictionaries use the opposite
+ column order, one surfaceForm, lemma, tag row
+ per entry. This is the layout of PoliMorf, the successor grammatical dictionary of
+ Polish, which is released under a permissive BSD 2-Clause license, and it is also
+ what exporting a morfologik dictionary to text produces.
+ PoliMorfDictionaryReader adapts such a table into a
+ DictionaryLemmatizer: it re-orders the columns, lower-cases each surface
+ form to match the case-insensitive lookup, and merges every lemma listed for the same
+ form and tag into one entry. The dictionary data is supplied by the caller and never
+ bundled. PoliMorfDictionaryReaderUsageExampleTest asserts the behavior
+ shown here over a small in-repo table in this format:
+ lemmatag, one row per entry
+String table = "psa\tpies\tsubst:sg:gen:m2\n"
+ + "psy\tpies\tsubst:pl:nom:m2\n"
+ + "kota\tkot\tsubst:sg:gen:m2\n";
+
+Lemmatizer lemmatizer = PoliMorfDictionaryReader.read(
+ new ByteArrayInputStream(table.getBytes(StandardCharsets.UTF_8)));
+
+lemmatizer.lemmatize(new String[] {"psa"}, new String[] {"subst:sg:gen:m2"}); // ["pies"]
+lemmatizer.lemmatize(new String[] {"Psa"}, new String[] {"subst:sg:gen:m2"}); // ["pies"], case-insensitive
+lemmatizer.lemmatize(new String[] {"psa"}, new String[] {"adj:sg:nom:m2:pos"}); // ["O"], tag miss]]>
+
+
+
+
+
+ Reading morfologik dictionaries
+
+ Many languages publish a morphological dictionary as a compact finite-state
+ automaton (a .dict file) paired with a .info metadata
+ file. MorfologikDictionaryReader reads both automaton formats,
+ FSA5 and CFSA2, with no third-party dependency, and decodes each entry, which is
+ a surface form, a base form stored relative to it, and a tag, into a
+ DictionaryLemmatizer. The four base-form encoders (none, suffix,
+ prefix, infix) are all decoded. The separator, character encoding, and encoder
+ are read from the .info file, or passed explicitly. The dictionary
+ data is supplied by the caller and never bundled.
+ MorfologikDictionaryReaderTest asserts the behavior shown here.
+
+
+ The companion PoliMorfDictionaryReader covers the same job for a
+ plain tab-separated table rather than an automaton, so a permissively licensed
+ dictionary such as PoliMorf and an existing automaton dictionary both reach the
+ same lemmatizer.
+
+
diff --git a/opennlp-docs/src/docbkx/opennlp.xml b/opennlp-docs/src/docbkx/opennlp.xml
index 36641c2c89..f7a8a39406 100644
--- a/opennlp-docs/src/docbkx/opennlp.xml
+++ b/opennlp-docs/src/docbkx/opennlp.xml
@@ -109,6 +109,7 @@ under the License.
+
diff --git a/opennlp-docs/src/docbkx/wordnet.xml b/opennlp-docs/src/docbkx/wordnet.xml
new file mode 100644
index 0000000000..7b1100820c
--- /dev/null
+++ b/opennlp-docs/src/docbkx/wordnet.xml
@@ -0,0 +1,129 @@
+
+
+
+
+
+
+ WordNet
+
+
+ Introduction
+
+ The opennlp-wordnet module loads a WordNet-style lexicon into
+ a LexicalKnowledgeBase and looks up synsets by lemma and part
+ of speech. Two readers are provided: WnLmfReader for the
+ Global WordNet Association WN-LMF XML interchange format, and
+ WndbReader for the classic Princeton WordNet database file
+ layout. Both return an immutable, thread-safe knowledge base. On top of
+ lookup, the module can Morphy-lemmatize, expand a term through synonym and
+ hypernym links, and score synset similarity on the hypernym graph.
+
+
+
+
+ Loading a lexicon
+
+ WN-LMF is the usual choice for Open English WordNet and other GWA
+ wordnets. WNDB remains available for a local Princeton-style
+ dict directory. The examples below use the miniature
+ fixtures from the module's tests; replace the paths with a full lexicon
+ in application code. WordNetUsageExampleTest asserts the
+ behavior shown here.
+
+
+
+
+
+
+ Lookup
+
+ Lookups are scoped by part of speech and fold case and underscores the
+ same way the readers index lemmas. Against the miniature WN-LMF fixture,
+ the noun dog has one sense:
+ senses = lexicon.lookup("dog", WordNetPOS.NOUN);
+// senses.size() = 1
+// senses.get(0).id() = "mini-n1"
+// senses.get(0).lemmas() = ["dog", "domestic dog"]
+// senses.get(0).gloss() = "a domesticated canid"]]>
+
+
+
+
+
+ Morphy lemmatization
+
+ MorphyLemmatizer implements the Morphy algorithm against a
+ loaded lexicon and the irregular-form exception lists
+ (noun.exc, verb.exc, adj.exc,
+ adv.exc). Exception hits are returned first; regular
+ detachments are kept only when the candidate is in the lexicon. Unknown
+ forms yield the marker O.
+
+
+
+
+
+
+ Lexical expansion
+
+ LexicalExpander turns a term into related terms from the
+ knowledge base: synonyms sharing its synsets, hypernym ancestors up to a
+ configured depth, and optionally direct hyponyms. Each
+ Expansion carries a heuristic weight in
+ (0, 1]: the first sense starts at 1.0, later
+ senses multiply by the sense decay, and each hypernym or hyponym step
+ multiplies by the depth decay. The input term itself is never returned.
+ Defaults use depth 1, sense decay 0.5, and
+ depth decay 0.5.
+ LexicalExpansionUsageExampleTest asserts the behavior shown
+ here.
+ expansions = LexicalExpander.builder(lexicon)
+ .build()
+ .expand("dog", WordNetPOS.NOUN);
+
+// "domestic dog": SYNONYM, senseRank 0, weight 1.0
+// "canid": HYPERNYM, depth 1, weight 0.5
+// "frank": SYNONYM, senseRank 1, weight 0.5]]>
+
+
+
+
+
+ Synset similarity
+
+ SynsetSimilarity scores two synset identifiers on the
+ hypernym graph. Path similarity is 1 / (1 + d) for the
+ shortest distance d through a common ancestor. Wu-Palmer
+ similarity relates the depth of the deepest common ancestor to the depths
+ of both synsets. Unrelated synsets score 0.
+ LexicalExpansionUsageExampleTest asserts the behavior shown
+ here.
+
+
+
+
+
diff --git a/opennlp-extensions/opennlp-wordnet/pom.xml b/opennlp-extensions/opennlp-wordnet/pom.xml
new file mode 100644
index 0000000000..0721fcfc98
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/pom.xml
@@ -0,0 +1,59 @@
+
+
+
+
+
+ 4.0.0
+
+ org.apache.opennlp
+ opennlp-extensions
+ 3.0.0-SNAPSHOT
+
+
+ opennlp-wordnet
+ jar
+ Apache OpenNLP :: Ext :: WordNet
+
+
+
+ org.apache.opennlp
+ opennlp-api
+
+
+
+ org.junit.jupiter
+ junit-jupiter-api
+ test
+
+
+
+ org.junit.jupiter
+ junit-jupiter-engine
+ test
+
+
+
+ org.junit.jupiter
+ junit-jupiter-params
+ test
+
+
+
+
diff --git a/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/HypernymTyper.java b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/HypernymTyper.java
new file mode 100644
index 0000000000..02756d0e1e
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/HypernymTyper.java
@@ -0,0 +1,167 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package opennlp.wordnet;
+
+import java.util.ArrayDeque;
+import java.util.Deque;
+import java.util.HashMap;
+import java.util.LinkedHashMap;
+import java.util.List;
+import java.util.Map;
+import java.util.Optional;
+
+import opennlp.tools.commons.ThreadSafe;
+import opennlp.tools.util.StringUtil;
+import opennlp.tools.wordnet.LexicalKnowledgeBase;
+import opennlp.tools.wordnet.Synset;
+import opennlp.tools.wordnet.WordNetPOS;
+import opennlp.tools.wordnet.WordNetRelation;
+
+/**
+ * Types a noun by walking its hypernym chain to the nearest registered anchor: the
+ * caller names anchor concepts by lemma, {@code person}, {@code organization},
+ * {@code location}, and any word whose senses lead up to an anchor's synsets receives
+ * that anchor's label. The nearest anchor wins, so a more specific registered concept
+ * beats a general one.
+ *
+ * Anchors are resolved against the knowledge base at construction and follow its
+ * sense inventory; nothing beyond the caller's anchor choice is built in. Words with no
+ * sense reaching an anchor get no type.
+ *
+ * The typer reads only immutable state and is safe to share between threads.
+ */
+@ThreadSafe
+public class HypernymTyper {
+
+ /** The relations that lead from a synset to its generalizations. */
+ private static final List UPWARD_RELATIONS =
+ List.of(WordNetRelation.HYPERNYM, WordNetRelation.INSTANCE_HYPERNYM);
+
+ private final LexicalKnowledgeBase knowledgeBase;
+ private final Map labelBySynsetId;
+
+ /**
+ * Initializes the typer.
+ *
+ * @param knowledgeBase The knowledge base to walk. Must not be {@code null}.
+ * @param anchors The anchor lemmas mapped to the labels they confer, for example
+ * {@code person} to {@code person}. Every lemma is resolved as a noun;
+ * all its senses anchor. Must not be {@code null} or empty, and no
+ * lemma or label may be blank.
+ * @throws IllegalArgumentException Thrown if a parameter is {@code null},
+ * {@code anchors} is empty or holds a blank entry, or an anchor lemma is
+ * unknown to the knowledge base.
+ */
+ public HypernymTyper(LexicalKnowledgeBase knowledgeBase, Map anchors) {
+ if (knowledgeBase == null) {
+ throw new IllegalArgumentException("knowledgeBase must not be null");
+ }
+ if (anchors == null || anchors.isEmpty()) {
+ throw new IllegalArgumentException("anchors must not be null or empty");
+ }
+ this.knowledgeBase = knowledgeBase;
+ final Map labels = new LinkedHashMap<>();
+ for (final Map.Entry anchor : anchors.entrySet()) {
+ if (anchor.getKey() == null || StringUtil.isBlank(anchor.getKey())
+ || anchor.getValue() == null || StringUtil.isBlank(anchor.getValue())) {
+ throw new IllegalArgumentException("anchors must not contain blank entries");
+ }
+ final List senses = knowledgeBase.lookup(anchor.getKey(), WordNetPOS.NOUN);
+ if (senses.isEmpty()) {
+ throw new IllegalArgumentException(
+ "anchor lemma is unknown to the knowledge base: " + anchor.getKey());
+ }
+ for (final Synset sense : senses) {
+ labels.putIfAbsent(sense.id(), anchor.getValue());
+ }
+ }
+ this.labelBySynsetId = Map.copyOf(labels);
+ }
+
+ /**
+ * Types a noun by its nearest anchored hypernym.
+ *
+ * @param lemma The noun lemma to type. Must not be {@code null} or blank.
+ * @return The label of the nearest anchor over all senses, or empty when no sense
+ * reaches an anchor.
+ * @throws IllegalArgumentException Thrown if {@code lemma} is {@code null} or blank.
+ */
+ public Optional type(String lemma) {
+ if (lemma == null || StringUtil.isBlank(lemma)) {
+ throw new IllegalArgumentException("lemma must not be null or blank");
+ }
+ String bestLabel = null;
+ int bestDistance = Integer.MAX_VALUE;
+ for (final Synset sense : knowledgeBase.lookup(lemma, WordNetPOS.NOUN)) {
+ final int[] distance = new int[1];
+ final String label = nearestAnchor(sense.id(), distance);
+ if (label != null && distance[0] < bestDistance) {
+ bestDistance = distance[0];
+ bestLabel = label;
+ }
+ }
+ return Optional.ofNullable(bestLabel);
+ }
+
+ /**
+ * Types a specific synset by its nearest anchored hypernym.
+ *
+ * @param synsetId The synset identifier. Must not be {@code null}.
+ * @return The nearest anchor's label, or empty when no ancestor is anchored.
+ * @throws IllegalArgumentException Thrown if {@code synsetId} is {@code null}.
+ */
+ public Optional typeSynset(String synsetId) {
+ if (synsetId == null) {
+ throw new IllegalArgumentException("synsetId must not be null");
+ }
+ return Optional.ofNullable(nearestAnchor(synsetId, new int[1]));
+ }
+
+ /**
+ * Walks up the hypernym graph breadth first to the closest anchored synset. Visiting each
+ * synset once bounds the walk even on cyclic data.
+ *
+ * @param synsetId The synset to start from. Must not be {@code null}.
+ * @param distanceOut A single-element array that receives the edge count to the anchor found;
+ * left untouched when no ancestor is anchored.
+ * @return The label of the nearest anchored synset, or {@code null} when none is reachable.
+ */
+ private String nearestAnchor(String synsetId, int[] distanceOut) {
+ final Deque queue = new ArrayDeque<>();
+ final Map depths = new HashMap<>();
+ queue.add(synsetId);
+ depths.put(synsetId, 0);
+ while (!queue.isEmpty()) {
+ final String current = queue.remove();
+ final String label = labelBySynsetId.get(current);
+ if (label != null) {
+ distanceOut[0] = depths.get(current);
+ return label;
+ }
+ final int parentDepth = depths.get(current) + 1;
+ for (final WordNetRelation relation : UPWARD_RELATIONS) {
+ for (final String parent : knowledgeBase.related(current, relation)) {
+ if (depths.putIfAbsent(parent, parentDepth) == null) {
+ queue.add(parent);
+ }
+ }
+ }
+ }
+ return null;
+ }
+}
diff --git a/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/InMemoryWordNetLexicon.java b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/InMemoryWordNetLexicon.java
new file mode 100644
index 0000000000..2401bfab6a
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/InMemoryWordNetLexicon.java
@@ -0,0 +1,156 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package opennlp.wordnet;
+
+import java.util.ArrayList;
+import java.util.Collection;
+import java.util.Collections;
+import java.util.HashMap;
+import java.util.List;
+import java.util.Map;
+import java.util.Optional;
+
+import opennlp.tools.commons.ThreadSafe;
+import opennlp.tools.wordnet.LexicalKnowledgeBase;
+import opennlp.tools.wordnet.Synset;
+import opennlp.tools.wordnet.WordNetPOS;
+import opennlp.tools.wordnet.WordNetRelation;
+
+/**
+ * The immutable in-memory {@link LexicalKnowledgeBase} both readers produce: a synset table plus a
+ * folded (lemma, part of speech) index. Package-private because it is a reader product, not a
+ * public entry point; consumers hold it as {@link LexicalKnowledgeBase}.
+ *
+ * Keys and queries are folded identically (see {@link LemmaFolding}). Construction verifies
+ * referential integrity: every relation target of every synset must resolve to a synset in the
+ * table, so a lexicon can never hand out a dangling identifier. After construction all state is
+ * immutable, making instances safe for concurrent lookups.
+ */
+@ThreadSafe
+final class InMemoryWordNetLexicon implements LexicalKnowledgeBase {
+
+ private final Map synsetsById;
+ private final Map> senseIndex;
+
+ /**
+ * Indexes the given synsets.
+ *
+ * @param synsetsById The synset table keyed by synset id; every key must equal its synset's
+ * {@link Synset#id() id}. Must not be {@code null}.
+ * @param senseOrder The sense order per folded (lemma, part of speech) key: for each key the
+ * ids of the synsets containing the lemma, most salient sense first, each
+ * id resolvable in {@code synsetsById} and free of duplicates per key.
+ * Must not be {@code null}.
+ * @throws IllegalArgumentException Thrown if a relation target or sense entry does not
+ * resolve, or a key disagrees with its synset id.
+ */
+ InMemoryWordNetLexicon(Map synsetsById, Map> senseOrder) {
+ if (synsetsById == null) {
+ throw new IllegalArgumentException("SynsetsById must not be null");
+ }
+ if (senseOrder == null) {
+ throw new IllegalArgumentException("SenseOrder must not be null");
+ }
+ final Map byId = new HashMap<>(synsetsById.size() * 2);
+ for (final Map.Entry entry : synsetsById.entrySet()) {
+ if (entry.getValue() == null || !entry.getValue().id().equals(entry.getKey())) {
+ throw new IllegalArgumentException(
+ "Synset table key " + entry.getKey() + " does not match its synset");
+ }
+ byId.put(entry.getKey(), entry.getValue());
+ }
+ for (final Synset synset : byId.values()) {
+ for (final Map.Entry> relation :
+ synset.relations().entrySet()) {
+ for (final String target : relation.getValue()) {
+ if (!byId.containsKey(target)) {
+ throw new IllegalArgumentException("Synset " + synset.id() + " has a "
+ + relation.getKey() + " relation to unknown synset " + target);
+ }
+ }
+ }
+ }
+ final Map> index = new HashMap<>(senseOrder.size() * 2);
+ for (final Map.Entry> entry : senseOrder.entrySet()) {
+ final List senses = new ArrayList<>(entry.getValue().size());
+ for (final String synsetId : entry.getValue()) {
+ final Synset synset = byId.get(synsetId);
+ if (synset == null) {
+ throw new IllegalArgumentException("Sense index entry " + entry.getKey().lemma()
+ + " (" + entry.getKey().pos() + ") references unknown synset " + synsetId);
+ }
+ senses.add(synset);
+ }
+ index.put(entry.getKey(), List.copyOf(senses));
+ }
+ this.synsetsById = byId;
+ this.senseIndex = index;
+ }
+
+ /** {@inheritDoc} */
+ @Override
+ public List lookup(String lemma, WordNetPOS pos) {
+ if (lemma == null) {
+ throw new IllegalArgumentException("Lemma must not be null");
+ }
+ if (pos == null) {
+ throw new IllegalArgumentException("Pos must not be null");
+ }
+ final List senses = senseIndex.get(LemmaKey.of(lemma, pos));
+ return senses == null ? List.of() : senses;
+ }
+
+ /** {@inheritDoc} */
+ @Override
+ public Optional synset(String synsetId) {
+ if (synsetId == null) {
+ throw new IllegalArgumentException("SynsetId must not be null");
+ }
+ return Optional.ofNullable(synsetsById.get(synsetId));
+ }
+
+ /** {@return the number of synsets in this lexicon} */
+ int size() {
+ return synsetsById.size();
+ }
+
+ /** {@return all synsets, for equivalence checks and diagnostics within this package} */
+ Collection synsets() {
+ return Collections.unmodifiableCollection(synsetsById.values());
+ }
+
+ /**
+ * A folded sense-index key. Build with {@link #of(String, WordNetPOS)} so every key passes
+ * through the same fold as every query.
+ *
+ * @param lemma The folded lemma.
+ * @param pos The part of speech.
+ */
+ record LemmaKey(String lemma, WordNetPOS pos) {
+
+ /**
+ * Folds a written form into a key: lowercase with the root locale, underscore as space.
+ *
+ * @param writtenForm The lemma as written in the source or query. Must not be {@code null}.
+ * @param pos The part of speech. Must not be {@code null}.
+ * @return The folded key.
+ */
+ static LemmaKey of(String writtenForm, WordNetPOS pos) {
+ return new LemmaKey(LemmaFolding.fold(writtenForm), pos);
+ }
+ }
+}
diff --git a/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/LemmaFolding.java b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/LemmaFolding.java
new file mode 100644
index 0000000000..4da713d520
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/LemmaFolding.java
@@ -0,0 +1,71 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package opennlp.wordnet;
+
+import java.util.ArrayList;
+import java.util.List;
+import java.util.Locale;
+
+/**
+ * The single home of the lemma fold and the space-separated field split this package relies on.
+ * {@link MorphyExceptions} keys, the {@link InMemoryWordNetLexicon.LemmaKey sense-index keys},
+ * and every query must fold through {@link #fold(String)} so their canonical forms agree.
+ */
+final class LemmaFolding {
+
+ /** Not instantiable. */
+ private LemmaFolding() {
+ }
+
+ /**
+ * Folds a written form into its canonical shape: lowercase with the root locale, with the
+ * underscore some formats store in multiword lemmas treated as a space.
+ *
+ * @param writtenForm The form as written in a source file or query. Must not be {@code null}.
+ * @return The folded form.
+ * @throws IllegalArgumentException Thrown if {@code writtenForm} is {@code null}.
+ */
+ static String fold(String writtenForm) {
+ if (writtenForm == null) {
+ throw new IllegalArgumentException("writtenForm must not be null");
+ }
+ return writtenForm.replace('_', ' ').toLowerCase(Locale.ROOT);
+ }
+
+ /**
+ * Splits a space-separated field list, collapsing runs of spaces.
+ *
+ * @param value The field list. Must not be {@code null}.
+ * @return The non-empty fields in order, never {@code null}.
+ */
+ static List splitOnSpaces(String value) {
+ final List parts = new ArrayList<>(4);
+ int start = 0;
+ while (start < value.length()) {
+ final int space = value.indexOf(' ', start);
+ if (space < 0) {
+ parts.add(value.substring(start));
+ break;
+ }
+ if (space > start) {
+ parts.add(value.substring(start, space));
+ }
+ start = space + 1;
+ }
+ return parts;
+ }
+}
diff --git a/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/LexicalExpander.java b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/LexicalExpander.java
new file mode 100644
index 0000000000..5a110cac84
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/LexicalExpander.java
@@ -0,0 +1,486 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package opennlp.wordnet;
+
+import java.util.ArrayDeque;
+import java.util.ArrayList;
+import java.util.Comparator;
+import java.util.HashMap;
+import java.util.HashSet;
+import java.util.List;
+import java.util.Map;
+import java.util.Set;
+
+import opennlp.tools.commons.ThreadSafe;
+import opennlp.tools.lemmatizer.Lemmatizer;
+import opennlp.tools.util.StringUtil;
+import opennlp.tools.wordnet.LexicalKnowledgeBase;
+import opennlp.tools.wordnet.Synset;
+import opennlp.tools.wordnet.WordNetPOS;
+import opennlp.tools.wordnet.WordNetRelation;
+
+/**
+ * Expands a term into related terms drawn from a {@link LexicalKnowledgeBase}: the synonyms
+ * sharing its synsets, the lemmas of its hypernym ancestors up to a configured depth, and
+ * optionally the lemmas of its direct hyponyms.
+ *
+ * Each {@link Expansion} carries a deterministic heuristic weight, not a probability: the
+ * first sense of a term starts at {@code 1.0}, each later sense is multiplied by the configurable
+ * sense decay, and every hypernym or hyponym step multiplies by the configurable depth decay.
+ * A decay product that underflows to zero in double arithmetic carries no ranking signal, so
+ * such expansions are dropped rather than emitted outside the {@code (0, 1]} weight range.
+ * When the term itself is not in the lexicon and a {@link Lemmatizer} is configured, the term is
+ * lemmatized and the lemma expanded instead; the lemmatizer is invoked with the
+ * {@link WordNetPOS} name as the tag.
+ *
+ * Hypernym walks follow both the direct and the instance relation, track visited synsets so
+ * malformed cyclic data cannot loop, and never report the term itself. Results are deduplicated
+ * case-insensitively, keeping the highest weight, and ordered by weight descending, then kind,
+ * then term, so output is stable across runs.
+ *
+ * Instances are immutable and safe for concurrent use when the configured lexicon and
+ * lemmatizer are.
+ */
+@ThreadSafe
+public final class LexicalExpander {
+
+ /** How an expansion relates to the input term. */
+ public enum Kind {
+
+ /** A member of one of the term's own synsets. */
+ SYNONYM,
+
+ /** A lemma of an ancestor synset, {@link Expansion#depth()} steps up. */
+ HYPERNYM,
+
+ /** A lemma of a direct child synset. */
+ HYPONYM
+ }
+
+ /**
+ * One expansion of a term.
+ *
+ * @param term The expanded term, in the lexicon's written form (multiword terms contain
+ * spaces).
+ * @param kind How the term relates to the input.
+ * @param depth The relation distance: {@code 0} for synonyms, the number of hypernym steps
+ * for {@link Kind#HYPERNYM}, {@code 1} for hyponyms.
+ * @param senseRank The zero-based rank of the input sense this expansion came from.
+ * @param weight The heuristic weight in {@code (0, 1]}; higher is closer to the input term.
+ */
+ public record Expansion(String term, Kind kind, int depth, int senseRank, double weight) {
+
+ /**
+ * Validates every component against the documented ranges.
+ *
+ * @throws IllegalArgumentException Thrown if {@code term} is {@code null} or blank,
+ * {@code kind} is {@code null}, {@code depth} or {@code senseRank} is negative, or
+ * {@code weight} is not in {@code (0, 1]}.
+ */
+ public Expansion {
+ if (term == null || StringUtil.isBlank(term)) {
+ throw new IllegalArgumentException("term must not be null or blank");
+ }
+ if (kind == null) {
+ throw new IllegalArgumentException("kind must not be null");
+ }
+ if (depth < 0) {
+ throw new IllegalArgumentException("depth must not be negative: " + depth);
+ }
+ if (senseRank < 0) {
+ throw new IllegalArgumentException("senseRank must not be negative: " + senseRank);
+ }
+ if (!(weight > 0.0 && weight <= 1.0)) {
+ throw new IllegalArgumentException("weight must be in (0, 1]: " + weight);
+ }
+ }
+ }
+
+ private final LexicalKnowledgeBase lexicon;
+ private final Lemmatizer lemmatizer;
+ private final int maxSenses;
+ private final int hypernymDepth;
+ private final boolean includeHyponyms;
+ private final int maxExpansions;
+ private final double senseDecay;
+ private final double depthDecay;
+
+ /**
+ * Creates an expander from a builder whose fields have already been validated.
+ *
+ * @param builder The configured builder. Must not be {@code null}.
+ */
+ private LexicalExpander(Builder builder) {
+ this.lexicon = builder.lexicon;
+ this.lemmatizer = builder.lemmatizer;
+ this.maxSenses = builder.maxSenses;
+ this.hypernymDepth = builder.hypernymDepth;
+ this.includeHyponyms = builder.includeHyponyms;
+ this.maxExpansions = builder.maxExpansions;
+ this.senseDecay = builder.senseDecay;
+ this.depthDecay = builder.depthDecay;
+ }
+
+ /**
+ * Starts a builder.
+ *
+ * @param lexicon The knowledge base to expand against. Must not be {@code null}.
+ * @return A builder with the default configuration.
+ * @throws IllegalArgumentException Thrown if {@code lexicon} is {@code null}.
+ */
+ public static Builder builder(LexicalKnowledgeBase lexicon) {
+ return new Builder(lexicon);
+ }
+
+ /**
+ * Expands a term for one part of speech.
+ *
+ * @param term The term to expand. Must not be {@code null} or blank.
+ * @param pos The part of speech to expand as. Must not be {@code null}.
+ * @return The expansions, deduplicated and ordered by descending weight; empty when the term
+ * (and its lemma, when a lemmatizer is configured) is not in the lexicon.
+ * @throws IllegalArgumentException Thrown if {@code term} is {@code null} or blank or
+ * {@code pos} is {@code null}.
+ */
+ public List expand(String term, WordNetPOS pos) {
+ if (pos == null) {
+ throw new IllegalArgumentException("pos must not be null");
+ }
+ return collect(term, List.of(pos));
+ }
+
+ /**
+ * Expands a term across all parts of speech.
+ *
+ * @param term The term to expand. Must not be {@code null} or blank.
+ * @return The expansions across every part of speech, deduplicated and ordered by descending
+ * weight; empty when the term is not in the lexicon.
+ * @throws IllegalArgumentException Thrown if {@code term} is {@code null} or blank.
+ */
+ public List expand(String term) {
+ return collect(term, List.of(WordNetPOS.values()));
+ }
+
+ /**
+ * Expands the term across the given parts of speech and returns the ranked, capped result.
+ *
+ * @param term The term to expand. Must not be {@code null} or blank.
+ * @param poses The parts of speech to expand as. Must not be {@code null}.
+ * @return The expansions, deduplicated, ordered by descending weight, and capped at the
+ * configured maximum; empty when the term is not in the lexicon.
+ * @throws IllegalArgumentException Thrown if {@code term} is {@code null} or blank.
+ */
+ private List collect(String term, List poses) {
+ if (term == null || StringUtil.isBlank(term)) {
+ throw new IllegalArgumentException("term must not be null or blank");
+ }
+ final Map best = new HashMap<>();
+ final Set excluded = new HashSet<>();
+ excluded.add(LemmaFolding.fold(term));
+
+ for (final WordNetPOS pos : poses) {
+ final String subject = resolveSubject(term, pos);
+ if (subject == null) {
+ continue;
+ }
+ final List senses = lexicon.lookup(subject, pos);
+ final int senseCount = Math.min(senses.size(), maxSenses);
+ for (int rank = 0; rank < senseCount; rank++) {
+ final double senseWeight = Math.pow(senseDecay, rank);
+ expandSense(senses.get(rank), rank, senseWeight, best, excluded);
+ }
+ }
+
+ final List ordered = new ArrayList<>(best.values());
+ ordered.sort(Comparator.comparingDouble(Expansion::weight).reversed()
+ .thenComparing(Expansion::kind)
+ .thenComparing(Expansion::term));
+ return ordered.size() > maxExpansions ? List.copyOf(ordered.subList(0, maxExpansions))
+ : List.copyOf(ordered);
+ }
+
+ /**
+ * Resolves the form actually expanded: the term when the lexicon knows it, otherwise its lemma
+ * when a lemmatizer is configured and produces a known lemma.
+ *
+ * @param term The input term. Must not be {@code null}.
+ * @param pos The part of speech to look the term up as. Must not be {@code null}.
+ * @return The known form to expand, or {@code null} when neither the term nor its lemma is in
+ * the lexicon.
+ */
+ private String resolveSubject(String term, WordNetPOS pos) {
+ if (lexicon.contains(term, pos)) {
+ return term;
+ }
+ if (lemmatizer == null) {
+ return null;
+ }
+ final String[] lemmas =
+ lemmatizer.lemmatize(new String[] {term}, new String[] {pos.name()});
+ if (lemmas.length == 0 || lemmas[0] == null) {
+ return null;
+ }
+ final String lemma = lemmas[0];
+ // Lemmatizers report an unresolvable token with the contract's unknown marker.
+ if (MorphyLemmatizer.UNKNOWN_LEMMA.equals(lemma) || !lexicon.contains(lemma, pos)) {
+ return null;
+ }
+ return lemma;
+ }
+
+ /**
+ * Offers the synonyms, hypernym ancestors, and optional hyponyms of one sense into the running
+ * best-expansion map.
+ *
+ * @param sense The sense to expand. Must not be {@code null}.
+ * @param rank The zero-based salience rank of the sense.
+ * @param senseWeight The weight of the sense itself; each relation step decays from it.
+ * @param best The best expansion seen so far per folded term; updated in place.
+ * @param excluded The folded terms that are never reported, such as the input term.
+ */
+ private void expandSense(Synset sense, int rank, double senseWeight,
+ Map best, Set excluded) {
+ if (senseWeight == 0.0) {
+ // Underflowed to zero: no ranking signal left, and zero is outside the documented range.
+ return;
+ }
+ for (final String lemma : sense.lemmas()) {
+ offer(best, excluded, new Expansion(lemma, Kind.SYNONYM, 0, rank, senseWeight));
+ }
+
+ // Breadth-first hypernym walk, visited-checked so cyclic data terminates.
+ if (hypernymDepth > 0) {
+ final Set visited = new HashSet<>();
+ visited.add(sense.id());
+ final ArrayDeque frontier = new ArrayDeque<>(hypernymsOf(sense));
+ final ArrayDeque next = new ArrayDeque<>();
+ for (int depth = 1; depth <= hypernymDepth && !frontier.isEmpty(); depth++) {
+ final double depthWeight = senseWeight * Math.pow(depthDecay, depth);
+ if (depthWeight == 0.0) {
+ // Deeper levels only shrink further, so the walk stops at the first underflow.
+ break;
+ }
+ while (!frontier.isEmpty()) {
+ final String id = frontier.poll();
+ if (!visited.add(id)) {
+ continue;
+ }
+ final Synset ancestor = lexicon.synset(id).orElse(null);
+ if (ancestor == null) {
+ continue;
+ }
+ for (final String lemma : ancestor.lemmas()) {
+ offer(best, excluded,
+ new Expansion(lemma, Kind.HYPERNYM, depth, rank, depthWeight));
+ }
+ next.addAll(hypernymsOf(ancestor));
+ }
+ frontier.addAll(next);
+ next.clear();
+ }
+ }
+
+ if (includeHyponyms && senseWeight * depthDecay > 0.0) {
+ final double hyponymWeight = senseWeight * depthDecay;
+ final List children = new ArrayList<>(sense.related(WordNetRelation.HYPONYM));
+ children.addAll(sense.related(WordNetRelation.INSTANCE_HYPONYM));
+ for (final String id : children) {
+ lexicon.synset(id).ifPresent(child -> {
+ for (final String lemma : child.lemmas()) {
+ offer(best, excluded, new Expansion(lemma, Kind.HYPONYM, 1, rank, hyponymWeight));
+ }
+ });
+ }
+ }
+ }
+
+ /**
+ * Collects the synset ids of both the direct and the instance hypernyms of a synset.
+ *
+ * @param synset The synset whose hypernyms are collected. Must not be {@code null}.
+ * @return The hypernym synset ids in source order, direct relations first.
+ */
+ private List hypernymsOf(Synset synset) {
+ final List direct = synset.related(WordNetRelation.HYPERNYM);
+ final List instance = synset.related(WordNetRelation.INSTANCE_HYPERNYM);
+ if (instance.isEmpty()) {
+ return direct;
+ }
+ final List all = new ArrayList<>(direct.size() + instance.size());
+ all.addAll(direct);
+ all.addAll(instance);
+ return all;
+ }
+
+ /**
+ * Records the candidate under its folded term when it is not excluded and it beats the current
+ * best weight for that term. Folding through {@link LemmaFolding#fold(String)} keeps the
+ * exclusion and deduplication keys aligned with the lexicon's own lemma keys.
+ *
+ * @param best The best expansion seen so far per folded term; updated in place.
+ * @param excluded The folded terms that are never reported.
+ * @param candidate The expansion to offer. Must not be {@code null}.
+ */
+ private void offer(Map best, Set excluded,
+ Expansion candidate) {
+ final String key = LemmaFolding.fold(candidate.term());
+ if (excluded.contains(key)) {
+ return;
+ }
+ final Expansion current = best.get(key);
+ if (current == null || candidate.weight() > current.weight()) {
+ best.put(key, candidate);
+ }
+ }
+
+ /** Configures and creates a {@link LexicalExpander}. */
+ public static final class Builder {
+
+ private final LexicalKnowledgeBase lexicon;
+ private Lemmatizer lemmatizer;
+ private int maxSenses = 3;
+ private int hypernymDepth = 1;
+ private boolean includeHyponyms = false;
+ private int maxExpansions = 20;
+ private double senseDecay = 0.5;
+ private double depthDecay = 0.5;
+
+ /**
+ * Creates a builder over the given lexicon; use {@link LexicalExpander#builder}.
+ *
+ * @param lexicon The knowledge base to expand against. Must not be {@code null}.
+ * @throws IllegalArgumentException Thrown if {@code lexicon} is {@code null}.
+ */
+ private Builder(LexicalKnowledgeBase lexicon) {
+ if (lexicon == null) {
+ throw new IllegalArgumentException("lexicon must not be null");
+ }
+ this.lexicon = lexicon;
+ }
+
+ /**
+ * Configures a lemmatizer used when the input term itself is not in the lexicon. It is
+ * invoked with the {@link WordNetPOS} name as the tag.
+ *
+ * @param lemmatizer The fallback lemmatizer. Must not be {@code null}.
+ * @return This builder.
+ * @throws IllegalArgumentException Thrown if {@code lemmatizer} is {@code null}.
+ */
+ public Builder lemmatizer(Lemmatizer lemmatizer) {
+ if (lemmatizer == null) {
+ throw new IllegalArgumentException("lemmatizer must not be null");
+ }
+ this.lemmatizer = lemmatizer;
+ return this;
+ }
+
+ /**
+ * Configures how many senses of the term are expanded, most salient first.
+ *
+ * @param maxSenses The sense count; must be positive. The default is {@code 3}.
+ * @return This builder.
+ * @throws IllegalArgumentException Thrown if {@code maxSenses} is not positive.
+ */
+ public Builder maxSenses(int maxSenses) {
+ if (maxSenses < 1) {
+ throw new IllegalArgumentException("maxSenses must be positive: " + maxSenses);
+ }
+ this.maxSenses = maxSenses;
+ return this;
+ }
+
+ /**
+ * Configures how many hypernym steps are walked; {@code 0} disables hypernym expansion.
+ *
+ * @param hypernymDepth The depth; must not be negative. The default is {@code 1}.
+ * @return This builder.
+ * @throws IllegalArgumentException Thrown if {@code hypernymDepth} is negative.
+ */
+ public Builder hypernymDepth(int hypernymDepth) {
+ if (hypernymDepth < 0) {
+ throw new IllegalArgumentException(
+ "hypernymDepth must not be negative: " + hypernymDepth);
+ }
+ this.hypernymDepth = hypernymDepth;
+ return this;
+ }
+
+ /**
+ * Configures whether direct hyponyms are included; off by default.
+ *
+ * @param includeHyponyms Whether to include direct hyponyms.
+ * @return This builder.
+ */
+ public Builder includeHyponyms(boolean includeHyponyms) {
+ this.includeHyponyms = includeHyponyms;
+ return this;
+ }
+
+ /**
+ * Configures the maximum number of expansions returned after ranking.
+ *
+ * @param maxExpansions The cap; must be positive. The default is {@code 20}.
+ * @return This builder.
+ * @throws IllegalArgumentException Thrown if {@code maxExpansions} is not positive.
+ */
+ public Builder maxExpansions(int maxExpansions) {
+ if (maxExpansions < 1) {
+ throw new IllegalArgumentException(
+ "maxExpansions must be positive: " + maxExpansions);
+ }
+ this.maxExpansions = maxExpansions;
+ return this;
+ }
+
+ /**
+ * Configures the weight multiplier applied per sense rank step.
+ *
+ * @param senseDecay The decay in {@code (0, 1]}. The default is {@code 0.5}.
+ * @return This builder.
+ * @throws IllegalArgumentException Thrown if {@code senseDecay} is outside {@code (0, 1]}.
+ */
+ public Builder senseDecay(double senseDecay) {
+ if (!(senseDecay > 0 && senseDecay <= 1)) {
+ throw new IllegalArgumentException(
+ "senseDecay must be in (0, 1]: " + senseDecay);
+ }
+ this.senseDecay = senseDecay;
+ return this;
+ }
+
+ /**
+ * Configures the weight multiplier applied per hypernym or hyponym step.
+ *
+ * @param depthDecay The decay in {@code (0, 1]}. The default is {@code 0.5}.
+ * @return This builder.
+ * @throws IllegalArgumentException Thrown if {@code depthDecay} is outside {@code (0, 1]}.
+ */
+ public Builder depthDecay(double depthDecay) {
+ if (!(depthDecay > 0 && depthDecay <= 1)) {
+ throw new IllegalArgumentException(
+ "depthDecay must be in (0, 1]: " + depthDecay);
+ }
+ this.depthDecay = depthDecay;
+ return this;
+ }
+
+ /** {@return the configured expander} */
+ public LexicalExpander build() {
+ return new LexicalExpander(this);
+ }
+ }
+}
diff --git a/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/MorphyExceptions.java b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/MorphyExceptions.java
new file mode 100644
index 0000000000..4f6c0b2934
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/MorphyExceptions.java
@@ -0,0 +1,145 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package opennlp.wordnet;
+
+import java.io.IOException;
+import java.nio.charset.StandardCharsets;
+import java.nio.file.Files;
+import java.nio.file.Path;
+import java.util.ArrayList;
+import java.util.EnumMap;
+import java.util.HashMap;
+import java.util.List;
+import java.util.Map;
+
+import opennlp.tools.commons.ThreadSafe;
+import opennlp.tools.util.InvalidFormatException;
+import opennlp.tools.wordnet.WordNetPOS;
+
+/**
+ * The Morphy exception lists: the per-part-of-speech tables of irregular inflected forms
+ * ({@code mice} to {@code mouse}, {@code went} to {@code go}) that the Morphy algorithm
+ * consults before its detachment rules.
+ *
+ * {@link #load(Path)} reads the four {@code *.exc} files ({@code noun.exc},
+ * {@code verb.exc}, {@code adj.exc}, {@code adv.exc}), which must all be present, in the WNDB
+ * format: one entry per line, the inflected form followed by one or more base forms, space
+ * separated, with underscores standing for spaces in multiword entries. No exception data is
+ * bundled; the caller supplies a directory.
+ *
+ * Lookups fold the queried word the same way the lexicon seam folds lemmas. Instances are
+ * immutable after loading and safe for concurrent lookups.
+ */
+@ThreadSafe
+public final class MorphyExceptions {
+
+ private final Map>> byPos;
+
+ /**
+ * Wraps the per-part-of-speech exception tables.
+ *
+ * @param byPos The loaded tables, one per part of speech.
+ */
+ private MorphyExceptions(Map>> byPos) {
+ this.byPos = byPos;
+ }
+
+ /**
+ * Loads the four exception lists from a directory.
+ *
+ * @param directory The directory containing {@code noun.exc}, {@code verb.exc},
+ * {@code adj.exc}, and {@code adv.exc}. Must not be {@code null} and must
+ * exist.
+ * @return The loaded exception lists.
+ * @throws IllegalArgumentException Thrown if {@code directory} is {@code null} or not a
+ * directory.
+ * @throws InvalidFormatException Thrown if one of the four files is missing or a line is
+ * malformed; the message names the file and line.
+ * @throws IOException Thrown if reading a file fails.
+ */
+ public static MorphyExceptions load(Path directory) throws IOException {
+ if (directory == null) {
+ throw new IllegalArgumentException("Directory must not be null");
+ }
+ if (!Files.isDirectory(directory)) {
+ throw new IllegalArgumentException(
+ "Directory does not exist or is not a directory: " + directory);
+ }
+ final Map>> byPos = new EnumMap<>(WordNetPOS.class);
+ byPos.put(WordNetPOS.NOUN, loadFile(directory, "noun.exc"));
+ byPos.put(WordNetPOS.VERB, loadFile(directory, "verb.exc"));
+ byPos.put(WordNetPOS.ADJECTIVE, loadFile(directory, "adj.exc"));
+ byPos.put(WordNetPOS.ADVERB, loadFile(directory, "adv.exc"));
+ return new MorphyExceptions(byPos);
+ }
+
+ /**
+ * Finds the base forms of an irregular inflected form.
+ *
+ * @param word The inflected form; folded before lookup. Must not be {@code null}.
+ * @param pos The part of speech. Must not be {@code null}.
+ * @return The base forms in file order, never {@code null}; empty when the word has no entry.
+ * @throws IllegalArgumentException Thrown if {@code word} or {@code pos} is {@code null}.
+ */
+ public List lookup(String word, WordNetPOS pos) {
+ if (word == null) {
+ throw new IllegalArgumentException("Word must not be null");
+ }
+ if (pos == null) {
+ throw new IllegalArgumentException("Pos must not be null");
+ }
+ final List lemmas = byPos.get(pos).get(LemmaFolding.fold(word));
+ return lemmas == null ? List.of() : lemmas;
+ }
+
+ /**
+ * Loads one {@code *.exc} file into a folded inflected-form to base-forms map.
+ *
+ * @param directory The directory holding the file.
+ * @param fileName The exception file name, for example {@code noun.exc}.
+ * @return The folded exception entries.
+ * @throws InvalidFormatException Thrown if the file is missing or a line is malformed.
+ * @throws IOException Thrown if reading the file fails.
+ */
+ private static Map> loadFile(Path directory, String fileName)
+ throws IOException {
+ final Path file = directory.resolve(fileName);
+ if (!Files.isRegularFile(file)) {
+ throw new InvalidFormatException("Missing exception list file: " + file);
+ }
+ final List lines = Files.readAllLines(file, StandardCharsets.ISO_8859_1);
+ final Map> entries = new HashMap<>(lines.size() * 2);
+ for (int i = 0; i < lines.size(); i++) {
+ final String line = lines.get(i);
+ if (line.isEmpty()) {
+ continue;
+ }
+ final List fields = LemmaFolding.splitOnSpaces(line);
+ if (fields.size() < 2) {
+ throw new InvalidFormatException("Malformed exception list " + fileName + " at line "
+ + (i + 1) + ": expected an inflected form and at least one base form, got: " + line);
+ }
+ final List lemmas = new ArrayList<>(fields.size() - 1);
+ for (final String lemma : fields.subList(1, fields.size())) {
+ lemmas.add(LemmaFolding.fold(lemma));
+ }
+ // A form listed twice keeps its first entry, matching first-match lookup semantics.
+ entries.putIfAbsent(LemmaFolding.fold(fields.get(0)), List.copyOf(lemmas));
+ }
+ return Map.copyOf(entries);
+ }
+}
diff --git a/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/MorphyLemmatizer.java b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/MorphyLemmatizer.java
new file mode 100644
index 0000000000..c4da01d94f
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/MorphyLemmatizer.java
@@ -0,0 +1,226 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package opennlp.wordnet;
+
+import java.util.ArrayList;
+import java.util.List;
+import java.util.Locale;
+
+import opennlp.tools.commons.ThreadSafe;
+import opennlp.tools.lemmatizer.Lemmatizer;
+import opennlp.tools.wordnet.LexicalKnowledgeBase;
+import opennlp.tools.wordnet.WordNetPOS;
+
+/**
+ * A {@link Lemmatizer} implementing the Morphy algorithm: exception-list lookup first, then the
+ * per-part-of-speech iterative detachment rules, with every rule-derived candidate validated
+ * against a {@link LexicalKnowledgeBase} before it is returned. A token is folded (lowercase with
+ * the root locale, underscore as space) before lookup, and returned lemmas are in that folded
+ * form.
+ *
+ * Part-of-speech tags map to a {@link WordNetPOS} by their conventional Penn Treebank
+ * prefixes ({@code N}, {@code V}, {@code J}, {@code R}), the names {@code ADJ} and {@code ADV},
+ * and the one-letter WordNet codes {@code n}, {@code v}, {@code a}, {@code r}, and {@code s}
+ * (satellite, treated as adjective), case-insensitively. A tag that maps to no part of speech
+ * yields the unknown-word result.
+ *
+ * Following {@code opennlp.tools.lemmatizer.DictionaryLemmatizer}, a token with no lemma
+ * yields {@link #UNKNOWN_LEMMA} from {@link #lemmatize(String[], String[])} and a singleton list
+ * of it from {@link #lemmatize(List, List)}. Both a lexicon and exception lists are required.
+ * Instances are immutable and safe for concurrent use.
+ */
+@ThreadSafe
+public final class MorphyLemmatizer implements Lemmatizer {
+
+ /**
+ * The output emitted for a token whose lemma is unknown, following the conventional
+ * {@link Lemmatizer} unknown marker also used by
+ * {@code opennlp.tools.lemmatizer.DictionaryLemmatizer}.
+ */
+ public static final String UNKNOWN_LEMMA = "O";
+
+ private static final String[][] NOUN_RULES = {
+ {"s", ""}, {"ses", "s"}, {"xes", "x"}, {"zes", "z"},
+ {"ches", "ch"}, {"shes", "sh"}, {"men", "man"}, {"ies", "y"},
+ };
+
+ private static final String[][] VERB_RULES = {
+ {"s", ""}, {"ies", "y"}, {"es", "e"}, {"es", ""},
+ {"ed", "e"}, {"ed", ""}, {"ing", "e"}, {"ing", ""},
+ };
+
+ private static final String[][] ADJECTIVE_RULES = {
+ {"er", ""}, {"est", ""}, {"er", "e"}, {"est", "e"},
+ };
+
+ private static final String[][] NO_RULES = {};
+
+ private final LexicalKnowledgeBase lexicon;
+ private final MorphyExceptions exceptions;
+
+ /**
+ * Creates a Morphy lemmatizer over a loaded lexicon and exception lists.
+ *
+ * @param lexicon The lexicon rule candidates are validated against. Must not be
+ * {@code null}.
+ * @param exceptions The irregular-form exception lists. Must not be {@code null}.
+ * @throws IllegalArgumentException Thrown if {@code lexicon} or {@code exceptions} is
+ * {@code null}.
+ */
+ public MorphyLemmatizer(LexicalKnowledgeBase lexicon, MorphyExceptions exceptions) {
+ if (lexicon == null) {
+ throw new IllegalArgumentException("Lexicon must not be null");
+ }
+ if (exceptions == null) {
+ throw new IllegalArgumentException("Exceptions must not be null");
+ }
+ this.lexicon = lexicon;
+ this.exceptions = exceptions;
+ }
+
+ /**
+ * {@inheritDoc}
+ *
+ * @throws IllegalArgumentException Thrown if {@code toks} or {@code tags} is {@code null},
+ * contains a {@code null} element, or the two differ in length.
+ */
+ @Override
+ public String[] lemmatize(String[] toks, String[] tags) {
+ if (toks == null || tags == null) {
+ throw new IllegalArgumentException("Toks and tags must not be null");
+ }
+ if (toks.length != tags.length) {
+ throw new IllegalArgumentException("Toks and tags must have the same length, got "
+ + toks.length + " and " + tags.length);
+ }
+ final String[] lemmas = new String[toks.length];
+ for (int i = 0; i < toks.length; i++) {
+ final List candidates = lemmasOf(toks[i], tags[i]);
+ lemmas[i] = candidates.isEmpty() ? UNKNOWN_LEMMA : candidates.get(0);
+ }
+ return lemmas;
+ }
+
+ /**
+ * {@inheritDoc}
+ *
+ * @throws IllegalArgumentException Thrown if {@code toks} or {@code tags} is {@code null},
+ * contains a {@code null} element, or the two differ in size.
+ */
+ @Override
+ public List> lemmatize(List toks, List tags) {
+ if (toks == null || tags == null) {
+ throw new IllegalArgumentException("Toks and tags must not be null");
+ }
+ if (toks.size() != tags.size()) {
+ throw new IllegalArgumentException("Toks and tags must have the same size, got "
+ + toks.size() + " and " + tags.size());
+ }
+ final List> lemmas = new ArrayList<>(toks.size());
+ for (int i = 0; i < toks.size(); i++) {
+ final List candidates = lemmasOf(toks.get(i), tags.get(i));
+ lemmas.add(candidates.isEmpty() ? List.of(UNKNOWN_LEMMA) : candidates);
+ }
+ return lemmas;
+ }
+
+ /**
+ * Finds all lemmas of one token, most preferred first.
+ *
+ * @param token The token to lemmatize.
+ * @param tag The part-of-speech tag.
+ * @return The candidate lemmas, empty when the word is unknown or the tag maps to no part of
+ * speech.
+ */
+ private List lemmasOf(String token, String tag) {
+ if (token == null || tag == null) {
+ throw new IllegalArgumentException("Tokens and tags must not contain null elements");
+ }
+ final WordNetPOS pos = posFromTag(tag);
+ if (pos == null) {
+ return List.of();
+ }
+ final String folded = LemmaFolding.fold(token);
+ final List irregular = exceptions.lookup(folded, pos);
+ if (!irregular.isEmpty()) {
+ return irregular;
+ }
+ final List candidates = new ArrayList<>(2);
+ if (lexicon.contains(folded, pos)) {
+ candidates.add(folded);
+ }
+ for (final String[] rule : rulesFor(pos)) {
+ final String suffix = rule[0];
+ if (folded.length() > suffix.length() && folded.endsWith(suffix)) {
+ final String candidate =
+ folded.substring(0, folded.length() - suffix.length()) + rule[1];
+ if (!candidates.contains(candidate) && lexicon.contains(candidate, pos)) {
+ candidates.add(candidate);
+ }
+ }
+ }
+ return candidates;
+ }
+
+ /**
+ * Selects the detachment-rule table for a part of speech.
+ *
+ * @param pos The part of speech.
+ * @return The suffix-substitution rules, empty for adverbs.
+ */
+ private static String[][] rulesFor(WordNetPOS pos) {
+ return switch (pos) {
+ case NOUN -> NOUN_RULES;
+ case VERB -> VERB_RULES;
+ case ADJECTIVE -> ADJECTIVE_RULES;
+ case ADVERB -> NO_RULES;
+ };
+ }
+
+ /**
+ * Maps a part-of-speech tag to a {@link WordNetPOS}. Package-private so tests can pin the
+ * mapping directly.
+ *
+ * @param tag The tag to map. Must not be {@code null}.
+ * @return The part of speech, or {@code null} when the tag names none.
+ * @throws IllegalArgumentException Thrown if {@code tag} is {@code null}.
+ */
+ static WordNetPOS posFromTag(String tag) {
+ if (tag == null) {
+ throw new IllegalArgumentException("Tag must not be null");
+ }
+ if (tag.isEmpty()) {
+ return null;
+ }
+ final String upper = tag.toUpperCase(Locale.ROOT);
+ if (upper.startsWith("ADJ")) {
+ return WordNetPOS.ADJECTIVE;
+ }
+ if (upper.startsWith("ADV")) {
+ return WordNetPOS.ADVERB;
+ }
+ return switch (upper.charAt(0)) {
+ case 'N' -> WordNetPOS.NOUN;
+ case 'V' -> WordNetPOS.VERB;
+ case 'J' -> WordNetPOS.ADJECTIVE;
+ // Codes a and s mean adjective only as one-letter tags; AUX, ADP and the like do not.
+ case 'A', 'S' -> tag.length() == 1 ? WordNetPOS.ADJECTIVE : null;
+ case 'R' -> WordNetPOS.ADVERB;
+ default -> null;
+ };
+ }
+}
diff --git a/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/SynsetSimilarity.java b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/SynsetSimilarity.java
new file mode 100644
index 0000000000..050f85a16c
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/SynsetSimilarity.java
@@ -0,0 +1,241 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package opennlp.wordnet;
+
+import java.util.ArrayDeque;
+import java.util.ArrayList;
+import java.util.Deque;
+import java.util.HashMap;
+import java.util.List;
+import java.util.Map;
+
+import opennlp.tools.commons.ThreadSafe;
+import opennlp.tools.wordnet.LexicalKnowledgeBase;
+import opennlp.tools.wordnet.WordNetRelation;
+
+/**
+ * Taxonomy-based similarity between synsets: measures over the hypernym graph of a
+ * {@link LexicalKnowledgeBase}, computed on demand with no precomputed tables.
+ *
+ * Path similarity is {@code 1 / (1 + d)} for the shortest hypernym-graph distance
+ * {@code d} through a common ancestor. Wu-Palmer similarity relates the depth of the
+ * deepest common ancestor to the depths of both synsets. Leacock-Chodorow scales the
+ * shortest path against a caller-supplied taxonomy depth, since the knowledge base
+ * interface does not enumerate the taxonomy. Unrelated synsets, those sharing no
+ * ancestor, score zero everywhere. Information-content measures need corpus counts and
+ * are not provided here.
+ *
+ * Both plain and instance hypernyms count as taxonomy edges. The measures read only
+ * the knowledge base and hold no mutable state, so instances are as thread-safe as
+ * their knowledge base.
+ */
+@ThreadSafe
+public class SynsetSimilarity {
+
+ private final LexicalKnowledgeBase knowledgeBase;
+
+ /**
+ * Initializes the measures.
+ *
+ * @param knowledgeBase The knowledge base to walk. Must not be {@code null}.
+ * @throws IllegalArgumentException Thrown if {@code knowledgeBase} is {@code null}.
+ */
+ public SynsetSimilarity(LexicalKnowledgeBase knowledgeBase) {
+ if (knowledgeBase == null) {
+ throw new IllegalArgumentException("knowledgeBase must not be null");
+ }
+ this.knowledgeBase = knowledgeBase;
+ }
+
+ /**
+ * Computes path similarity: {@code 1 / (1 + d)} over the shortest hypernym-graph
+ * distance.
+ *
+ * @param synsetId The first synset identifier. Must not be {@code null}.
+ * @param otherSynsetId The second synset identifier. Must not be {@code null}.
+ * @return The similarity in {@code (0, 1]}, or {@code 0} when the synsets share no
+ * ancestor.
+ * @throws IllegalArgumentException Thrown if {@code synsetId} or
+ * {@code otherSynsetId} is {@code null}.
+ */
+ public double path(String synsetId, String otherSynsetId) {
+ final int distance = shortestDistance(synsetId, otherSynsetId);
+ return distance < 0 ? 0.0 : 1.0 / (1.0 + distance);
+ }
+
+ /**
+ * Computes Wu-Palmer similarity: {@code 2 * depth(lcs) / (depth(a) + depth(b))},
+ * with depths counted from the taxonomy root and the deepest common ancestor as the
+ * lcs.
+ *
+ * @param synsetId The first synset identifier. Must not be {@code null}.
+ * @param otherSynsetId The second synset identifier. Must not be {@code null}.
+ * @return The similarity in {@code (0, 1]}, or {@code 0} when the synsets share no
+ * ancestor.
+ * @throws IllegalArgumentException Thrown if {@code synsetId} or
+ * {@code otherSynsetId} is {@code null}.
+ */
+ public double wuPalmer(String synsetId, String otherSynsetId) {
+ validateIds(synsetId, otherSynsetId);
+ final Map up = depthsAbove(synsetId);
+ final Map otherUp = depthsAbove(otherSynsetId);
+ double best = 0.0;
+ for (final Map.Entry common : up.entrySet()) {
+ final Integer otherDistance = otherUp.get(common.getKey());
+ if (otherDistance == null) {
+ continue;
+ }
+ final int rootDepth = depthFromRoot(common.getKey());
+ final int depthA = rootDepth + common.getValue();
+ final int depthB = rootDepth + otherDistance;
+ if (depthA + depthB == 0) {
+ continue;
+ }
+ final double score = 2.0 * rootDepth / (depthA + depthB);
+ best = Math.max(best, score);
+ }
+ return best;
+ }
+
+ /**
+ * Computes Leacock-Chodorow similarity:
+ * {@code -log((d + 1) / (2 * taxonomyDepth))} over the shortest hypernym-graph
+ * distance {@code d}.
+ *
+ * @param synsetId The first synset identifier. Must not be {@code null}.
+ * @param otherSynsetId The second synset identifier. Must not be {@code null}.
+ * @param taxonomyDepth The maximum depth of the taxonomy the synsets live in. Must
+ * be positive.
+ * @return The similarity, higher for closer synsets, or {@code 0} when the synsets
+ * share no ancestor.
+ * @throws IllegalArgumentException Thrown if {@code synsetId} or
+ * {@code otherSynsetId} is {@code null}, or {@code taxonomyDepth} is not
+ * positive.
+ */
+ public double leacockChodorow(String synsetId, String otherSynsetId,
+ int taxonomyDepth) {
+ validateIds(synsetId, otherSynsetId);
+ if (taxonomyDepth <= 0) {
+ throw new IllegalArgumentException(
+ "taxonomyDepth must be positive: " + taxonomyDepth);
+ }
+ final int distance = shortestDistance(synsetId, otherSynsetId);
+ if (distance < 0) {
+ return 0.0;
+ }
+ return -Math.log((distance + 1.0) / (2.0 * taxonomyDepth));
+ }
+
+ /**
+ * Finds the shortest distance between two synsets through a common ancestor.
+ *
+ * @param synsetId The first synset identifier. Must not be {@code null}.
+ * @param otherSynsetId The second synset identifier. Must not be {@code null}.
+ * @return The edge count of the shortest connecting path, or {@code -1} when no
+ * common ancestor exists.
+ * @throws IllegalArgumentException Thrown if {@code synsetId} or
+ * {@code otherSynsetId} is {@code null}.
+ */
+ public int shortestDistance(String synsetId, String otherSynsetId) {
+ validateIds(synsetId, otherSynsetId);
+ final Map up = depthsAbove(synsetId);
+ final Map otherUp = depthsAbove(otherSynsetId);
+ int best = -1;
+ for (final Map.Entry common : up.entrySet()) {
+ final Integer otherDistance = otherUp.get(common.getKey());
+ if (otherDistance != null) {
+ final int total = common.getValue() + otherDistance;
+ if (best < 0 || total < best) {
+ best = total;
+ }
+ }
+ }
+ return best;
+ }
+
+ /**
+ * Validates the identifier arguments of the public measures.
+ *
+ * @param synsetId The first synset identifier.
+ * @param otherSynsetId The second synset identifier.
+ * @throws IllegalArgumentException Thrown if {@code synsetId} or
+ * {@code otherSynsetId} is {@code null}.
+ */
+ private void validateIds(String synsetId, String otherSynsetId) {
+ if (synsetId == null) {
+ throw new IllegalArgumentException("synsetId must not be null");
+ }
+ if (otherSynsetId == null) {
+ throw new IllegalArgumentException("otherSynsetId must not be null");
+ }
+ }
+
+ /**
+ * Collects every ancestor with its minimal upward distance, the synset itself included at
+ * distance zero.
+ *
+ * @param synsetId The synset to walk up from.
+ * @return The upward distance to each reachable ancestor, keyed by synset identifier.
+ */
+ private Map depthsAbove(String synsetId) {
+ final Map depths = new HashMap<>();
+ final Deque queue = new ArrayDeque<>();
+ depths.put(synsetId, 0);
+ queue.add(synsetId);
+ while (!queue.isEmpty()) {
+ final String current = queue.remove();
+ final int depth = depths.get(current);
+ for (final String parent : hypernyms(current)) {
+ if (!depths.containsKey(parent) || depths.get(parent) > depth + 1) {
+ depths.put(parent, depth + 1);
+ queue.add(parent);
+ }
+ }
+ }
+ return depths;
+ }
+
+ /**
+ * Measures a synset's depth as the distance to its farthest ancestor, which is the taxonomy
+ * root reached the long way round when several paths lead up.
+ *
+ * @param synsetId The synset to measure.
+ * @return The edge count to the farthest ancestor, {@code 0} for a root.
+ */
+ private int depthFromRoot(String synsetId) {
+ final Map above = depthsAbove(synsetId);
+ int deepest = 0;
+ for (final int distance : above.values()) {
+ deepest = Math.max(deepest, distance);
+ }
+ return deepest;
+ }
+
+ /**
+ * Collects the synsets one taxonomy edge above a synset.
+ *
+ * @param synsetId The synset whose parents are collected.
+ * @return The plain hypernyms followed by the instance hypernyms.
+ */
+ private Iterable hypernyms(String synsetId) {
+ final List parents = new ArrayList<>(
+ knowledgeBase.related(synsetId, WordNetRelation.HYPERNYM));
+ parents.addAll(knowledgeBase.related(synsetId, WordNetRelation.INSTANCE_HYPERNYM));
+ return parents;
+ }
+}
diff --git a/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/WnLmfReader.java b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/WnLmfReader.java
new file mode 100644
index 0000000000..1222cfbfd1
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/WnLmfReader.java
@@ -0,0 +1,600 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package opennlp.wordnet;
+
+import java.io.BufferedInputStream;
+import java.io.IOException;
+import java.io.InputStream;
+import java.nio.file.Files;
+import java.nio.file.Path;
+import java.util.ArrayList;
+import java.util.HashMap;
+import java.util.HashSet;
+import java.util.LinkedHashMap;
+import java.util.LinkedHashSet;
+import java.util.List;
+import java.util.Map;
+import java.util.Set;
+import javax.xml.XMLConstants;
+import javax.xml.stream.Location;
+import javax.xml.stream.XMLInputFactory;
+import javax.xml.stream.XMLStreamConstants;
+import javax.xml.stream.XMLStreamException;
+import javax.xml.stream.XMLStreamReader;
+
+import opennlp.tools.util.InvalidFormatException;
+import opennlp.tools.wordnet.LexicalKnowledgeBase;
+import opennlp.tools.wordnet.Synset;
+import opennlp.tools.wordnet.WordNetPOS;
+import opennlp.tools.wordnet.WordNetRelation;
+
+/**
+ * Reads a WN-LMF XML document (the Global WordNet Association
+ * interchange format, used by
+ * Open English WordNet and many
+ * other language wordnets) into a {@link LexicalKnowledgeBase} using the JDK StAX parser.
+ *
+ * It reads the subset of the format the contract serves: lexical entries, synsets with their
+ * definitions and typed relations, and sense relations, which are lifted to the synset level as
+ * documented on {@link WordNetRelation}. Elements outside that subset are skipped, as are
+ * relations of type {@code other} (the format's untyped escape hatch); any other unknown
+ * relation type fails loud.
+ *
+ * The parser is hardened against XXE: DTD processing and external entities are disabled, so a
+ * DOCTYPE is skipped but nothing it names is fetched or resolved.
+ *
+ * Malformed structure fails loud with an {@link InvalidFormatException} naming the resource
+ * and, where the parser provides one, the line; I/O failures propagate as {@link IOException}.
+ * Part-of-speech code {@code s} normalizes to {@link WordNetPOS#ADJECTIVE}, and a {@code similar}
+ * relation on a verb synset maps to {@link WordNetRelation#VERB_GROUP} rather than
+ * {@link WordNetRelation#SIMILAR_TO}. The returned lexicon is immutable and safe for concurrent
+ * lookups.
+ */
+public final class WnLmfReader {
+
+ private static final Map RELATION_NAMES = relationNames();
+
+ /** The format's escape-hatch relation type; carries no type the contract can express. */
+ private static final String OTHER_RELATION = "other";
+
+ /** The element declaring a lexical entry; opened and closed by the same handlers. */
+ private static final String LEXICAL_ENTRY_ELEMENT = "LexicalEntry";
+
+ /** The element declaring a sense; opened and closed by the same handlers. */
+ private static final String SENSE_ELEMENT = "Sense";
+
+ /** The element declaring a synset; opened and closed by the same handlers. */
+ private static final String SYNSET_ELEMENT = "Synset";
+
+ /** Not instantiable. */
+ private WnLmfReader() {
+ }
+
+ /**
+ * Reads a WN-LMF XML file.
+ *
+ * @param file The XML file. Must not be {@code null} and must exist.
+ * @return The loaded lexicon.
+ * @throws IllegalArgumentException Thrown if {@code file} is {@code null} or missing.
+ * @throws InvalidFormatException Thrown if the document is malformed; the message names the
+ * file and, where available, the line.
+ * @throws IOException Thrown if reading the file fails.
+ */
+ public static LexicalKnowledgeBase read(Path file) throws IOException {
+ if (file == null) {
+ throw new IllegalArgumentException("File must not be null");
+ }
+ if (!Files.isRegularFile(file)) {
+ throw new IllegalArgumentException("File does not exist or is not a regular file: " + file);
+ }
+ try (InputStream in = new BufferedInputStream(Files.newInputStream(file))) {
+ return read(in, file.toString());
+ }
+ }
+
+ /**
+ * Reads a WN-LMF XML document from a stream. The stream is not closed.
+ *
+ * @param in The document stream. Must not be {@code null}.
+ * @param resourceName The name used in error messages. Must not be {@code null}.
+ * @return The loaded lexicon.
+ * @throws IllegalArgumentException Thrown if an argument is {@code null}.
+ * @throws InvalidFormatException Thrown if the document is malformed; the message names the
+ * resource and, where available, the line.
+ * @throws IOException Thrown if reading the stream fails.
+ */
+ public static LexicalKnowledgeBase read(InputStream in, String resourceName) throws IOException {
+ if (in == null) {
+ throw new IllegalArgumentException("In must not be null");
+ }
+ if (resourceName == null) {
+ throw new IllegalArgumentException("ResourceName must not be null");
+ }
+ final Parser parser = new Parser(resourceName);
+ try {
+ final XMLStreamReader reader = hardenedFactory().createXMLStreamReader(in);
+ try {
+ parser.parse(reader);
+ } finally {
+ reader.close();
+ }
+ } catch (XMLStreamException e) {
+ // StAX wraps a failing stream read in an XMLStreamException; surface it as the I/O failure.
+ final Throwable nested = e.getNestedException() == null ? e.getCause()
+ : e.getNestedException();
+ if (nested instanceof IOException io) {
+ throw io;
+ }
+ throw parser.malformed(e.getLocation(), "XML error: " + e.getMessage(), e);
+ }
+ return parser.build();
+ }
+
+ /**
+ * Builds an XXE-hardened StAX factory: the DTD internal subset is not processed and external
+ * entities and the external DTD subset are denied, so a DOCTYPE is skipped but never resolved.
+ *
+ * @return The hardened factory.
+ */
+ private static XMLInputFactory hardenedFactory() {
+ final XMLInputFactory factory = XMLInputFactory.newFactory();
+ factory.setProperty(XMLInputFactory.SUPPORT_DTD, Boolean.FALSE);
+ factory.setProperty(XMLInputFactory.IS_SUPPORTING_EXTERNAL_ENTITIES, Boolean.FALSE);
+ factory.setProperty(XMLConstants.ACCESS_EXTERNAL_DTD, "");
+ factory.setProperty(XMLInputFactory.IS_COALESCING, Boolean.TRUE);
+ factory.setXMLResolver((publicId, systemId, baseUri, namespace) -> {
+ throw new XMLStreamException("External entity resolution is disabled, refusing " + systemId);
+ });
+ return factory;
+ }
+
+ /** Holds the streaming parse state and performs post-parse resolution. */
+ private static final class Parser {
+
+ private final String resourceName;
+
+ // Entry state.
+ private final Set entryIds = new HashSet<>();
+ private final Map lemmaByEntryId = new HashMap<>();
+ private final Map posByEntryId = new HashMap<>();
+ private final Map synsetBySenseId = new HashMap<>();
+ private final Map> senseOrder =
+ new LinkedHashMap<>();
+ private final List senseRelations = new ArrayList<>();
+ private final Map rawSynsets = new LinkedHashMap<>();
+ // Fallback membership (entry ids per synset in document order) when members is absent.
+ private final Map> entryIdsBySynset = new HashMap<>();
+
+ // Cursor state.
+ private String currentEntryId;
+ private String currentEntryLemma;
+ private WordNetPOS currentEntryPos;
+ private String currentSenseId;
+ private RawSynset currentSynset;
+
+ /**
+ * Creates a parser.
+ *
+ * @param resourceName The name used in error messages.
+ */
+ Parser(String resourceName) {
+ this.resourceName = resourceName;
+ }
+
+ /**
+ * Streams the document, dispatching start and end elements.
+ *
+ * @param reader The StAX reader.
+ * @throws XMLStreamException Thrown if the stream read fails.
+ * @throws InvalidFormatException Thrown if the document is malformed.
+ */
+ void parse(XMLStreamReader reader) throws XMLStreamException, InvalidFormatException {
+ while (reader.hasNext()) {
+ final int event = reader.next();
+ // A DTD event carries nothing that can affect parsing once the factory is hardened.
+ if (event == XMLStreamConstants.START_ELEMENT) {
+ startElement(reader);
+ } else if (event == XMLStreamConstants.END_ELEMENT) {
+ endElement(reader.getLocalName());
+ }
+ }
+ }
+
+ /**
+ * Handles one start element, updating cursor state and collecting raw entries, senses, and
+ * synsets.
+ *
+ * @param reader The StAX reader positioned on the start element.
+ * @throws XMLStreamException Thrown if reading element text fails.
+ * @throws InvalidFormatException Thrown if the element violates the format.
+ */
+ private void startElement(XMLStreamReader reader)
+ throws XMLStreamException, InvalidFormatException {
+ final String name = reader.getLocalName();
+ switch (name) {
+ case LEXICAL_ENTRY_ELEMENT -> {
+ currentEntryId = requireAttribute(reader, "id");
+ if (!entryIds.add(currentEntryId)) {
+ throw malformed(reader.getLocation(),
+ "Duplicate lexical entry id " + currentEntryId, null);
+ }
+ currentEntryLemma = null;
+ currentEntryPos = null;
+ }
+ case "Lemma" -> {
+ if (currentEntryId == null) {
+ throw malformed(reader.getLocation(), "Lemma outside a LexicalEntry", null);
+ }
+ currentEntryLemma = requireAttribute(reader, "writtenForm");
+ currentEntryPos = parsePos(requireAttribute(reader, "partOfSpeech"),
+ reader.getLocation());
+ lemmaByEntryId.put(currentEntryId, currentEntryLemma);
+ posByEntryId.put(currentEntryId, currentEntryPos);
+ }
+ case SENSE_ELEMENT -> {
+ if (currentEntryLemma == null) {
+ throw malformed(reader.getLocation(),
+ "Sense before its entry's Lemma in LexicalEntry " + currentEntryId, null);
+ }
+ currentSenseId = requireAttribute(reader, "id");
+ final String synsetId = requireAttribute(reader, "synset");
+ if (synsetBySenseId.putIfAbsent(currentSenseId, synsetId) != null) {
+ throw malformed(reader.getLocation(), "Duplicate sense id " + currentSenseId, null);
+ }
+ entryIdsBySynset.computeIfAbsent(synsetId, unused -> new ArrayList<>(2))
+ .add(currentEntryId);
+ final List order = senseOrder.computeIfAbsent(
+ InMemoryWordNetLexicon.LemmaKey.of(currentEntryLemma, currentEntryPos),
+ unused -> new ArrayList<>(2));
+ if (!order.contains(synsetId)) {
+ order.add(synsetId);
+ }
+ }
+ case "SenseRelation" -> {
+ if (currentSenseId == null) {
+ throw malformed(reader.getLocation(), "SenseRelation outside a Sense", null);
+ }
+ senseRelations.add(new RawSenseRelation(currentSenseId,
+ requireAttribute(reader, "relType"), requireAttribute(reader, "target"),
+ line(reader.getLocation())));
+ }
+ case SYNSET_ELEMENT -> {
+ final String id = requireAttribute(reader, "id");
+ final WordNetPOS pos = parsePos(requireAttribute(reader, "partOfSpeech"),
+ reader.getLocation());
+ currentSynset = new RawSynset(id, pos, reader.getAttributeValue(null, "members"),
+ line(reader.getLocation()));
+ if (rawSynsets.putIfAbsent(id, currentSynset) != null) {
+ throw malformed(reader.getLocation(), "Duplicate synset id " + id, null);
+ }
+ }
+ case "Definition" -> {
+ if (currentSynset != null && currentSynset.gloss == null) {
+ currentSynset.gloss = reader.getElementText();
+ }
+ }
+ case "SynsetRelation" -> {
+ if (currentSynset == null) {
+ throw malformed(reader.getLocation(), "SynsetRelation outside a Synset", null);
+ }
+ final String relType = requireAttribute(reader, "relType");
+ final String target = requireAttribute(reader, "target");
+ // The escape-hatch type is a documented skip, not a rejection.
+ if (!OTHER_RELATION.equals(relType)) {
+ currentSynset.relations.add(
+ new RawRelation(relType, target, line(reader.getLocation())));
+ }
+ }
+ default -> {
+ // Pronunciation, Form, Example, SyntacticBehaviour, ILIDefinition, and other
+ // elements outside the contract subset are skipped.
+ }
+ }
+ }
+
+ /**
+ * Clears cursor state when a tracked element closes.
+ *
+ * @param name The local name of the closing element.
+ */
+ private void endElement(String name) {
+ switch (name) {
+ case LEXICAL_ENTRY_ELEMENT -> {
+ currentEntryId = null;
+ currentEntryLemma = null;
+ currentEntryPos = null;
+ }
+ case SENSE_ELEMENT -> currentSenseId = null;
+ case SYNSET_ELEMENT -> currentSynset = null;
+ default -> {
+ // Nothing to close for skipped elements.
+ }
+ }
+ }
+
+ /**
+ * Resolves the collected raw state into an immutable lexicon: validates sense targets, lifts
+ * sense relations to the synset level, and materializes the contract synsets.
+ *
+ * @return The loaded lexicon.
+ * @throws InvalidFormatException Thrown if a sense or relation references an undeclared
+ * target, or a synset has no members.
+ */
+ LexicalKnowledgeBase build() throws InvalidFormatException {
+ // Every sense must point to a declared synset, with a consistent part of speech.
+ for (final Map.Entry sense : synsetBySenseId.entrySet()) {
+ final RawSynset target = rawSynsets.get(sense.getValue());
+ if (target == null) {
+ throw malformed(null,
+ "Sense " + sense.getKey() + " references undeclared synset " + sense.getValue(),
+ null);
+ }
+ }
+ // Lift sense relations to the synset level.
+ for (final RawSenseRelation relation : senseRelations) {
+ if (OTHER_RELATION.equals(relation.relType)) {
+ continue;
+ }
+ final String sourceSynsetId = synsetBySenseId.get(relation.sourceSenseId);
+ final String targetSynsetId = synsetBySenseId.get(relation.targetSenseId);
+ if (targetSynsetId == null) {
+ throw malformed(null, "SenseRelation at line " + relation.line + " from sense "
+ + relation.sourceSenseId + " references undeclared sense " + relation.targetSenseId,
+ null);
+ }
+ final RawSynset source = rawSynsets.get(sourceSynsetId);
+ source.relations.add(new RawRelation(relation.relType, targetSynsetId, relation.line));
+ }
+ // Resolve raw synsets into contract synsets.
+ final Map synsetsById = new LinkedHashMap<>(rawSynsets.size() * 2);
+ for (final RawSynset raw : rawSynsets.values()) {
+ final Map> relations = resolveRelations(raw);
+ synsetsById.put(raw.id,
+ new Synset(raw.id, raw.pos, memberLemmas(raw), raw.gloss == null ? "" : raw.gloss,
+ relations));
+ }
+ return new InMemoryWordNetLexicon(synsetsById, senseOrder);
+ }
+
+ /**
+ * Resolves a raw synset's relations into typed target-id lists, deduplicated in source order.
+ *
+ * @param raw The raw synset.
+ * @return The typed relations for the contract synset.
+ * @throws InvalidFormatException Thrown if a relation type is unknown or its target is
+ * undeclared.
+ */
+ private Map> resolveRelations(RawSynset raw)
+ throws InvalidFormatException {
+ final Map> typed = new LinkedHashMap<>();
+ for (final RawRelation relation : raw.relations) {
+ final WordNetRelation type = parseRelation(relation.relType, raw.pos, relation.line);
+ final RawSynset target = rawSynsets.get(relation.target);
+ if (target == null) {
+ throw malformed(null, "Relation " + relation.relType + " at line " + relation.line
+ + " on synset " + raw.id + " references undeclared synset " + relation.target, null);
+ }
+ // Share the synset table's id instance so only one copy of each id is retained.
+ typed.computeIfAbsent(type, unused -> new LinkedHashSet<>()).add(target.id);
+ }
+ final Map> relations = new LinkedHashMap<>(typed.size() * 2);
+ for (final Map.Entry> entry : typed.entrySet()) {
+ relations.put(entry.getKey(), List.copyOf(entry.getValue()));
+ }
+ return relations;
+ }
+
+ /**
+ * Resolves a synset's member entry ids to their lemmas, from the {@code members} attribute
+ * when present and otherwise from the senses that pointed at the synset.
+ *
+ * @param raw The raw synset.
+ * @return The member lemmas in source order, deduplicated.
+ * @throws InvalidFormatException Thrown if the synset has no members, names an undeclared
+ * entry, or a member's part of speech disagrees with the synset's.
+ */
+ private List memberLemmas(RawSynset raw) throws InvalidFormatException {
+ final List entryIds;
+ if (raw.members != null && !raw.members.isEmpty()) {
+ entryIds = LemmaFolding.splitOnSpaces(raw.members);
+ } else {
+ final List fromSenses = entryIdsBySynset.get(raw.id);
+ entryIds = fromSenses == null ? List.of() : fromSenses;
+ }
+ if (entryIds.isEmpty()) {
+ throw malformed(null, "Synset " + raw.id + " at line " + raw.line
+ + " has no member entries", null);
+ }
+ final List lemmas = new ArrayList<>(entryIds.size());
+ for (final String entryId : entryIds) {
+ final String lemma = lemmaByEntryId.get(entryId);
+ if (lemma == null) {
+ throw malformed(null, "Synset " + raw.id + " at line " + raw.line
+ + " lists undeclared member entry " + entryId, null);
+ }
+ if (raw.pos != posByEntryId.get(entryId)) {
+ throw malformed(null, "Synset " + raw.id + " at line " + raw.line
+ + " has part of speech " + raw.pos + " but member entry " + entryId
+ + " has " + posByEntryId.get(entryId), null);
+ }
+ if (!lemmas.contains(lemma)) {
+ lemmas.add(lemma);
+ }
+ }
+ return lemmas;
+ }
+
+ /**
+ * Maps a WN-LMF part-of-speech code to a {@link WordNetPOS}; code {@code s} normalizes to
+ * {@link WordNetPOS#ADJECTIVE}.
+ *
+ * @param code The part-of-speech code.
+ * @param location The parser location, for error reporting.
+ * @return The part of speech.
+ * @throws InvalidFormatException Thrown if the code is unknown.
+ */
+ private WordNetPOS parsePos(String code, Location location) throws InvalidFormatException {
+ return switch (code) {
+ case "n" -> WordNetPOS.NOUN;
+ case "v" -> WordNetPOS.VERB;
+ case "a", "s" -> WordNetPOS.ADJECTIVE;
+ case "r" -> WordNetPOS.ADVERB;
+ default -> throw malformed(location, "Unknown part-of-speech code: " + code, null);
+ };
+ }
+
+ /**
+ * Maps a WN-LMF relation name to a {@link WordNetRelation}. A {@code similar} relation on a
+ * verb synset maps to {@link WordNetRelation#VERB_GROUP}, otherwise to
+ * {@link WordNetRelation#SIMILAR_TO}.
+ *
+ * @param relType The relation name.
+ * @param sourcePos The part of speech of the source synset.
+ * @param line The document line, for error reporting.
+ * @return The mapped relation.
+ * @throws InvalidFormatException Thrown if the relation name is unknown.
+ */
+ private WordNetRelation parseRelation(String relType, WordNetPOS sourcePos, int line)
+ throws InvalidFormatException {
+ if ("similar".equals(relType)) {
+ return sourcePos == WordNetPOS.VERB ? WordNetRelation.VERB_GROUP
+ : WordNetRelation.SIMILAR_TO;
+ }
+ final WordNetRelation relation = RELATION_NAMES.get(relType);
+ if (relation == null) {
+ throw malformed(null, "Unknown relation type " + relType + " at line " + line, null);
+ }
+ return relation;
+ }
+
+ /**
+ * Reads a required attribute from the current element.
+ *
+ * @param reader The StAX reader.
+ * @param attribute The attribute name.
+ * @return The non-empty attribute value.
+ * @throws InvalidFormatException Thrown if the attribute is absent or empty.
+ */
+ private String requireAttribute(XMLStreamReader reader, String attribute)
+ throws InvalidFormatException {
+ final String value = reader.getAttributeValue(null, attribute);
+ if (value == null || value.isEmpty()) {
+ throw malformed(reader.getLocation(), "Element " + reader.getLocalName()
+ + " is missing required attribute " + attribute, null);
+ }
+ return value;
+ }
+
+ /**
+ * Builds a malformed-document exception naming the resource and, when known, the line.
+ *
+ * @param location The parser location, or {@code null} when unavailable.
+ * @param message The failure detail.
+ * @param cause The underlying cause, or {@code null}.
+ * @return The exception to throw.
+ */
+ InvalidFormatException malformed(Location location, String message, Throwable cause) {
+ final int line = line(location);
+ final String prefix = line < 0 ? "Malformed WN-LMF document " + resourceName + ": "
+ : "Malformed WN-LMF document " + resourceName + " at line " + line + ": ";
+ return cause == null ? new InvalidFormatException(prefix + message)
+ : new InvalidFormatException(prefix + message, cause);
+ }
+
+ /**
+ * Extracts a line number from a parser location.
+ *
+ * @param location The location, or {@code null}.
+ * @return The line number, or {@code -1} when unknown.
+ */
+ private static int line(Location location) {
+ return location == null ? -1 : location.getLineNumber();
+ }
+ }
+
+ private static final class RawSynset {
+ private final String id;
+ private final WordNetPOS pos;
+ private final String members;
+ private final int line;
+ private final List relations = new ArrayList<>(4);
+ private String gloss;
+
+ /**
+ * Creates a raw synset gathered during parsing.
+ *
+ * @param id The synset id.
+ * @param pos The part of speech.
+ * @param members The {@code members} attribute value, or {@code null} when absent.
+ * @param line The document line.
+ */
+ RawSynset(String id, WordNetPOS pos, String members, int line) {
+ this.id = id;
+ this.pos = pos;
+ this.members = members;
+ this.line = line;
+ }
+ }
+
+ /** A parsed synset relation, kept until the target synset is known. */
+ private record RawRelation(String relType, String target, int line) {
+ }
+
+ /** A parsed sense relation, kept until both sense ids are known. */
+ private record RawSenseRelation(String sourceSenseId, String relType, String targetSenseId,
+ int line) {
+ }
+
+ /**
+ * Builds the WN-LMF relation-name to {@link WordNetRelation} table.
+ *
+ * @return The immutable name table.
+ */
+ private static Map relationNames() {
+ final Map names = new HashMap<>();
+ names.put("antonym", WordNetRelation.ANTONYM);
+ names.put("hypernym", WordNetRelation.HYPERNYM);
+ names.put("instance_hypernym", WordNetRelation.INSTANCE_HYPERNYM);
+ names.put("hyponym", WordNetRelation.HYPONYM);
+ names.put("instance_hyponym", WordNetRelation.INSTANCE_HYPONYM);
+ names.put("holo_member", WordNetRelation.MEMBER_HOLONYM);
+ names.put("holo_substance", WordNetRelation.SUBSTANCE_HOLONYM);
+ names.put("holo_part", WordNetRelation.PART_HOLONYM);
+ names.put("mero_member", WordNetRelation.MEMBER_MERONYM);
+ names.put("mero_substance", WordNetRelation.SUBSTANCE_MERONYM);
+ names.put("mero_part", WordNetRelation.PART_MERONYM);
+ names.put("attribute", WordNetRelation.ATTRIBUTE);
+ names.put("derivation", WordNetRelation.DERIVATIONALLY_RELATED);
+ names.put("entails", WordNetRelation.ENTAILMENT);
+ names.put("is_entailed_by", WordNetRelation.ENTAILED_BY);
+ names.put("causes", WordNetRelation.CAUSE);
+ names.put("is_caused_by", WordNetRelation.CAUSED_BY);
+ names.put("also", WordNetRelation.ALSO_SEE);
+ names.put("participle", WordNetRelation.PARTICIPLE);
+ names.put("pertainym", WordNetRelation.PERTAINYM);
+ names.put("domain_topic", WordNetRelation.DOMAIN_TOPIC);
+ names.put("has_domain_topic", WordNetRelation.MEMBER_OF_DOMAIN_TOPIC);
+ names.put("domain_region", WordNetRelation.DOMAIN_REGION);
+ names.put("has_domain_region", WordNetRelation.MEMBER_OF_DOMAIN_REGION);
+ // The usage domain carries both its current WN-LMF name and the legacy alias.
+ names.put("exemplifies", WordNetRelation.DOMAIN_USAGE);
+ names.put("domain_usage", WordNetRelation.DOMAIN_USAGE);
+ names.put("is_exemplified_by", WordNetRelation.MEMBER_OF_DOMAIN_USAGE);
+ names.put("has_domain_usage", WordNetRelation.MEMBER_OF_DOMAIN_USAGE);
+ return Map.copyOf(names);
+ }
+}
diff --git a/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/WndbReader.java b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/WndbReader.java
new file mode 100644
index 0000000000..ed6b18ea8b
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/WndbReader.java
@@ -0,0 +1,627 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package opennlp.wordnet;
+
+import java.io.IOException;
+import java.nio.charset.StandardCharsets;
+import java.nio.file.Files;
+import java.nio.file.Path;
+import java.util.ArrayList;
+import java.util.HashMap;
+import java.util.LinkedHashMap;
+import java.util.LinkedHashSet;
+import java.util.List;
+import java.util.Map;
+
+import opennlp.tools.util.InvalidFormatException;
+import opennlp.tools.wordnet.LexicalKnowledgeBase;
+import opennlp.tools.wordnet.Synset;
+import opennlp.tools.wordnet.WordNetPOS;
+import opennlp.tools.wordnet.WordNetRelation;
+
+/**
+ * Reads a Princeton WordNet database directory in the
+ * WNDB format
+ * ({@code index.noun}, {@code data.noun}, and the corresponding pairs for verbs, adjectives, and
+ * adverbs) into a {@link LexicalKnowledgeBase}.
+ *
+ * All eight index and data files must be present. License preamble lines (which begin with a
+ * space in the released files) are skipped. {@code index.sense} is not read, and the
+ * {@code *.exc} exception lists are the {@link MorphyLemmatizer} companion input, read
+ * separately.
+ *
+ * Synset ids are minted as {@code wndb-}offset{@code -}pos from the data file's
+ * 8-digit byte offset and part-of-speech letter, for example {@code wndb-00001740-n}; the id is
+ * opaque to consumers. Adjective satellite lines normalize to {@link WordNetPOS#ADJECTIVE}, the
+ * syntactic markers the adjective files append ({@code (p)}, {@code (a)}, {@code (ip)}) are
+ * stripped, and underscores in lemmas become spaces. Sense order per lemma follows the index
+ * file's offset order.
+ *
+ * Malformed content fails loud with an {@link InvalidFormatException} naming the file and
+ * line; I/O failures propagate as {@link IOException}. The returned lexicon is immutable and safe
+ * for concurrent lookups.
+ */
+public final class WndbReader {
+
+ private static final Map POINTER_SYMBOLS = pointerSymbols();
+
+ /** The prefix of every synset id this reader mints. */
+ private static final String SYNSET_ID_PREFIX = "wndb-";
+
+ /** Not instantiable. */
+ private WndbReader() {
+ }
+
+ /**
+ * Mints a synset id in this reader's scheme: the {@code wndb-} prefix, the 8-digit data-file
+ * byte offset, a hyphen, and the part-of-speech letter, for example {@code wndb-00001740-n}.
+ *
+ * @param offset The 8-digit synset offset field.
+ * @param posChar The WNDB part-of-speech letter.
+ * @return The minted synset id.
+ */
+ private static String synsetId(String offset, char posChar) {
+ return SYNSET_ID_PREFIX + offset + '-' + posChar;
+ }
+
+ /**
+ * Reads a WNDB database directory.
+ *
+ * @param directory The directory containing the eight index and data files. Must not be
+ * {@code null} and must exist.
+ * @return The loaded lexicon.
+ * @throws IllegalArgumentException Thrown if {@code directory} is {@code null} or not a
+ * directory.
+ * @throws InvalidFormatException Thrown if a database file is missing or any file is
+ * malformed; the message names the file and line.
+ * @throws IOException Thrown if reading a file fails.
+ */
+ public static LexicalKnowledgeBase read(Path directory) throws IOException {
+ if (directory == null) {
+ throw new IllegalArgumentException("Directory must not be null");
+ }
+ if (!Files.isDirectory(directory)) {
+ throw new IllegalArgumentException(
+ "Directory does not exist or is not a directory: " + directory);
+ }
+ final Map rawSynsets = new LinkedHashMap<>();
+ for (final FilePos filePos : FilePos.values()) {
+ parseDataFile(directory, filePos, rawSynsets);
+ }
+ final Map synsetsById = resolve(rawSynsets);
+ final Map> senseOrder = new LinkedHashMap<>();
+ for (final FilePos filePos : FilePos.values()) {
+ parseIndexFile(directory, filePos, rawSynsets, senseOrder);
+ }
+ return new InMemoryWordNetLexicon(synsetsById, senseOrder);
+ }
+
+ /** The four part-of-speech file pairs of a WNDB directory. */
+ private enum FilePos {
+ NOUN("noun", 'n', WordNetPOS.NOUN),
+ VERB("verb", 'v', WordNetPOS.VERB),
+ ADJECTIVE("adj", 'a', WordNetPOS.ADJECTIVE),
+ ADVERB("adv", 'r', WordNetPOS.ADVERB);
+
+ private final String suffix;
+ private final char posChar;
+ private final WordNetPOS pos;
+
+ /**
+ * Binds a part of speech to its file suffix and WNDB letter.
+ *
+ * @param suffix The file suffix, for example {@code noun}.
+ * @param posChar The WNDB part-of-speech letter.
+ * @param pos The mapped part of speech.
+ */
+ FilePos(String suffix, char posChar, WordNetPOS pos) {
+ this.suffix = suffix;
+ this.posChar = posChar;
+ this.pos = pos;
+ }
+ }
+
+ /**
+ * Parses one {@code data.*} file, collecting its synsets keyed by minted id.
+ *
+ * @param directory The database directory.
+ * @param filePos The part-of-speech file pair.
+ * @param rawSynsets The accumulating synset table.
+ * @throws IOException Thrown if the file is missing, malformed, or unreadable.
+ */
+ private static void parseDataFile(Path directory, FilePos filePos,
+ Map rawSynsets) throws IOException {
+ final String fileName = "data." + filePos.suffix;
+ final byte[] bytes = readAll(directory.resolve(fileName), fileName);
+ int lineStart = 0;
+ int lineNumber = 0;
+ while (lineStart < bytes.length) {
+ lineNumber++;
+ int lineEnd = lineStart;
+ while (lineEnd < bytes.length && bytes[lineEnd] != '\n') {
+ lineEnd++;
+ }
+ // ISO-8859-1 decodes bytes one-to-one, keeping offsets exact for any released file.
+ final String line =
+ new String(bytes, lineStart, lineEnd - lineStart, StandardCharsets.ISO_8859_1);
+ if (!line.isEmpty() && line.charAt(0) != ' ') {
+ parseDataLine(line, lineStart, fileName, lineNumber, filePos, rawSynsets);
+ }
+ lineStart = lineEnd + 1;
+ }
+ }
+
+ /**
+ * Parses one data-file synset line into a raw synset.
+ *
+ * @param line The decoded line, without its trailing newline.
+ * @param byteOffset The line's byte offset, matched against the line's own offset field.
+ * @param fileName The data file name, for error reporting.
+ * @param lineNumber The 1-based line number.
+ * @param filePos The part-of-speech file pair.
+ * @param rawSynsets The accumulating synset table.
+ * @throws InvalidFormatException Thrown if the line is malformed or its offset field disagrees
+ * with its byte position.
+ */
+ private static void parseDataLine(String line, int byteOffset, String fileName, int lineNumber,
+ FilePos filePos, Map rawSynsets)
+ throws InvalidFormatException {
+ final Tokenizer tokens = new Tokenizer(line, fileName, lineNumber);
+ final String offsetField = tokens.next("synset_offset");
+ if (parseOffset(offsetField, tokens) != byteOffset) {
+ throw malformed(fileName, lineNumber, "Synset offset field " + offsetField
+ + " disagrees with the actual byte position " + byteOffset);
+ }
+ tokens.next("lex_filenum (lexicographer file number)");
+ final String ssType = tokens.next("ss_type (synset type)");
+ final boolean validType = switch (filePos) {
+ case ADJECTIVE -> "a".equals(ssType) || "s".equals(ssType);
+ default -> ssType.length() == 1 && ssType.charAt(0) == filePos.posChar;
+ };
+ if (!validType) {
+ throw malformed(fileName, lineNumber,
+ "Synset type " + ssType + " does not belong in " + fileName);
+ }
+ final int wordCount = tokens.nextInt("w_cnt (word count)", 16);
+ if (wordCount < 1) {
+ throw malformed(fileName, lineNumber, "Word count must be at least 1, got: " + wordCount);
+ }
+ final List lemmas = new ArrayList<>(wordCount);
+ for (int i = 0; i < wordCount; i++) {
+ final String lemma = cleanLemma(tokens.next("word"), fileName, lineNumber);
+ tokens.nextInt("lex_id (sense id within the lexicographer file)", 16);
+ if (!lemmas.contains(lemma)) {
+ lemmas.add(lemma);
+ }
+ }
+ final int pointerCount = tokens.nextInt("p_cnt (pointer count)", 10);
+ final List pointers = new ArrayList<>(pointerCount);
+ for (int i = 0; i < pointerCount; i++) {
+ final String symbol = tokens.next("pointer_symbol");
+ final WordNetRelation relation = POINTER_SYMBOLS.get(symbol);
+ if (relation == null) {
+ throw malformed(fileName, lineNumber, "Undeclared pointer symbol: " + symbol);
+ }
+ final String targetOffset = tokens.next("pointer synset_offset");
+ parseOffset(targetOffset, tokens);
+ final char targetPos = posChar(tokens.next("pointer pos"), tokens);
+ tokens.next("pointer source/target");
+ pointers.add(new RawPointer(relation, synsetId(targetOffset, targetPos), lineNumber));
+ }
+ if (filePos == FilePos.VERB) {
+ final int frameCount = tokens.nextInt("f_cnt (verb frame count)", 10);
+ for (int i = 0; i < frameCount; i++) {
+ tokens.next("frame marker");
+ tokens.next("f_num (verb frame number)");
+ tokens.next("w_num (word number)");
+ }
+ }
+ final String gloss = tokens.gloss();
+ final String id = synsetId(offsetField, filePos.posChar);
+ rawSynsets.put(id, new RawSynset(id, filePos.pos, lemmas, gloss, pointers,
+ fileName, lineNumber));
+ }
+
+ /**
+ * Resolves raw synsets into contract synsets, validating every pointer target.
+ *
+ * @param rawSynsets The parsed synsets keyed by id.
+ * @return The contract synsets keyed by id.
+ * @throws InvalidFormatException Thrown if a pointer targets a nonexistent synset.
+ */
+ private static Map resolve(Map rawSynsets)
+ throws InvalidFormatException {
+ final Map synsetsById = new LinkedHashMap<>(rawSynsets.size() * 2);
+ for (final RawSynset raw : rawSynsets.values()) {
+ final Map> typed = new LinkedHashMap<>();
+ for (final RawPointer pointer : raw.pointers) {
+ final RawSynset target = rawSynsets.get(pointer.targetId);
+ if (target == null) {
+ throw malformed(raw.fileName, pointer.lineNumber, "Synset " + raw.id + " has a "
+ + pointer.relation + " pointer to nonexistent synset " + pointer.targetId);
+ }
+ // Share the synset table's id instance so only one copy of each id is retained.
+ typed.computeIfAbsent(pointer.relation, unused -> new LinkedHashSet<>())
+ .add(target.id);
+ }
+ final Map> relations = new LinkedHashMap<>(typed.size() * 2);
+ for (final Map.Entry> entry : typed.entrySet()) {
+ relations.put(entry.getKey(), List.copyOf(entry.getValue()));
+ }
+ synsetsById.put(raw.id, new Synset(raw.id, raw.pos, raw.lemmas, raw.gloss, relations));
+ }
+ return synsetsById;
+ }
+
+ /**
+ * Parses one {@code index.*} file, building the sense order per folded lemma key.
+ *
+ * @param directory The database directory.
+ * @param filePos The part-of-speech file pair.
+ * @param rawSynsets The resolved synset table, for offset validation.
+ * @param senses The accumulating sense-order map.
+ * @throws IOException Thrown if the file is missing, malformed, or unreadable.
+ */
+ private static void parseIndexFile(Path directory, FilePos filePos,
+ Map rawSynsets,
+ Map> senses)
+ throws IOException {
+ final String fileName = "index." + filePos.suffix;
+ final byte[] bytes = readAll(directory.resolve(fileName), fileName);
+ final String content = new String(bytes, StandardCharsets.ISO_8859_1);
+ int lineNumber = 0;
+ int lineStart = 0;
+ while (lineStart < content.length()) {
+ lineNumber++;
+ int lineEnd = content.indexOf('\n', lineStart);
+ if (lineEnd < 0) {
+ lineEnd = content.length();
+ }
+ final String line = content.substring(lineStart, lineEnd);
+ if (!line.isEmpty() && line.charAt(0) != ' ') {
+ parseIndexLine(line, fileName, lineNumber, filePos, rawSynsets, senses);
+ }
+ lineStart = lineEnd + 1;
+ }
+ }
+
+ /**
+ * Parses one index-file line into a lemma's sense order.
+ *
+ * @param line The line to parse.
+ * @param fileName The index file name, for error reporting.
+ * @param lineNumber The 1-based line number.
+ * @param filePos The part-of-speech file pair.
+ * @param rawSynsets The resolved synset table, for offset validation.
+ * @param senses The accumulating sense-order map.
+ * @throws InvalidFormatException Thrown if the line is malformed or references an unknown
+ * offset.
+ */
+ private static void parseIndexLine(String line, String fileName, int lineNumber,
+ FilePos filePos, Map rawSynsets,
+ Map> senses)
+ throws InvalidFormatException {
+ final Tokenizer tokens = new Tokenizer(line, fileName, lineNumber);
+ final String lemma = tokens.next("lemma");
+ final String pos = tokens.next("pos");
+ if (pos.length() != 1 || pos.charAt(0) != filePos.posChar) {
+ throw malformed(fileName, lineNumber, "Index pos " + pos + " does not belong in "
+ + fileName);
+ }
+ final int synsetCount = tokens.nextInt("synset_cnt (synset count)", 10);
+ if (synsetCount < 1) {
+ throw malformed(fileName, lineNumber,
+ "Synset count must be at least 1, got: " + synsetCount);
+ }
+ final int pointerTypeCount = tokens.nextInt("p_cnt (pointer count)", 10);
+ for (int i = 0; i < pointerTypeCount; i++) {
+ // The summary symbols are informational; the data file's pointers are authoritative.
+ tokens.next("ptr_symbol (pointer symbol)");
+ }
+ tokens.next("sense_cnt (sense count)");
+ tokens.next("tagsense_cnt (tagged-sense count)");
+ final List order = new ArrayList<>(synsetCount);
+ for (int i = 0; i < synsetCount; i++) {
+ final String offset = tokens.next("synset_offset");
+ parseOffset(offset, tokens);
+ final String synsetId = synsetId(offset, filePos.posChar);
+ if (!rawSynsets.containsKey(synsetId)) {
+ throw malformed(fileName, lineNumber, "Lemma " + lemma + " references offset " + offset
+ + " with no data." + filePos.suffix + " line");
+ }
+ if (!order.contains(synsetId)) {
+ order.add(synsetId);
+ }
+ }
+ final InMemoryWordNetLexicon.LemmaKey key =
+ InMemoryWordNetLexicon.LemmaKey.of(lemma, filePos.pos);
+ final List existing = senses.get(key);
+ if (existing == null) {
+ senses.put(key, order);
+ } else {
+ // Two index lemmas can fold to one key; keep first-listed order and append the rest.
+ for (final String synsetId : order) {
+ if (!existing.contains(synsetId)) {
+ existing.add(synsetId);
+ }
+ }
+ }
+ }
+
+ /**
+ * Strips the adjective syntactic markers ({@code (p)}, {@code (a)}, {@code (ip)}) and turns
+ * underscores into spaces.
+ *
+ * @param word The raw word field.
+ * @param fileName The data file name, for error reporting.
+ * @param lineNumber The 1-based line number.
+ * @return The cleaned lemma.
+ * @throws InvalidFormatException Thrown if the word carries an unknown marker or is empty.
+ */
+ private static String cleanLemma(String word, String fileName, int lineNumber)
+ throws InvalidFormatException {
+ String cleaned = word;
+ if (cleaned.endsWith(")")) {
+ final int open = cleaned.lastIndexOf('(');
+ final String marker = open < 0 ? "" : cleaned.substring(open);
+ if (!"(p)".equals(marker) && !"(a)".equals(marker) && !"(ip)".equals(marker)) {
+ throw malformed(fileName, lineNumber, "Unknown syntactic marker on word: " + word);
+ }
+ cleaned = cleaned.substring(0, open);
+ }
+ if (cleaned.isEmpty()) {
+ throw malformed(fileName, lineNumber, "Empty word field");
+ }
+ return cleaned.replace('_', ' ');
+ }
+
+ /**
+ * Parses an 8-digit synset offset.
+ *
+ * @param offset The offset field.
+ * @param tokens The tokenizer, for error reporting.
+ * @return The offset as an integer.
+ * @throws InvalidFormatException Thrown if the field is not 8 digits.
+ */
+ private static int parseOffset(String offset, Tokenizer tokens) throws InvalidFormatException {
+ if (offset.length() != 8) {
+ throw tokens.malformedToken("Synset offset must be 8 digits, got: " + offset);
+ }
+ int value = 0;
+ for (int i = 0; i < 8; i++) {
+ final char c = offset.charAt(i);
+ if (c < '0' || c > '9') {
+ throw tokens.malformedToken("Synset offset must be 8 digits, got: " + offset);
+ }
+ value = value * 10 + (c - '0');
+ }
+ return value;
+ }
+
+ /**
+ * Parses a pointer's one-letter part-of-speech code.
+ *
+ * @param pos The code field.
+ * @param tokens The tokenizer, for error reporting.
+ * @return One of {@code n}, {@code v}, {@code a}, {@code r}.
+ * @throws InvalidFormatException Thrown if the code is not one of those letters.
+ */
+ private static char posChar(String pos, Tokenizer tokens) throws InvalidFormatException {
+ if (pos.length() == 1) {
+ final char c = pos.charAt(0);
+ if (c == 'n' || c == 'v' || c == 'a' || c == 'r') {
+ return c;
+ }
+ }
+ throw tokens.malformedToken("Pointer pos must be one of n, v, a, r, got: " + pos);
+ }
+
+ /**
+ * Reads a required database file in full.
+ *
+ * @param file The file path.
+ * @param fileName The file name, for error reporting.
+ * @return The file bytes.
+ * @throws InvalidFormatException Thrown if the file is missing.
+ * @throws IOException Thrown if reading fails.
+ */
+ private static byte[] readAll(Path file, String fileName) throws IOException {
+ if (!Files.isRegularFile(file)) {
+ throw new InvalidFormatException("Missing WNDB database file: " + file);
+ }
+ return Files.readAllBytes(file);
+ }
+
+ /**
+ * Builds a malformed-file exception naming the file and line.
+ *
+ * @param fileName The file name.
+ * @param lineNumber The 1-based line number.
+ * @param message The failure detail.
+ * @return The exception to throw.
+ */
+ private static InvalidFormatException malformed(String fileName, int lineNumber,
+ String message) {
+ return new InvalidFormatException(
+ "Malformed WNDB file " + fileName + " at line " + lineNumber + ": " + message);
+ }
+
+ /** A cursor over one line's space-separated fields. */
+ private static final class Tokenizer {
+
+ private final String line;
+ private final String fileName;
+ private final int lineNumber;
+ private int position;
+
+ /**
+ * Creates a tokenizer over one line.
+ *
+ * @param line The line to tokenize.
+ * @param fileName The file name, for error reporting.
+ * @param lineNumber The 1-based line number.
+ */
+ Tokenizer(String line, String fileName, int lineNumber) {
+ this.line = line;
+ this.fileName = fileName;
+ this.lineNumber = lineNumber;
+ }
+
+ /**
+ * Reads the next space-separated field.
+ *
+ * @param field The field name, for error reporting.
+ * @return The field value.
+ * @throws InvalidFormatException Thrown if the line is truncated before the field.
+ */
+ String next(String field) throws InvalidFormatException {
+ while (position < line.length() && line.charAt(position) == ' ') {
+ position++;
+ }
+ if (position >= line.length()) {
+ throw malformed(fileName, lineNumber, "Truncated line, missing field: " + field);
+ }
+ final int start = position;
+ while (position < line.length() && line.charAt(position) != ' ') {
+ position++;
+ }
+ return line.substring(start, position);
+ }
+
+ /**
+ * Reads the next field as an integer in the given radix.
+ *
+ * @param field The field name, for error reporting.
+ * @param radix The numeric radix.
+ * @return The parsed value.
+ * @throws InvalidFormatException Thrown if the field is missing or not a valid integer.
+ */
+ int nextInt(String field, int radix) throws InvalidFormatException {
+ final String token = next(field);
+ try {
+ return Integer.parseInt(token, radix);
+ } catch (NumberFormatException e) {
+ throw new InvalidFormatException(malformed(fileName, lineNumber,
+ "Field " + field + " is not a base-" + radix + " integer: " + token).getMessage(), e);
+ }
+ }
+
+ /**
+ * Reads the gloss: the remainder after the pipe separator, trimmed of surrounding spaces.
+ *
+ * @return The gloss text.
+ * @throws InvalidFormatException Thrown if the pipe separator is missing.
+ */
+ String gloss() throws InvalidFormatException {
+ final String separator = next("gloss separator");
+ if (!"|".equals(separator)) {
+ throw malformed(fileName, lineNumber, "Expected the | gloss separator, got: " + separator);
+ }
+ int start = position;
+ while (start < line.length() && line.charAt(start) == ' ') {
+ start++;
+ }
+ int end = line.length();
+ while (end > start && line.charAt(end - 1) == ' ') {
+ end--;
+ }
+ return line.substring(start, end);
+ }
+
+ /**
+ * Builds a malformed-file exception at this tokenizer's line.
+ *
+ * @param message The failure detail.
+ * @return The exception to throw.
+ */
+ InvalidFormatException malformedToken(String message) {
+ return malformed(fileName, lineNumber, message);
+ }
+ }
+
+ /** A parsed pointer line, kept until the target synset is known. */
+ private record RawPointer(WordNetRelation relation, String targetId, int lineNumber) {
+ }
+
+ private static final class RawSynset {
+ private final String id;
+ private final WordNetPOS pos;
+ private final List lemmas;
+ private final String gloss;
+ private final List pointers;
+ private final String fileName;
+ private final int lineNumber;
+
+ /**
+ * Creates a raw synset gathered while parsing a data file.
+ *
+ * @param id The minted synset id.
+ * @param pos The part of speech.
+ * @param lemmas The member lemmas.
+ * @param gloss The gloss text.
+ * @param pointers The raw pointers to resolve.
+ * @param fileName The source file name.
+ * @param lineNumber The source line number.
+ */
+ RawSynset(String id, WordNetPOS pos, List lemmas, String gloss,
+ List pointers, String fileName, int lineNumber) {
+ this.id = id;
+ this.pos = pos;
+ this.lemmas = lemmas;
+ this.gloss = gloss;
+ this.pointers = pointers;
+ this.fileName = fileName;
+ this.lineNumber = lineNumber;
+ }
+ }
+
+ /**
+ * Builds the WNDB pointer-symbol to {@link WordNetRelation} table.
+ *
+ * @return The immutable symbol table.
+ */
+ private static Map pointerSymbols() {
+ final Map symbols = new HashMap<>();
+ symbols.put("!", WordNetRelation.ANTONYM);
+ symbols.put("@", WordNetRelation.HYPERNYM);
+ symbols.put("@i", WordNetRelation.INSTANCE_HYPERNYM);
+ symbols.put("~", WordNetRelation.HYPONYM);
+ symbols.put("~i", WordNetRelation.INSTANCE_HYPONYM);
+ symbols.put("#m", WordNetRelation.MEMBER_HOLONYM);
+ symbols.put("#s", WordNetRelation.SUBSTANCE_HOLONYM);
+ symbols.put("#p", WordNetRelation.PART_HOLONYM);
+ symbols.put("%m", WordNetRelation.MEMBER_MERONYM);
+ symbols.put("%s", WordNetRelation.SUBSTANCE_MERONYM);
+ symbols.put("%p", WordNetRelation.PART_MERONYM);
+ symbols.put("=", WordNetRelation.ATTRIBUTE);
+ symbols.put("+", WordNetRelation.DERIVATIONALLY_RELATED);
+ symbols.put("*", WordNetRelation.ENTAILMENT);
+ symbols.put(">", WordNetRelation.CAUSE);
+ symbols.put("^", WordNetRelation.ALSO_SEE);
+ symbols.put("$", WordNetRelation.VERB_GROUP);
+ symbols.put("&", WordNetRelation.SIMILAR_TO);
+ symbols.put("<", WordNetRelation.PARTICIPLE);
+ symbols.put("\\", WordNetRelation.PERTAINYM);
+ symbols.put(";c", WordNetRelation.DOMAIN_TOPIC);
+ symbols.put("-c", WordNetRelation.MEMBER_OF_DOMAIN_TOPIC);
+ symbols.put(";r", WordNetRelation.DOMAIN_REGION);
+ symbols.put("-r", WordNetRelation.MEMBER_OF_DOMAIN_REGION);
+ symbols.put(";u", WordNetRelation.DOMAIN_USAGE);
+ symbols.put("-u", WordNetRelation.MEMBER_OF_DOMAIN_USAGE);
+ return Map.copyOf(symbols);
+ }
+}
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/ExpansionAssertions.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/ExpansionAssertions.java
new file mode 100644
index 0000000000..c67d7c425e
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/ExpansionAssertions.java
@@ -0,0 +1,42 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package opennlp.wordnet;
+
+import java.util.List;
+
+import opennlp.wordnet.LexicalExpander.Expansion;
+
+/**
+ * Shared lookup helpers for tests that assert on {@link LexicalExpander} output.
+ */
+final class ExpansionAssertions {
+
+ /** Not instantiable. */
+ private ExpansionAssertions() {
+ }
+
+ /**
+ * Finds the first expansion of a term in an expansion list.
+ *
+ * @param expansions The expansions to search. Must not be {@code null}.
+ * @param term The exact term to find. Must not be {@code null}.
+ * @return The first expansion whose term equals {@code term}, or {@code null} when absent.
+ */
+ static Expansion find(List expansions, String term) {
+ return expansions.stream().filter(e -> e.term().equals(term)).findFirst().orElse(null);
+ }
+}
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/HypernymTyperTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/HypernymTyperTest.java
new file mode 100644
index 0000000000..cba6828fc5
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/HypernymTyperTest.java
@@ -0,0 +1,118 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package opennlp.wordnet;
+
+import java.util.LinkedHashMap;
+import java.util.Map;
+import java.util.Optional;
+
+import org.junit.jupiter.api.Assertions;
+import org.junit.jupiter.api.Test;
+
+/**
+ * Tests that {@link HypernymTyper} labels a word by its nearest anchored hypernym over
+ * the fixture taxonomy of {@link SynsetSimilarityTest}, follows instance hypernymy,
+ * prefers the closer of two anchors, and validates its arguments.
+ */
+public class HypernymTyperTest {
+
+ /**
+ * @return A typer over the shared fixture taxonomy with person and location anchors.
+ * Never {@code null}.
+ */
+ private static HypernymTyper typer() {
+ return new HypernymTyper(taxonomy(),
+ Map.of("person", "person", "location", "location"));
+ }
+
+ /**
+ * @return The taxonomy shared with {@link SynsetSimilarityTest}. Never {@code null}.
+ */
+ private static SynsetSimilarityTest.FixtureKnowledgeBase taxonomy() {
+ return SynsetSimilarityTest.taxonomy();
+ }
+
+ /**
+ * Verifies that a noun whose hypernym chain reaches an anchor receives that anchor's
+ * label, both through plain and instance hypernymy, and that the anchor lemma itself
+ * is typed with its own label at distance zero.
+ */
+ @Test
+ void testTypesThroughHypernymAndInstanceChains() {
+ final HypernymTyper typer = typer();
+ Assertions.assertEquals(Optional.of("person"), typer.type("chemist"));
+ Assertions.assertEquals(Optional.of("location"), typer.type("paris"));
+ Assertions.assertEquals(Optional.of("person"), typer.type("person"));
+ Assertions.assertEquals(Optional.of("location"), typer.typeSynset("n8"));
+ }
+
+ /**
+ * Verifies that no label is produced when no sense of the word, or no ancestor of
+ * the synset, reaches an anchor.
+ */
+ @Test
+ void testUnreachableAnchorsYieldEmpty() {
+ final HypernymTyper typer = typer();
+ Assertions.assertEquals(Optional.empty(), typer.type("abstract"));
+ Assertions.assertEquals(Optional.empty(), typer.type("unknownword"));
+ Assertions.assertEquals(Optional.empty(), typer.typeSynset("n12"));
+ Assertions.assertEquals(Optional.empty(), typer.typeSynset("missing"));
+ }
+
+ /**
+ * Verifies that the nearest anchor wins: with scientist registered as its own
+ * anchor, a chemist is a scientist rather than the more distant person.
+ */
+ @Test
+ void testNearestAnchorWins() {
+ final Map anchors = new LinkedHashMap<>();
+ anchors.put("person", "person");
+ anchors.put("scientist", "scientist");
+ final HypernymTyper typer = new HypernymTyper(taxonomy(), anchors);
+ Assertions.assertEquals(Optional.of("scientist"), typer.type("chemist"));
+ Assertions.assertEquals(Optional.of("scientist"), typer.type("scientist"));
+ // the walk is upward only, so an ancestor of an anchor is never typed by it
+ Assertions.assertEquals(Optional.empty(), typer.type("organism"));
+ }
+
+ /**
+ * Verifies that invalid construction and query arguments are rejected: null or
+ * empty inputs, blank anchor entries, and an anchor lemma the knowledge base does
+ * not know.
+ */
+ @Test
+ void testInvalidArguments() {
+ final Map anchors = Map.of("person", "person");
+ Assertions.assertThrows(IllegalArgumentException.class,
+ () -> new HypernymTyper(null, anchors));
+ Assertions.assertThrows(IllegalArgumentException.class,
+ () -> new HypernymTyper(taxonomy(), null));
+ Assertions.assertThrows(IllegalArgumentException.class,
+ () -> new HypernymTyper(taxonomy(), Map.of()));
+ Assertions.assertThrows(IllegalArgumentException.class,
+ () -> new HypernymTyper(taxonomy(), Map.of(" ", "person")));
+ Assertions.assertThrows(IllegalArgumentException.class,
+ () -> new HypernymTyper(taxonomy(), Map.of("person", "\u00A0")));
+ Assertions.assertThrows(IllegalArgumentException.class,
+ () -> new HypernymTyper(taxonomy(), Map.of("notaword", "label")));
+ final HypernymTyper typer = typer();
+ Assertions.assertThrows(IllegalArgumentException.class, () -> typer.type(null));
+ Assertions.assertThrows(IllegalArgumentException.class, () -> typer.type(" "));
+ Assertions.assertThrows(IllegalArgumentException.class, () -> typer.typeSynset(null));
+ }
+}
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/InMemoryWordNetLexiconTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/InMemoryWordNetLexiconTest.java
new file mode 100644
index 0000000000..8a238832fb
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/InMemoryWordNetLexiconTest.java
@@ -0,0 +1,89 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package opennlp.wordnet;
+
+import java.util.List;
+import java.util.Map;
+
+import org.junit.jupiter.api.Test;
+
+import opennlp.tools.wordnet.Synset;
+import opennlp.tools.wordnet.WordNetPOS;
+import opennlp.tools.wordnet.WordNetRelation;
+
+import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assertions.assertThrows;
+import static org.junit.jupiter.api.Assertions.assertTrue;
+
+/**
+ * Exercises the constructor's referential-integrity validation directly, with deliberately
+ * inconsistent maps a reader would never produce: any future reader relies on these checks,
+ * so they are pinned independently of both existing readers.
+ */
+public class InMemoryWordNetLexiconTest {
+
+ private static Synset synset(String id, Map> relations) {
+ return new Synset(id, WordNetPOS.NOUN, List.of("lemma"), "a gloss", relations);
+ }
+
+ @Test
+ void testAcceptsConsistentMaps() {
+ final Synset a = synset("a", Map.of(WordNetRelation.HYPERNYM, List.of("b")));
+ final Synset b = synset("b", Map.of());
+ final InMemoryWordNetLexicon lexicon = new InMemoryWordNetLexicon(
+ Map.of("a", a, "b", b),
+ Map.of(InMemoryWordNetLexicon.LemmaKey.of("lemma", WordNetPOS.NOUN), List.of("a", "b")));
+ assertEquals(2, lexicon.size());
+ assertEquals(List.of(a, b), lexicon.lookup("lemma", WordNetPOS.NOUN));
+ }
+
+ @Test
+ void testRejectsKeyThatDoesNotMatchSynsetId() {
+ final Map table = Map.of("wrong-key", synset("real-id", Map.of()));
+ final IllegalArgumentException e = assertThrows(IllegalArgumentException.class,
+ () -> new InMemoryWordNetLexicon(table, Map.of()));
+ assertTrue(e.getMessage().contains("wrong-key"));
+ }
+
+ @Test
+ void testRejectsDanglingRelationTarget() {
+ final Map table =
+ Map.of("a", synset("a", Map.of(WordNetRelation.HYPERNYM, List.of("nope"))));
+ final IllegalArgumentException e = assertThrows(IllegalArgumentException.class,
+ () -> new InMemoryWordNetLexicon(table, Map.of()));
+ assertTrue(e.getMessage().contains("nope"));
+ assertTrue(e.getMessage().contains("HYPERNYM"));
+ }
+
+ @Test
+ void testRejectsSenseOrderEntryWithUnknownSynset() {
+ final Map table = Map.of("a", synset("a", Map.of()));
+ final Map> senseOrder =
+ Map.of(InMemoryWordNetLexicon.LemmaKey.of("lemma", WordNetPOS.NOUN), List.of("missing"));
+ final IllegalArgumentException e = assertThrows(IllegalArgumentException.class,
+ () -> new InMemoryWordNetLexicon(table, senseOrder));
+ assertTrue(e.getMessage().contains("missing"));
+ assertTrue(e.getMessage().contains("lemma"));
+ }
+
+ @Test
+ void testRejectsNullMaps() {
+ assertThrows(IllegalArgumentException.class, () -> new InMemoryWordNetLexicon(null, Map.of()));
+ assertThrows(IllegalArgumentException.class,
+ () -> new InMemoryWordNetLexicon(Map.of(), null));
+ }
+}
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LemmaFoldingTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LemmaFoldingTest.java
new file mode 100644
index 0000000000..3d32eb4431
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LemmaFoldingTest.java
@@ -0,0 +1,66 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package opennlp.wordnet;
+
+import java.util.List;
+
+import org.junit.jupiter.api.Test;
+
+import opennlp.tools.wordnet.WordNetPOS;
+
+import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assertions.assertThrows;
+
+/**
+ * Pins the shared fold and split behavior every user of {@link LemmaFolding} depends on:
+ * the exception lists, the sense index keys, and the WN-LMF members parsing all fold and
+ * split through this one implementation.
+ */
+public class LemmaFoldingTest {
+
+ @Test
+ void testFoldLowercasesWithRootLocaleAndTreatsUnderscoreAsSpace() {
+ assertEquals("mice", LemmaFolding.fold("MICE"));
+ assertEquals("domestic dog", LemmaFolding.fold("Domestic_Dog"));
+ assertEquals("attorney general", LemmaFolding.fold("attorney_general"));
+ assertEquals("dog", LemmaFolding.fold("dog"));
+ assertEquals("", LemmaFolding.fold(""));
+ }
+
+ @Test
+ void testSplitOnSpacesCollapsesRunsAndIgnoresEdges() {
+ assertEquals(List.of("a", "b", "c"), LemmaFolding.splitOnSpaces("a b c"));
+ assertEquals(List.of("a", "b"), LemmaFolding.splitOnSpaces("a b"));
+ assertEquals(List.of("a"), LemmaFolding.splitOnSpaces("a"));
+ assertEquals(List.of("a"), LemmaFolding.splitOnSpaces(" a "));
+ assertEquals(List.of(), LemmaFolding.splitOnSpaces(""));
+ assertEquals(List.of(), LemmaFolding.splitOnSpaces(" "));
+ }
+
+ @Test
+ void testLemmaKeyAndExceptionLookupAgreeOnTheFold() {
+ // The agreement that makes Morphy correct: a key built from a stored written form and a
+ // query folded at lookup time land on the same canonical shape.
+ assertEquals(InMemoryWordNetLexicon.LemmaKey.of("Domestic_Dog", WordNetPOS.NOUN),
+ InMemoryWordNetLexicon.LemmaKey.of(LemmaFolding.fold("DOMESTIC_DOG"), WordNetPOS.NOUN));
+ }
+
+ @Test
+ void testFoldRejectsNull() {
+ assertThrows(IllegalArgumentException.class, () -> LemmaFolding.fold(null));
+ }
+}
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LexicalExpanderLexiconTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LexicalExpanderLexiconTest.java
new file mode 100644
index 0000000000..176c1142d7
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LexicalExpanderLexiconTest.java
@@ -0,0 +1,106 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package opennlp.wordnet;
+
+import java.util.List;
+
+import org.junit.jupiter.api.Test;
+
+import opennlp.tools.wordnet.LexicalKnowledgeBase;
+import opennlp.tools.wordnet.WordNetPOS;
+import opennlp.wordnet.LexicalExpander.Expansion;
+import opennlp.wordnet.LexicalExpander.Kind;
+
+import static opennlp.wordnet.ExpansionAssertions.find;
+import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assertions.assertNotNull;
+
+/**
+ * End-to-end expansion over the miniature lexicon fixtures: the WN-LMF and WNDB readers each
+ * feed the expander, and the Morphy lemmatizer bridges inflected input, exercising the whole
+ * stack the way a consumer wires it.
+ */
+public class LexicalExpanderLexiconTest {
+
+ @Test
+ void testExpansionOverTheWnLmfLexicon() {
+ final LexicalExpander expander =
+ LexicalExpander.builder(WnLmfReaderTest.fixture()).build();
+
+ final List expansions = expander.expand("dog", WordNetPOS.NOUN);
+ final Expansion domesticDog = find(expansions, "domestic dog");
+ assertEquals(Kind.SYNONYM, domesticDog.kind());
+ assertEquals(1.0, domesticDog.weight());
+ final Expansion canid = find(expansions, "canid");
+ assertEquals(Kind.HYPERNYM, canid.kind());
+ assertEquals(0.5, canid.weight());
+ }
+
+ @Test
+ void testExpansionOverTheWndbLexicon() {
+ final LexicalExpander expander =
+ LexicalExpander.builder(WndbReaderTest.fixture()).build();
+
+ final List expansions = expander.expand("dog", WordNetPOS.NOUN);
+ assertNotNull(find(expansions, "domestic dog"), "got " + expansions);
+ final Expansion canid = find(expansions, "canid");
+ assertEquals(Kind.HYPERNYM, canid.kind());
+ }
+
+ @Test
+ void testUnderscoreQueryFoldsToTheMultiwordLexiconEntry() {
+ // The WNDB index stores the entry as "domestic_dog"; the reader folds it to "domestic dog"
+ // at load time. The expander must fold the underscore query the same way, so the entry's
+ // own synset expands and the query never surfaces as its own synonym.
+ final List expansions = LexicalExpander.builder(WndbReaderTest.fixture())
+ .build().expand("domestic_dog", WordNetPOS.NOUN);
+
+ assertEquals(List.of(
+ new Expansion("dog", Kind.SYNONYM, 0, 0, 1.0),
+ new Expansion("canid", Kind.HYPERNYM, 1, 0, 0.5)), expansions);
+ }
+
+ @Test
+ void testReadersAgreeOnExpansions() {
+ final List lmf = LexicalExpander.builder(WnLmfReaderTest.fixture())
+ .hypernymDepth(2).build().expand("mouse", WordNetPOS.NOUN);
+ final List wndb = LexicalExpander.builder(WndbReaderTest.fixture())
+ .hypernymDepth(2).build().expand("mouse", WordNetPOS.NOUN);
+
+ assertEquals(
+ lmf.stream().map(e -> e.term() + "|" + e.kind() + "|" + e.weight()).toList(),
+ wndb.stream().map(e -> e.term() + "|" + e.kind() + "|" + e.weight()).toList());
+ assertNotNull(find(lmf, "rodent"), "got " + lmf);
+ }
+
+ @Test
+ void testMorphyBridgesInflectedInput() {
+ final LexicalKnowledgeBase lexicon = WnLmfReaderTest.fixture();
+ final LexicalExpander expander = LexicalExpander.builder(lexicon)
+ .lemmatizer(new MorphyLemmatizer(lexicon, MorphyExceptionsTest.fixture()))
+ .build();
+
+ // A regular inflection resolves by rule, an irregular one by the exception list.
+ final List dogs = expander.expand("dogs", WordNetPOS.NOUN);
+ assertEquals(Kind.SYNONYM, find(dogs, "dog").kind());
+ assertNotNull(find(dogs, "canid"), "got " + dogs);
+
+ final List mice = expander.expand("mice", WordNetPOS.NOUN);
+ assertEquals(Kind.SYNONYM, find(mice, "mouse").kind());
+ assertNotNull(find(mice, "rodent"), "got " + mice);
+ }
+}
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LexicalExpanderTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LexicalExpanderTest.java
new file mode 100644
index 0000000000..85f8c700bb
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LexicalExpanderTest.java
@@ -0,0 +1,349 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package opennlp.wordnet;
+
+import java.util.HashMap;
+import java.util.List;
+import java.util.Map;
+import java.util.Optional;
+import java.util.stream.Stream;
+
+import org.junit.jupiter.api.Test;
+import org.junit.jupiter.params.ParameterizedTest;
+import org.junit.jupiter.params.provider.Arguments;
+import org.junit.jupiter.params.provider.MethodSource;
+
+import opennlp.tools.lemmatizer.Lemmatizer;
+import opennlp.tools.wordnet.LexicalKnowledgeBase;
+import opennlp.tools.wordnet.Synset;
+import opennlp.tools.wordnet.WordNetPOS;
+import opennlp.tools.wordnet.WordNetRelation;
+import opennlp.wordnet.LexicalExpander.Expansion;
+import opennlp.wordnet.LexicalExpander.Kind;
+
+import static opennlp.wordnet.ExpansionAssertions.find;
+import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assertions.assertFalse;
+import static org.junit.jupiter.api.Assertions.assertNotNull;
+import static org.junit.jupiter.api.Assertions.assertNull;
+import static org.junit.jupiter.api.Assertions.assertThrows;
+import static org.junit.jupiter.api.Assertions.assertTrue;
+
+/**
+ * Behavioral tests over a hand-built lexicon whose graph shape is fully controlled: sense
+ * ranking, hypernym depth and decay, hyponym opt-in, deduplication, exclusion of the input,
+ * cycle termination, and configuration validation.
+ */
+public class LexicalExpanderTest {
+
+ // dog: sense 1 = {dog, domestic dog} -> canid -> carnivore, with hyponym puppy;
+ // sense 2 = {dog, frank, hot dog} -> sausage. The verb sense = {dog, chase}.
+ // hot dog: the standalone multiword sense {hot dog, red hot}.
+ // alpha <-> beta form a malformed hypernym cycle.
+ // Lookups fold through LemmaFolding, exactly as the readers fold their keys at load time.
+ private static LexicalKnowledgeBase lexicon() {
+ final Map synsets = new HashMap<>();
+ final Map> senses = new HashMap<>();
+
+ final Synset n1 = new Synset("n1", WordNetPOS.NOUN, List.of("dog", "domestic dog"), "canine",
+ Map.of(WordNetRelation.HYPERNYM, List.of("n2"),
+ WordNetRelation.HYPONYM, List.of("n4")));
+ final Synset n2 = new Synset("n2", WordNetPOS.NOUN, List.of("canid"), "canid family",
+ Map.of(WordNetRelation.HYPERNYM, List.of("n3")));
+ final Synset n3 = new Synset("n3", WordNetPOS.NOUN, List.of("carnivore"), "meat eater",
+ Map.of());
+ final Synset n4 = new Synset("n4", WordNetPOS.NOUN, List.of("puppy"), "young dog", Map.of());
+ final Synset n5 = new Synset("n5", WordNetPOS.NOUN, List.of("dog", "frank", "hot dog"),
+ "sausage in a bun", Map.of(WordNetRelation.HYPERNYM, List.of("n6")));
+ final Synset n6 = new Synset("n6", WordNetPOS.NOUN, List.of("sausage"), "ground meat",
+ Map.of());
+ final Synset v1 = new Synset("v1", WordNetPOS.VERB, List.of("dog", "chase"), "follow",
+ Map.of());
+ final Synset m1 = new Synset("m1", WordNetPOS.NOUN, List.of("hot dog", "red hot"),
+ "grilled sausage", Map.of());
+ final Synset c1 = new Synset("c1", WordNetPOS.NOUN, List.of("alpha"), "cycle start",
+ Map.of(WordNetRelation.HYPERNYM, List.of("c2")));
+ final Synset c2 = new Synset("c2", WordNetPOS.NOUN, List.of("beta"), "cycle end",
+ Map.of(WordNetRelation.HYPERNYM, List.of("c1")));
+
+ for (final Synset synset : List.of(n1, n2, n3, n4, n5, n6, v1, m1, c1, c2)) {
+ synsets.put(synset.id(), synset);
+ }
+ senses.put("dog|NOUN", List.of(n1, n5));
+ senses.put("dog|VERB", List.of(v1));
+ senses.put("domestic dog|NOUN", List.of(n1));
+ senses.put("hot dog|NOUN", List.of(m1));
+ senses.put("alpha|NOUN", List.of(c1));
+
+ return new LexicalKnowledgeBase() {
+ @Override
+ public List lookup(String lemma, WordNetPOS pos) {
+ if (lemma == null || pos == null) {
+ throw new IllegalArgumentException("null");
+ }
+ return senses.getOrDefault(LemmaFolding.fold(lemma) + "|" + pos, List.of());
+ }
+
+ @Override
+ public Optional synset(String synsetId) {
+ return Optional.ofNullable(synsets.get(synsetId));
+ }
+ };
+ }
+
+ @Test
+ void testSynonymsAndHypernymsWithDefaultConfiguration() {
+ final List expansions =
+ LexicalExpander.builder(lexicon()).build().expand("dog", WordNetPOS.NOUN);
+
+ final Expansion domesticDog = find(expansions, "domestic dog");
+ assertEquals(Kind.SYNONYM, domesticDog.kind());
+ assertEquals(1.0, domesticDog.weight());
+ assertEquals(0, domesticDog.senseRank());
+
+ final Expansion canid = find(expansions, "canid");
+ assertEquals(Kind.HYPERNYM, canid.kind());
+ assertEquals(1, canid.depth());
+ assertEquals(0.5, canid.weight());
+
+ final Expansion frank = find(expansions, "frank");
+ assertEquals(Kind.SYNONYM, frank.kind());
+ assertEquals(1, frank.senseRank());
+ assertEquals(0.5, frank.weight());
+
+ // Depth 1 by default: the grandparent stays out, and so do hyponyms.
+ assertNull(find(expansions, "carnivore"));
+ assertNull(find(expansions, "puppy"));
+ }
+
+ @Test
+ void testUnderscoreInputReachesTheSpaceFoldedLexiconEntry() {
+ // The lexicon keys "hot_dog" under its folded form "hot dog". The expander must fold the
+ // input the same way, so "hot dog" is excluded as the input itself and cannot displace the
+ // other member "red hot" from the capped result.
+ final List expansions = LexicalExpander.builder(lexicon())
+ .maxExpansions(1).build().expand("hot_dog", WordNetPOS.NOUN);
+
+ assertEquals(List.of(new Expansion("red hot", Kind.SYNONYM, 0, 0, 1.0)), expansions);
+ }
+
+ @Test
+ void testTheInputTermIsNeverAnExpansion() {
+ for (final Expansion expansion :
+ LexicalExpander.builder(lexicon()).build().expand("dog", WordNetPOS.NOUN)) {
+ assertFalse(expansion.term().equalsIgnoreCase("dog"), "got " + expansion);
+ }
+ }
+
+ @Test
+ void testDeeperHypernymWalkDecaysPerStep() {
+ final List expansions = LexicalExpander.builder(lexicon())
+ .hypernymDepth(2).build().expand("dog", WordNetPOS.NOUN);
+
+ final Expansion carnivore = find(expansions, "carnivore");
+ assertEquals(2, carnivore.depth());
+ assertEquals(0.25, carnivore.weight());
+ }
+
+ @Test
+ void testHyponymsAreOptIn() {
+ final List expansions = LexicalExpander.builder(lexicon())
+ .includeHyponyms(true).build().expand("dog", WordNetPOS.NOUN);
+
+ final Expansion puppy = find(expansions, "puppy");
+ assertEquals(Kind.HYPONYM, puppy.kind());
+ assertEquals(0.5, puppy.weight());
+ }
+
+ @Test
+ void testMaxSensesLimitsToTheMostSalient() {
+ final List expansions = LexicalExpander.builder(lexicon())
+ .maxSenses(1).build().expand("dog", WordNetPOS.NOUN);
+
+ assertNull(find(expansions, "frank"));
+ assertNotNull(find(expansions, "domestic dog"));
+ }
+
+ @Test
+ void testAllPosExpansionIncludesVerbSynonyms() {
+ final List expansions =
+ LexicalExpander.builder(lexicon()).build().expand("dog");
+
+ assertNotNull(find(expansions, "chase"));
+ assertNotNull(find(expansions, "domestic dog"));
+ }
+
+ @Test
+ void testCyclicHypernymDataTerminates() {
+ final List expansions = LexicalExpander.builder(lexicon())
+ .hypernymDepth(10).build().expand("alpha", WordNetPOS.NOUN);
+
+ final Expansion beta = find(expansions, "beta");
+ assertEquals(1, beta.depth());
+ // The cycle leads back to alpha's own synset, which is visited and the input besides.
+ assertEquals(1, expansions.size());
+ }
+
+ @Test
+ void testDeduplicationKeepsTheHighestWeight() {
+ // "domestic dog" reaches n1 at rank 0; its synonym "dog" is the only other member and
+ // must appear once with the rank-0 weight even though deeper paths could yield it again.
+ final List expansions = LexicalExpander.builder(lexicon())
+ .hypernymDepth(2).build().expand("domestic dog", WordNetPOS.NOUN);
+
+ final Expansion dog = find(expansions, "dog");
+ assertEquals(1.0, dog.weight());
+ assertEquals(1, expansions.stream().filter(e -> e.term().equals("dog")).count());
+ }
+
+ @Test
+ void testOrderingIsWeightDescendingAndStable() {
+ final List expansions = LexicalExpander.builder(lexicon())
+ .hypernymDepth(2).includeHyponyms(true).build().expand("dog", WordNetPOS.NOUN);
+
+ for (int i = 1; i < expansions.size(); i++) {
+ assertTrue(expansions.get(i - 1).weight() >= expansions.get(i).weight(),
+ "weights must not increase: " + expansions);
+ }
+ assertEquals(expansions,
+ LexicalExpander.builder(lexicon()).hypernymDepth(2).includeHyponyms(true).build()
+ .expand("dog", WordNetPOS.NOUN));
+ }
+
+ @Test
+ void testMaxExpansionsCapsAfterRanking() {
+ final List expansions = LexicalExpander.builder(lexicon())
+ .hypernymDepth(2).includeHyponyms(true).maxExpansions(2).build()
+ .expand("dog", WordNetPOS.NOUN);
+
+ assertEquals(2, expansions.size());
+ assertEquals(1.0, expansions.get(0).weight());
+ }
+
+ @Test
+ void testUnknownTermExpandsToNothing() {
+ assertEquals(List.of(),
+ LexicalExpander.builder(lexicon()).build().expand("xyzzy", WordNetPOS.NOUN));
+ }
+
+ @Test
+ void testLemmatizerFallbackExpandsInflectedInput() {
+ final LexicalExpander expander = LexicalExpander.builder(lexicon())
+ .lemmatizer(new Lemmatizer() {
+ @Override
+ public String[] lemmatize(String[] tokens, String[] tags) {
+ final String[] lemmas = new String[tokens.length];
+ for (int i = 0; i < tokens.length; i++) {
+ lemmas[i] = "dogs".equals(tokens[i]) ? "dog" : "O";
+ }
+ return lemmas;
+ }
+
+ @Override
+ public List> lemmatize(List tokens, List tags) {
+ throw new UnsupportedOperationException();
+ }
+ })
+ .build();
+
+ final List expansions = expander.expand("dogs", WordNetPOS.NOUN);
+ // The lemma itself surfaces as a synonym, along with the rest of its synsets.
+ final Expansion dog = find(expansions, "dog");
+ assertEquals(Kind.SYNONYM, dog.kind());
+ assertEquals(1.0, dog.weight());
+ assertNotNull(find(expansions, "domestic dog"));
+
+ assertEquals(List.of(), expander.expand("cats", WordNetPOS.NOUN));
+ }
+
+ @Test
+ void testValidationFailsLoudly() {
+ assertThrows(IllegalArgumentException.class, () -> LexicalExpander.builder(null));
+ final LexicalExpander.Builder builder = LexicalExpander.builder(lexicon());
+ assertThrows(IllegalArgumentException.class, () -> builder.lemmatizer(null));
+ assertThrows(IllegalArgumentException.class, () -> builder.maxSenses(0));
+ assertThrows(IllegalArgumentException.class, () -> builder.hypernymDepth(-1));
+ assertThrows(IllegalArgumentException.class, () -> builder.maxExpansions(0));
+ assertThrows(IllegalArgumentException.class, () -> builder.senseDecay(0));
+ assertThrows(IllegalArgumentException.class, () -> builder.senseDecay(1.5));
+ assertThrows(IllegalArgumentException.class, () -> builder.depthDecay(0));
+ assertThrows(IllegalArgumentException.class, () -> builder.depthDecay(1.5));
+
+ final LexicalExpander expander = builder.build();
+ assertThrows(IllegalArgumentException.class, () -> expander.expand(null));
+ assertThrows(IllegalArgumentException.class, () -> expander.expand(" "));
+ assertThrows(IllegalArgumentException.class, () -> expander.expand("dog", null));
+ }
+
+ /**
+ * Verifies that decay products which underflow to zero are dropped instead of emitted:
+ * with the smallest positive depth decay, the first hypernym level keeps the smallest
+ * positive weight while the second level underflows to zero and never appears, and no
+ * reported expansion carries a weight outside the documented range.
+ */
+ @Test
+ void testUnderflowedWeightsAreDropped() {
+ final List expansions = LexicalExpander.builder(lexicon())
+ .depthDecay(Double.MIN_VALUE).hypernymDepth(2).maxSenses(1).build()
+ .expand("domestic dog", WordNetPOS.NOUN);
+
+ final Expansion canid = find(expansions, "canid");
+ assertNotNull(canid);
+ assertEquals(Double.MIN_VALUE, canid.weight(), 0.0);
+ assertNull(find(expansions, "carnivore"));
+ for (final Expansion expansion : expansions) {
+ assertTrue(expansion.weight() > 0.0 && expansion.weight() <= 1.0,
+ "weight out of (0, 1]: " + expansion);
+ }
+ }
+
+ private static Stream invalidExpansions() {
+ return Stream.of(
+ Arguments.of(null, Kind.SYNONYM, 0, 0, 1.0),
+ Arguments.of(" ", Kind.SYNONYM, 0, 0, 1.0),
+ Arguments.of("dog", null, 0, 0, 1.0),
+ Arguments.of("dog", Kind.SYNONYM, -1, 0, 1.0),
+ Arguments.of("dog", Kind.SYNONYM, 0, -1, 1.0),
+ Arguments.of("dog", Kind.SYNONYM, 0, 0, 0.0),
+ Arguments.of("dog", Kind.SYNONYM, 0, 0, 1.5),
+ Arguments.of("dog", Kind.SYNONYM, 0, 0, Double.NaN));
+ }
+
+ /**
+ * Verifies that the {@link Expansion} record rejects every component
+ * outside its documented range with a loud exception.
+ */
+ @ParameterizedTest
+ @MethodSource("invalidExpansions")
+ void testExpansionValidatesItsComponents(String term, Kind kind, int depth, int senseRank,
+ double weight) {
+ assertThrows(IllegalArgumentException.class,
+ () -> new Expansion(term, kind, depth, senseRank, weight));
+ }
+
+ /** Verifies that a fully valid component set is accepted. */
+ @Test
+ void testExpansionAcceptsValidComponents() {
+ final Expansion expansion = new Expansion("dog", Kind.HYPERNYM, 2, 1, 0.25);
+
+ assertEquals("dog", expansion.term());
+ assertEquals(Kind.HYPERNYM, expansion.kind());
+ assertEquals(2, expansion.depth());
+ assertEquals(1, expansion.senseRank());
+ assertEquals(0.25, expansion.weight());
+ }
+}
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LexicalExpansionUsageExampleTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LexicalExpansionUsageExampleTest.java
new file mode 100644
index 0000000000..c89a62499a
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LexicalExpansionUsageExampleTest.java
@@ -0,0 +1,154 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package opennlp.wordnet;
+
+import java.util.HashMap;
+import java.util.List;
+import java.util.Map;
+import java.util.Optional;
+
+import org.junit.jupiter.api.Assertions;
+import org.junit.jupiter.api.Test;
+
+import opennlp.tools.wordnet.LexicalKnowledgeBase;
+import opennlp.tools.wordnet.Synset;
+import opennlp.tools.wordnet.WordNetPOS;
+import opennlp.tools.wordnet.WordNetRelation;
+import opennlp.wordnet.LexicalExpander.Expansion;
+import opennlp.wordnet.LexicalExpander.Kind;
+
+import static opennlp.wordnet.ExpansionAssertions.find;
+
+/**
+ * Runs the manual's lexical expansion and synset similarity examples (docbkx
+ * {@code wordnet.xml}) verbatim: every value the chapter states is asserted here, so a
+ * change breaking this test breaks the manual. The taxonomy is a hand-built miniature
+ * matching the shapes used elsewhere in this module's tests.
+ */
+public class LexicalExpansionUsageExampleTest {
+
+ /**
+ * dog sense 1 = {dog, domestic dog} -> canid; sense 2 = {dog, frank, hot dog} -> sausage.
+ */
+ private static LexicalKnowledgeBase dogTaxonomy() {
+ final Map synsets = new HashMap<>();
+ final Map> senses = new HashMap<>();
+
+ final Synset n1 = new Synset("n1", WordNetPOS.NOUN, List.of("dog", "domestic dog"), "canine",
+ Map.of(WordNetRelation.HYPERNYM, List.of("n2")));
+ final Synset n2 = new Synset("n2", WordNetPOS.NOUN, List.of("canid"), "canid family", Map.of());
+ final Synset n5 = new Synset("n5", WordNetPOS.NOUN, List.of("dog", "frank", "hot dog"),
+ "sausage in a bun", Map.of(WordNetRelation.HYPERNYM, List.of("n6")));
+ final Synset n6 = new Synset("n6", WordNetPOS.NOUN, List.of("sausage"), "ground meat",
+ Map.of());
+
+ for (final Synset synset : List.of(n1, n2, n5, n6)) {
+ synsets.put(synset.id(), synset);
+ }
+ senses.put("dog|NOUN", List.of(n1, n5));
+ senses.put("domestic dog|NOUN", List.of(n1));
+
+ return new LexicalKnowledgeBase() {
+ @Override
+ public List lookup(String lemma, WordNetPOS pos) {
+ if (lemma == null) {
+ throw new IllegalArgumentException("lemma must not be null");
+ }
+ if (pos == null) {
+ throw new IllegalArgumentException("pos must not be null");
+ }
+ return senses.getOrDefault(LemmaFolding.fold(lemma) + "|" + pos, List.of());
+ }
+
+ @Override
+ public Optional synset(String synsetId) {
+ return Optional.ofNullable(synsets.get(synsetId));
+ }
+ };
+ }
+
+ /**
+ * chemist -> scientist -> person -> organism -> physical -> entity; city -> location ->
+ * physical.
+ */
+ private static LexicalKnowledgeBase similarityTaxonomy() {
+ final Map byId = new HashMap<>();
+ add(byId, "n1", "entity");
+ add(byId, "n2", "physical", "n1");
+ add(byId, "n3", "organism", "n2");
+ add(byId, "n4", "person", "n3");
+ add(byId, "n5", "scientist", "n4");
+ add(byId, "n6", "chemist", "n5");
+ add(byId, "n7", "location", "n2");
+ add(byId, "n8", "city", "n7");
+ return new LexicalKnowledgeBase() {
+ @Override
+ public List lookup(String lemma, WordNetPOS pos) {
+ return List.of();
+ }
+
+ @Override
+ public Optional synset(String synsetId) {
+ return Optional.ofNullable(byId.get(synsetId));
+ }
+ };
+ }
+
+ private static void add(Map byId, String id, String lemma, String... parents) {
+ final Map> relations = parents.length == 0
+ ? Map.of() : Map.of(WordNetRelation.HYPERNYM, List.of(parents));
+ byId.put(id, new Synset(id, WordNetPOS.NOUN, List.of(lemma), "fixture", relations));
+ }
+
+ /**
+ * Default expansion of noun {@code dog}: synonym and depth-1 hypernym weights.
+ */
+ @Test
+ void testExpandDogNoun() {
+ final List expansions =
+ LexicalExpander.builder(dogTaxonomy()).build().expand("dog", WordNetPOS.NOUN);
+
+ final Expansion domesticDog = find(expansions, "domestic dog");
+ Assertions.assertNotNull(domesticDog);
+ Assertions.assertEquals(Kind.SYNONYM, domesticDog.kind());
+ Assertions.assertEquals(1.0, domesticDog.weight());
+ Assertions.assertEquals(0, domesticDog.senseRank());
+
+ final Expansion canid = find(expansions, "canid");
+ Assertions.assertNotNull(canid);
+ Assertions.assertEquals(Kind.HYPERNYM, canid.kind());
+ Assertions.assertEquals(1, canid.depth());
+ Assertions.assertEquals(0.5, canid.weight());
+
+ final Expansion frank = find(expansions, "frank");
+ Assertions.assertNotNull(frank);
+ Assertions.assertEquals(Kind.SYNONYM, frank.kind());
+ Assertions.assertEquals(1, frank.senseRank());
+ Assertions.assertEquals(0.5, frank.weight());
+ }
+
+ /**
+ * Path and Wu-Palmer scores on the miniature scientist/city taxonomy.
+ */
+ @Test
+ void testSynsetSimilarityScores() {
+ final SynsetSimilarity similarity = new SynsetSimilarity(similarityTaxonomy());
+ Assertions.assertEquals(0.5, similarity.path("n6", "n5"), 1e-9);
+ Assertions.assertEquals(8.0 / 9.0, similarity.wuPalmer("n5", "n6"), 1e-9);
+ }
+}
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LexiconConcurrencyTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LexiconConcurrencyTest.java
new file mode 100644
index 0000000000..0292f7ffe2
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LexiconConcurrencyTest.java
@@ -0,0 +1,91 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package opennlp.wordnet;
+
+import java.util.List;
+import java.util.Queue;
+import java.util.concurrent.ConcurrentLinkedQueue;
+import java.util.concurrent.CountDownLatch;
+import java.util.concurrent.TimeUnit;
+
+import org.junit.jupiter.api.Test;
+
+import opennlp.tools.wordnet.LexicalKnowledgeBase;
+import opennlp.tools.wordnet.WordNetPOS;
+import opennlp.tools.wordnet.WordNetRelation;
+
+import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assertions.assertTrue;
+
+/**
+ * Exercises the immutable-after-load contract: one loaded lexicon serves many threads issuing
+ * concurrent lookups, and every thread observes exactly the single-threaded results.
+ */
+public class LexiconConcurrencyTest {
+
+ private static final int THREADS = 8;
+ private static final int ITERATIONS = 500;
+
+ @Test
+ void testConcurrentLookupsSeeConsistentResults() throws InterruptedException {
+ final LexicalKnowledgeBase lexicon = WndbReaderTest.fixture();
+ final CountDownLatch start = new CountDownLatch(1);
+ final CountDownLatch done = new CountDownLatch(THREADS);
+ final Queue problems = new ConcurrentLinkedQueue<>();
+ for (int t = 0; t < THREADS; t++) {
+ final Thread thread = new Thread(() -> {
+ try {
+ start.await();
+ for (int i = 0; i < ITERATIONS; i++) {
+ verifyOnce(lexicon, problems);
+ }
+ } catch (InterruptedException e) {
+ Thread.currentThread().interrupt();
+ problems.add("Interrupted: " + e);
+ } catch (RuntimeException e) {
+ problems.add("Unexpected exception: " + e);
+ } finally {
+ done.countDown();
+ }
+ });
+ thread.setDaemon(true);
+ thread.start();
+ }
+ start.countDown();
+ assertTrue(done.await(60, TimeUnit.SECONDS), "Worker threads must finish in time");
+ assertEquals(List.of(), List.copyOf(problems));
+ }
+
+ private static void verifyOnce(LexicalKnowledgeBase lexicon, Queue problems) {
+ if (!"wndb-00001075-n".equals(lexicon.lookup("dog", WordNetPOS.NOUN).get(0).id())) {
+ problems.add("Wrong dog lookup");
+ }
+ if (lexicon.lookup("run", WordNetPOS.NOUN).size() != 2) {
+ problems.add("Wrong run sense count");
+ }
+ if (!List.of("wndb-00001160-n")
+ .equals(lexicon.related("wndb-00001075-n", WordNetRelation.HYPERNYM))) {
+ problems.add("Wrong dog hypernym");
+ }
+ if (lexicon.contains("zebra", WordNetPOS.NOUN)) {
+ problems.add("Phantom zebra");
+ }
+ if (!lexicon.contains("walk", WordNetPOS.VERB)) {
+ problems.add("Missing walk verb");
+ }
+ }
+}
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/MorphyExceptionsTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/MorphyExceptionsTest.java
new file mode 100644
index 0000000000..b3aaf78ba2
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/MorphyExceptionsTest.java
@@ -0,0 +1,124 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package opennlp.wordnet;
+
+import java.io.IOException;
+import java.nio.file.Files;
+import java.nio.file.Path;
+import java.util.List;
+
+import org.junit.jupiter.api.Test;
+import org.junit.jupiter.api.io.TempDir;
+import org.junit.jupiter.params.ParameterizedTest;
+import org.junit.jupiter.params.provider.CsvSource;
+
+import opennlp.tools.util.InvalidFormatException;
+import opennlp.tools.wordnet.WordNetPOS;
+
+import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assertions.assertThrows;
+import static org.junit.jupiter.api.Assertions.assertTrue;
+
+public class MorphyExceptionsTest {
+
+ static MorphyExceptions fixture() {
+ try {
+ return MorphyExceptions.load(WndbReaderTest.fixtureDirectory());
+ } catch (IOException e) {
+ throw new IllegalStateException("Unexpected IOException reading the fixture lists", e);
+ }
+ }
+
+ /**
+ * Writes the standard one-entry exception list for each part of speech into
+ * {@code directory}. Tests that need a variation overwrite or delete individual files
+ * afterwards.
+ *
+ * @param directory The directory to receive {@code noun.exc}, {@code verb.exc},
+ * {@code adj.exc}, and {@code adv.exc}.
+ * @throws IOException Thrown if writing a file fails.
+ */
+ private static void writeStandardLists(Path directory) throws IOException {
+ Files.writeString(directory.resolve("noun.exc"), "mice mouse\n");
+ Files.writeString(directory.resolve("verb.exc"), "went go\n");
+ Files.writeString(directory.resolve("adj.exc"), "better good\n");
+ Files.writeString(directory.resolve("adv.exc"), "best well\n");
+ }
+
+ @ParameterizedTest
+ @CsvSource(nullValues = "unknown", value = {
+ "mice, NOUN, mouse",
+ "went, VERB, go",
+ "better, ADJECTIVE, good",
+ "best, ADVERB, well",
+ // Entries are part-of-speech scoped: went is only a verb exception.
+ "went, NOUN, unknown",
+ "dog, NOUN, unknown",
+ })
+ void testLookupPerPartOfSpeech(String form, WordNetPOS pos, String lemma) {
+ final List expected = lemma == null ? List.of() : List.of(lemma);
+ assertEquals(expected, fixture().lookup(form, pos));
+ }
+
+ @Test
+ void testLookupFoldsCase() {
+ assertEquals(List.of("mouse"), fixture().lookup("Mice", WordNetPOS.NOUN));
+ assertEquals(List.of("mouse"), fixture().lookup("MICE", WordNetPOS.NOUN));
+ }
+
+ @Test
+ void testLookupRejectsNulls() {
+ final MorphyExceptions exceptions = fixture();
+ assertThrows(IllegalArgumentException.class,
+ () -> exceptions.lookup(null, WordNetPOS.NOUN));
+ assertThrows(IllegalArgumentException.class, () -> exceptions.lookup("mice", null));
+ }
+
+ @Test
+ void testLoadRejectsNullAndMissingDirectory(@TempDir Path tempDir) {
+ assertThrows(IllegalArgumentException.class, () -> MorphyExceptions.load(null));
+ assertThrows(IllegalArgumentException.class,
+ () -> MorphyExceptions.load(tempDir.resolve("absent")));
+ }
+
+ @Test
+ void testLoadRejectsMissingFile(@TempDir Path tempDir) throws IOException {
+ writeStandardLists(tempDir);
+ Files.delete(tempDir.resolve("adv.exc"));
+ final InvalidFormatException e = assertThrows(InvalidFormatException.class,
+ () -> MorphyExceptions.load(tempDir));
+ assertTrue(e.getMessage().contains("adv.exc"));
+ }
+
+ @Test
+ void testLoadRejectsMalformedLine(@TempDir Path tempDir) throws IOException {
+ writeStandardLists(tempDir);
+ Files.writeString(tempDir.resolve("noun.exc"), "mice mouse\nlonely\n");
+ final InvalidFormatException e = assertThrows(InvalidFormatException.class,
+ () -> MorphyExceptions.load(tempDir));
+ assertTrue(e.getMessage().contains("noun.exc"));
+ assertTrue(e.getMessage().contains("line 2"));
+ }
+
+ @Test
+ void testMultipleBaseFormsKeepFileOrder(@TempDir Path tempDir) throws IOException {
+ writeStandardLists(tempDir);
+ Files.writeString(tempDir.resolve("noun.exc"), "axes axis ax\n");
+ assertEquals(List.of("axis", "ax"),
+ MorphyExceptions.load(tempDir).lookup("axes", WordNetPOS.NOUN));
+ }
+}
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/MorphyLemmatizerTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/MorphyLemmatizerTest.java
new file mode 100644
index 0000000000..24df815548
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/MorphyLemmatizerTest.java
@@ -0,0 +1,188 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package opennlp.wordnet;
+
+import java.util.List;
+
+import org.junit.jupiter.api.Test;
+import org.junit.jupiter.params.ParameterizedTest;
+import org.junit.jupiter.params.provider.CsvSource;
+
+import opennlp.tools.wordnet.WordNetPOS;
+
+import static org.junit.jupiter.api.Assertions.assertArrayEquals;
+import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assertions.assertThrows;
+
+public class MorphyLemmatizerTest {
+
+ private static MorphyLemmatizer morphy() {
+ return new MorphyLemmatizer(WndbReaderTest.fixture(), MorphyExceptionsTest.fixture());
+ }
+
+ private static String one(String token, String tag) {
+ return morphy().lemmatize(new String[] {token}, new String[] {tag})[0];
+ }
+
+ @ParameterizedTest
+ @CsvSource({
+ // Irregular forms resolve through the exception lists.
+ "mice, NN, mouse",
+ "Mice, NNS, mouse",
+ "men, NNS, man",
+ "ran, VBD, run",
+ "running, VBG, run",
+ "went, VBD, go",
+ "gone, VBN, go",
+ "best, RBS, well",
+ // Regular detachments, validated against the lexicon.
+ "dogs, NNS, dog",
+ "boxes, NNS, box",
+ "berries, NNS, berry",
+ "runs, NNS, run",
+ "runs, VBZ, run",
+ "walked, VBD, walk",
+ "walking, VBG, walk",
+ "walks, VBZ, walk",
+ "moved, VBD, move",
+ "taller, JJR, tall",
+ "tallest, JJS, tall",
+ "larger, JJR, large",
+ // A word that is already a lemma comes back as itself.
+ "dog, NN, dog",
+ "quickly, RB, quickly",
+ // WordNet letter tags are accepted alongside Penn tags.
+ "dogs, n, dog",
+ "walked, v, walk",
+ "taller, a, tall",
+ "best, r, well",
+ })
+ void testLemmatizesToken(String token, String tag, String lemma) {
+ assertEquals(lemma, one(token, tag));
+ }
+
+ @ParameterizedTest
+ @CsvSource({
+ // Rule candidates not in the lexicon are rejected, not returned.
+ "dogged, VBD",
+ "boxes, VBZ",
+ "glarbs, NNS",
+ // A known word under the wrong part of speech is unknown.
+ "walk, NN",
+ // Tags outside the mapping yield the unknown marker.
+ "dog, DT",
+ "dog, XYZ",
+ "dogs, ''",
+ // Multi-letter closed-class tags that merely begin with a WordNet letter code are not
+ // adjective lookups: AUX was must be unknown, and AUX taller must not detach to tall.
+ "was, AUX",
+ "taller, AUX",
+ })
+ void testUnknownYieldsMarker(String token, String tag) {
+ assertEquals("O", one(token, tag));
+ }
+
+ @Test
+ void testExceptionHitsAreReturnedWithoutLexiconValidation() {
+ // oxen maps to ox, which the miniature lexicon does not contain; the exception list is
+ // authoritative for irregulars, so the lemma is returned anyway.
+ assertEquals("ox", one("oxen", "NNS"));
+ // better maps to good, also absent from the miniature lexicon.
+ assertEquals("good", one("better", "JJR"));
+ }
+
+ @Test
+ void testArrayFormKeepsPositions() {
+ final String[] lemmas = morphy().lemmatize(
+ new String[] {"The", "mice", "ran", "quickly"},
+ new String[] {"DT", "NNS", "VBD", "RB"});
+ assertArrayEquals(new String[] {"O", "mouse", "run", "quickly"}, lemmas);
+ }
+
+ @Test
+ void testListFormReturnsAllCandidates() {
+ final List> lemmas = morphy().lemmatize(
+ List.of("glarbs", "berries"), List.of("NNS", "NNS"));
+ assertEquals(List.of("O"), lemmas.get(0));
+ assertEquals(List.of("berry"), lemmas.get(1));
+ }
+
+ @Test
+ void testWorksIdenticallyOverTheWnLmfLexicon() {
+ final MorphyLemmatizer lmfMorphy =
+ new MorphyLemmatizer(WnLmfReaderTest.fixture(), MorphyExceptionsTest.fixture());
+ assertArrayEquals(new String[] {"mouse", "box", "walk", "large", "O"},
+ lmfMorphy.lemmatize(
+ new String[] {"mice", "boxes", "walking", "larger", "dogged"},
+ new String[] {"NNS", "NNS", "VBG", "JJR", "VBD"}));
+ }
+
+ @ParameterizedTest
+ @CsvSource(nullValues = "none", value = {
+ "NNP, NOUN",
+ "VBZ, VERB",
+ "JJ, ADJECTIVE",
+ "RBR, ADVERB",
+ "a, ADJECTIVE",
+ "s, ADJECTIVE",
+ "ADJ, ADJECTIVE",
+ "ADV, ADVERB",
+ "r, ADVERB",
+ "DT, none",
+ "'', none",
+ // The letter codes a and s match only as one-letter tags: multi-letter tags beginning
+ // with those letters are closed-class or symbol tags, never adjectives.
+ "AUX, none",
+ "ADP, none",
+ "SCONJ, none",
+ "SYM, none",
+ })
+ void testPosFromTagMapping(String tag, WordNetPOS pos) {
+ assertEquals(pos, MorphyLemmatizer.posFromTag(tag));
+ }
+
+ @Test
+ void testPosFromTagRejectsNull() {
+ assertThrows(IllegalArgumentException.class, () -> MorphyLemmatizer.posFromTag(null));
+ }
+
+ @Test
+ void testConstructorFailsLoudWithoutInputs() {
+ final MorphyExceptions exceptions = MorphyExceptionsTest.fixture();
+ assertThrows(IllegalArgumentException.class,
+ () -> new MorphyLemmatizer(null, exceptions));
+ assertThrows(IllegalArgumentException.class,
+ () -> new MorphyLemmatizer(WndbReaderTest.fixture(), null));
+ }
+
+ @Test
+ void testRejectsNullOrMismatchedSequences() {
+ final MorphyLemmatizer morphy = morphy();
+ assertThrows(IllegalArgumentException.class,
+ () -> morphy.lemmatize((String[]) null, new String[0]));
+ assertThrows(IllegalArgumentException.class,
+ () -> morphy.lemmatize(new String[0], (String[]) null));
+ assertThrows(IllegalArgumentException.class,
+ () -> morphy.lemmatize(new String[] {"a", "b"}, new String[] {"NN"}));
+ assertThrows(IllegalArgumentException.class,
+ () -> morphy.lemmatize(List.of("a"), List.of("NN", "NN")));
+ assertThrows(IllegalArgumentException.class,
+ () -> morphy.lemmatize(new String[] {null}, new String[] {"NN"}));
+ assertThrows(IllegalArgumentException.class,
+ () -> morphy.lemmatize(new String[] {"dog"}, new String[] {null}));
+ }
+}
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/ReaderEquivalenceTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/ReaderEquivalenceTest.java
new file mode 100644
index 0000000000..b9f3523cbc
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/ReaderEquivalenceTest.java
@@ -0,0 +1,121 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package opennlp.wordnet;
+
+import java.util.HashMap;
+import java.util.HashSet;
+import java.util.List;
+import java.util.Map;
+import java.util.Set;
+
+import org.junit.jupiter.api.Test;
+
+import opennlp.tools.wordnet.LexicalKnowledgeBase;
+import opennlp.tools.wordnet.Synset;
+import opennlp.tools.wordnet.WordNetPOS;
+import opennlp.tools.wordnet.WordNetRelation;
+
+import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assertions.assertNotNull;
+
+/**
+ * Asserts that the WN-LMF fixture and the WNDB fixture, which encode the same miniature
+ * wordnet, load into equivalent lexicon views. Synset ids are reader-minted and intentionally
+ * differ, so the comparison is structural, joining synsets on their glosses (unique within the
+ * fixtures) and comparing everything else through that join.
+ */
+public class ReaderEquivalenceTest {
+
+ @Test
+ void testBothReadersProduceEquivalentViews() {
+ final InMemoryWordNetLexicon lmf = (InMemoryWordNetLexicon) WnLmfReaderTest.fixture();
+ final InMemoryWordNetLexicon wndb = (InMemoryWordNetLexicon) WndbReaderTest.fixture();
+ assertEquals(lmf.size(), wndb.size(), "Both fixtures encode the same synsets");
+
+ final Map wndbByGloss = byGloss(wndb);
+ assertEquals(byGloss(lmf).keySet(), wndbByGloss.keySet(), "Same glosses on both sides");
+
+ for (final Synset expected : lmf.synsets()) {
+ final Synset actual = wndbByGloss.get(expected.gloss());
+ assertNotNull(actual, "WNDB view has a synset for gloss: " + expected.gloss());
+ assertEquals(expected.pos(), actual.pos(), "Part of speech for: " + expected.gloss());
+ assertEquals(expected.lemmas(), actual.lemmas(), "Lemmas for: " + expected.gloss());
+ assertEquals(relationsByGloss(expected, lmf), relationsByGloss(actual, wndb),
+ "Relations for: " + expected.gloss());
+ }
+ }
+
+ @Test
+ void testLookupAgreesForEveryLemmaAndPos() {
+ final LexicalKnowledgeBase lmf = WnLmfReaderTest.fixture();
+ final InMemoryWordNetLexicon wndb = (InMemoryWordNetLexicon) WndbReaderTest.fixture();
+ final Set checked = new HashSet<>();
+ for (final Synset synset : wndb.synsets()) {
+ for (final String lemma : synset.lemmas()) {
+ if (!checked.add(lemma + "/" + synset.pos())) {
+ continue;
+ }
+ assertEquals(
+ glosses(lmf.lookup(lemma, synset.pos())),
+ glosses(wndb.lookup(lemma, synset.pos())),
+ "Sense sequence for " + lemma + " as " + synset.pos());
+ }
+ }
+ for (final WordNetPOS pos : WordNetPOS.values()) {
+ assertEquals(lmf.contains("dog", pos), wndb.contains("dog", pos));
+ }
+ }
+
+ @Test
+ void testSenseOrderAgreesForMultiSenseLemma() {
+ final LexicalKnowledgeBase lmf = WnLmfReaderTest.fixture();
+ final LexicalKnowledgeBase wndb = WndbReaderTest.fixture();
+ final List lmfOrder = glosses(lmf.lookup("run", WordNetPOS.NOUN));
+ final List wndbOrder = glosses(wndb.lookup("run", WordNetPOS.NOUN));
+ assertEquals(2, lmfOrder.size());
+ assertEquals(lmfOrder, wndbOrder);
+ }
+
+ private static Map byGloss(InMemoryWordNetLexicon lexicon) {
+ final Map byGloss = new HashMap<>();
+ for (final Synset synset : lexicon.synsets()) {
+ final Synset previous = byGloss.put(synset.gloss(), synset);
+ assertEquals(null, previous, "Fixture glosses must be unique, duplicated: "
+ + synset.gloss());
+ }
+ return byGloss;
+ }
+
+ // A synset's relations with targets replaced by their glosses, id-scheme independent.
+ private static Map> relationsByGloss(Synset synset,
+ LexicalKnowledgeBase lexicon) {
+ final Map> result = new HashMap<>();
+ for (final Map.Entry> relation :
+ synset.relations().entrySet()) {
+ final Set targetGlosses = new HashSet<>();
+ for (final String targetId : relation.getValue()) {
+ targetGlosses.add(lexicon.synset(targetId).orElseThrow().gloss());
+ }
+ result.put(relation.getKey(), targetGlosses);
+ }
+ return result;
+ }
+
+ private static List glosses(List synsets) {
+ return synsets.stream().map(Synset::gloss).toList();
+ }
+}
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/SynsetSimilarityTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/SynsetSimilarityTest.java
new file mode 100644
index 0000000000..48accbc230
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/SynsetSimilarityTest.java
@@ -0,0 +1,145 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package opennlp.wordnet;
+
+import java.util.ArrayList;
+import java.util.HashMap;
+import java.util.List;
+import java.util.Map;
+import java.util.Optional;
+
+import org.junit.jupiter.api.Assertions;
+import org.junit.jupiter.api.Test;
+
+import opennlp.tools.wordnet.LexicalKnowledgeBase;
+import opennlp.tools.wordnet.Synset;
+import opennlp.tools.wordnet.WordNetPOS;
+import opennlp.tools.wordnet.WordNetRelation;
+
+/**
+ * Tests the taxonomy measures against a project-authored miniature taxonomy; no external
+ * lexicon data is involved. {@link HypernymTyperTest} shares the same taxonomy.
+ */
+public class SynsetSimilarityTest {
+
+ /** A tiny in-memory knowledge base over a hand-built noun taxonomy. */
+ static final class FixtureKnowledgeBase implements LexicalKnowledgeBase {
+ private final Map byId = new HashMap<>();
+ private final Map> byLemma = new HashMap<>();
+
+ void add(String id, String lemma, WordNetRelation relation, String... parents) {
+ final Map> relations = parents.length == 0
+ ? Map.of() : Map.of(relation, List.of(parents));
+ final Synset synset =
+ new Synset(id, WordNetPOS.NOUN, List.of(lemma), "fixture", relations);
+ byId.put(id, synset);
+ byLemma.computeIfAbsent(lemma, key -> new ArrayList<>()).add(synset);
+ }
+
+ @Override
+ public List lookup(String lemma, WordNetPOS pos) {
+ return byLemma.getOrDefault(lemma, List.of());
+ }
+
+ @Override
+ public Optional synset(String synsetId) {
+ return Optional.ofNullable(byId.get(synsetId));
+ }
+ }
+
+ /** {@return the taxonomy both this test and {@link HypernymTyperTest} assert against} */
+ static FixtureKnowledgeBase taxonomy() {
+ final FixtureKnowledgeBase kb = new FixtureKnowledgeBase();
+ kb.add("n1", "entity", WordNetRelation.HYPERNYM);
+ kb.add("n2", "physical", WordNetRelation.HYPERNYM, "n1");
+ kb.add("n3", "organism", WordNetRelation.HYPERNYM, "n2");
+ kb.add("n4", "person", WordNetRelation.HYPERNYM, "n3");
+ kb.add("n5", "scientist", WordNetRelation.HYPERNYM, "n4");
+ kb.add("n6", "chemist", WordNetRelation.HYPERNYM, "n5");
+ kb.add("n7", "location", WordNetRelation.HYPERNYM, "n2");
+ kb.add("n8", "city", WordNetRelation.HYPERNYM, "n7");
+ kb.add("n9", "organization", WordNetRelation.HYPERNYM, "n1");
+ kb.add("n10", "company", WordNetRelation.HYPERNYM, "n9");
+ kb.add("n11", "paris", WordNetRelation.INSTANCE_HYPERNYM, "n8");
+ kb.add("n12", "abstract", WordNetRelation.HYPERNYM);
+ return kb;
+ }
+
+ @Test
+ void testPathSimilarity() {
+ final SynsetSimilarity similarity = new SynsetSimilarity(taxonomy());
+ Assertions.assertEquals(1.0, similarity.path("n5", "n5"), 1e-9);
+ Assertions.assertEquals(0.5, similarity.path("n6", "n5"), 1e-9);
+ // chemist up four to physical, city up two: six edges apart
+ Assertions.assertEquals(1.0 / 7.0, similarity.path("n6", "n8"), 1e-9);
+ Assertions.assertEquals(0.0, similarity.path("n6", "n12"), 1e-9);
+ }
+
+ @Test
+ void testWuPalmerRewardsDeepSharedAncestry() {
+ final SynsetSimilarity similarity = new SynsetSimilarity(taxonomy());
+ // scientist and chemist share scientist itself at depth four
+ Assertions.assertEquals(8.0 / 9.0, similarity.wuPalmer("n5", "n6"), 1e-9);
+ final double siblingBranches = similarity.wuPalmer("n6", "n8");
+ Assertions.assertTrue(siblingBranches < similarity.wuPalmer("n5", "n6"));
+ Assertions.assertTrue(siblingBranches > 0.0);
+ Assertions.assertEquals(0.0, similarity.wuPalmer("n6", "n12"), 1e-9);
+ }
+
+ @Test
+ void testLeacockChodorow() {
+ final SynsetSimilarity similarity = new SynsetSimilarity(taxonomy());
+ Assertions.assertEquals(Math.log(10.0),
+ similarity.leacockChodorow("n5", "n6", 10), 1e-9);
+ Assertions.assertEquals(0.0, similarity.leacockChodorow("n6", "n12", 10), 1e-9);
+ Assertions.assertThrows(IllegalArgumentException.class,
+ () -> similarity.leacockChodorow("n5", "n6", 0));
+ }
+
+ @Test
+ void testInstanceHypernymsCountAsEdges() {
+ final SynsetSimilarity similarity = new SynsetSimilarity(taxonomy());
+ Assertions.assertEquals(0.5, similarity.path("n11", "n8"), 1e-9);
+ }
+
+ @Test
+ void testInvalidArguments() {
+ Assertions.assertThrows(IllegalArgumentException.class,
+ () -> new SynsetSimilarity(null));
+ final SynsetSimilarity similarity = new SynsetSimilarity(taxonomy());
+ Assertions.assertThrows(IllegalArgumentException.class,
+ () -> similarity.path(null, "n1"));
+ Assertions.assertThrows(IllegalArgumentException.class,
+ () -> similarity.path("n1", null));
+ Assertions.assertThrows(IllegalArgumentException.class,
+ () -> similarity.wuPalmer(null, "n1"));
+ Assertions.assertThrows(IllegalArgumentException.class,
+ () -> similarity.shortestDistance("n1", null));
+ Assertions.assertThrows(IllegalArgumentException.class,
+ () -> similarity.leacockChodorow(null, "n1", 10));
+ }
+
+ @Test
+ void testUnknownSynsetsAreUnrelatedRatherThanFatal() {
+ final SynsetSimilarity similarity = new SynsetSimilarity(taxonomy());
+ Assertions.assertEquals(-1, similarity.shortestDistance("n5", "missing"));
+ Assertions.assertEquals(0.0, similarity.path("n5", "missing"), 1e-9);
+ Assertions.assertEquals(0.0, similarity.wuPalmer("n5", "missing"), 1e-9);
+ Assertions.assertEquals(0.0, similarity.leacockChodorow("n5", "missing", 10), 1e-9);
+ }
+}
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/WnLmfReaderTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/WnLmfReaderTest.java
new file mode 100644
index 0000000000..5d48d17969
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/WnLmfReaderTest.java
@@ -0,0 +1,405 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package opennlp.wordnet;
+
+import java.io.ByteArrayInputStream;
+import java.io.IOException;
+import java.io.InputStream;
+import java.nio.charset.StandardCharsets;
+import java.nio.file.Files;
+import java.nio.file.Path;
+import java.util.List;
+
+import org.junit.jupiter.api.Test;
+import org.junit.jupiter.api.io.TempDir;
+
+import opennlp.tools.util.InvalidFormatException;
+import opennlp.tools.wordnet.LexicalKnowledgeBase;
+import opennlp.tools.wordnet.Synset;
+import opennlp.tools.wordnet.WordNetPOS;
+import opennlp.tools.wordnet.WordNetRelation;
+
+import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assertions.assertFalse;
+import static org.junit.jupiter.api.Assertions.assertNotNull;
+import static org.junit.jupiter.api.Assertions.assertSame;
+import static org.junit.jupiter.api.Assertions.assertThrows;
+import static org.junit.jupiter.api.Assertions.assertTrue;
+
+public class WnLmfReaderTest {
+
+ static LexicalKnowledgeBase fixture() {
+ try (InputStream in = WnLmfReaderTest.class.getResourceAsStream("mini-wn-lmf.xml")) {
+ assertNotNull(in, "Fixture mini-wn-lmf.xml must be on the test classpath");
+ return WnLmfReader.read(in, "mini-wn-lmf.xml");
+ } catch (IOException e) {
+ throw new IllegalStateException("Unexpected IOException from a classpath stream", e);
+ }
+ }
+
+ private static LexicalKnowledgeBase parse(String document) throws IOException {
+ return WnLmfReader.read(
+ new ByteArrayInputStream(document.getBytes(StandardCharsets.UTF_8)), "inline.xml");
+ }
+
+ private static String wrap(String body) {
+ return "\n\n"
+ + "\n"
+ + body + "\n\n\n";
+ }
+
+ @Test
+ void testLookupReturnsSynsetWithAllComponents() {
+ final List senses = fixture().lookup("dog", WordNetPOS.NOUN);
+ assertEquals(1, senses.size());
+ final Synset dog = senses.get(0);
+ assertEquals("mini-n1", dog.id());
+ assertEquals(WordNetPOS.NOUN, dog.pos());
+ assertEquals(List.of("dog", "domestic dog"), dog.lemmas());
+ assertEquals("a domesticated canid", dog.gloss());
+ assertEquals(List.of("mini-n2"), dog.related(WordNetRelation.HYPERNYM));
+ }
+
+ @Test
+ void testLookupFoldsCaseAndUnderscore() {
+ final LexicalKnowledgeBase lexicon = fixture();
+ assertEquals("mini-n1", lexicon.lookup("Domestic_Dog", WordNetPOS.NOUN).get(0).id());
+ assertEquals("mini-n1", lexicon.lookup("DOG", WordNetPOS.NOUN).get(0).id());
+ }
+
+ @Test
+ void testLookupKeepsSenseOrder() {
+ final List runSenses = fixture().lookup("run", WordNetPOS.NOUN);
+ assertEquals(List.of("mini-n5", "mini-n9"),
+ runSenses.stream().map(Synset::id).toList());
+ }
+
+ @Test
+ void testLookupIsPosScoped() {
+ final LexicalKnowledgeBase lexicon = fixture();
+ assertEquals(1, lexicon.lookup("run", WordNetPOS.VERB).size());
+ assertTrue(lexicon.lookup("dog", WordNetPOS.VERB).isEmpty());
+ assertFalse(lexicon.contains("walk", WordNetPOS.NOUN));
+ assertTrue(lexicon.contains("walk", WordNetPOS.VERB));
+ }
+
+ @Test
+ void testRelationNavigation() {
+ final LexicalKnowledgeBase lexicon = fixture();
+ assertEquals(List.of("mini-n1"), lexicon.related("mini-n2", WordNetRelation.HYPONYM));
+ assertEquals(List.of("mini-v1", "mini-v2"),
+ lexicon.related("mini-v4", WordNetRelation.HYPONYM));
+ assertEquals(List.of("mini-v4"), lexicon.related("mini-v1", WordNetRelation.HYPERNYM));
+ }
+
+ @Test
+ void testRelationTargetSharesCanonicalIdInstance() {
+ final LexicalKnowledgeBase lexicon = fixture();
+ final String target = lexicon.synset("mini-n1").orElseThrow()
+ .related(WordNetRelation.HYPERNYM).get(0);
+ // Not just equal: the identical instance from the synset table, so a loaded lexicon keeps
+ // one copy of each id no matter how many relations point at it.
+ assertSame(lexicon.synset("mini-n2").orElseThrow().id(), target);
+ }
+
+ @Test
+ void testSenseRelationsAreLiftedToSynsetLevel() {
+ final LexicalKnowledgeBase lexicon = fixture();
+ assertEquals(List.of("mini-a2"), lexicon.related("mini-a1", WordNetRelation.ANTONYM));
+ assertEquals(List.of("mini-a1"), lexicon.related("mini-a2", WordNetRelation.ANTONYM));
+ assertEquals(List.of("mini-v1"),
+ lexicon.related("mini-n5", WordNetRelation.DERIVATIONALLY_RELATED));
+ assertEquals(List.of("mini-n5"),
+ lexicon.related("mini-v1", WordNetRelation.DERIVATIONALLY_RELATED));
+ }
+
+ @Test
+ void testSatelliteNormalizesToAdjective() {
+ final List senses = fixture().lookup("large", WordNetPOS.ADJECTIVE);
+ assertEquals(1, senses.size());
+ assertEquals(WordNetPOS.ADJECTIVE, senses.get(0).pos());
+ assertEquals(List.of("mini-a4"), fixture().related("mini-a3", WordNetRelation.SIMILAR_TO));
+ assertEquals(List.of("mini-a3"), fixture().related("mini-a4", WordNetRelation.SIMILAR_TO));
+ }
+
+ @Test
+ void testSimilarOnVerbSynsetMapsToVerbGroup() throws IOException {
+ // Documents derived from Princeton data express verb groups as similar on verb synsets;
+ // the fixture only carries similar on adjectives, so this pins the verb branch directly.
+ final LexicalKnowledgeBase lexicon = parse(wrap(
+ ""
+ + ""
+ + ""
+ + ""
+ + ""
+ + "produce musical tones"
+ + ""
+ + ""
+ + "sing monotonously"));
+ assertEquals(List.of("t-v2"), lexicon.related("t-v1", WordNetRelation.VERB_GROUP));
+ assertTrue(lexicon.related("t-v1", WordNetRelation.SIMILAR_TO).isEmpty());
+ }
+
+ @Test
+ void testUnknownLemmaOrSynsetIsEmpty() {
+ final LexicalKnowledgeBase lexicon = fixture();
+ assertTrue(lexicon.lookup("zebra", WordNetPOS.NOUN).isEmpty());
+ assertTrue(lexicon.synset("mini-n99").isEmpty());
+ }
+
+ @Test
+ void testReadPath(@TempDir Path tempDir) throws IOException {
+ final Path file = tempDir.resolve("tiny.xml");
+ Files.writeString(file, wrap(
+ ""
+ + ""
+ + "a feline"));
+ final LexicalKnowledgeBase lexicon = WnLmfReader.read(file);
+ assertEquals("a feline", lexicon.lookup("cat", WordNetPOS.NOUN).get(0).gloss());
+ }
+
+ @Test
+ void testReadPathRejectsNullAndMissing(@TempDir Path tempDir) {
+ assertThrows(IllegalArgumentException.class, () -> WnLmfReader.read((Path) null));
+ assertThrows(IllegalArgumentException.class,
+ () -> WnLmfReader.read(tempDir.resolve("absent.xml")));
+ }
+
+ @Test
+ void testReadStreamRejectsNulls() {
+ assertThrows(IllegalArgumentException.class, () -> WnLmfReader.read(null, "x"));
+ assertThrows(IllegalArgumentException.class,
+ () -> WnLmfReader.read(new ByteArrayInputStream(new byte[0]), null));
+ }
+
+ @Test
+ void testStreamReadFailurePropagatesAsIOException() {
+ final InputStream failing = new InputStream() {
+ @Override
+ public int read() throws IOException {
+ throw new IOException("Simulated stream failure");
+ }
+ };
+ final IOException e =
+ assertThrows(IOException.class, () -> WnLmfReader.read(failing, "failing.xml"));
+ // The I/O failure must surface as itself, not be misreported as a malformed document.
+ assertFalse(e instanceof InvalidFormatException);
+ }
+
+ @Test
+ void testSkipsDoctypeDeclaration() throws IOException {
+ // Real Open English WordNet releases ship exactly this shape: a DOCTYPE naming the schema
+ // DTD by an unreachable SYSTEM identifier (example.invalid is the RFC 2606 reserved domain
+ // that must never resolve). The reader must parse past it without attempting to fetch it.
+ final String document = "\n"
+ + "\n"
+ + ""
+ + ""
+ + ""
+ + "a feline"
+ + "";
+ final LexicalKnowledgeBase lexicon = parse(document);
+ assertEquals("a feline", lexicon.lookup("cat", WordNetPOS.NOUN).get(0).gloss());
+ }
+
+ @Test
+ void testInternalSubsetEntityIsNeverExpanded(@TempDir Path tempDir) throws IOException {
+ // A DOCTYPE-declared internal-subset entity is the classic XXE payload: if the parser ever
+ // honored it, the entity reference below would be replaced by the target file's content.
+ // With SUPPORT_DTD disabled the declaration itself is never registered, so the reference is
+ // undefined and parsing must fail loud rather than silently expand it.
+ final Path secret = tempDir.resolve("secret.txt");
+ Files.writeString(secret, "xxe-marker-should-never-appear");
+ final String document = "\n"
+ + "]>\n"
+ + ""
+ + ""
+ + ""
+ + "a feline"
+ + "";
+ final InvalidFormatException e =
+ assertThrows(InvalidFormatException.class, () -> parse(document));
+ assertFalse(e.getMessage().contains("xxe-marker-should-never-appear"));
+ }
+
+ @Test
+ void testRejectsTruncatedDocument() {
+ final InvalidFormatException e = assertThrows(InvalidFormatException.class,
+ () -> parse("\n parse(
+ wrap(""
+ + "")));
+ assertTrue(e.getMessage().contains("synset"));
+ }
+
+ @Test
+ void testRejectsSenseToUndeclaredSynset() {
+ final InvalidFormatException e = assertThrows(InvalidFormatException.class, () -> parse(
+ wrap(""
+ + "")));
+ assertTrue(e.getMessage().contains("t-9"));
+ }
+
+ @Test
+ void testRejectsRelationToUndeclaredSynset() {
+ final InvalidFormatException e = assertThrows(InvalidFormatException.class, () -> parse(
+ wrap(""
+ + ""
+ + "a feline"
+ + "")));
+ assertTrue(e.getMessage().contains("t-9"));
+ }
+
+ @Test
+ void testRejectsUnknownRelationType() {
+ final InvalidFormatException e = assertThrows(InvalidFormatException.class, () -> parse(
+ wrap(""
+ + ""
+ + "a feline"
+ + "")));
+ assertTrue(e.getMessage().contains("quasi_synonym"));
+ }
+
+ @Test
+ void testSkipsOtherRelationTypeOnSenseRelation() throws IOException {
+ final LexicalKnowledgeBase lexicon = parse(
+ wrap(""
+ + ""
+ + ""
+ + "a feline"));
+ assertTrue(lexicon.synset("t-1").orElseThrow().relations().isEmpty());
+ }
+
+ @Test
+ void testSkipsOtherRelationTypeOnSynsetRelation() throws IOException {
+ // The DTD permits relType="other" on SynsetRelation too, and several OMW-family wordnets
+ // emit it; it is skipped exactly like the SenseRelation case, not rejected.
+ final LexicalKnowledgeBase lexicon = parse(
+ wrap(""
+ + ""
+ + "a feline"
+ + ""));
+ assertTrue(lexicon.synset("t-1").orElseThrow().relations().isEmpty());
+ }
+
+ @Test
+ void testRejectsUnknownPartOfSpeech() {
+ final InvalidFormatException e = assertThrows(InvalidFormatException.class, () -> parse(
+ wrap(""
+ + ""
+ + "a feline")));
+ assertTrue(e.getMessage().contains("x"));
+ }
+
+ @Test
+ void testRejectsSynsetWithoutMembers() {
+ final InvalidFormatException e = assertThrows(InvalidFormatException.class, () -> parse(
+ wrap("orphan")));
+ assertTrue(e.getMessage().contains("t-1"));
+ }
+
+ @Test
+ void testRejectsDuplicateSynsetId() {
+ final InvalidFormatException e = assertThrows(InvalidFormatException.class, () -> parse(
+ wrap(""
+ + ""
+ + "a feline"
+ + "a repeat")));
+ assertTrue(e.getMessage().contains("Duplicate synset id t-1"));
+ }
+
+ @Test
+ void testRejectsDuplicateLexicalEntryId() {
+ final InvalidFormatException e = assertThrows(InvalidFormatException.class, () -> parse(
+ wrap(""
+ + ""
+ + ""
+ + ""
+ + "a feline")));
+ assertTrue(e.getMessage().contains("Duplicate lexical entry id t-cat-n"));
+ }
+
+ @Test
+ void testRejectsDuplicateSenseId() {
+ final InvalidFormatException e = assertThrows(InvalidFormatException.class, () -> parse(
+ wrap(""
+ + ""
+ + ""
+ + "a feline"
+ + "a second")));
+ assertTrue(e.getMessage().contains("Duplicate sense id t-cat-n-1"));
+ }
+
+ @Test
+ void testRejectsSynsetMemberPosMismatch() {
+ final InvalidFormatException e = assertThrows(InvalidFormatException.class, () -> parse(
+ wrap(""
+ + ""
+ + "a feline")));
+ assertTrue(e.getMessage().contains("t-cat-n"));
+ assertTrue(e.getMessage().contains("VERB"));
+ assertTrue(e.getMessage().contains("NOUN"));
+ }
+
+ @Test
+ void testRejectsSenseRelationToUndeclaredSense() {
+ final InvalidFormatException e = assertThrows(InvalidFormatException.class, () -> parse(
+ wrap(""
+ + ""
+ + ""
+ + "a feline")));
+ assertTrue(e.getMessage().contains("t-ghost-1"));
+ }
+
+ @Test
+ void testRejectsLemmaOutsideLexicalEntry() {
+ final InvalidFormatException e = assertThrows(InvalidFormatException.class,
+ () -> parse(wrap("")));
+ assertTrue(e.getMessage().contains("Lemma outside a LexicalEntry"));
+ }
+
+ @Test
+ void testRejectsSenseBeforeLemma() {
+ final InvalidFormatException e = assertThrows(InvalidFormatException.class, () -> parse(
+ wrap(""
+ + ""
+ + "a feline")));
+ assertTrue(e.getMessage().contains("Sense before its entry's Lemma"));
+ }
+
+ @Test
+ void testRejectsSenseRelationOutsideSense() {
+ final InvalidFormatException e = assertThrows(InvalidFormatException.class, () -> parse(
+ wrap(""
+ + ""
+ + ""
+ + "a feline")));
+ assertTrue(e.getMessage().contains("SenseRelation outside a Sense"));
+ }
+
+ @Test
+ void testRejectsSynsetRelationOutsideSynset() {
+ final InvalidFormatException e = assertThrows(InvalidFormatException.class,
+ () -> parse(wrap("")));
+ assertTrue(e.getMessage().contains("SynsetRelation outside a Synset"));
+ }
+}
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/WndbReaderTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/WndbReaderTest.java
new file mode 100644
index 0000000000..5931957e3b
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/WndbReaderTest.java
@@ -0,0 +1,273 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+package opennlp.wordnet;
+
+import java.io.IOException;
+import java.net.URISyntaxException;
+import java.net.URL;
+import java.nio.charset.StandardCharsets;
+import java.nio.file.Files;
+import java.nio.file.Path;
+import java.util.List;
+import java.util.Locale;
+import java.util.function.UnaryOperator;
+
+import org.junit.jupiter.api.Test;
+import org.junit.jupiter.api.io.TempDir;
+
+import opennlp.tools.util.InvalidFormatException;
+import opennlp.tools.wordnet.LexicalKnowledgeBase;
+import opennlp.tools.wordnet.Synset;
+import opennlp.tools.wordnet.WordNetPOS;
+import opennlp.tools.wordnet.WordNetRelation;
+
+import static org.junit.jupiter.api.Assertions.assertEquals;
+import static org.junit.jupiter.api.Assertions.assertNotNull;
+import static org.junit.jupiter.api.Assertions.assertSame;
+import static org.junit.jupiter.api.Assertions.assertThrows;
+import static org.junit.jupiter.api.Assertions.assertTrue;
+
+public class WndbReaderTest {
+
+ private static final String DOG_ID = "wndb-00001075-n";
+ private static final String CANID_ID = "wndb-00001160-n";
+
+ static Path fixtureDirectory() {
+ final URL url = WndbReaderTest.class.getResource("mini-wndb");
+ assertNotNull(url, "Fixture directory mini-wndb must be on the test classpath");
+ try {
+ return Path.of(url.toURI());
+ } catch (URISyntaxException e) {
+ throw new IllegalStateException("Unexpected fixture URI: " + url, e);
+ }
+ }
+
+ static LexicalKnowledgeBase fixture() {
+ try {
+ return WndbReader.read(fixtureDirectory());
+ } catch (IOException e) {
+ throw new IllegalStateException("Unexpected IOException reading the WNDB fixture", e);
+ }
+ }
+
+ @Test
+ void testLookupReturnsSynsetWithAllComponents() {
+ final List senses = fixture().lookup("dog", WordNetPOS.NOUN);
+ assertEquals(1, senses.size());
+ final Synset dog = senses.get(0);
+ assertEquals(DOG_ID, dog.id());
+ assertEquals(WordNetPOS.NOUN, dog.pos());
+ assertEquals(List.of("dog", "domestic dog"), dog.lemmas());
+ assertEquals("a domesticated canid", dog.gloss());
+ assertEquals(List.of(CANID_ID), dog.related(WordNetRelation.HYPERNYM));
+ }
+
+ @Test
+ void testLookupFoldsCaseAndUnderscore() {
+ final LexicalKnowledgeBase lexicon = fixture();
+ assertEquals(DOG_ID, lexicon.lookup("Domestic_Dog", WordNetPOS.NOUN).get(0).id());
+ assertEquals(DOG_ID, lexicon.lookup("DOG", WordNetPOS.NOUN).get(0).id());
+ }
+
+ @Test
+ void testLookupKeepsIndexSenseOrder() {
+ assertEquals(List.of("wndb-00001427-n", "wndb-00001669-n"),
+ fixture().lookup("run", WordNetPOS.NOUN).stream().map(Synset::id).toList());
+ }
+
+ @Test
+ void testRelationNavigation() {
+ final LexicalKnowledgeBase lexicon = fixture();
+ assertEquals(List.of(DOG_ID), lexicon.related(CANID_ID, WordNetRelation.HYPONYM));
+ assertEquals(List.of("wndb-00001075-v", "wndb-00001171-v"),
+ lexicon.related("wndb-00001324-v", WordNetRelation.HYPONYM));
+ assertEquals(List.of("wndb-00001075-v"),
+ lexicon.related("wndb-00001427-n", WordNetRelation.DERIVATIONALLY_RELATED));
+ }
+
+ @Test
+ void testRelationTargetSharesCanonicalIdInstance() {
+ final LexicalKnowledgeBase lexicon = fixture();
+ final String target = lexicon.synset(DOG_ID).orElseThrow()
+ .related(WordNetRelation.HYPERNYM).get(0);
+ // Not just equal: the identical instance from the synset table, so a loaded lexicon keeps
+ // one copy of each id no matter how many pointers reference it.
+ assertSame(lexicon.synset(CANID_ID).orElseThrow().id(), target);
+ }
+
+ @Test
+ void testLexicalPointersSurfaceAtSynsetLevel() {
+ final LexicalKnowledgeBase lexicon = fixture();
+ assertEquals(List.of("wndb-00001141-a"),
+ lexicon.related("wndb-00001075-a", WordNetRelation.ANTONYM));
+ assertEquals(List.of("wndb-00001075-a"),
+ lexicon.related("wndb-00001141-a", WordNetRelation.ANTONYM));
+ }
+
+ @Test
+ void testSatelliteNormalizesToAdjectiveAndMarkerIsStripped() {
+ final LexicalKnowledgeBase lexicon = fixture();
+ final Synset large = lexicon.lookup("large", WordNetPOS.ADJECTIVE).get(0);
+ assertEquals(WordNetPOS.ADJECTIVE, large.pos());
+ assertEquals(List.of("wndb-00001211-a"), large.related(WordNetRelation.SIMILAR_TO));
+ // short is stored as short(p); the syntactic marker is not part of the lemma.
+ assertEquals(List.of("short"),
+ lexicon.lookup("short", WordNetPOS.ADJECTIVE).get(0).lemmas());
+ }
+
+ @Test
+ void testVerbGroupPointerMapsToVerbGroup(@TempDir Path tempDir) throws IOException {
+ // The fixture has no $ pointer, so the VERB_GROUP mapping is pinned against a minimal
+ // constructed database whose byte offsets are computed, not hard-coded: every offset field
+ // is exactly eight digits, so the second line's position is independent of the digit values.
+ writeEmptyDb(tempDir, "noun", "adj", "adv");
+ final String template =
+ "00000000 29 v 01 sing 0 001 $ XXXXXXXX v 0000 00 | produce musical tones";
+ final String off2 = String.format(Locale.ROOT, "%08d", template.length() + 1);
+ final String line1 = template.replace("XXXXXXXX", off2);
+ final String line2 = off2 + " 29 v 01 chant 0 001 $ 00000000 v 0000 00 | sing monotonously";
+ Files.writeString(tempDir.resolve("data.verb"), line1 + "\n" + line2 + "\n",
+ StandardCharsets.ISO_8859_1);
+ Files.writeString(tempDir.resolve("index.verb"),
+ "chant v 1 1 $ 1 0 " + off2 + "\nsing v 1 1 $ 1 0 00000000\n",
+ StandardCharsets.ISO_8859_1);
+ final LexicalKnowledgeBase lexicon = WndbReader.read(tempDir);
+ assertEquals(List.of("wndb-" + off2 + "-v"),
+ lexicon.related("wndb-00000000-v", WordNetRelation.VERB_GROUP));
+ assertEquals(List.of("wndb-00000000-v"),
+ lexicon.related("wndb-" + off2 + "-v", WordNetRelation.VERB_GROUP));
+ }
+
+ @Test
+ void testUnknownLemmaOrSynsetIsEmpty() {
+ final LexicalKnowledgeBase lexicon = fixture();
+ assertTrue(lexicon.lookup("zebra", WordNetPOS.NOUN).isEmpty());
+ assertTrue(lexicon.synset("wndb-99999999-n").isEmpty());
+ }
+
+ @Test
+ void testRejectsNullAndMissingDirectory(@TempDir Path tempDir) {
+ assertThrows(IllegalArgumentException.class, () -> WndbReader.read(null));
+ assertThrows(IllegalArgumentException.class,
+ () -> WndbReader.read(tempDir.resolve("absent")));
+ }
+
+ @Test
+ void testRejectsMissingDatabaseFile(@TempDir Path tempDir) throws IOException {
+ copyFixture(tempDir);
+ Files.delete(tempDir.resolve("data.verb"));
+ final InvalidFormatException e =
+ assertThrows(InvalidFormatException.class, () -> WndbReader.read(tempDir));
+ assertTrue(e.getMessage().contains("data.verb"));
+ }
+
+ @Test
+ void testRejectsIndexOffsetWithoutDataLine(@TempDir Path tempDir) throws IOException {
+ copyFixture(tempDir);
+ mutate(tempDir, "index.noun", line -> line.startsWith("berry ")
+ ? line.replace("00001564", "00001565") : line);
+ final InvalidFormatException e =
+ assertThrows(InvalidFormatException.class, () -> WndbReader.read(tempDir));
+ assertTrue(e.getMessage().contains("berry"));
+ assertTrue(e.getMessage().contains("00001565"));
+ }
+
+ @Test
+ void testRejectsDataOffsetFieldMismatch(@TempDir Path tempDir) throws IOException {
+ copyFixture(tempDir);
+ mutate(tempDir, "data.noun",
+ line -> line.replace("00001503 03 n 01 box", "00001504 03 n 01 box"));
+ final InvalidFormatException e =
+ assertThrows(InvalidFormatException.class, () -> WndbReader.read(tempDir));
+ assertTrue(e.getMessage().contains("disagrees"));
+ }
+
+ @Test
+ void testRejectsTruncatedDataLine(@TempDir Path tempDir) throws IOException {
+ copyFixture(tempDir);
+ mutate(tempDir, "data.noun", line -> line.startsWith("00001564")
+ ? line.substring(0, line.indexOf(" 000 |")) : line);
+ final InvalidFormatException e =
+ assertThrows(InvalidFormatException.class, () -> WndbReader.read(tempDir));
+ assertTrue(e.getMessage().contains("data.noun"));
+ assertTrue(e.getMessage().contains("Truncated"));
+ }
+
+ @Test
+ void testRejectsUndeclaredPointerSymbol(@TempDir Path tempDir) throws IOException {
+ copyFixture(tempDir);
+ mutate(tempDir, "data.noun", line -> line.replace("001 @ 00001160 n 0000",
+ "001 ? 00001160 n 0000"));
+ final InvalidFormatException e =
+ assertThrows(InvalidFormatException.class, () -> WndbReader.read(tempDir));
+ assertTrue(e.getMessage().contains("Undeclared pointer symbol: ?"));
+ }
+
+ @Test
+ void testRejectsPointerToNonexistentSynset(@TempDir Path tempDir) throws IOException {
+ copyFixture(tempDir);
+ mutate(tempDir, "data.noun", line -> line.replace("001 @ 00001160 n 0000",
+ "001 @ 00009999 n 0000"));
+ final InvalidFormatException e =
+ assertThrows(InvalidFormatException.class, () -> WndbReader.read(tempDir));
+ assertTrue(e.getMessage().contains("wndb-00009999-n"));
+ }
+
+ @Test
+ void testDanglingPointerErrorNamesPointerLine(@TempDir Path tempDir) throws IOException {
+ // A constructed database with no preamble, so the dangling pointer sits on a known line
+ // and the error message can be pinned to name it.
+ writeEmptyDb(tempDir, "noun", "adj", "adv");
+ Files.writeString(tempDir.resolve("data.verb"),
+ "00000000 29 v 01 sing 0 001 $ 00009999 v 0000 00 | produce musical tones\n",
+ StandardCharsets.ISO_8859_1);
+ Files.writeString(tempDir.resolve("index.verb"), "sing v 1 1 $ 1 0 00000000\n",
+ StandardCharsets.ISO_8859_1);
+ final InvalidFormatException e =
+ assertThrows(InvalidFormatException.class, () -> WndbReader.read(tempDir));
+ assertTrue(e.getMessage().contains("wndb-00009999-v"));
+ assertTrue(e.getMessage().contains("line 1"));
+ }
+
+ private static void writeEmptyDb(Path directory, String... suffixes) throws IOException {
+ for (final String suffix : suffixes) {
+ Files.writeString(directory.resolve("data." + suffix), "");
+ Files.writeString(directory.resolve("index." + suffix), "");
+ }
+ }
+
+ private static void copyFixture(Path target) throws IOException {
+ try (var files = Files.list(fixtureDirectory())) {
+ for (final Path file : files.toList()) {
+ Files.copy(file, target.resolve(file.getFileName().toString()));
+ }
+ }
+ }
+
+ // Applies a line transformation to one fixture file. The mutations only ever keep or shrink
+ // line lengths of the affected line's own fields, so surrounding offsets stay valid.
+ private static void mutate(Path directory, String fileName, UnaryOperator edit)
+ throws IOException {
+ final Path file = directory.resolve(fileName);
+ final List lines = Files.readAllLines(file, StandardCharsets.ISO_8859_1);
+ final StringBuilder out = new StringBuilder();
+ for (final String line : lines) {
+ out.append(edit.apply(line)).append('\n');
+ }
+ Files.writeString(file, out.toString(), StandardCharsets.ISO_8859_1);
+ }
+}
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/WordNetUsageExampleTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/WordNetUsageExampleTest.java
new file mode 100644
index 0000000000..e4f7a31916
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/WordNetUsageExampleTest.java
@@ -0,0 +1,78 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one or more
+ * contributor license agreements. See the NOTICE file distributed with
+ * this work for additional information regarding copyright ownership.
+ * The ASF licenses this file to You under the Apache License, Version 2.0
+ * (the "License"); you may not use this file except in compliance with
+ * the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+package opennlp.wordnet;
+
+import java.io.IOException;
+import java.io.InputStream;
+import java.net.URISyntaxException;
+import java.net.URL;
+import java.nio.file.Path;
+import java.util.List;
+
+import org.junit.jupiter.api.Assertions;
+import org.junit.jupiter.api.Test;
+
+import opennlp.tools.wordnet.LexicalKnowledgeBase;
+import opennlp.tools.wordnet.Synset;
+import opennlp.tools.wordnet.WordNetPOS;
+
+/**
+ * Runs the manual's WordNet load and lookup examples (docbkx {@code wordnet.xml})
+ * verbatim: every value the chapter states is asserted here, so a change breaking this
+ * test breaks the manual. The lexicon is the classpath fixture {@code mini-wn-lmf.xml};
+ * exception lists come from the sibling {@code mini-wndb} directory.
+ */
+public class WordNetUsageExampleTest {
+
+ private static LexicalKnowledgeBase loadMiniWnLmf() throws IOException {
+ try (InputStream in = WordNetUsageExampleTest.class.getResourceAsStream("mini-wn-lmf.xml")) {
+ Assertions.assertNotNull(in, "Fixture mini-wn-lmf.xml must be on the test classpath");
+ return WnLmfReader.read(in, "mini-wn-lmf.xml");
+ }
+ }
+
+ private static Path miniWndbDirectory() {
+ final URL url = WordNetUsageExampleTest.class.getResource("mini-wndb");
+ Assertions.assertNotNull(url, "Fixture directory mini-wndb must be on the test classpath");
+ try {
+ return Path.of(url.toURI());
+ } catch (URISyntaxException e) {
+ throw new IllegalStateException("Unexpected fixture URI: " + url, e);
+ }
+ }
+
+ /**
+ * Load, lookup, and Morphy lemmatize as the chapter shows.
+ */
+ @Test
+ void testLoadLookupAndLemmatize() throws IOException {
+ final LexicalKnowledgeBase lexicon = loadMiniWnLmf();
+ final List senses = lexicon.lookup("dog", WordNetPOS.NOUN);
+ Assertions.assertEquals(1, senses.size());
+ Assertions.assertEquals("mini-n1", senses.get(0).id());
+ Assertions.assertEquals(List.of("dog", "domestic dog"), senses.get(0).lemmas());
+ Assertions.assertEquals("a domesticated canid", senses.get(0).gloss());
+
+ final MorphyLemmatizer lemmatizer =
+ new MorphyLemmatizer(lexicon, MorphyExceptions.load(miniWndbDirectory()));
+ Assertions.assertEquals("mouse",
+ lemmatizer.lemmatize(new String[] {"mice"}, new String[] {"NNS"})[0]);
+ Assertions.assertEquals("dog",
+ lemmatizer.lemmatize(new String[] {"dogs"}, new String[] {"NNS"})[0]);
+ }
+}
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wn-lmf.xml b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wn-lmf.xml
new file mode 100644
index 0000000000..10e039214b
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wn-lmf.xml
@@ -0,0 +1,183 @@
+
+
+
+
+
+
+ dog
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+ a domesticated canid
+
+ the dog barked
+
+
+ a carnivorous mammal with nonretractile claws
+
+
+
+ a small rodent with a long tail
+
+
+
+ a gnawing mammal with chisel teeth
+
+
+
+ an act of running at speed
+
+
+ a rigid rectangular container
+
+
+ a small juicy fruit
+
+
+ an adult male person
+
+
+ a score made in baseball
+
+
+ move fast on foot
+
+
+
+ move at a regular pace
+
+
+
+ change location or position
+
+
+ change position in space
+
+
+
+
+ of great height
+
+
+ of small height
+
+
+ of great size
+
+
+
+ above average in size
+
+
+
+ with speed
+
+
+ in a good or proper manner
+
+
+
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/.gitattributes b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/.gitattributes
new file mode 100644
index 0000000000..3868b4d103
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/.gitattributes
@@ -0,0 +1,7 @@
+# WNDB is a byte-offset format: each data line embeds its own byte position in the
+# file, and WndbReader validates that offset against the actual position. Line-ending
+# normalization on checkout (the repo root's `* text=auto`) would insert a CR before
+# every LF on Windows, shifting every offset after the first line and breaking every
+# fixture that has more than one line. Disable it here so checkout is byte-identical
+# on every platform.
+* -text
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/adj.exc b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/adj.exc
new file mode 100644
index 0000000000..404a2e4ddb
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/adj.exc
@@ -0,0 +1 @@
+better good
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/adv.exc b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/adv.exc
new file mode 100644
index 0000000000..c43a2cd529
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/adv.exc
@@ -0,0 +1 @@
+best well
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/data.adj b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/data.adj
new file mode 100644
index 0000000000..1340d16a45
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/data.adj
@@ -0,0 +1,22 @@
+ 1 Licensed to the Apache Software Foundation (ASF) under one or more
+ 2 contributor license agreements. See the NOTICE file distributed with
+ 3 this work for additional information regarding copyright ownership.
+ 4 The ASF licenses this file to You under the Apache License, Version 2.0
+ 5 (the "License"); you may not use this file except in compliance with
+ 6 the License. You may obtain a copy of the License at
+ 7
+ 8 http://www.apache.org/licenses/LICENSE-2.0
+ 9
+ 10 Unless required by applicable law or agreed to in writing, software
+ 11 distributed under the License is distributed on an "AS IS" BASIS,
+ 12 WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ 13 See the License for the specific language governing permissions and
+ 14 limitations under the License.
+ 15
+ 16 Project-authored miniature WNDB fixture mirroring mini-wn-lmf.xml.
+ 17 License preamble lines begin with two spaces, as in released WNDB files,
+ 18 so readers skip them; data line offsets include this preamble.
+00001075 00 a 01 tall 0 001 ! 00001141 a 0101 | of great height
+00001141 00 a 01 short(p) 0 001 ! 00001075 a 0101 | of small height
+00001211 00 a 01 big 0 001 & 00001274 a 0000 | of great size
+00001274 00 s 01 large 0 001 & 00001211 a 0000 | above average in size
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/data.adv b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/data.adv
new file mode 100644
index 0000000000..732c21dfd5
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/data.adv
@@ -0,0 +1,20 @@
+ 1 Licensed to the Apache Software Foundation (ASF) under one or more
+ 2 contributor license agreements. See the NOTICE file distributed with
+ 3 this work for additional information regarding copyright ownership.
+ 4 The ASF licenses this file to You under the Apache License, Version 2.0
+ 5 (the "License"); you may not use this file except in compliance with
+ 6 the License. You may obtain a copy of the License at
+ 7
+ 8 http://www.apache.org/licenses/LICENSE-2.0
+ 9
+ 10 Unless required by applicable law or agreed to in writing, software
+ 11 distributed under the License is distributed on an "AS IS" BASIS,
+ 12 WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ 13 See the License for the specific language governing permissions and
+ 14 limitations under the License.
+ 15
+ 16 Project-authored miniature WNDB fixture mirroring mini-wn-lmf.xml.
+ 17 License preamble lines begin with two spaces, as in released WNDB files,
+ 18 so readers skip them; data line offsets include this preamble.
+00001075 02 r 01 quickly 0 000 | with speed
+00001121 02 r 01 well 0 000 | in a good or proper manner
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/data.noun b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/data.noun
new file mode 100644
index 0000000000..0598111bf1
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/data.noun
@@ -0,0 +1,27 @@
+ 1 Licensed to the Apache Software Foundation (ASF) under one or more
+ 2 contributor license agreements. See the NOTICE file distributed with
+ 3 this work for additional information regarding copyright ownership.
+ 4 The ASF licenses this file to You under the Apache License, Version 2.0
+ 5 (the "License"); you may not use this file except in compliance with
+ 6 the License. You may obtain a copy of the License at
+ 7
+ 8 http://www.apache.org/licenses/LICENSE-2.0
+ 9
+ 10 Unless required by applicable law or agreed to in writing, software
+ 11 distributed under the License is distributed on an "AS IS" BASIS,
+ 12 WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ 13 See the License for the specific language governing permissions and
+ 14 limitations under the License.
+ 15
+ 16 Project-authored miniature WNDB fixture mirroring mini-wn-lmf.xml.
+ 17 License preamble lines begin with two spaces, as in released WNDB files,
+ 18 so readers skip them; data line offsets include this preamble.
+00001075 03 n 02 dog 0 domestic_dog 0 001 @ 00001160 n 0000 | a domesticated canid
+00001160 03 n 01 canid 0 001 ~ 00001075 n 0000 | a carnivorous mammal with nonretractile claws
+00001257 03 n 01 mouse 0 001 @ 00001340 n 0000 | a small rodent with a long tail
+00001340 03 n 01 rodent 0 001 ~ 00001257 n 0000 | a gnawing mammal with chisel teeth
+00001427 03 n 01 run 0 001 + 00001075 v 0101 | an act of running at speed
+00001503 03 n 01 box 0 000 | a rigid rectangular container
+00001564 03 n 01 berry 0 000 | a small juicy fruit
+00001617 03 n 01 man 0 000 | an adult male person
+00001669 03 n 01 run 0 000 | a score made in baseball
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/data.verb b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/data.verb
new file mode 100644
index 0000000000..048546ed71
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/data.verb
@@ -0,0 +1,22 @@
+ 1 Licensed to the Apache Software Foundation (ASF) under one or more
+ 2 contributor license agreements. See the NOTICE file distributed with
+ 3 this work for additional information regarding copyright ownership.
+ 4 The ASF licenses this file to You under the Apache License, Version 2.0
+ 5 (the "License"); you may not use this file except in compliance with
+ 6 the License. You may obtain a copy of the License at
+ 7
+ 8 http://www.apache.org/licenses/LICENSE-2.0
+ 9
+ 10 Unless required by applicable law or agreed to in writing, software
+ 11 distributed under the License is distributed on an "AS IS" BASIS,
+ 12 WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ 13 See the License for the specific language governing permissions and
+ 14 limitations under the License.
+ 15
+ 16 Project-authored miniature WNDB fixture mirroring mini-wn-lmf.xml.
+ 17 License preamble lines begin with two spaces, as in released WNDB files,
+ 18 so readers skip them; data line offsets include this preamble.
+00001075 29 v 01 run 0 002 @ 00001324 v 0000 + 00001427 n 0101 01 + 02 00 | move fast on foot
+00001171 29 v 01 walk 0 001 @ 00001324 v 0000 01 + 02 00 | move at a regular pace
+00001255 29 v 01 go 0 000 01 + 02 00 | change location or position
+00001324 29 v 01 move 0 002 ~ 00001075 v 0000 ~ 00001171 v 0000 01 + 02 00 | change position in space
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/index.adj b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/index.adj
new file mode 100644
index 0000000000..827a988a7d
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/index.adj
@@ -0,0 +1,22 @@
+ 1 Licensed to the Apache Software Foundation (ASF) under one or more
+ 2 contributor license agreements. See the NOTICE file distributed with
+ 3 this work for additional information regarding copyright ownership.
+ 4 The ASF licenses this file to You under the Apache License, Version 2.0
+ 5 (the "License"); you may not use this file except in compliance with
+ 6 the License. You may obtain a copy of the License at
+ 7
+ 8 http://www.apache.org/licenses/LICENSE-2.0
+ 9
+ 10 Unless required by applicable law or agreed to in writing, software
+ 11 distributed under the License is distributed on an "AS IS" BASIS,
+ 12 WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ 13 See the License for the specific language governing permissions and
+ 14 limitations under the License.
+ 15
+ 16 Project-authored miniature WNDB fixture mirroring mini-wn-lmf.xml.
+ 17 License preamble lines begin with two spaces, as in released WNDB files,
+ 18 so readers skip them; data line offsets include this preamble.
+big a 1 1 & 1 0 00001211
+large a 1 1 & 1 0 00001274
+short a 1 1 ! 1 0 00001141
+tall a 1 1 ! 1 0 00001075
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/index.adv b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/index.adv
new file mode 100644
index 0000000000..da20fe1193
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/index.adv
@@ -0,0 +1,20 @@
+ 1 Licensed to the Apache Software Foundation (ASF) under one or more
+ 2 contributor license agreements. See the NOTICE file distributed with
+ 3 this work for additional information regarding copyright ownership.
+ 4 The ASF licenses this file to You under the Apache License, Version 2.0
+ 5 (the "License"); you may not use this file except in compliance with
+ 6 the License. You may obtain a copy of the License at
+ 7
+ 8 http://www.apache.org/licenses/LICENSE-2.0
+ 9
+ 10 Unless required by applicable law or agreed to in writing, software
+ 11 distributed under the License is distributed on an "AS IS" BASIS,
+ 12 WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ 13 See the License for the specific language governing permissions and
+ 14 limitations under the License.
+ 15
+ 16 Project-authored miniature WNDB fixture mirroring mini-wn-lmf.xml.
+ 17 License preamble lines begin with two spaces, as in released WNDB files,
+ 18 so readers skip them; data line offsets include this preamble.
+quickly r 1 0 1 0 00001075
+well r 1 0 1 0 00001121
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/index.noun b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/index.noun
new file mode 100644
index 0000000000..41a8a4317b
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/index.noun
@@ -0,0 +1,27 @@
+ 1 Licensed to the Apache Software Foundation (ASF) under one or more
+ 2 contributor license agreements. See the NOTICE file distributed with
+ 3 this work for additional information regarding copyright ownership.
+ 4 The ASF licenses this file to You under the Apache License, Version 2.0
+ 5 (the "License"); you may not use this file except in compliance with
+ 6 the License. You may obtain a copy of the License at
+ 7
+ 8 http://www.apache.org/licenses/LICENSE-2.0
+ 9
+ 10 Unless required by applicable law or agreed to in writing, software
+ 11 distributed under the License is distributed on an "AS IS" BASIS,
+ 12 WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ 13 See the License for the specific language governing permissions and
+ 14 limitations under the License.
+ 15
+ 16 Project-authored miniature WNDB fixture mirroring mini-wn-lmf.xml.
+ 17 License preamble lines begin with two spaces, as in released WNDB files,
+ 18 so readers skip them; data line offsets include this preamble.
+berry n 1 0 1 0 00001564
+box n 1 0 1 0 00001503
+canid n 1 1 ~ 1 0 00001160
+dog n 1 1 @ 1 0 00001075
+domestic_dog n 1 1 @ 1 0 00001075
+man n 1 0 1 0 00001617
+mouse n 1 1 @ 1 0 00001257
+rodent n 1 1 ~ 1 0 00001340
+run n 2 1 + 2 1 00001427 00001669
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/index.verb b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/index.verb
new file mode 100644
index 0000000000..2b380478de
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/index.verb
@@ -0,0 +1,22 @@
+ 1 Licensed to the Apache Software Foundation (ASF) under one or more
+ 2 contributor license agreements. See the NOTICE file distributed with
+ 3 this work for additional information regarding copyright ownership.
+ 4 The ASF licenses this file to You under the Apache License, Version 2.0
+ 5 (the "License"); you may not use this file except in compliance with
+ 6 the License. You may obtain a copy of the License at
+ 7
+ 8 http://www.apache.org/licenses/LICENSE-2.0
+ 9
+ 10 Unless required by applicable law or agreed to in writing, software
+ 11 distributed under the License is distributed on an "AS IS" BASIS,
+ 12 WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ 13 See the License for the specific language governing permissions and
+ 14 limitations under the License.
+ 15
+ 16 Project-authored miniature WNDB fixture mirroring mini-wn-lmf.xml.
+ 17 License preamble lines begin with two spaces, as in released WNDB files,
+ 18 so readers skip them; data line offsets include this preamble.
+go v 1 0 1 0 00001255
+move v 1 1 ~ 1 0 00001324
+run v 1 2 @ + 1 1 00001075
+walk v 1 1 @ 1 0 00001171
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/noun.exc b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/noun.exc
new file mode 100644
index 0000000000..e5b3080a88
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/noun.exc
@@ -0,0 +1,3 @@
+men man
+mice mouse
+oxen ox
diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/verb.exc b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/verb.exc
new file mode 100644
index 0000000000..486d0c7851
--- /dev/null
+++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/verb.exc
@@ -0,0 +1,4 @@
+gone go
+ran run
+running run
+went go
diff --git a/opennlp-extensions/pom.xml b/opennlp-extensions/pom.xml
index 9afcd3fe3c..8c0e432a58 100644
--- a/opennlp-extensions/pom.xml
+++ b/opennlp-extensions/pom.xml
@@ -41,6 +41,7 @@
opennlp-morfologik
opennlp-spellcheck
opennlp-uima
+ opennlp-wordnet
\ No newline at end of file
diff --git a/opennlp-tools/src/test/java/opennlp/tools/util/StringUtilTest.java b/opennlp-tools/src/test/java/opennlp/tools/util/StringUtilTest.java
index a47306d3d4..73f81c209e 100644
--- a/opennlp-tools/src/test/java/opennlp/tools/util/StringUtilTest.java
+++ b/opennlp-tools/src/test/java/opennlp/tools/util/StringUtilTest.java
@@ -679,4 +679,23 @@ void testLowercaseBeyondBMP() {
String lc = StringUtil.toLowerCase(input);
Assertions.assertArrayEquals(expectedCodePoints, lc.codePoints().toArray());
}
+
+ /**
+ * Verifies the blank check against the toolkit's whitespace definition: the
+ * no-break space is blank here although the JDK's own check does not cover it,
+ * whitespace-only and empty values are blank, and any non-whitespace code point,
+ * supplementary ones included, makes a value non-blank.
+ */
+ @Test
+ void testIsBlankFollowsTheToolkitWhitespaceDefinition() {
+ Assertions.assertTrue(StringUtil.isBlank(""));
+ Assertions.assertTrue(StringUtil.isBlank(" \t\n"));
+ // U+00A0 no-break space and U+2007 figure space: JDK String.isBlank says false
+ Assertions.assertTrue(StringUtil.isBlank("\u00A0"));
+ Assertions.assertTrue(StringUtil.isBlank(" \u00A0\u2007 "));
+ Assertions.assertFalse(StringUtil.isBlank("a"));
+ Assertions.assertFalse(StringUtil.isBlank(" a "));
+ // U+10428, a supplementary-plane letter read as one code point, not two chars
+ Assertions.assertFalse(StringUtil.isBlank("\uD801\uDC28"));
+ }
}
diff --git a/pom.xml b/pom.xml
index 168f44b845..9e30dd9fad 100644
--- a/pom.xml
+++ b/pom.xml
@@ -216,6 +216,12 @@
${project.version}
+
+ opennlp-wordnet
+ ${project.groupId}
+ ${project.version}
+
+
opennlp-uima
${project.groupId}
diff --git a/rat-excludes b/rat-excludes
index 5a5d86b90c..561869bd46 100644
--- a/rat-excludes
+++ b/rat-excludes
@@ -70,3 +70,12 @@ src/main/resources/opennlp/tools/tokenize/uax29/WordBreakProperty.txt
src/main/resources/opennlp/tools/tokenize/uax29/ExtendedPictographic.txt
src/main/resources/opennlp/tools/util/normalizer/confusables.txt
src/test/resources/opennlp/tools/tokenize/uax29/WordBreakTest.txt
+
+
+src/test/resources/opennlp/wordnet/mini-wndb/noun.exc
+src/test/resources/opennlp/wordnet/mini-wndb/verb.exc
+src/test/resources/opennlp/wordnet/mini-wndb/adj.exc
+src/test/resources/opennlp/wordnet/mini-wndb/adv.exc