diff --git a/.gitignore b/.gitignore index 965000f400..611df9b3c1 100644 --- a/.gitignore +++ b/.gitignore @@ -1,3 +1,4 @@ +.claude *.iml .idea target diff --git a/opennlp-api/src/main/java/opennlp/tools/util/StringUtil.java b/opennlp-api/src/main/java/opennlp/tools/util/StringUtil.java index 98cca59891..1842556720 100644 --- a/opennlp-api/src/main/java/opennlp/tools/util/StringUtil.java +++ b/opennlp-api/src/main/java/opennlp/tools/util/StringUtil.java @@ -268,6 +268,27 @@ public static boolean isEmpty(CharSequence theString) { return theString.length() == 0; } + /** + * Determines whether a {@link CharSequence} is blank: empty, or made up entirely of + * code points that {@link #isWhitespace(int)} accepts. Unlike + * {@link String#isBlank()}, this follows the toolkit's whitespace definition, which + * includes the no-break spaces the JDK predicate leaves out, so a value spelled + * entirely from them cannot pass a blank check as content. + * + * @param theString The {@link CharSequence} to examine. Must not be {@code null}. + * @return {@code true} if {@code theString} is empty or all whitespace. + */ + public static boolean isBlank(CharSequence theString) { + for (int i = 0; i < theString.length(); ) { + final int codePoint = Character.codePointAt(theString, i); + if (!isWhitespace(codePoint)) { + return false; + } + i += Character.charCount(codePoint); + } + return true; + } + /** * Get the minimum of three values. * diff --git a/opennlp-api/src/main/java/opennlp/tools/wordnet/LexicalKnowledgeBase.java b/opennlp-api/src/main/java/opennlp/tools/wordnet/LexicalKnowledgeBase.java new file mode 100644 index 0000000000..214ef74db1 --- /dev/null +++ b/opennlp-api/src/main/java/opennlp/tools/wordnet/LexicalKnowledgeBase.java @@ -0,0 +1,89 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package opennlp.tools.wordnet; + +import java.util.List; +import java.util.Optional; + +/** + * Lemma and synset lookup over a loaded lexical-semantic resource in the WordNet family. Synset + * identity is opaque and source-qualified (see {@link Synset#id()}). Lookups return their matches + * in the source's sense order and never return {@code null}. + * + *

Lemma matching semantics are the implementation's concern. The reference implementations + * match case-insensitively (case folding with the root locale) and treat the underscore some + * formats store in multiword lemmas as a space; an implementation with different semantics must + * document them. Returned {@link Synset#lemmas() lemmas} preserve the source's written forms, + * with spaces in multiword lemmas.

+ * + *

Implementations must be immutable and thread-safe after loading: one instance is meant to + * be shared across an application's threads for concurrent lookups.

+ */ +public interface LexicalKnowledgeBase { + + /** + * Finds the synsets containing a lemma with a part of speech, in the source's sense order + * (the most salient sense first when the source ranks senses). + * + * @param lemma The lemma to look up. Must not be {@code null}. + * @param pos The part of speech to look it up as. Must not be {@code null}. + * @return The matching synsets, never {@code null}; empty when the lexicon does not contain + * the lemma with that part of speech. + * @throws IllegalArgumentException Thrown if {@code lemma} or {@code pos} is {@code null}. + */ + List lookup(String lemma, WordNetPOS pos); + + /** + * Finds a synset by its opaque identifier. + * + * @param synsetId The synset identifier, as minted by this lexicon. Must not be {@code null}. + * @return The synset, or empty when this lexicon has no synset with that identifier. + * @throws IllegalArgumentException Thrown if {@code synsetId} is {@code null}. + */ + Optional synset(String synsetId); + + /** + * Navigates one typed relation from a synset. + * + * @param synsetId The source synset identifier. Must not be {@code null}. + * @param relation The relation type to follow. Must not be {@code null}. + * @return The target synset ids in source order, never {@code null}; empty when the synset is + * unknown or has no relation of that type. + * @throws IllegalArgumentException Thrown if {@code synsetId} or {@code relation} is + * {@code null}. + */ + default List related(String synsetId, WordNetRelation relation) { + if (relation == null) { + throw new IllegalArgumentException("Relation must not be null"); + } + return synset(synsetId).map(s -> s.related(relation)).orElse(List.of()); + } + + /** + * Tests whether the lexicon contains a lemma with a part of speech. This is the membership + * check morphological rules validate their candidates against; implementations may override + * it with a cheaper check than {@link #lookup(String, WordNetPOS)}. + * + * @param lemma The lemma to test. Must not be {@code null}. + * @param pos The part of speech to test it as. Must not be {@code null}. + * @return {@code true} if the lexicon contains the lemma with that part of speech. + * @throws IllegalArgumentException Thrown if {@code lemma} or {@code pos} is {@code null}. + */ + default boolean contains(String lemma, WordNetPOS pos) { + return !lookup(lemma, pos).isEmpty(); + } +} diff --git a/opennlp-api/src/main/java/opennlp/tools/wordnet/Synset.java b/opennlp-api/src/main/java/opennlp/tools/wordnet/Synset.java new file mode 100644 index 0000000000..477fa38d90 --- /dev/null +++ b/opennlp-api/src/main/java/opennlp/tools/wordnet/Synset.java @@ -0,0 +1,123 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package opennlp.tools.wordnet; + +import java.util.Collections; +import java.util.EnumMap; +import java.util.List; +import java.util.Map; + +import opennlp.tools.commons.ThreadSafe; + +/** + * One synonym set: a single lexicalized concept with its member lemmas, gloss, and typed + * relations to other synsets. + * + *

The {@link #id() id} is an opaque, source-qualified string minted by the reader that + * produced the synset; consumers must not parse it, only pass it back to + * {@link LexicalKnowledgeBase#synset(String)} and compare it for equality. Relations map each + * {@link WordNetRelation} present on this synset to the target synset ids in source order.

+ * + *

Instances are immutable and thread-safe: the list and map components are defensively + * copied to immutable views at construction.

+ * + * @param id The opaque, source-qualified synset identifier. Must not be {@code null} or + * empty. + * @param pos The part of speech. Must not be {@code null}. + * @param lemmas The member lemmas in source order, human-readable (multiword lemmas use + * spaces, not the underscores some formats store). Must not be {@code null} or + * empty, and must not contain {@code null} or empty elements. + * @param gloss The definition text, possibly empty when the source has none. Must not be + * {@code null}. + * @param relations The typed relations, each mapping to the target synset ids in source order. + * Must not be {@code null}; keys must not be {@code null}; each value must be + * a non-empty list of non-{@code null}, non-empty target ids. + */ +@ThreadSafe +public record Synset( + String id, + WordNetPOS pos, + List lemmas, + String gloss, + Map> relations) { + + /** + * Creates a synset. + * + * @throws IllegalArgumentException Thrown if any component violates its documented constraint. + */ + public Synset { + if (id == null || id.isEmpty()) { + throw new IllegalArgumentException("Id must not be null or empty"); + } + if (pos == null) { + throw new IllegalArgumentException("Pos must not be null"); + } + if (lemmas == null || lemmas.isEmpty()) { + throw new IllegalArgumentException("Lemmas must not be null or empty for synset " + id); + } + for (final String lemma : lemmas) { + if (lemma == null || lemma.isEmpty()) { + throw new IllegalArgumentException( + "Lemmas must not contain a null or empty element for synset " + id); + } + } + if (gloss == null) { + throw new IllegalArgumentException("Gloss must not be null for synset " + id); + } + if (relations == null) { + throw new IllegalArgumentException("Relations must not be null for synset " + id); + } + final Map> copiedRelations = + new EnumMap<>(WordNetRelation.class); + for (final Map.Entry> relation : relations.entrySet()) { + if (relation.getKey() == null) { + throw new IllegalArgumentException("Relations must not contain a null key for synset " + id); + } + final List targets = relation.getValue(); + if (targets == null || targets.isEmpty()) { + throw new IllegalArgumentException("Relation " + relation.getKey() + + " must map to a non-empty target list for synset " + id); + } + for (final String target : targets) { + if (target == null || target.isEmpty()) { + throw new IllegalArgumentException("Relation " + relation.getKey() + + " must not contain a null or empty target id for synset " + id); + } + } + copiedRelations.put(relation.getKey(), List.copyOf(targets)); + } + lemmas = List.copyOf(lemmas); + relations = Collections.unmodifiableMap(copiedRelations); + } + + /** + * Finds the target synset ids of one relation type. + * + * @param relation The relation type. Must not be {@code null}. + * @return The target synset ids in source order, never {@code null}; empty when this synset + * has no relation of that type. + * @throws IllegalArgumentException Thrown if {@code relation} is {@code null}. + */ + public List related(WordNetRelation relation) { + if (relation == null) { + throw new IllegalArgumentException("Relation must not be null"); + } + final List targets = relations.get(relation); + return targets == null ? List.of() : targets; + } +} diff --git a/opennlp-api/src/main/java/opennlp/tools/wordnet/WordNetPOS.java b/opennlp-api/src/main/java/opennlp/tools/wordnet/WordNetPOS.java new file mode 100644 index 0000000000..65f6e691ec --- /dev/null +++ b/opennlp-api/src/main/java/opennlp/tools/wordnet/WordNetPOS.java @@ -0,0 +1,40 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package opennlp.tools.wordnet; + +/** + * The four parts of speech a wordnet-style lexicon distinguishes. + * + *

The enum carries none of the single-letter codes the on-disk formats use; readers own the + * mapping from their format's codes to these values. Adjective satellites normalize to + * {@link #ADJECTIVE}, with the cluster structure preserved through + * {@link WordNetRelation#SIMILAR_TO}.

+ */ +public enum WordNetPOS { + + /** Nouns. */ + NOUN, + + /** Verbs. */ + VERB, + + /** Adjectives, including adjective satellites. */ + ADJECTIVE, + + /** Adverbs. */ + ADVERB +} diff --git a/opennlp-api/src/main/java/opennlp/tools/wordnet/WordNetRelation.java b/opennlp-api/src/main/java/opennlp/tools/wordnet/WordNetRelation.java new file mode 100644 index 0000000000..10da3cea34 --- /dev/null +++ b/opennlp-api/src/main/java/opennlp/tools/wordnet/WordNetRelation.java @@ -0,0 +1,115 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package opennlp.tools.wordnet; + +/** + * The typed relations a wordnet-style lexicon draws between {@link Synset synsets}. Readers map + * their source format's relation names onto these values. + * + *

Relations that a source format draws between individual word senses (antonymy and + * derivation, for example) surface here at the synset level: the synset containing the source + * sense carries the relation to the synset containing the target sense.

+ */ +public enum WordNetRelation { + + /** Opposition in meaning, for example between the adjectives for tall and short. */ + ANTONYM, + + /** The more general concept: a dog is a kind of canid. */ + HYPERNYM, + + /** The class a named instance belongs to: a specific river is an instance of river. */ + INSTANCE_HYPERNYM, + + /** The more specific concept: canid has the hyponym dog. */ + HYPONYM, + + /** A named instance of this class. */ + INSTANCE_HYPONYM, + + /** The group this synset is a member of. */ + MEMBER_HOLONYM, + + /** The whole this synset is a substance of. */ + SUBSTANCE_HOLONYM, + + /** The whole this synset is a part of. */ + PART_HOLONYM, + + /** A member of this group. */ + MEMBER_MERONYM, + + /** A substance this synset is made of. */ + SUBSTANCE_MERONYM, + + /** A part of this synset. */ + PART_MERONYM, + + /** The attribute a value expresses, or a value of this attribute. */ + ATTRIBUTE, + + /** A derivationally related form, typically across parts of speech. */ + DERIVATIONALLY_RELATED, + + /** An action entailed by this verb: snoring entails sleeping. */ + ENTAILMENT, + + /** The verb that entails this one; the inverse of {@link #ENTAILMENT}. */ + ENTAILED_BY, + + /** An effect this verb causes. */ + CAUSE, + + /** The cause of this verb; the inverse of {@link #CAUSE}. */ + CAUSED_BY, + + /** A related synset worth consulting. */ + ALSO_SEE, + + /** A verb sense grouped with this one. */ + VERB_GROUP, + + /** A satellite or head adjective in the same similarity cluster. */ + SIMILAR_TO, + + /** The verb an adjective is the participle of. */ + PARTICIPLE, + + /** + * The noun an adjective pertains to, or the adjective an adverb derives from. The source + * formats use one pointer for both directions of derivation, so this value does too. + */ + PERTAINYM, + + /** The topical domain this synset belongs to. */ + DOMAIN_TOPIC, + + /** A synset belonging to this topical domain. */ + MEMBER_OF_DOMAIN_TOPIC, + + /** The regional domain this synset belongs to. */ + DOMAIN_REGION, + + /** A synset belonging to this regional domain. */ + MEMBER_OF_DOMAIN_REGION, + + /** The usage domain this synset belongs to, for example slang or archaism. */ + DOMAIN_USAGE, + + /** A synset belonging to this usage domain. */ + MEMBER_OF_DOMAIN_USAGE +} diff --git a/opennlp-api/src/test/java/opennlp/tools/wordnet/LexicalKnowledgeBaseTest.java b/opennlp-api/src/test/java/opennlp/tools/wordnet/LexicalKnowledgeBaseTest.java new file mode 100644 index 0000000000..2b698797f1 --- /dev/null +++ b/opennlp-api/src/test/java/opennlp/tools/wordnet/LexicalKnowledgeBaseTest.java @@ -0,0 +1,105 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package opennlp.tools.wordnet; + +import java.util.List; +import java.util.Map; +import java.util.Optional; + +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertFalse; +import static org.junit.jupiter.api.Assertions.assertThrows; +import static org.junit.jupiter.api.Assertions.assertTrue; + +/** + * Exercises the {@link LexicalKnowledgeBase} default methods against a minimal in-memory + * implementation, so the defaults are validated independently of any reader. + */ +public class LexicalKnowledgeBaseTest { + + private static final Synset DOG = new Synset("test-1-n", WordNetPOS.NOUN, List.of("dog"), + "a domesticated canid", Map.of(WordNetRelation.HYPERNYM, List.of("test-2-n"))); + + private static final Synset CANID = new Synset("test-2-n", WordNetPOS.NOUN, List.of("canid"), + "a carnivorous mammal", Map.of(WordNetRelation.HYPONYM, List.of("test-1-n"))); + + // A deliberately tiny implementation of only the two abstract methods. + private static final LexicalKnowledgeBase LEXICON = new LexicalKnowledgeBase() { + + @Override + public List lookup(String lemma, WordNetPOS pos) { + if (lemma == null) { + throw new IllegalArgumentException("Lemma must not be null"); + } + if (pos == null) { + throw new IllegalArgumentException("Pos must not be null"); + } + if (pos == WordNetPOS.NOUN && "dog".equals(lemma)) { + return List.of(DOG); + } + return List.of(); + } + + @Override + public Optional synset(String synsetId) { + if (synsetId == null) { + throw new IllegalArgumentException("SynsetId must not be null"); + } + if (DOG.id().equals(synsetId)) { + return Optional.of(DOG); + } + if (CANID.id().equals(synsetId)) { + return Optional.of(CANID); + } + return Optional.empty(); + } + }; + + @Test + void testRelatedNavigatesThroughSynset() { + assertEquals(List.of("test-2-n"), LEXICON.related("test-1-n", WordNetRelation.HYPERNYM)); + assertEquals(List.of("test-1-n"), LEXICON.related("test-2-n", WordNetRelation.HYPONYM)); + } + + @Test + void testRelatedIsEmptyForAbsentRelationOrUnknownSynset() { + assertTrue(LEXICON.related("test-1-n", WordNetRelation.ANTONYM).isEmpty()); + assertTrue(LEXICON.related("test-99-n", WordNetRelation.HYPERNYM).isEmpty()); + } + + @Test + void testRelatedRejectsNulls() { + assertThrows(IllegalArgumentException.class, + () -> LEXICON.related(null, WordNetRelation.HYPERNYM)); + assertThrows(IllegalArgumentException.class, () -> LEXICON.related("test-1-n", null)); + } + + @Test + void testContainsFollowsLookup() { + assertTrue(LEXICON.contains("dog", WordNetPOS.NOUN)); + assertFalse(LEXICON.contains("dog", WordNetPOS.VERB)); + assertFalse(LEXICON.contains("cat", WordNetPOS.NOUN)); + } + + @Test + void testContainsRejectsNulls() { + assertThrows(IllegalArgumentException.class, () -> LEXICON.contains(null, WordNetPOS.NOUN)); + assertThrows(IllegalArgumentException.class, () -> LEXICON.contains("dog", null)); + } +} diff --git a/opennlp-api/src/test/java/opennlp/tools/wordnet/SynsetTest.java b/opennlp-api/src/test/java/opennlp/tools/wordnet/SynsetTest.java new file mode 100644 index 0000000000..e9cbb4c063 --- /dev/null +++ b/opennlp-api/src/test/java/opennlp/tools/wordnet/SynsetTest.java @@ -0,0 +1,150 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package opennlp.tools.wordnet; + +import java.util.ArrayList; +import java.util.HashMap; +import java.util.List; +import java.util.Map; + +import org.junit.jupiter.api.Test; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertThrows; +import static org.junit.jupiter.api.Assertions.assertTrue; + +public class SynsetTest { + + private static Synset dog() { + return new Synset("test-1-n", WordNetPOS.NOUN, List.of("dog", "domestic dog"), + "a domesticated canid", + Map.of(WordNetRelation.HYPERNYM, List.of("test-2-n"))); + } + + @Test + void testComponents() { + final Synset synset = dog(); + assertEquals("test-1-n", synset.id()); + assertEquals(WordNetPOS.NOUN, synset.pos()); + assertEquals(List.of("dog", "domestic dog"), synset.lemmas()); + assertEquals("a domesticated canid", synset.gloss()); + assertEquals(Map.of(WordNetRelation.HYPERNYM, List.of("test-2-n")), synset.relations()); + } + + @Test + void testRelatedReturnsTargetsInOrder() { + final Synset synset = new Synset("test-1-n", WordNetPOS.NOUN, List.of("dog"), "", + Map.of(WordNetRelation.HYPONYM, List.of("test-3-n", "test-2-n"))); + assertEquals(List.of("test-3-n", "test-2-n"), synset.related(WordNetRelation.HYPONYM)); + } + + @Test + void testRelatedIsEmptyForAbsentRelation() { + assertTrue(dog().related(WordNetRelation.ANTONYM).isEmpty()); + } + + @Test + void testRelatedRejectsNull() { + assertThrows(IllegalArgumentException.class, () -> dog().related(null)); + } + + @Test + void testEmptyGlossAndNoRelationsAreValid() { + final Synset synset = new Synset("test-9-r", WordNetPOS.ADVERB, List.of("well"), "", Map.of()); + assertEquals("", synset.gloss()); + assertTrue(synset.relations().isEmpty()); + } + + @Test + void testDefensiveCopies() { + final List lemmas = new ArrayList<>(List.of("dog")); + final List targets = new ArrayList<>(List.of("test-2-n")); + final Map> relations = new HashMap<>(); + relations.put(WordNetRelation.HYPERNYM, targets); + final Synset synset = new Synset("test-1-n", WordNetPOS.NOUN, lemmas, "gloss", relations); + lemmas.add("mutated"); + targets.add("mutated"); + relations.put(WordNetRelation.ANTONYM, List.of("test-3-n")); + assertEquals(List.of("dog"), synset.lemmas()); + assertEquals(List.of("test-2-n"), synset.related(WordNetRelation.HYPERNYM)); + assertEquals(1, synset.relations().size()); + } + + @Test + void testReturnedCollectionsAreImmutable() { + final Synset synset = dog(); + assertThrows(UnsupportedOperationException.class, () -> synset.lemmas().add("x")); + assertThrows(UnsupportedOperationException.class, + () -> synset.relations().put(WordNetRelation.ANTONYM, List.of("x"))); + assertThrows(UnsupportedOperationException.class, + () -> synset.related(WordNetRelation.HYPERNYM).add("x")); + } + + @Test + void testRejectsNullOrEmptyId() { + assertThrows(IllegalArgumentException.class, + () -> new Synset(null, WordNetPOS.NOUN, List.of("dog"), "", Map.of())); + assertThrows(IllegalArgumentException.class, + () -> new Synset("", WordNetPOS.NOUN, List.of("dog"), "", Map.of())); + } + + @Test + void testRejectsNullPos() { + assertThrows(IllegalArgumentException.class, + () -> new Synset("test-1-n", null, List.of("dog"), "", Map.of())); + } + + @Test + void testRejectsNullOrEmptyLemmas() { + assertThrows(IllegalArgumentException.class, + () -> new Synset("test-1-n", WordNetPOS.NOUN, null, "", Map.of())); + assertThrows(IllegalArgumentException.class, + () -> new Synset("test-1-n", WordNetPOS.NOUN, List.of(), "", Map.of())); + final List withNull = new ArrayList<>(); + withNull.add(null); + assertThrows(IllegalArgumentException.class, + () -> new Synset("test-1-n", WordNetPOS.NOUN, withNull, "", Map.of())); + assertThrows(IllegalArgumentException.class, + () -> new Synset("test-1-n", WordNetPOS.NOUN, List.of(""), "", Map.of())); + } + + @Test + void testRejectsNullGloss() { + assertThrows(IllegalArgumentException.class, + () -> new Synset("test-1-n", WordNetPOS.NOUN, List.of("dog"), null, Map.of())); + } + + @Test + void testRejectsInvalidRelations() { + assertThrows(IllegalArgumentException.class, + () -> new Synset("test-1-n", WordNetPOS.NOUN, List.of("dog"), "", null)); + final Map> nullKey = new HashMap<>(); + nullKey.put(null, List.of("test-2-n")); + assertThrows(IllegalArgumentException.class, + () -> new Synset("test-1-n", WordNetPOS.NOUN, List.of("dog"), "", nullKey)); + final Map> nullTargets = new HashMap<>(); + nullTargets.put(WordNetRelation.HYPERNYM, null); + assertThrows(IllegalArgumentException.class, + () -> new Synset("test-1-n", WordNetPOS.NOUN, List.of("dog"), "", nullTargets)); + assertThrows(IllegalArgumentException.class, + () -> new Synset("test-1-n", WordNetPOS.NOUN, List.of("dog"), "", + Map.of(WordNetRelation.HYPERNYM, List.of()))); + assertThrows(IllegalArgumentException.class, + () -> new Synset("test-1-n", WordNetPOS.NOUN, List.of("dog"), "", + Map.of(WordNetRelation.HYPERNYM, List.of("")))); + } +} diff --git a/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/CFSA2Reader.java b/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/CFSA2Reader.java new file mode 100644 index 0000000000..f429e129a5 --- /dev/null +++ b/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/CFSA2Reader.java @@ -0,0 +1,253 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package opennlp.tools.formats; + +import java.io.IOException; +import java.io.InputStream; +import java.util.Arrays; +import java.util.function.Consumer; + +/** + * Reads a CFSA2 finite-state automaton and enumerates the byte sequences it accepts. + * + *

CFSA2 is the compact automaton format defined by the morfologik project (BSD); the + * {@code .dict} files distributed for many languages are CFSA2 automata paired with a plain + * {@code .info} metadata file. This is a clean-room reader written from the published format, + * with no dependency on that library. It exposes the raw accepted byte sequences; interpreting + * them as morphological entries (surface form, separator, encoded base form, separator, tag) is + * left to the caller, which also owns the character encoding declared by the dictionary.

+ * + *

A constructed instance holds only immutable state, so {@link #forEachSequence(Consumer)} may + * be called concurrently.

+ */ +public final class CFSA2Reader implements FsaSequenceReader { + + private static final byte VERSION = (byte) 0xc6; + + private static final int FLAG_NUMBERS = 0x0100; + + private static final int BIT_TARGET_NEXT = 0x80; + private static final int BIT_LAST_ARC = 0x40; + private static final int BIT_FINAL_ARC = 0x20; + private static final int LABEL_INDEX_MASK = 0x1f; + + private static final int HEADER_SIZE = 8; + private static final int TERMINAL_NODE = 0; + private static final int NO_ARC = 0; + + /** Guards against runaway recursion on a malformed automaton. */ + private static final int MAX_SEQUENCE_LENGTH = 8192; + + private final byte[] arcs; + private final byte[] labelMapping; + private final boolean hasNumbers; + private final int rootNode; + + /** + * Initializes the reader over the automaton's arc block. + * + * @param arcs The arc block, the automaton bytes after the header and label table. + * @param labelMapping The label table indexed by an arc's label index. + * @param hasNumbers Whether each node is prefixed with a perfect-hash number to skip. + */ + private CFSA2Reader(byte[] arcs, byte[] labelMapping, boolean hasNumbers) { + this.arcs = arcs; + this.labelMapping = labelMapping; + this.hasNumbers = hasNumbers; + this.rootNode = destinationNode(firstArc(TERMINAL_NODE)); + } + + /** + * Reads a CFSA2 automaton from a stream. + * + * @param in The automaton bytes, referenced by an open {@link InputStream}. Must not be + * {@code null}. + * @return A reader over the automaton. + * @throws IllegalArgumentException Thrown if {@code in} is {@code null}. + * @throws IOException Thrown on IO errors, or if the stream is not a CFSA2 automaton. + */ + public static CFSA2Reader read(InputStream in) throws IOException { + if (in == null) { + throw new IllegalArgumentException("in must not be null"); + } + return fromBytes(in.readAllBytes()); + } + + /** + * Reads an automaton from a byte block already held in memory. + * + * @param bytes The automaton bytes, header included. + * @return A reader over the automaton. + * @throws IOException Thrown if the block is not a CFSA2 automaton or its header is truncated. + */ + static CFSA2Reader fromBytes(byte[] bytes) throws IOException { + FsaSequenceReader.requireFsaHeader(bytes); + if (bytes.length < HEADER_SIZE) { + throw new IOException("truncated CFSA2 header: fewer than " + HEADER_SIZE + " bytes"); + } + if (bytes[4] != VERSION) { + throw new IOException("unsupported FSA version 0x" + + Integer.toHexString(bytes[4] & 0xff) + "; only CFSA2 (0xc6) is read"); + } + final int flags = ((bytes[5] & 0xff) << 8) | (bytes[6] & 0xff); + final int labelMappingSize = bytes[7] & 0xff; + final int arcsStart = HEADER_SIZE + labelMappingSize; + if (arcsStart > bytes.length) { + throw new IOException("truncated CFSA2 header: label table runs past end of data"); + } + final byte[] labelMapping = Arrays.copyOfRange(bytes, HEADER_SIZE, arcsStart); + final byte[] arcs = Arrays.copyOfRange(bytes, arcsStart, bytes.length); + return new CFSA2Reader(arcs, labelMapping, (flags & FLAG_NUMBERS) != 0); + } + + /** + * {@inheritDoc} + * + *

The stored order of a CFSA2 automaton is lexicographic.

+ * + * @throws IllegalStateException Thrown if a path exceeds {@value #MAX_SEQUENCE_LENGTH} bytes, + * which indicates a malformed automaton. + */ + @Override + public void forEachSequence(Consumer action) { + if (action == null) { + throw new IllegalArgumentException("action must not be null"); + } + enumerate(rootNode, new GrowableByteSequence(), action); + } + + /** + * Walks the automaton depth first, reporting the path of every arc that ends a word. + * + * @param node The node whose arcs are walked. + * @param path The labels of the arcs walked so far; pushed and popped in place. + * @param action The action to run for each accepted sequence. + */ + private void enumerate(int node, GrowableByteSequence path, Consumer action) { + if (path.length() > MAX_SEQUENCE_LENGTH) { + throw new IllegalStateException( + "CFSA2 sequence exceeds " + MAX_SEQUENCE_LENGTH + " bytes; automaton may be malformed"); + } + for (int arc = firstArc(node); arc != NO_ARC; arc = nextArc(arc)) { + path.push(arcLabel(arc)); + if ((arcs[arc] & BIT_FINAL_ARC) != 0) { + action.accept(path.toByteArray()); + } + final int destination = destinationNode(arc); + if (destination != TERMINAL_NODE) { + enumerate(destination, path, action); + } + path.pop(); + } + } + + /** + * Locates the first arc of a node, skipping the node number when the automaton stores one. + * + * @param node The offset of the node. + * @return The offset of the node's first arc. + */ + private int firstArc(int node) { + return hasNumbers ? skipVInt(node) : node; + } + + /** + * Locates the arc following one within the same node. + * + * @param arc The offset of the current arc. + * @return The offset of the next arc, or {@link #NO_ARC} when the current arc is the last. + */ + private int nextArc(int arc) { + return (arcs[arc] & BIT_LAST_ARC) != 0 ? NO_ARC : skipArc(arc); + } + + /** + * Reads the label of an arc, from the label table or from the arc itself. + * + * @param arc The offset of the arc. + * @return The label byte the arc consumes. + */ + private byte arcLabel(int arc) { + final int index = arcs[arc] & LABEL_INDEX_MASK; + return index > 0 ? labelMapping[index] : arcs[arc + 1]; + } + + /** + * Reads the node an arc leads to, following either its goto address or the next-node bit. + * + * @param arc The offset of the arc. + * @return The offset of the destination node, {@link #TERMINAL_NODE} when the arc ends a word + * without continuing. + */ + private int destinationNode(int arc) { + if ((arcs[arc] & BIT_TARGET_NEXT) != 0) { + int last = arc; + while ((arcs[last] & BIT_LAST_ARC) == 0) { + last = skipArc(last); + } + return skipArc(last); + } + return readVInt(arc + ((arcs[arc] & LABEL_INDEX_MASK) == 0 ? 2 : 1)); + } + + /** + * Skips over one arc, whose width depends on its flags. + * + * @param offset The offset of the arc. + * @return The offset just past the arc. + */ + private int skipArc(int offset) { + final int flag = arcs[offset++]; + if ((flag & LABEL_INDEX_MASK) == 0) { + offset++; + } + if ((flag & BIT_TARGET_NEXT) == 0) { + offset = skipVInt(offset); + } + return offset; + } + + /** + * Reads a variable-length integer, least significant group first. + * + * @param offset The offset of the first byte of the integer. + * @return The decoded value. + */ + private int readVInt(int offset) { + byte b = arcs[offset]; + int value = b & 0x7f; + for (int shift = 7; b < 0; shift += 7) { + b = arcs[++offset]; + value |= (b & 0x7f) << shift; + } + return value; + } + + /** + * Skips over a variable-length integer. + * + * @param offset The offset of the first byte of the integer. + * @return The offset just past the integer. + */ + private int skipVInt(int offset) { + while (arcs[offset] < 0) { + offset++; + } + return offset + 1; + } +} diff --git a/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/FSA5Reader.java b/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/FSA5Reader.java new file mode 100644 index 0000000000..7b7f685c4b --- /dev/null +++ b/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/FSA5Reader.java @@ -0,0 +1,209 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package opennlp.tools.formats; + +import java.io.IOException; +import java.io.InputStream; +import java.util.Arrays; +import java.util.function.Consumer; + +/** + * Reads an FSA5 finite-state automaton and enumerates the byte sequences it accepts. + * + *

FSA5 is the older of the two morfologik automaton formats (the newer being + * {@link CFSA2Reader}); its arcs use a fixed-width goto address rather than a variable-length one. + * This is a clean-room reader written from the published format, with no dependency on that + * library. Interpreting the accepted sequences as morphological entries is left to the caller.

+ * + *

A constructed instance holds only immutable state, so {@link #forEachSequence(Consumer)} may + * be called concurrently.

+ */ +public final class FSA5Reader implements FsaSequenceReader { + + private static final byte VERSION = (byte) 0x05; + + private static final int BIT_FINAL_ARC = 0x01; + private static final int BIT_LAST_ARC = 0x02; + private static final int BIT_TARGET_NEXT = 0x04; + + /** Offset of the flags/goto field within an arc; the label occupies the byte before it. */ + private static final int ADDRESS_OFFSET = 1; + private static final int HEADER_SIZE = 8; + private static final int TERMINAL_NODE = 0; + private static final int NO_ARC = 0; + + /** Guards against runaway recursion on a malformed automaton. */ + private static final int MAX_SEQUENCE_LENGTH = 8192; + + private final byte[] arcs; + private final int gotoLength; + private final int nodeDataLength; + private final int rootNode; + + /** + * Initializes the reader over the automaton's arc block. + * + * @param arcs The arc block, the automaton bytes after the header. + * @param gotoLength The width in bytes of an arc's flags and goto address field. + * @param nodeDataLength The width in bytes of the optional data preceding a node's arcs. + */ + private FSA5Reader(byte[] arcs, int gotoLength, int nodeDataLength) { + this.arcs = arcs; + this.gotoLength = gotoLength; + this.nodeDataLength = nodeDataLength; + // FSA5 keeps a dummy node ahead of the epsilon node: skip the dummy's arc to reach the + // epsilon node, then the root is its single arc's destination. + final int epsilonNode = skipArc(firstArc(TERMINAL_NODE)); + this.rootNode = destinationNode(firstArc(epsilonNode)); + } + + /** + * Reads an FSA5 automaton from a stream. + * + * @param in The automaton bytes, referenced by an open {@link InputStream}. Must not be + * {@code null}. + * @return A reader over the automaton. + * @throws IllegalArgumentException Thrown if {@code in} is {@code null}. + * @throws IOException Thrown on IO errors, or if the stream is not an FSA5 automaton. + */ + public static FSA5Reader read(InputStream in) throws IOException { + if (in == null) { + throw new IllegalArgumentException("in must not be null"); + } + return fromBytes(in.readAllBytes()); + } + + /** + * Reads an automaton from a byte block already held in memory. + * + * @param bytes The automaton bytes, header included. + * @return A reader over the automaton. + * @throws IOException Thrown if the block is not an FSA5 automaton or its header is truncated. + */ + static FSA5Reader fromBytes(byte[] bytes) throws IOException { + FsaSequenceReader.requireFsaHeader(bytes); + if (bytes.length < HEADER_SIZE) { + throw new IOException("truncated FSA5 header: fewer than " + HEADER_SIZE + " bytes"); + } + if (bytes[4] != VERSION) { + throw new IOException("unsupported FSA version 0x" + + Integer.toHexString(bytes[4] & 0xff) + "; only FSA5 (0x05) is read here"); + } + // One header byte packs both widths: the goto length low, the node data length high. + final int packedLengths = bytes[7] & 0xff; + final int gotoLength = packedLengths & 0x0f; + final int nodeDataLength = (packedLengths >>> 4) & 0x0f; + if (gotoLength < 1) { + throw new IOException("invalid FSA5 goto length: " + gotoLength); + } + final byte[] arcs = Arrays.copyOfRange(bytes, HEADER_SIZE, bytes.length); + return new FSA5Reader(arcs, gotoLength, nodeDataLength); + } + + /** + * {@inheritDoc} + * + *

The stored order of an FSA5 automaton is lexicographic.

+ * + * @throws IllegalStateException Thrown if a path exceeds {@value #MAX_SEQUENCE_LENGTH} bytes, + * which indicates a malformed automaton. + */ + @Override + public void forEachSequence(Consumer action) { + if (action == null) { + throw new IllegalArgumentException("action must not be null"); + } + enumerate(rootNode, new GrowableByteSequence(), action); + } + + /** + * Walks the automaton depth first, reporting the path of every arc that ends a word. + * + * @param node The node whose arcs are walked. + * @param path The labels of the arcs walked so far; pushed and popped in place. + * @param action The action to run for each accepted sequence. + */ + private void enumerate(int node, GrowableByteSequence path, Consumer action) { + if (path.length() > MAX_SEQUENCE_LENGTH) { + throw new IllegalStateException( + "FSA5 sequence exceeds " + MAX_SEQUENCE_LENGTH + " bytes; automaton may be malformed"); + } + for (int arc = firstArc(node); arc != NO_ARC; arc = nextArc(arc)) { + path.push(arcs[arc]); + if ((arcs[arc + ADDRESS_OFFSET] & BIT_FINAL_ARC) != 0) { + action.accept(path.toByteArray()); + } + final int destination = destinationNode(arc); + if (destination != TERMINAL_NODE) { + enumerate(destination, path, action); + } + path.pop(); + } + } + + /** + * Locates the first arc of a node, skipping the node data when the automaton stores any. + * + * @param node The offset of the node. + * @return The offset of the node's first arc. + */ + private int firstArc(int node) { + return nodeDataLength + node; + } + + /** + * Locates the arc following one within the same node. + * + * @param arc The offset of the current arc. + * @return The offset of the next arc, or {@link #NO_ARC} when the current arc is the last. + */ + private int nextArc(int arc) { + return (arcs[arc + ADDRESS_OFFSET] & BIT_LAST_ARC) != 0 ? NO_ARC : skipArc(arc); + } + + /** + * Reads the node an arc leads to, following either its goto address or the next-node bit. + * + * @param arc The offset of the arc. + * @return The offset of the destination node, {@link #TERMINAL_NODE} when the arc ends a word + * without continuing. + */ + private int destinationNode(int arc) { + if ((arcs[arc + ADDRESS_OFFSET] & BIT_TARGET_NEXT) != 0) { + return skipArc(arc); + } + int value = 0; + for (int i = gotoLength - 1; i >= 0; i--) { + value = (value << 8) | (arcs[arc + ADDRESS_OFFSET + i] & 0xff); + } + // The three flag bits share the low end of the goto field. + return value >>> 3; + } + + /** + * Skips over one arc, whose width depends on whether it carries a goto address. + * + * @param arc The offset of the arc. + * @return The offset just past the arc. + */ + private int skipArc(int arc) { + return (arcs[arc + ADDRESS_OFFSET] & BIT_TARGET_NEXT) != 0 + ? arc + ADDRESS_OFFSET + 1 + : arc + ADDRESS_OFFSET + gotoLength; + } +} diff --git a/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/FsaSequenceReader.java b/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/FsaSequenceReader.java new file mode 100644 index 0000000000..6bea9716d3 --- /dev/null +++ b/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/FsaSequenceReader.java @@ -0,0 +1,94 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package opennlp.tools.formats; + +import java.io.IOException; +import java.io.InputStream; +import java.util.function.Consumer; + +/** + * Enumerates the byte sequences accepted by a finite-state automaton, independent of which + * on-disk format encodes it. The two morfologik automaton formats are read: FSA5 (version + * {@code 0x05}) and CFSA2 (version {@code 0xc6}). + * + *

Thread safety is implementation specific.

+ */ +public interface FsaSequenceReader { + + /** ASCII {@code \fsa}, the shared magic header of both automaton formats. */ + String MAGIC = "\\fsa"; + + /** Version byte of the FSA5 format. */ + int VERSION_FSA5 = 0x05; + + /** Version byte of the CFSA2 format. */ + int VERSION_CFSA2 = 0xc6; + + /** + * Passes every accepted byte sequence to {@code action}, in the automaton's stored order. Each + * sequence is a fresh array owned by the callee. + * + * @param action The action to run for each accepted sequence. Must not be {@code null}. + * @throws IllegalArgumentException Thrown if {@code action} is {@code null}. + */ + void forEachSequence(Consumer action); + + /** + * Checks that a byte block starts with {@link #MAGIC} and carries a version byte. + * + * @param bytes The automaton bytes. Must not be {@code null}. + * @throws IOException Thrown if the block is too short or does not start with {@link #MAGIC}. + */ + static void requireFsaHeader(byte[] bytes) throws IOException { + if (bytes.length <= MAGIC.length()) { + throw new IOException("not an FSA automaton: bad magic header"); + } + for (int i = 0; i < MAGIC.length(); i++) { + if (bytes[i] != (byte) MAGIC.charAt(i)) { + throw new IOException("not an FSA automaton: bad magic header"); + } + } + } + + /** + * Reads an FSA5 or CFSA2 automaton, dispatching on the version byte. + * + * @param in The automaton bytes, referenced by an open {@link InputStream}. Must not be + * {@code null}. + * @return A reader over the automaton. + * @throws IllegalArgumentException Thrown if {@code in} is {@code null}. + * @throws IOException Thrown on IO errors, or if the stream is not a supported FSA automaton. + */ + static FsaSequenceReader read(InputStream in) throws IOException { + if (in == null) { + throw new IllegalArgumentException("in must not be null"); + } + final byte[] bytes = in.readAllBytes(); + requireFsaHeader(bytes); + final int version = bytes[4] & 0xff; + switch (version) { + case VERSION_CFSA2: + return CFSA2Reader.fromBytes(bytes); + case VERSION_FSA5: + return FSA5Reader.fromBytes(bytes); + default: + throw new IOException("unsupported FSA version 0x" + Integer.toHexString(version) + + "; only FSA5 (0x05) and CFSA2 (0xc6) are read"); + } + } +} diff --git a/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/GrowableByteSequence.java b/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/GrowableByteSequence.java new file mode 100644 index 0000000000..46a4a63322 --- /dev/null +++ b/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/GrowableByteSequence.java @@ -0,0 +1,59 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package opennlp.tools.formats; + +import java.util.Arrays; + +/** + * A growable byte buffer used as the path stack while walking a finite-state automaton. One + * instance is reused for a whole traversal: labels are {@link #push(byte)}ed on the way down and + * {@link #pop()}ped on the way back up, so only accepted sequences allocate, via + * {@link #toByteArray()}. Not thread-safe; each traversal creates its own. + */ +final class GrowableByteSequence { + + private byte[] data = new byte[64]; + private int length; + + /** {@return the number of bytes currently on the stack} */ + int length() { + return length; + } + + /** + * Appends one byte, growing the buffer when it is full. + * + * @param value The byte to append. + */ + void push(byte value) { + if (length == data.length) { + data = Arrays.copyOf(data, data.length << 1); + } + data[length++] = value; + } + + /** Drops the last byte. The caller must not pop more bytes than it pushed. */ + void pop() { + length--; + } + + /** {@return a copy of the bytes currently on the stack} */ + byte[] toByteArray() { + return Arrays.copyOf(data, length); + } +} diff --git a/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/MorfologikDictionaryReader.java b/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/MorfologikDictionaryReader.java new file mode 100644 index 0000000000..ba8f2c9329 --- /dev/null +++ b/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/MorfologikDictionaryReader.java @@ -0,0 +1,370 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package opennlp.tools.formats; + +import java.io.ByteArrayInputStream; +import java.io.IOException; +import java.io.InputStream; +import java.io.UncheckedIOException; +import java.nio.charset.Charset; +import java.nio.charset.StandardCharsets; +import java.util.LinkedHashMap; +import java.util.LinkedHashSet; +import java.util.Locale; +import java.util.Map; +import java.util.Properties; + +import opennlp.tools.lemmatizer.DictionaryLemmatizer; + +/** + * Builds a {@link DictionaryLemmatizer} from a morfologik-format morphological dictionary: an + * FSA5 or CFSA2 automaton (read by {@link FsaSequenceReader}) whose accepted byte sequences are + * {@code surfaceForm SEP encodedBase SEP tag}, paired with the {@code .info} metadata that + * declares the separator byte, the character encoding, and the base-form encoder. + * + *

This is a clean-room reader with no dependency on the morfologik library. The base form + * (lemma) is stored relative to the surface form to save space; the four encoders are decoded + * here. Each is a run of control bytes offset by {@code 'A'}, followed by literal bytes to + * append:

+ *
    + *
  • {@code NONE}: the encoded bytes are the base form verbatim.
  • + *
  • {@code SUFFIX} ({@code K}): drop {@code K} bytes from the end of the form, then append.
  • + *
  • {@code PREFIX} ({@code P},{@code K}): drop {@code P} from the front and {@code K} from the + * end of the form, then append.
  • + *
  • {@code INFIX} ({@code I},{@code L},{@code K}): drop {@code L} bytes at offset {@code I} and + * {@code K} from the end of the form, then append.
  • + *
+ * + *

Surface forms are lower-cased on load, because {@link DictionaryLemmatizer} lower-cases the + * queried token before lookup. Dictionary data is supplied by the caller and never bundled.

+ */ +public final class MorfologikDictionaryReader { + + /** The base-form encoder declared by a dictionary's {@code fsa.dict.encoder}. */ + public enum BaseFormEncoding { + + /** The encoded bytes are the base form verbatim. */ + NONE, + + /** The base form drops trailing bytes of the surface form, then appends the encoded rest. */ + SUFFIX, + + /** As {@link #SUFFIX}, and leading bytes of the surface form are dropped too. */ + PREFIX, + + /** As {@link #SUFFIX}, and a run of bytes inside the surface form is dropped too. */ + INFIX + } + + private static final int OFFSET = 'A'; + private static final String KEY_SEPARATOR = "fsa.dict.separator"; + private static final String KEY_ENCODING = "fsa.dict.encoding"; + private static final String KEY_ENCODER = "fsa.dict.encoder"; + + private static final String FIELD_SEPARATOR = "\t"; + private static final String LEMMA_SEPARATOR = "#"; + + /** Not instantiable. */ + private MorfologikDictionaryReader() { + } + + /** + * Reads a morfologik CFSA2 dictionary into a {@link DictionaryLemmatizer} using an explicit + * separator, encoder, and charset. + * + * @param dictionary The CFSA2 automaton, referenced by an open {@link InputStream}. Must not be + * {@code null}. + * @param separator The byte separating the form, encoded base, and tag fields. + * @param encoding The base-form encoder. Must not be {@code null}. + * @param charset The character encoding of the dictionary bytes. Must not be {@code null}. + * @return A {@link DictionaryLemmatizer} over the decoded entries. + * @throws IllegalArgumentException Thrown if {@code dictionary}, {@code encoding}, or + * {@code charset} is {@code null}. + * @throws IOException Thrown on IO errors, if the stream is not a CFSA2 automaton, or if an + * entry cannot be split into a form and encoded base. + */ + public static DictionaryLemmatizer read(InputStream dictionary, byte separator, + BaseFormEncoding encoding, Charset charset) throws IOException { + if (dictionary == null) { + throw new IllegalArgumentException("dictionary must not be null"); + } + if (encoding == null) { + throw new IllegalArgumentException("encoding must not be null"); + } + if (charset == null) { + throw new IllegalArgumentException("charset must not be null"); + } + + final FsaSequenceReader automaton = FsaSequenceReader.read(dictionary); + final Map> entries = new LinkedHashMap<>(); + try { + automaton.forEachSequence( + sequence -> addEntry(sequence, separator, encoding, charset, entries)); + } catch (UncheckedIOException e) { + throw e.getCause(); + } + + final StringBuilder adapted = new StringBuilder(); + for (final Map.Entry> entry : entries.entrySet()) { + adapted.append(entry.getKey()) + .append(FIELD_SEPARATOR) + .append(String.join(LEMMA_SEPARATOR, entry.getValue())) + .append('\n'); + } + final byte[] bytes = adapted.toString().getBytes(StandardCharsets.UTF_8); + return new DictionaryLemmatizer(new ByteArrayInputStream(bytes), StandardCharsets.UTF_8); + } + + /** + * Reads a morfologik CFSA2 dictionary into a {@link DictionaryLemmatizer}, taking the separator, + * charset, and encoder from the dictionary's {@code .info} metadata. + * + * @param dictionary The CFSA2 automaton, referenced by an open {@link InputStream}. Must not be + * {@code null}. + * @param info The {@code .info} metadata properties, referenced by an open + * {@link InputStream}. Must not be {@code null} and must declare + * {@code fsa.dict.separator}, {@code fsa.dict.encoding}, and + * {@code fsa.dict.encoder}. + * @return A {@link DictionaryLemmatizer} over the decoded entries. + * @throws IllegalArgumentException Thrown if an argument is {@code null} or a required metadata + * key is missing or invalid. + * @throws IOException Thrown on IO errors or invalid dictionary content. + */ + public static DictionaryLemmatizer read(InputStream dictionary, InputStream info) + throws IOException { + if (info == null) { + throw new IllegalArgumentException("info must not be null"); + } + final Properties properties = new Properties(); + properties.load(info); + + final String separator = required(properties, KEY_SEPARATOR); + if (separator.length() != 1) { + throw new IllegalArgumentException(KEY_SEPARATOR + " must be a single character"); + } + final Charset charset = Charset.forName(required(properties, KEY_ENCODING)); + final BaseFormEncoding encoding = BaseFormEncoding.valueOf( + required(properties, KEY_ENCODER).toUpperCase(Locale.ROOT)); + return read(dictionary, (byte) separator.charAt(0), encoding, charset); + } + + /** + * Reads a metadata value that the dictionary must declare. + * + * @param properties The parsed {@code .info} metadata. + * @param key The metadata key to read. + * @return The declared value. + * @throws IllegalArgumentException Thrown if the key is not declared. + */ + private static String required(Properties properties, String key) { + final String value = properties.getProperty(key); + if (value == null) { + throw new IllegalArgumentException("missing required metadata key: " + key); + } + return value; + } + + /** + * Splits one accepted sequence into form, base form, and tag, and records it. + * + * @param sequence The accepted byte sequence, the fields joined by {@code separator}. + * @param separator The byte separating the fields. + * @param encoding The encoder the base form is stored with. + * @param charset The character encoding of the dictionary bytes. + * @param entries The entries collected so far, keyed by form and tag; updated in place. + * @throws UncheckedIOException Thrown if the sequence carries no separator or its base form + * cannot be decoded. + */ + private static void addEntry(byte[] sequence, byte separator, BaseFormEncoding encoding, + Charset charset, Map> entries) { + final int firstSeparator = indexOf(sequence, separator, 0); + if (firstSeparator < 0) { + throw new UncheckedIOException(new IOException( + "morfologik entry has no separator: " + new String(sequence, charset))); + } + final int secondSeparator = indexOf(sequence, separator, firstSeparator + 1); + final int baseEnd = secondSeparator < 0 ? sequence.length : secondSeparator; + + final byte[] form = slice(sequence, 0, firstSeparator); + final byte[] encodedBase = slice(sequence, firstSeparator + 1, baseEnd); + final String tag = secondSeparator < 0 ? "" + : new String(sequence, secondSeparator + 1, sequence.length - secondSeparator - 1, charset); + + final byte[] base; + try { + base = decodeBaseForm(form, encodedBase, encoding); + } catch (IllegalArgumentException e) { + throw new UncheckedIOException(new IOException( + "malformed morfologik entry: " + new String(sequence, charset), e)); + } + + final String key = new String(form, charset).toLowerCase() + FIELD_SEPARATOR + tag; + entries.computeIfAbsent(key, k -> new LinkedHashSet<>()).add(new String(base, charset)); + } + + /** + * Recovers a base form from a surface form and its encoded representation. + * + * @param form The surface form bytes. + * @param encoded The encoded base bytes: control bytes followed by literal bytes to append. + * @param encoding The encoder that produced {@code encoded}. + * @return The decoded base form bytes. + * @throws IllegalArgumentException Thrown if {@code encoded} is too short for the encoder or the + * control bytes address positions outside {@code form}. + */ + static byte[] decodeBaseForm(byte[] form, byte[] encoded, BaseFormEncoding encoding) { + switch (encoding) { + case NONE: + return encoded.clone(); + case SUFFIX: { + require(encoded, 1, encoding); + final int keep = bounded(form.length - control(encoded[0]), form); + return join(form, 0, keep, encoded, 1); + } + case PREFIX: { + require(encoded, 2, encoding); + final int start = bounded(control(encoded[0]), form); + final int end = bounded(form.length - control(encoded[1]), form); + return join(form, Math.min(start, end), end, encoded, 2); + } + case INFIX: { + require(encoded, 3, encoding); + final int cut = bounded(control(encoded[0]), form); + final int resume = bounded(cut + control(encoded[1]), form); + final int end = bounded(form.length - control(encoded[2]), form); + return join3(form, cut, Math.max(resume, cut), Math.max(end, resume), encoded, 3); + } + default: + throw new IllegalArgumentException("unknown encoder: " + encoding); + } + } + + /** + * Reads one control byte as the offset it encodes. + * + * @param b The control byte. + * @return The encoded offset, the byte value less {@code 'A'}. + */ + private static int control(byte b) { + return (b & 0xff) - OFFSET; + } + + /** + * Checks that an encoded base form carries all the control bytes its encoder needs. + * + * @param encoded The encoded base bytes. + * @param prefixBytes The number of control bytes the encoder needs. + * @param encoding The encoder, named in the failure message. + * @throws IllegalArgumentException Thrown if fewer bytes are present. + */ + private static void require(byte[] encoded, int prefixBytes, BaseFormEncoding encoding) { + if (encoded.length < prefixBytes) { + throw new IllegalArgumentException( + encoding + " encoded base needs at least " + prefixBytes + " control byte(s)"); + } + } + + /** + * Checks that a decoded offset addresses a position within a surface form. + * + * @param index The offset to check. + * @param form The surface form bytes. + * @return The offset itself. + * @throws IllegalArgumentException Thrown if the offset lies outside {@code form}. + */ + private static int bounded(int index, byte[] form) { + if (index < 0 || index > form.length) { + throw new IllegalArgumentException( + "encoded base addresses byte " + index + " outside a form of length " + form.length); + } + return index; + } + + /** + * Joins one run of the surface form with the literal tail of an encoded base form. + * + * @param form The surface form bytes. + * @param from The first byte of the run to keep. + * @param to The byte after the run to keep. + * @param encoded The encoded base bytes. + * @param appendFrom The first literal byte of {@code encoded}, past its control bytes. + * @return The decoded base form bytes. + */ + private static byte[] join(byte[] form, int from, int to, byte[] encoded, int appendFrom) { + final int kept = to - from; + final int appended = encoded.length - appendFrom; + final byte[] out = new byte[kept + appended]; + System.arraycopy(form, from, out, 0, kept); + System.arraycopy(encoded, appendFrom, out, kept, appended); + return out; + } + + /** + * Joins two runs of the surface form, the second past a dropped infix, with the literal tail of + * an encoded base form. + * + * @param form The surface form bytes. + * @param headEnd The byte after the leading run to keep, which starts at zero. + * @param tailFrom The first byte of the trailing run to keep. + * @param tailEnd The byte after the trailing run to keep. + * @param encoded The encoded base bytes. + * @param appendFrom The first literal byte of {@code encoded}, past its control bytes. + * @return The decoded base form bytes. + */ + private static byte[] join3(byte[] form, int headEnd, int tailFrom, int tailEnd, + byte[] encoded, int appendFrom) { + final int tail = tailEnd - tailFrom; + final int appended = encoded.length - appendFrom; + final byte[] out = new byte[headEnd + tail + appended]; + System.arraycopy(form, 0, out, 0, headEnd); + System.arraycopy(form, tailFrom, out, headEnd, tail); + System.arraycopy(encoded, appendFrom, out, headEnd + tail, appended); + return out; + } + + /** + * Finds the next occurrence of a byte. + * + * @param array The bytes to search. + * @param value The byte to find. + * @param from The index to start at. + * @return The index of the first occurrence at or after {@code from}, or {@code -1} when absent. + */ + private static int indexOf(byte[] array, byte value, int from) { + for (int i = from; i < array.length; i++) { + if (array[i] == value) { + return i; + } + } + return -1; + } + + /** + * Copies a run of bytes. + * + * @param array The bytes to copy from. + * @param from The first byte to copy. + * @param to The byte after the last one to copy. + * @return The copied run. + */ + private static byte[] slice(byte[] array, int from, int to) { + final byte[] out = new byte[to - from]; + System.arraycopy(array, from, out, 0, to - from); + return out; + } +} diff --git a/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/PoliMorfDictionaryReader.java b/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/PoliMorfDictionaryReader.java new file mode 100644 index 0000000000..e155bb3408 --- /dev/null +++ b/opennlp-core/opennlp-formats/src/main/java/opennlp/tools/formats/PoliMorfDictionaryReader.java @@ -0,0 +1,129 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package opennlp.tools.formats; + +import java.io.BufferedReader; +import java.io.ByteArrayInputStream; +import java.io.IOException; +import java.io.InputStream; +import java.io.InputStreamReader; +import java.nio.charset.Charset; +import java.nio.charset.StandardCharsets; +import java.util.LinkedHashMap; +import java.util.LinkedHashSet; +import java.util.Map; + +import opennlp.tools.lemmatizer.DictionaryLemmatizer; +import opennlp.tools.util.StringUtil; + +/** + * Builds a {@link DictionaryLemmatizer} from a morphological dictionary laid out as one + * tab-separated {@code surfaceForm\tlemma\ttag} row per entry. + * + *

This is the layout published by PoliMorf, the successor grammatical dictionary of Polish + * (BSD 2-Clause), and emitted by exporting a morfologik dictionary to text; it is otherwise + * language-agnostic. {@link DictionaryLemmatizer} expects the opposite column order, + * {@code word\tpostag\tlemma}, and a single row per {@code (word, postag)} key with alternative + * lemmas joined by {@code #}. This reader performs that adaptation: it re-orders the columns and + * merges every lemma seen for the same form and tag into one entry, preserving first-seen order. + * The bundled dictionary data itself is never shipped; callers supply it.

+ * + *

Surface forms are lower-cased on load because {@link DictionaryLemmatizer} lower-cases the + * queried token before lookup, so an entry keyed on a mixed-case form would otherwise be + * unreachable. Tags are kept verbatim and must match the tags the caller's tagger emits.

+ */ +public final class PoliMorfDictionaryReader { + + private static final String FIELD_SEPARATOR = "\t"; + private static final String LEMMA_SEPARATOR = "#"; + private static final int MIN_FIELDS = 3; + + /** Not instantiable. */ + private PoliMorfDictionaryReader() { + } + + /** + * Reads a UTF-8 {@code surfaceForm\tlemma\ttag} dictionary into a {@link DictionaryLemmatizer}. + * + * @param dictionary The dictionary referenced by an open {@link InputStream}. Must not be + * {@code null}. + * @return A {@link DictionaryLemmatizer} over the adapted entries. + * @throws IllegalArgumentException Thrown if {@code dictionary} is {@code null}. + * @throws IOException Thrown if IO errors occur while reading, or a non-blank line carries + * fewer than three tab-separated fields. + */ + public static DictionaryLemmatizer read(InputStream dictionary) throws IOException { + return read(dictionary, StandardCharsets.UTF_8); + } + + /** + * Reads a {@code surfaceForm\tlemma\ttag} dictionary into a {@link DictionaryLemmatizer}. + * + * @param dictionary The dictionary referenced by an open {@link InputStream}. Must not be + * {@code null}. + * @param charset The character encoding of the dictionary. Must not be {@code null}. + * @return A {@link DictionaryLemmatizer} over the adapted entries. + * @throws IllegalArgumentException Thrown if {@code dictionary} or {@code charset} is + * {@code null}. + * @throws IOException Thrown if IO errors occur while reading, or a non-blank line carries + * fewer than three tab-separated fields. + */ + public static DictionaryLemmatizer read(InputStream dictionary, Charset charset) + throws IOException { + if (dictionary == null) { + throw new IllegalArgumentException("dictionary must not be null"); + } + if (charset == null) { + throw new IllegalArgumentException("charset must not be null"); + } + + // key "form\ttag" -> alternative lemmas, both maps ordered so the output is deterministic. + final Map> entries = new LinkedHashMap<>(); + try (BufferedReader reader = new BufferedReader(new InputStreamReader(dictionary, charset))) { + String line; + int lineNumber = 0; + while ((line = reader.readLine()) != null) { + lineNumber++; + if (StringUtil.isBlank(line)) { + continue; + } + final String[] fields = line.split(FIELD_SEPARATOR, -1); + if (fields.length < MIN_FIELDS) { + throw new IOException("PoliMorf line " + lineNumber + + " has fewer than " + MIN_FIELDS + " tab-separated fields: " + line); + } + final String form = fields[0].toLowerCase(); + final String lemma = fields[1]; + final String tag = fields[2]; + entries.computeIfAbsent(form + FIELD_SEPARATOR + tag, key -> new LinkedHashSet<>()) + .add(lemma); + } + } + + final StringBuilder adapted = new StringBuilder(); + for (final Map.Entry> entry : entries.entrySet()) { + adapted.append(entry.getKey()) + .append(FIELD_SEPARATOR) + .append(String.join(LEMMA_SEPARATOR, entry.getValue())) + .append('\n'); + } + + final byte[] bytes = adapted.toString().getBytes(StandardCharsets.UTF_8); + return new DictionaryLemmatizer(new ByteArrayInputStream(bytes), StandardCharsets.UTF_8); + } +} diff --git a/opennlp-core/opennlp-formats/src/test/java/opennlp/tools/formats/CFSA2ReaderTest.java b/opennlp-core/opennlp-formats/src/test/java/opennlp/tools/formats/CFSA2ReaderTest.java new file mode 100644 index 0000000000..558969315f --- /dev/null +++ b/opennlp-core/opennlp-formats/src/test/java/opennlp/tools/formats/CFSA2ReaderTest.java @@ -0,0 +1,75 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package opennlp.tools.formats; + +import java.io.ByteArrayInputStream; +import java.io.IOException; +import java.nio.charset.StandardCharsets; +import java.util.ArrayList; +import java.util.Base64; +import java.util.List; + +import org.junit.jupiter.api.Assertions; +import org.junit.jupiter.api.Test; + +/** + * Validates the clean-room {@link CFSA2Reader} against a ground-truth automaton. The fixture below + * was generated once by morfologik's {@code CFSA2Serializer} from the words + * {@code {cat, cats, do, dog, dogs}} and committed as the expected encoding; the reader must + * recover exactly those sequences. + */ +public class CFSA2ReaderTest { + + private static final String FIXTURE_BASE64 = "XGZzYcYABwgAdHNvZ2RjYUBeAwYKxePkYgDHYQg="; + + private static List sequences(byte[] cfsa2) throws IOException { + final CFSA2Reader reader = CFSA2Reader.read(new ByteArrayInputStream(cfsa2)); + final List out = new ArrayList<>(); + reader.forEachSequence(bytes -> out.add(new String(bytes, StandardCharsets.UTF_8))); + return out; + } + + /** Every accepted sequence is recovered, in the automaton's stored lexicographic order. */ + @Test + void testEnumeratesAllAcceptedSequences() throws IOException { + Assertions.assertEquals(List.of("cat", "cats", "do", "dog", "dogs"), + sequences(Base64.getDecoder().decode(FIXTURE_BASE64))); + } + + /** A stream that is not an FSA automaton fails loudly. */ + @Test + void testRejectsNonFsaMagic() { + Assertions.assertThrows(IOException.class, () -> CFSA2Reader.read( + new ByteArrayInputStream("not an fsa header".getBytes(StandardCharsets.UTF_8)))); + } + + /** An FSA of a different version than CFSA2 fails loudly rather than misreading. */ + @Test + void testRejectsUnsupportedVersion() { + final byte[] altered = Base64.getDecoder().decode(FIXTURE_BASE64); + altered[4] = 0x05; + Assertions.assertThrows(IOException.class, + () -> CFSA2Reader.read(new ByteArrayInputStream(altered))); + } + + /** A null stream is rejected at the boundary. */ + @Test + void testNullStreamRejected() { + Assertions.assertThrows(IllegalArgumentException.class, () -> CFSA2Reader.read(null)); + } +} diff --git a/opennlp-core/opennlp-formats/src/test/java/opennlp/tools/formats/FSA5ReaderTest.java b/opennlp-core/opennlp-formats/src/test/java/opennlp/tools/formats/FSA5ReaderTest.java new file mode 100644 index 0000000000..39423cdb7b --- /dev/null +++ b/opennlp-core/opennlp-formats/src/test/java/opennlp/tools/formats/FSA5ReaderTest.java @@ -0,0 +1,74 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package opennlp.tools.formats; + +import java.io.ByteArrayInputStream; +import java.io.IOException; +import java.nio.charset.StandardCharsets; +import java.util.ArrayList; +import java.util.Base64; +import java.util.List; + +import org.junit.jupiter.api.Assertions; +import org.junit.jupiter.api.Test; + +/** + * Validates the clean-room {@link FSA5Reader} against a ground-truth automaton generated once by + * morfologik's {@code FSA5Serializer} from the words {@code {cat, cats, do, dog, dogs}}. + */ +public class FSA5ReaderTest { + + private static final String FIXTURE_BASE64 = "XGZzYQVfKwEAAF4GY3BkBm8HZwdzA2EGdGM="; + + private static List sequences(FsaSequenceReader reader) { + final List out = new ArrayList<>(); + reader.forEachSequence(bytes -> out.add(new String(bytes, StandardCharsets.UTF_8))); + return out; + } + + /** Every accepted sequence is recovered, in stored lexicographic order. */ + @Test + void testEnumeratesAllAcceptedSequences() throws IOException { + final FSA5Reader reader = FSA5Reader.read( + new ByteArrayInputStream(Base64.getDecoder().decode(FIXTURE_BASE64))); + Assertions.assertEquals(List.of("cat", "cats", "do", "dog", "dogs"), sequences(reader)); + } + + /** The format-agnostic dispatcher recognizes and reads FSA5 by its version byte. */ + @Test + void testDispatcherReadsFsa5() throws IOException { + final FsaSequenceReader reader = FsaSequenceReader.read( + new ByteArrayInputStream(Base64.getDecoder().decode(FIXTURE_BASE64))); + Assertions.assertEquals(List.of("cat", "cats", "do", "dog", "dogs"), sequences(reader)); + } + + /** A different FSA version fails loudly rather than misreading. */ + @Test + void testRejectsUnsupportedVersion() { + final byte[] altered = Base64.getDecoder().decode(FIXTURE_BASE64); + altered[4] = (byte) 0x99; + Assertions.assertThrows(IOException.class, + () -> FSA5Reader.read(new ByteArrayInputStream(altered))); + } + + /** A null stream is rejected at the boundary. */ + @Test + void testNullStreamRejected() { + Assertions.assertThrows(IllegalArgumentException.class, () -> FSA5Reader.read(null)); + } +} diff --git a/opennlp-core/opennlp-formats/src/test/java/opennlp/tools/formats/MorfologikDictionaryReaderTest.java b/opennlp-core/opennlp-formats/src/test/java/opennlp/tools/formats/MorfologikDictionaryReaderTest.java new file mode 100644 index 0000000000..40eb404f3c --- /dev/null +++ b/opennlp-core/opennlp-formats/src/test/java/opennlp/tools/formats/MorfologikDictionaryReaderTest.java @@ -0,0 +1,144 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package opennlp.tools.formats; + +import java.io.ByteArrayInputStream; +import java.io.IOException; +import java.io.InputStream; +import java.nio.charset.StandardCharsets; +import java.util.Base64; + +import org.junit.jupiter.api.Assertions; +import org.junit.jupiter.api.Test; +import org.junit.jupiter.params.ParameterizedTest; +import org.junit.jupiter.params.provider.CsvSource; + +import opennlp.tools.formats.MorfologikDictionaryReader.BaseFormEncoding; +import opennlp.tools.lemmatizer.DictionaryLemmatizer; + +/** + * Validates the clean-room {@link MorfologikDictionaryReader}. The base-form decode cases and the + * end-to-end CFSA2 dictionary below were generated by morfologik's own encoder and serializer, so + * the reader is checked against that library's output rather than against restated assumptions. + */ +public class MorfologikDictionaryReaderTest { + + // A CFSA2 dictionary of cat+A+NN, cats+B+NNS, dogs+B+NNS, mice+Douse+NN (SUFFIX encoder, '+' + // separator, ISO-8859-1), built by morfologik's CFSA2Serializer. + private static final String DICTIONARY_BASE64 = + "XGZzYcYABxIAdXRtaWdkYVNEQkFvZWNzTitAXgMOHwYVw8TOzdHJzMHPzdHQcADMxc/RytHQ0GgAx8KRTxhLEQ=="; + + // The same dictionary serialized in the older FSA5 format, to prove both formats are read. + private static final String FSA5_DICTIONARY_BASE64 = + "XGZzYQVfKwIAAABeBmPIAWQwAW0GaQZjBmUGKwZEBm8GdQZzBmUGKwZOBk4DAG8G" + + "ZwZzBisGQgYrBk4GTgZTAwBhBnQGKxgCc2IBQfoA"; + + private static byte[] iso(String s) { + return s.getBytes(StandardCharsets.ISO_8859_1); + } + + /** Each encoder recovers the base form morfologik encoded. */ + @ParameterizedTest + @CsvSource({ + "SUFFIX, cats, B, cat", + "SUFFIX, mice, Douse, mouse", + "SUFFIX, foobar, Gbar, bar", + "SUFFIX, cat, A, cat", + "PREFIX, cats, AB, cat", + "PREFIX, foobar, DA, bar", + "PREFIX, unhappy, CA, happy", + "PREFIX, mice, ADouse, mouse", + "INFIX, cats, AAB, cat", + "INFIX, foobar, ADA, bar", + "INFIX, unhappy, ACA, happy", + "INFIX, mice, AADouse, mouse", + "NONE, cats, cat, cat", + }) + void testDecodeBaseForm(String encoder, String form, String encoded, String base) { + final byte[] decoded = MorfologikDictionaryReader.decodeBaseForm( + iso(form), iso(encoded), BaseFormEncoding.valueOf(encoder)); + Assertions.assertEquals(base, new String(decoded, StandardCharsets.ISO_8859_1)); + } + + private static DictionaryLemmatizer dictionary() throws IOException { + return MorfologikDictionaryReader.read( + new ByteArrayInputStream(Base64.getDecoder().decode(DICTIONARY_BASE64)), + (byte) '+', BaseFormEncoding.SUFFIX, StandardCharsets.ISO_8859_1); + } + + /** The CFSA2 dictionary lemmatizes each surface form under its tag; misses yield "O". */ + @Test + void testReadsCfsa2DictionaryEndToEnd() throws IOException { + final DictionaryLemmatizer lemmatizer = dictionary(); + + Assertions.assertArrayEquals(new String[] {"cat"}, + lemmatizer.lemmatize(new String[] {"cats"}, new String[] {"NNS"})); + Assertions.assertArrayEquals(new String[] {"dog"}, + lemmatizer.lemmatize(new String[] {"dogs"}, new String[] {"NNS"})); + Assertions.assertArrayEquals(new String[] {"mouse"}, + lemmatizer.lemmatize(new String[] {"mice"}, new String[] {"NN"})); + Assertions.assertArrayEquals(new String[] {"cat"}, + lemmatizer.lemmatize(new String[] {"cat"}, new String[] {"NN"})); + Assertions.assertArrayEquals(new String[] {"O"}, + lemmatizer.lemmatize(new String[] {"mice"}, new String[] {"NNS"})); + } + + /** The same dictionary in the older FSA5 format is read identically. */ + @Test + void testReadsFsa5DictionaryEndToEnd() throws IOException { + final DictionaryLemmatizer lemmatizer = MorfologikDictionaryReader.read( + new ByteArrayInputStream(Base64.getDecoder().decode(FSA5_DICTIONARY_BASE64)), + (byte) '+', BaseFormEncoding.SUFFIX, StandardCharsets.ISO_8859_1); + + Assertions.assertArrayEquals(new String[] {"mouse"}, + lemmatizer.lemmatize(new String[] {"mice"}, new String[] {"NN"})); + Assertions.assertArrayEquals(new String[] {"cat"}, + lemmatizer.lemmatize(new String[] {"cats"}, new String[] {"NNS"})); + } + + /** The separator, encoding, and encoder are taken from .info metadata when supplied. */ + @Test + void testReadsWithInfoMetadata() throws IOException { + final String info = "fsa.dict.separator=+\n" + + "fsa.dict.encoding=iso-8859-1\n" + + "fsa.dict.encoder=suffix\n"; + final InputStream dict = new ByteArrayInputStream(Base64.getDecoder().decode(DICTIONARY_BASE64)); + + final DictionaryLemmatizer lemmatizer = MorfologikDictionaryReader.read( + dict, new ByteArrayInputStream(info.getBytes(StandardCharsets.ISO_8859_1))); + + Assertions.assertArrayEquals(new String[] {"mouse"}, + lemmatizer.lemmatize(new String[] {"mice"}, new String[] {"NN"})); + } + + /** Missing required metadata fails loudly. */ + @Test + void testMissingMetadataKeyRejected() { + final String info = "fsa.dict.separator=+\n"; + final InputStream dict = new ByteArrayInputStream(Base64.getDecoder().decode(DICTIONARY_BASE64)); + Assertions.assertThrows(IllegalArgumentException.class, () -> MorfologikDictionaryReader.read( + dict, new ByteArrayInputStream(info.getBytes(StandardCharsets.ISO_8859_1)))); + } + + /** A null dictionary stream is rejected at the boundary. */ + @Test + void testNullDictionaryRejected() { + Assertions.assertThrows(IllegalArgumentException.class, () -> MorfologikDictionaryReader.read( + null, (byte) '+', BaseFormEncoding.SUFFIX, StandardCharsets.ISO_8859_1)); + } +} diff --git a/opennlp-core/opennlp-formats/src/test/java/opennlp/tools/formats/PoliMorfDictionaryReaderUsageExampleTest.java b/opennlp-core/opennlp-formats/src/test/java/opennlp/tools/formats/PoliMorfDictionaryReaderUsageExampleTest.java new file mode 100644 index 0000000000..e46dfaba5e --- /dev/null +++ b/opennlp-core/opennlp-formats/src/test/java/opennlp/tools/formats/PoliMorfDictionaryReaderUsageExampleTest.java @@ -0,0 +1,113 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package opennlp.tools.formats; + +import java.io.ByteArrayInputStream; +import java.io.IOException; +import java.io.InputStream; +import java.nio.charset.StandardCharsets; +import java.util.List; + +import org.junit.jupiter.api.Assertions; +import org.junit.jupiter.api.Test; + +import opennlp.tools.lemmatizer.DictionaryLemmatizer; + +/** + * Runs the manual's PoliMorf reader examples (docbkx {@code lemmatizer.xml}) verbatim over a + * small illustrative dictionary in PoliMorf's {@code surfaceForm\tlemma\ttag} format. It is a + * hand-built fixture, not the PoliMorf distribution; every value the chapter states is asserted + * here, so a change breaking this test breaks the manual. + */ +public class PoliMorfDictionaryReaderUsageExampleTest { + + private static final String[] ROWS = { + "pies\tpies\tsubst:sg:nom:m2", + "psa\tpies\tsubst:sg:gen:m2", + "psy\tpies\tsubst:pl:nom:m2", + "kota\tkot\tsubst:sg:gen:m2", + // Same (form, tag) listed with two lemmas: the reader merges them into one entry. + "formy\tforma\tsubst:pl:nom:f", + "formy\tform\tsubst:pl:nom:f", + }; + + private static InputStream dictionary(String text) { + return new ByteArrayInputStream(text.getBytes(StandardCharsets.UTF_8)); + } + + private static DictionaryLemmatizer fixture() throws IOException { + return PoliMorfDictionaryReader.read(dictionary(String.join("\n", ROWS) + "\n")); + } + + /** A form and its tag resolve to the base form; a homograph tag or unknown form yields "O". */ + @Test + void testResolvesFormAndTagToLemma() throws IOException { + final DictionaryLemmatizer lemmatizer = fixture(); + + Assertions.assertArrayEquals(new String[] {"pies"}, + lemmatizer.lemmatize(new String[] {"psa"}, new String[] {"subst:sg:gen:m2"})); + Assertions.assertArrayEquals(new String[] {"kot"}, + lemmatizer.lemmatize(new String[] {"kota"}, new String[] {"subst:sg:gen:m2"})); + // A known form under a tag it never carries is a miss. + Assertions.assertArrayEquals(new String[] {"O"}, + lemmatizer.lemmatize(new String[] {"psa"}, new String[] {"adj:sg:nom:m2:pos"})); + // An unknown form is a miss. + Assertions.assertArrayEquals(new String[] {"O"}, + lemmatizer.lemmatize(new String[] {"kanapa"}, new String[] {"subst:sg:nom:f"})); + } + + /** Lookup is case-insensitive because forms are lower-cased on load. */ + @Test + void testLookupIsCaseInsensitive() throws IOException { + Assertions.assertArrayEquals(new String[] {"pies"}, + fixture().lemmatize(new String[] {"Psa"}, new String[] {"subst:sg:gen:m2"})); + } + + /** Every lemma listed for a form and tag is kept, in first-seen order. */ + @Test + void testAlternativeLemmasAreMerged() throws IOException { + final List> lemmas = + fixture().lemmatize(List.of("formy"), List.of("subst:pl:nom:f")); + + Assertions.assertEquals(List.of(List.of("forma", "form")), lemmas); + } + + /** Blank lines are skipped rather than treated as entries. */ + @Test + void testBlankLinesAreSkipped() throws IOException { + final DictionaryLemmatizer lemmatizer = PoliMorfDictionaryReader.read( + dictionary("\npsa\tpies\tsubst:sg:gen:m2\n \n")); + + Assertions.assertArrayEquals(new String[] {"pies"}, + lemmatizer.lemmatize(new String[] {"psa"}, new String[] {"subst:sg:gen:m2"})); + } + + /** A non-blank line with fewer than three fields fails loudly. */ + @Test + void testTooFewFieldsThrows() { + Assertions.assertThrows(IOException.class, + () -> PoliMorfDictionaryReader.read(dictionary("psa\tpies\n"))); + } + + /** A null dictionary stream is rejected at the boundary. */ + @Test + void testNullDictionaryRejected() { + Assertions.assertThrows(IllegalArgumentException.class, + () -> PoliMorfDictionaryReader.read(null)); + } +} diff --git a/opennlp-distr/pom.xml b/opennlp-distr/pom.xml index e9092d8821..39269575e8 100644 --- a/opennlp-distr/pom.xml +++ b/opennlp-distr/pom.xml @@ -91,6 +91,10 @@ org.apache.opennlp opennlp-spellcheck + + org.apache.opennlp + opennlp-wordnet + diff --git a/opennlp-distr/src/main/assembly/bin.xml b/opennlp-distr/src/main/assembly/bin.xml index 2db4eafc65..36c556d473 100644 --- a/opennlp-distr/src/main/assembly/bin.xml +++ b/opennlp-distr/src/main/assembly/bin.xml @@ -239,6 +239,13 @@ docs/apidocs/opennlp-spellcheck + + ../opennlp-extensions/opennlp-wordnet/target/reports/apidocs + 644 + 755 + docs/apidocs/opennlp-wordnet + + ../opennlp-extensions/opennlp-uima/target/reports/apidocs 644 diff --git a/opennlp-docs/src/docbkx/lemmatizer.xml b/opennlp-docs/src/docbkx/lemmatizer.xml index 4d33948e1a..09158161ed 100644 --- a/opennlp-docs/src/docbkx/lemmatizer.xml +++ b/opennlp-docs/src/docbkx/lemmatizer.xml @@ -316,4 +316,60 @@ Accuracy: 0.9659110277825124]]> +
+ Dictionary lemmatization from PoliMorf tables + + DictionaryLemmatizer reads a word, postag, + lemma table. Many published morphological dictionaries use the opposite + column order, one surfaceForm, lemma, tag row + per entry. This is the layout of PoliMorf, the successor grammatical dictionary of + Polish, which is released under a permissive BSD 2-Clause license, and it is also + what exporting a morfologik dictionary to text produces. + PoliMorfDictionaryReader adapts such a table into a + DictionaryLemmatizer: it re-orders the columns, lower-cases each surface + form to match the case-insensitive lookup, and merges every lemma listed for the same + form and tag into one entry. The dictionary data is supplied by the caller and never + bundled. PoliMorfDictionaryReaderUsageExampleTest asserts the behavior + shown here over a small in-repo table in this format: + lemmatag, one row per entry +String table = "psa\tpies\tsubst:sg:gen:m2\n" + + "psy\tpies\tsubst:pl:nom:m2\n" + + "kota\tkot\tsubst:sg:gen:m2\n"; + +Lemmatizer lemmatizer = PoliMorfDictionaryReader.read( + new ByteArrayInputStream(table.getBytes(StandardCharsets.UTF_8))); + +lemmatizer.lemmatize(new String[] {"psa"}, new String[] {"subst:sg:gen:m2"}); // ["pies"] +lemmatizer.lemmatize(new String[] {"Psa"}, new String[] {"subst:sg:gen:m2"}); // ["pies"], case-insensitive +lemmatizer.lemmatize(new String[] {"psa"}, new String[] {"adj:sg:nom:m2:pos"}); // ["O"], tag miss]]> + + +
+ +
+ Reading morfologik dictionaries + + Many languages publish a morphological dictionary as a compact finite-state + automaton (a .dict file) paired with a .info metadata + file. MorfologikDictionaryReader reads both automaton formats, + FSA5 and CFSA2, with no third-party dependency, and decodes each entry, which is + a surface form, a base form stored relative to it, and a tag, into a + DictionaryLemmatizer. The four base-form encoders (none, suffix, + prefix, infix) are all decoded. The separator, character encoding, and encoder + are read from the .info file, or passed explicitly. The dictionary + data is supplied by the caller and never bundled. + MorfologikDictionaryReaderTest asserts the behavior shown here. + + + The companion PoliMorfDictionaryReader covers the same job for a + plain tab-separated table rather than an automaton, so a permissively licensed + dictionary such as PoliMorf and an existing automaton dictionary both reach the + same lemmatizer. + +
diff --git a/opennlp-docs/src/docbkx/opennlp.xml b/opennlp-docs/src/docbkx/opennlp.xml index 36641c2c89..f7a8a39406 100644 --- a/opennlp-docs/src/docbkx/opennlp.xml +++ b/opennlp-docs/src/docbkx/opennlp.xml @@ -109,6 +109,7 @@ under the License. + diff --git a/opennlp-docs/src/docbkx/wordnet.xml b/opennlp-docs/src/docbkx/wordnet.xml new file mode 100644 index 0000000000..7b1100820c --- /dev/null +++ b/opennlp-docs/src/docbkx/wordnet.xml @@ -0,0 +1,129 @@ + + + + + + + WordNet + +
+ Introduction + + The opennlp-wordnet module loads a WordNet-style lexicon into + a LexicalKnowledgeBase and looks up synsets by lemma and part + of speech. Two readers are provided: WnLmfReader for the + Global WordNet Association WN-LMF XML interchange format, and + WndbReader for the classic Princeton WordNet database file + layout. Both return an immutable, thread-safe knowledge base. On top of + lookup, the module can Morphy-lemmatize, expand a term through synonym and + hypernym links, and score synset similarity on the hypernym graph. + +
+ +
+ Loading a lexicon + + WN-LMF is the usual choice for Open English WordNet and other GWA + wordnets. WNDB remains available for a local Princeton-style + dict directory. The examples below use the miniature + fixtures from the module's tests; replace the paths with a full lexicon + in application code. WordNetUsageExampleTest asserts the + behavior shown here. + + + +
+ +
+ Lookup + + Lookups are scoped by part of speech and fold case and underscores the + same way the readers index lemmas. Against the miniature WN-LMF fixture, + the noun dog has one sense: + senses = lexicon.lookup("dog", WordNetPOS.NOUN); +// senses.size() = 1 +// senses.get(0).id() = "mini-n1" +// senses.get(0).lemmas() = ["dog", "domestic dog"] +// senses.get(0).gloss() = "a domesticated canid"]]> + + +
+ +
+ Morphy lemmatization + + MorphyLemmatizer implements the Morphy algorithm against a + loaded lexicon and the irregular-form exception lists + (noun.exc, verb.exc, adj.exc, + adv.exc). Exception hits are returned first; regular + detachments are kept only when the candidate is in the lexicon. Unknown + forms yield the marker O. + + + +
+ +
+ Lexical expansion + + LexicalExpander turns a term into related terms from the + knowledge base: synonyms sharing its synsets, hypernym ancestors up to a + configured depth, and optionally direct hyponyms. Each + Expansion carries a heuristic weight in + (0, 1]: the first sense starts at 1.0, later + senses multiply by the sense decay, and each hypernym or hyponym step + multiplies by the depth decay. The input term itself is never returned. + Defaults use depth 1, sense decay 0.5, and + depth decay 0.5. + LexicalExpansionUsageExampleTest asserts the behavior shown + here. + expansions = LexicalExpander.builder(lexicon) + .build() + .expand("dog", WordNetPOS.NOUN); + +// "domestic dog": SYNONYM, senseRank 0, weight 1.0 +// "canid": HYPERNYM, depth 1, weight 0.5 +// "frank": SYNONYM, senseRank 1, weight 0.5]]> + + +
+ +
+ Synset similarity + + SynsetSimilarity scores two synset identifiers on the + hypernym graph. Path similarity is 1 / (1 + d) for the + shortest distance d through a common ancestor. Wu-Palmer + similarity relates the depth of the deepest common ancestor to the depths + of both synsets. Unrelated synsets score 0. + LexicalExpansionUsageExampleTest asserts the behavior shown + here. + + + +
+
diff --git a/opennlp-extensions/opennlp-wordnet/pom.xml b/opennlp-extensions/opennlp-wordnet/pom.xml new file mode 100644 index 0000000000..0721fcfc98 --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/pom.xml @@ -0,0 +1,59 @@ + + + + + + 4.0.0 + + org.apache.opennlp + opennlp-extensions + 3.0.0-SNAPSHOT + + + opennlp-wordnet + jar + Apache OpenNLP :: Ext :: WordNet + + + + org.apache.opennlp + opennlp-api + + + + org.junit.jupiter + junit-jupiter-api + test + + + + org.junit.jupiter + junit-jupiter-engine + test + + + + org.junit.jupiter + junit-jupiter-params + test + + + + diff --git a/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/HypernymTyper.java b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/HypernymTyper.java new file mode 100644 index 0000000000..02756d0e1e --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/HypernymTyper.java @@ -0,0 +1,167 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package opennlp.wordnet; + +import java.util.ArrayDeque; +import java.util.Deque; +import java.util.HashMap; +import java.util.LinkedHashMap; +import java.util.List; +import java.util.Map; +import java.util.Optional; + +import opennlp.tools.commons.ThreadSafe; +import opennlp.tools.util.StringUtil; +import opennlp.tools.wordnet.LexicalKnowledgeBase; +import opennlp.tools.wordnet.Synset; +import opennlp.tools.wordnet.WordNetPOS; +import opennlp.tools.wordnet.WordNetRelation; + +/** + * Types a noun by walking its hypernym chain to the nearest registered anchor: the + * caller names anchor concepts by lemma, {@code person}, {@code organization}, + * {@code location}, and any word whose senses lead up to an anchor's synsets receives + * that anchor's label. The nearest anchor wins, so a more specific registered concept + * beats a general one. + * + *

Anchors are resolved against the knowledge base at construction and follow its + * sense inventory; nothing beyond the caller's anchor choice is built in. Words with no + * sense reaching an anchor get no type.

+ * + *

The typer reads only immutable state and is safe to share between threads.

+ */ +@ThreadSafe +public class HypernymTyper { + + /** The relations that lead from a synset to its generalizations. */ + private static final List UPWARD_RELATIONS = + List.of(WordNetRelation.HYPERNYM, WordNetRelation.INSTANCE_HYPERNYM); + + private final LexicalKnowledgeBase knowledgeBase; + private final Map labelBySynsetId; + + /** + * Initializes the typer. + * + * @param knowledgeBase The knowledge base to walk. Must not be {@code null}. + * @param anchors The anchor lemmas mapped to the labels they confer, for example + * {@code person} to {@code person}. Every lemma is resolved as a noun; + * all its senses anchor. Must not be {@code null} or empty, and no + * lemma or label may be blank. + * @throws IllegalArgumentException Thrown if a parameter is {@code null}, + * {@code anchors} is empty or holds a blank entry, or an anchor lemma is + * unknown to the knowledge base. + */ + public HypernymTyper(LexicalKnowledgeBase knowledgeBase, Map anchors) { + if (knowledgeBase == null) { + throw new IllegalArgumentException("knowledgeBase must not be null"); + } + if (anchors == null || anchors.isEmpty()) { + throw new IllegalArgumentException("anchors must not be null or empty"); + } + this.knowledgeBase = knowledgeBase; + final Map labels = new LinkedHashMap<>(); + for (final Map.Entry anchor : anchors.entrySet()) { + if (anchor.getKey() == null || StringUtil.isBlank(anchor.getKey()) + || anchor.getValue() == null || StringUtil.isBlank(anchor.getValue())) { + throw new IllegalArgumentException("anchors must not contain blank entries"); + } + final List senses = knowledgeBase.lookup(anchor.getKey(), WordNetPOS.NOUN); + if (senses.isEmpty()) { + throw new IllegalArgumentException( + "anchor lemma is unknown to the knowledge base: " + anchor.getKey()); + } + for (final Synset sense : senses) { + labels.putIfAbsent(sense.id(), anchor.getValue()); + } + } + this.labelBySynsetId = Map.copyOf(labels); + } + + /** + * Types a noun by its nearest anchored hypernym. + * + * @param lemma The noun lemma to type. Must not be {@code null} or blank. + * @return The label of the nearest anchor over all senses, or empty when no sense + * reaches an anchor. + * @throws IllegalArgumentException Thrown if {@code lemma} is {@code null} or blank. + */ + public Optional type(String lemma) { + if (lemma == null || StringUtil.isBlank(lemma)) { + throw new IllegalArgumentException("lemma must not be null or blank"); + } + String bestLabel = null; + int bestDistance = Integer.MAX_VALUE; + for (final Synset sense : knowledgeBase.lookup(lemma, WordNetPOS.NOUN)) { + final int[] distance = new int[1]; + final String label = nearestAnchor(sense.id(), distance); + if (label != null && distance[0] < bestDistance) { + bestDistance = distance[0]; + bestLabel = label; + } + } + return Optional.ofNullable(bestLabel); + } + + /** + * Types a specific synset by its nearest anchored hypernym. + * + * @param synsetId The synset identifier. Must not be {@code null}. + * @return The nearest anchor's label, or empty when no ancestor is anchored. + * @throws IllegalArgumentException Thrown if {@code synsetId} is {@code null}. + */ + public Optional typeSynset(String synsetId) { + if (synsetId == null) { + throw new IllegalArgumentException("synsetId must not be null"); + } + return Optional.ofNullable(nearestAnchor(synsetId, new int[1])); + } + + /** + * Walks up the hypernym graph breadth first to the closest anchored synset. Visiting each + * synset once bounds the walk even on cyclic data. + * + * @param synsetId The synset to start from. Must not be {@code null}. + * @param distanceOut A single-element array that receives the edge count to the anchor found; + * left untouched when no ancestor is anchored. + * @return The label of the nearest anchored synset, or {@code null} when none is reachable. + */ + private String nearestAnchor(String synsetId, int[] distanceOut) { + final Deque queue = new ArrayDeque<>(); + final Map depths = new HashMap<>(); + queue.add(synsetId); + depths.put(synsetId, 0); + while (!queue.isEmpty()) { + final String current = queue.remove(); + final String label = labelBySynsetId.get(current); + if (label != null) { + distanceOut[0] = depths.get(current); + return label; + } + final int parentDepth = depths.get(current) + 1; + for (final WordNetRelation relation : UPWARD_RELATIONS) { + for (final String parent : knowledgeBase.related(current, relation)) { + if (depths.putIfAbsent(parent, parentDepth) == null) { + queue.add(parent); + } + } + } + } + return null; + } +} diff --git a/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/InMemoryWordNetLexicon.java b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/InMemoryWordNetLexicon.java new file mode 100644 index 0000000000..2401bfab6a --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/InMemoryWordNetLexicon.java @@ -0,0 +1,156 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package opennlp.wordnet; + +import java.util.ArrayList; +import java.util.Collection; +import java.util.Collections; +import java.util.HashMap; +import java.util.List; +import java.util.Map; +import java.util.Optional; + +import opennlp.tools.commons.ThreadSafe; +import opennlp.tools.wordnet.LexicalKnowledgeBase; +import opennlp.tools.wordnet.Synset; +import opennlp.tools.wordnet.WordNetPOS; +import opennlp.tools.wordnet.WordNetRelation; + +/** + * The immutable in-memory {@link LexicalKnowledgeBase} both readers produce: a synset table plus a + * folded (lemma, part of speech) index. Package-private because it is a reader product, not a + * public entry point; consumers hold it as {@link LexicalKnowledgeBase}. + * + *

Keys and queries are folded identically (see {@link LemmaFolding}). Construction verifies + * referential integrity: every relation target of every synset must resolve to a synset in the + * table, so a lexicon can never hand out a dangling identifier. After construction all state is + * immutable, making instances safe for concurrent lookups.

+ */ +@ThreadSafe +final class InMemoryWordNetLexicon implements LexicalKnowledgeBase { + + private final Map synsetsById; + private final Map> senseIndex; + + /** + * Indexes the given synsets. + * + * @param synsetsById The synset table keyed by synset id; every key must equal its synset's + * {@link Synset#id() id}. Must not be {@code null}. + * @param senseOrder The sense order per folded (lemma, part of speech) key: for each key the + * ids of the synsets containing the lemma, most salient sense first, each + * id resolvable in {@code synsetsById} and free of duplicates per key. + * Must not be {@code null}. + * @throws IllegalArgumentException Thrown if a relation target or sense entry does not + * resolve, or a key disagrees with its synset id. + */ + InMemoryWordNetLexicon(Map synsetsById, Map> senseOrder) { + if (synsetsById == null) { + throw new IllegalArgumentException("SynsetsById must not be null"); + } + if (senseOrder == null) { + throw new IllegalArgumentException("SenseOrder must not be null"); + } + final Map byId = new HashMap<>(synsetsById.size() * 2); + for (final Map.Entry entry : synsetsById.entrySet()) { + if (entry.getValue() == null || !entry.getValue().id().equals(entry.getKey())) { + throw new IllegalArgumentException( + "Synset table key " + entry.getKey() + " does not match its synset"); + } + byId.put(entry.getKey(), entry.getValue()); + } + for (final Synset synset : byId.values()) { + for (final Map.Entry> relation : + synset.relations().entrySet()) { + for (final String target : relation.getValue()) { + if (!byId.containsKey(target)) { + throw new IllegalArgumentException("Synset " + synset.id() + " has a " + + relation.getKey() + " relation to unknown synset " + target); + } + } + } + } + final Map> index = new HashMap<>(senseOrder.size() * 2); + for (final Map.Entry> entry : senseOrder.entrySet()) { + final List senses = new ArrayList<>(entry.getValue().size()); + for (final String synsetId : entry.getValue()) { + final Synset synset = byId.get(synsetId); + if (synset == null) { + throw new IllegalArgumentException("Sense index entry " + entry.getKey().lemma() + + " (" + entry.getKey().pos() + ") references unknown synset " + synsetId); + } + senses.add(synset); + } + index.put(entry.getKey(), List.copyOf(senses)); + } + this.synsetsById = byId; + this.senseIndex = index; + } + + /** {@inheritDoc} */ + @Override + public List lookup(String lemma, WordNetPOS pos) { + if (lemma == null) { + throw new IllegalArgumentException("Lemma must not be null"); + } + if (pos == null) { + throw new IllegalArgumentException("Pos must not be null"); + } + final List senses = senseIndex.get(LemmaKey.of(lemma, pos)); + return senses == null ? List.of() : senses; + } + + /** {@inheritDoc} */ + @Override + public Optional synset(String synsetId) { + if (synsetId == null) { + throw new IllegalArgumentException("SynsetId must not be null"); + } + return Optional.ofNullable(synsetsById.get(synsetId)); + } + + /** {@return the number of synsets in this lexicon} */ + int size() { + return synsetsById.size(); + } + + /** {@return all synsets, for equivalence checks and diagnostics within this package} */ + Collection synsets() { + return Collections.unmodifiableCollection(synsetsById.values()); + } + + /** + * A folded sense-index key. Build with {@link #of(String, WordNetPOS)} so every key passes + * through the same fold as every query. + * + * @param lemma The folded lemma. + * @param pos The part of speech. + */ + record LemmaKey(String lemma, WordNetPOS pos) { + + /** + * Folds a written form into a key: lowercase with the root locale, underscore as space. + * + * @param writtenForm The lemma as written in the source or query. Must not be {@code null}. + * @param pos The part of speech. Must not be {@code null}. + * @return The folded key. + */ + static LemmaKey of(String writtenForm, WordNetPOS pos) { + return new LemmaKey(LemmaFolding.fold(writtenForm), pos); + } + } +} diff --git a/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/LemmaFolding.java b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/LemmaFolding.java new file mode 100644 index 0000000000..4da713d520 --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/LemmaFolding.java @@ -0,0 +1,71 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package opennlp.wordnet; + +import java.util.ArrayList; +import java.util.List; +import java.util.Locale; + +/** + * The single home of the lemma fold and the space-separated field split this package relies on. + * {@link MorphyExceptions} keys, the {@link InMemoryWordNetLexicon.LemmaKey sense-index keys}, + * and every query must fold through {@link #fold(String)} so their canonical forms agree. + */ +final class LemmaFolding { + + /** Not instantiable. */ + private LemmaFolding() { + } + + /** + * Folds a written form into its canonical shape: lowercase with the root locale, with the + * underscore some formats store in multiword lemmas treated as a space. + * + * @param writtenForm The form as written in a source file or query. Must not be {@code null}. + * @return The folded form. + * @throws IllegalArgumentException Thrown if {@code writtenForm} is {@code null}. + */ + static String fold(String writtenForm) { + if (writtenForm == null) { + throw new IllegalArgumentException("writtenForm must not be null"); + } + return writtenForm.replace('_', ' ').toLowerCase(Locale.ROOT); + } + + /** + * Splits a space-separated field list, collapsing runs of spaces. + * + * @param value The field list. Must not be {@code null}. + * @return The non-empty fields in order, never {@code null}. + */ + static List splitOnSpaces(String value) { + final List parts = new ArrayList<>(4); + int start = 0; + while (start < value.length()) { + final int space = value.indexOf(' ', start); + if (space < 0) { + parts.add(value.substring(start)); + break; + } + if (space > start) { + parts.add(value.substring(start, space)); + } + start = space + 1; + } + return parts; + } +} diff --git a/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/LexicalExpander.java b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/LexicalExpander.java new file mode 100644 index 0000000000..5a110cac84 --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/LexicalExpander.java @@ -0,0 +1,486 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package opennlp.wordnet; + +import java.util.ArrayDeque; +import java.util.ArrayList; +import java.util.Comparator; +import java.util.HashMap; +import java.util.HashSet; +import java.util.List; +import java.util.Map; +import java.util.Set; + +import opennlp.tools.commons.ThreadSafe; +import opennlp.tools.lemmatizer.Lemmatizer; +import opennlp.tools.util.StringUtil; +import opennlp.tools.wordnet.LexicalKnowledgeBase; +import opennlp.tools.wordnet.Synset; +import opennlp.tools.wordnet.WordNetPOS; +import opennlp.tools.wordnet.WordNetRelation; + +/** + * Expands a term into related terms drawn from a {@link LexicalKnowledgeBase}: the synonyms + * sharing its synsets, the lemmas of its hypernym ancestors up to a configured depth, and + * optionally the lemmas of its direct hyponyms. + * + *

Each {@link Expansion} carries a deterministic heuristic weight, not a probability: the + * first sense of a term starts at {@code 1.0}, each later sense is multiplied by the configurable + * sense decay, and every hypernym or hyponym step multiplies by the configurable depth decay. + * A decay product that underflows to zero in double arithmetic carries no ranking signal, so + * such expansions are dropped rather than emitted outside the {@code (0, 1]} weight range. + * When the term itself is not in the lexicon and a {@link Lemmatizer} is configured, the term is + * lemmatized and the lemma expanded instead; the lemmatizer is invoked with the + * {@link WordNetPOS} name as the tag.

+ * + *

Hypernym walks follow both the direct and the instance relation, track visited synsets so + * malformed cyclic data cannot loop, and never report the term itself. Results are deduplicated + * case-insensitively, keeping the highest weight, and ordered by weight descending, then kind, + * then term, so output is stable across runs.

+ * + *

Instances are immutable and safe for concurrent use when the configured lexicon and + * lemmatizer are.

+ */ +@ThreadSafe +public final class LexicalExpander { + + /** How an expansion relates to the input term. */ + public enum Kind { + + /** A member of one of the term's own synsets. */ + SYNONYM, + + /** A lemma of an ancestor synset, {@link Expansion#depth()} steps up. */ + HYPERNYM, + + /** A lemma of a direct child synset. */ + HYPONYM + } + + /** + * One expansion of a term. + * + * @param term The expanded term, in the lexicon's written form (multiword terms contain + * spaces). + * @param kind How the term relates to the input. + * @param depth The relation distance: {@code 0} for synonyms, the number of hypernym steps + * for {@link Kind#HYPERNYM}, {@code 1} for hyponyms. + * @param senseRank The zero-based rank of the input sense this expansion came from. + * @param weight The heuristic weight in {@code (0, 1]}; higher is closer to the input term. + */ + public record Expansion(String term, Kind kind, int depth, int senseRank, double weight) { + + /** + * Validates every component against the documented ranges. + * + * @throws IllegalArgumentException Thrown if {@code term} is {@code null} or blank, + * {@code kind} is {@code null}, {@code depth} or {@code senseRank} is negative, or + * {@code weight} is not in {@code (0, 1]}. + */ + public Expansion { + if (term == null || StringUtil.isBlank(term)) { + throw new IllegalArgumentException("term must not be null or blank"); + } + if (kind == null) { + throw new IllegalArgumentException("kind must not be null"); + } + if (depth < 0) { + throw new IllegalArgumentException("depth must not be negative: " + depth); + } + if (senseRank < 0) { + throw new IllegalArgumentException("senseRank must not be negative: " + senseRank); + } + if (!(weight > 0.0 && weight <= 1.0)) { + throw new IllegalArgumentException("weight must be in (0, 1]: " + weight); + } + } + } + + private final LexicalKnowledgeBase lexicon; + private final Lemmatizer lemmatizer; + private final int maxSenses; + private final int hypernymDepth; + private final boolean includeHyponyms; + private final int maxExpansions; + private final double senseDecay; + private final double depthDecay; + + /** + * Creates an expander from a builder whose fields have already been validated. + * + * @param builder The configured builder. Must not be {@code null}. + */ + private LexicalExpander(Builder builder) { + this.lexicon = builder.lexicon; + this.lemmatizer = builder.lemmatizer; + this.maxSenses = builder.maxSenses; + this.hypernymDepth = builder.hypernymDepth; + this.includeHyponyms = builder.includeHyponyms; + this.maxExpansions = builder.maxExpansions; + this.senseDecay = builder.senseDecay; + this.depthDecay = builder.depthDecay; + } + + /** + * Starts a builder. + * + * @param lexicon The knowledge base to expand against. Must not be {@code null}. + * @return A builder with the default configuration. + * @throws IllegalArgumentException Thrown if {@code lexicon} is {@code null}. + */ + public static Builder builder(LexicalKnowledgeBase lexicon) { + return new Builder(lexicon); + } + + /** + * Expands a term for one part of speech. + * + * @param term The term to expand. Must not be {@code null} or blank. + * @param pos The part of speech to expand as. Must not be {@code null}. + * @return The expansions, deduplicated and ordered by descending weight; empty when the term + * (and its lemma, when a lemmatizer is configured) is not in the lexicon. + * @throws IllegalArgumentException Thrown if {@code term} is {@code null} or blank or + * {@code pos} is {@code null}. + */ + public List expand(String term, WordNetPOS pos) { + if (pos == null) { + throw new IllegalArgumentException("pos must not be null"); + } + return collect(term, List.of(pos)); + } + + /** + * Expands a term across all parts of speech. + * + * @param term The term to expand. Must not be {@code null} or blank. + * @return The expansions across every part of speech, deduplicated and ordered by descending + * weight; empty when the term is not in the lexicon. + * @throws IllegalArgumentException Thrown if {@code term} is {@code null} or blank. + */ + public List expand(String term) { + return collect(term, List.of(WordNetPOS.values())); + } + + /** + * Expands the term across the given parts of speech and returns the ranked, capped result. + * + * @param term The term to expand. Must not be {@code null} or blank. + * @param poses The parts of speech to expand as. Must not be {@code null}. + * @return The expansions, deduplicated, ordered by descending weight, and capped at the + * configured maximum; empty when the term is not in the lexicon. + * @throws IllegalArgumentException Thrown if {@code term} is {@code null} or blank. + */ + private List collect(String term, List poses) { + if (term == null || StringUtil.isBlank(term)) { + throw new IllegalArgumentException("term must not be null or blank"); + } + final Map best = new HashMap<>(); + final Set excluded = new HashSet<>(); + excluded.add(LemmaFolding.fold(term)); + + for (final WordNetPOS pos : poses) { + final String subject = resolveSubject(term, pos); + if (subject == null) { + continue; + } + final List senses = lexicon.lookup(subject, pos); + final int senseCount = Math.min(senses.size(), maxSenses); + for (int rank = 0; rank < senseCount; rank++) { + final double senseWeight = Math.pow(senseDecay, rank); + expandSense(senses.get(rank), rank, senseWeight, best, excluded); + } + } + + final List ordered = new ArrayList<>(best.values()); + ordered.sort(Comparator.comparingDouble(Expansion::weight).reversed() + .thenComparing(Expansion::kind) + .thenComparing(Expansion::term)); + return ordered.size() > maxExpansions ? List.copyOf(ordered.subList(0, maxExpansions)) + : List.copyOf(ordered); + } + + /** + * Resolves the form actually expanded: the term when the lexicon knows it, otherwise its lemma + * when a lemmatizer is configured and produces a known lemma. + * + * @param term The input term. Must not be {@code null}. + * @param pos The part of speech to look the term up as. Must not be {@code null}. + * @return The known form to expand, or {@code null} when neither the term nor its lemma is in + * the lexicon. + */ + private String resolveSubject(String term, WordNetPOS pos) { + if (lexicon.contains(term, pos)) { + return term; + } + if (lemmatizer == null) { + return null; + } + final String[] lemmas = + lemmatizer.lemmatize(new String[] {term}, new String[] {pos.name()}); + if (lemmas.length == 0 || lemmas[0] == null) { + return null; + } + final String lemma = lemmas[0]; + // Lemmatizers report an unresolvable token with the contract's unknown marker. + if (MorphyLemmatizer.UNKNOWN_LEMMA.equals(lemma) || !lexicon.contains(lemma, pos)) { + return null; + } + return lemma; + } + + /** + * Offers the synonyms, hypernym ancestors, and optional hyponyms of one sense into the running + * best-expansion map. + * + * @param sense The sense to expand. Must not be {@code null}. + * @param rank The zero-based salience rank of the sense. + * @param senseWeight The weight of the sense itself; each relation step decays from it. + * @param best The best expansion seen so far per folded term; updated in place. + * @param excluded The folded terms that are never reported, such as the input term. + */ + private void expandSense(Synset sense, int rank, double senseWeight, + Map best, Set excluded) { + if (senseWeight == 0.0) { + // Underflowed to zero: no ranking signal left, and zero is outside the documented range. + return; + } + for (final String lemma : sense.lemmas()) { + offer(best, excluded, new Expansion(lemma, Kind.SYNONYM, 0, rank, senseWeight)); + } + + // Breadth-first hypernym walk, visited-checked so cyclic data terminates. + if (hypernymDepth > 0) { + final Set visited = new HashSet<>(); + visited.add(sense.id()); + final ArrayDeque frontier = new ArrayDeque<>(hypernymsOf(sense)); + final ArrayDeque next = new ArrayDeque<>(); + for (int depth = 1; depth <= hypernymDepth && !frontier.isEmpty(); depth++) { + final double depthWeight = senseWeight * Math.pow(depthDecay, depth); + if (depthWeight == 0.0) { + // Deeper levels only shrink further, so the walk stops at the first underflow. + break; + } + while (!frontier.isEmpty()) { + final String id = frontier.poll(); + if (!visited.add(id)) { + continue; + } + final Synset ancestor = lexicon.synset(id).orElse(null); + if (ancestor == null) { + continue; + } + for (final String lemma : ancestor.lemmas()) { + offer(best, excluded, + new Expansion(lemma, Kind.HYPERNYM, depth, rank, depthWeight)); + } + next.addAll(hypernymsOf(ancestor)); + } + frontier.addAll(next); + next.clear(); + } + } + + if (includeHyponyms && senseWeight * depthDecay > 0.0) { + final double hyponymWeight = senseWeight * depthDecay; + final List children = new ArrayList<>(sense.related(WordNetRelation.HYPONYM)); + children.addAll(sense.related(WordNetRelation.INSTANCE_HYPONYM)); + for (final String id : children) { + lexicon.synset(id).ifPresent(child -> { + for (final String lemma : child.lemmas()) { + offer(best, excluded, new Expansion(lemma, Kind.HYPONYM, 1, rank, hyponymWeight)); + } + }); + } + } + } + + /** + * Collects the synset ids of both the direct and the instance hypernyms of a synset. + * + * @param synset The synset whose hypernyms are collected. Must not be {@code null}. + * @return The hypernym synset ids in source order, direct relations first. + */ + private List hypernymsOf(Synset synset) { + final List direct = synset.related(WordNetRelation.HYPERNYM); + final List instance = synset.related(WordNetRelation.INSTANCE_HYPERNYM); + if (instance.isEmpty()) { + return direct; + } + final List all = new ArrayList<>(direct.size() + instance.size()); + all.addAll(direct); + all.addAll(instance); + return all; + } + + /** + * Records the candidate under its folded term when it is not excluded and it beats the current + * best weight for that term. Folding through {@link LemmaFolding#fold(String)} keeps the + * exclusion and deduplication keys aligned with the lexicon's own lemma keys. + * + * @param best The best expansion seen so far per folded term; updated in place. + * @param excluded The folded terms that are never reported. + * @param candidate The expansion to offer. Must not be {@code null}. + */ + private void offer(Map best, Set excluded, + Expansion candidate) { + final String key = LemmaFolding.fold(candidate.term()); + if (excluded.contains(key)) { + return; + } + final Expansion current = best.get(key); + if (current == null || candidate.weight() > current.weight()) { + best.put(key, candidate); + } + } + + /** Configures and creates a {@link LexicalExpander}. */ + public static final class Builder { + + private final LexicalKnowledgeBase lexicon; + private Lemmatizer lemmatizer; + private int maxSenses = 3; + private int hypernymDepth = 1; + private boolean includeHyponyms = false; + private int maxExpansions = 20; + private double senseDecay = 0.5; + private double depthDecay = 0.5; + + /** + * Creates a builder over the given lexicon; use {@link LexicalExpander#builder}. + * + * @param lexicon The knowledge base to expand against. Must not be {@code null}. + * @throws IllegalArgumentException Thrown if {@code lexicon} is {@code null}. + */ + private Builder(LexicalKnowledgeBase lexicon) { + if (lexicon == null) { + throw new IllegalArgumentException("lexicon must not be null"); + } + this.lexicon = lexicon; + } + + /** + * Configures a lemmatizer used when the input term itself is not in the lexicon. It is + * invoked with the {@link WordNetPOS} name as the tag. + * + * @param lemmatizer The fallback lemmatizer. Must not be {@code null}. + * @return This builder. + * @throws IllegalArgumentException Thrown if {@code lemmatizer} is {@code null}. + */ + public Builder lemmatizer(Lemmatizer lemmatizer) { + if (lemmatizer == null) { + throw new IllegalArgumentException("lemmatizer must not be null"); + } + this.lemmatizer = lemmatizer; + return this; + } + + /** + * Configures how many senses of the term are expanded, most salient first. + * + * @param maxSenses The sense count; must be positive. The default is {@code 3}. + * @return This builder. + * @throws IllegalArgumentException Thrown if {@code maxSenses} is not positive. + */ + public Builder maxSenses(int maxSenses) { + if (maxSenses < 1) { + throw new IllegalArgumentException("maxSenses must be positive: " + maxSenses); + } + this.maxSenses = maxSenses; + return this; + } + + /** + * Configures how many hypernym steps are walked; {@code 0} disables hypernym expansion. + * + * @param hypernymDepth The depth; must not be negative. The default is {@code 1}. + * @return This builder. + * @throws IllegalArgumentException Thrown if {@code hypernymDepth} is negative. + */ + public Builder hypernymDepth(int hypernymDepth) { + if (hypernymDepth < 0) { + throw new IllegalArgumentException( + "hypernymDepth must not be negative: " + hypernymDepth); + } + this.hypernymDepth = hypernymDepth; + return this; + } + + /** + * Configures whether direct hyponyms are included; off by default. + * + * @param includeHyponyms Whether to include direct hyponyms. + * @return This builder. + */ + public Builder includeHyponyms(boolean includeHyponyms) { + this.includeHyponyms = includeHyponyms; + return this; + } + + /** + * Configures the maximum number of expansions returned after ranking. + * + * @param maxExpansions The cap; must be positive. The default is {@code 20}. + * @return This builder. + * @throws IllegalArgumentException Thrown if {@code maxExpansions} is not positive. + */ + public Builder maxExpansions(int maxExpansions) { + if (maxExpansions < 1) { + throw new IllegalArgumentException( + "maxExpansions must be positive: " + maxExpansions); + } + this.maxExpansions = maxExpansions; + return this; + } + + /** + * Configures the weight multiplier applied per sense rank step. + * + * @param senseDecay The decay in {@code (0, 1]}. The default is {@code 0.5}. + * @return This builder. + * @throws IllegalArgumentException Thrown if {@code senseDecay} is outside {@code (0, 1]}. + */ + public Builder senseDecay(double senseDecay) { + if (!(senseDecay > 0 && senseDecay <= 1)) { + throw new IllegalArgumentException( + "senseDecay must be in (0, 1]: " + senseDecay); + } + this.senseDecay = senseDecay; + return this; + } + + /** + * Configures the weight multiplier applied per hypernym or hyponym step. + * + * @param depthDecay The decay in {@code (0, 1]}. The default is {@code 0.5}. + * @return This builder. + * @throws IllegalArgumentException Thrown if {@code depthDecay} is outside {@code (0, 1]}. + */ + public Builder depthDecay(double depthDecay) { + if (!(depthDecay > 0 && depthDecay <= 1)) { + throw new IllegalArgumentException( + "depthDecay must be in (0, 1]: " + depthDecay); + } + this.depthDecay = depthDecay; + return this; + } + + /** {@return the configured expander} */ + public LexicalExpander build() { + return new LexicalExpander(this); + } + } +} diff --git a/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/MorphyExceptions.java b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/MorphyExceptions.java new file mode 100644 index 0000000000..4f6c0b2934 --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/MorphyExceptions.java @@ -0,0 +1,145 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package opennlp.wordnet; + +import java.io.IOException; +import java.nio.charset.StandardCharsets; +import java.nio.file.Files; +import java.nio.file.Path; +import java.util.ArrayList; +import java.util.EnumMap; +import java.util.HashMap; +import java.util.List; +import java.util.Map; + +import opennlp.tools.commons.ThreadSafe; +import opennlp.tools.util.InvalidFormatException; +import opennlp.tools.wordnet.WordNetPOS; + +/** + * The Morphy exception lists: the per-part-of-speech tables of irregular inflected forms + * ({@code mice} to {@code mouse}, {@code went} to {@code go}) that the Morphy algorithm + * consults before its detachment rules. + * + *

{@link #load(Path)} reads the four {@code *.exc} files ({@code noun.exc}, + * {@code verb.exc}, {@code adj.exc}, {@code adv.exc}), which must all be present, in the WNDB + * format: one entry per line, the inflected form followed by one or more base forms, space + * separated, with underscores standing for spaces in multiword entries. No exception data is + * bundled; the caller supplies a directory.

+ * + *

Lookups fold the queried word the same way the lexicon seam folds lemmas. Instances are + * immutable after loading and safe for concurrent lookups.

+ */ +@ThreadSafe +public final class MorphyExceptions { + + private final Map>> byPos; + + /** + * Wraps the per-part-of-speech exception tables. + * + * @param byPos The loaded tables, one per part of speech. + */ + private MorphyExceptions(Map>> byPos) { + this.byPos = byPos; + } + + /** + * Loads the four exception lists from a directory. + * + * @param directory The directory containing {@code noun.exc}, {@code verb.exc}, + * {@code adj.exc}, and {@code adv.exc}. Must not be {@code null} and must + * exist. + * @return The loaded exception lists. + * @throws IllegalArgumentException Thrown if {@code directory} is {@code null} or not a + * directory. + * @throws InvalidFormatException Thrown if one of the four files is missing or a line is + * malformed; the message names the file and line. + * @throws IOException Thrown if reading a file fails. + */ + public static MorphyExceptions load(Path directory) throws IOException { + if (directory == null) { + throw new IllegalArgumentException("Directory must not be null"); + } + if (!Files.isDirectory(directory)) { + throw new IllegalArgumentException( + "Directory does not exist or is not a directory: " + directory); + } + final Map>> byPos = new EnumMap<>(WordNetPOS.class); + byPos.put(WordNetPOS.NOUN, loadFile(directory, "noun.exc")); + byPos.put(WordNetPOS.VERB, loadFile(directory, "verb.exc")); + byPos.put(WordNetPOS.ADJECTIVE, loadFile(directory, "adj.exc")); + byPos.put(WordNetPOS.ADVERB, loadFile(directory, "adv.exc")); + return new MorphyExceptions(byPos); + } + + /** + * Finds the base forms of an irregular inflected form. + * + * @param word The inflected form; folded before lookup. Must not be {@code null}. + * @param pos The part of speech. Must not be {@code null}. + * @return The base forms in file order, never {@code null}; empty when the word has no entry. + * @throws IllegalArgumentException Thrown if {@code word} or {@code pos} is {@code null}. + */ + public List lookup(String word, WordNetPOS pos) { + if (word == null) { + throw new IllegalArgumentException("Word must not be null"); + } + if (pos == null) { + throw new IllegalArgumentException("Pos must not be null"); + } + final List lemmas = byPos.get(pos).get(LemmaFolding.fold(word)); + return lemmas == null ? List.of() : lemmas; + } + + /** + * Loads one {@code *.exc} file into a folded inflected-form to base-forms map. + * + * @param directory The directory holding the file. + * @param fileName The exception file name, for example {@code noun.exc}. + * @return The folded exception entries. + * @throws InvalidFormatException Thrown if the file is missing or a line is malformed. + * @throws IOException Thrown if reading the file fails. + */ + private static Map> loadFile(Path directory, String fileName) + throws IOException { + final Path file = directory.resolve(fileName); + if (!Files.isRegularFile(file)) { + throw new InvalidFormatException("Missing exception list file: " + file); + } + final List lines = Files.readAllLines(file, StandardCharsets.ISO_8859_1); + final Map> entries = new HashMap<>(lines.size() * 2); + for (int i = 0; i < lines.size(); i++) { + final String line = lines.get(i); + if (line.isEmpty()) { + continue; + } + final List fields = LemmaFolding.splitOnSpaces(line); + if (fields.size() < 2) { + throw new InvalidFormatException("Malformed exception list " + fileName + " at line " + + (i + 1) + ": expected an inflected form and at least one base form, got: " + line); + } + final List lemmas = new ArrayList<>(fields.size() - 1); + for (final String lemma : fields.subList(1, fields.size())) { + lemmas.add(LemmaFolding.fold(lemma)); + } + // A form listed twice keeps its first entry, matching first-match lookup semantics. + entries.putIfAbsent(LemmaFolding.fold(fields.get(0)), List.copyOf(lemmas)); + } + return Map.copyOf(entries); + } +} diff --git a/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/MorphyLemmatizer.java b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/MorphyLemmatizer.java new file mode 100644 index 0000000000..c4da01d94f --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/MorphyLemmatizer.java @@ -0,0 +1,226 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package opennlp.wordnet; + +import java.util.ArrayList; +import java.util.List; +import java.util.Locale; + +import opennlp.tools.commons.ThreadSafe; +import opennlp.tools.lemmatizer.Lemmatizer; +import opennlp.tools.wordnet.LexicalKnowledgeBase; +import opennlp.tools.wordnet.WordNetPOS; + +/** + * A {@link Lemmatizer} implementing the Morphy algorithm: exception-list lookup first, then the + * per-part-of-speech iterative detachment rules, with every rule-derived candidate validated + * against a {@link LexicalKnowledgeBase} before it is returned. A token is folded (lowercase with + * the root locale, underscore as space) before lookup, and returned lemmas are in that folded + * form. + * + *

Part-of-speech tags map to a {@link WordNetPOS} by their conventional Penn Treebank + * prefixes ({@code N}, {@code V}, {@code J}, {@code R}), the names {@code ADJ} and {@code ADV}, + * and the one-letter WordNet codes {@code n}, {@code v}, {@code a}, {@code r}, and {@code s} + * (satellite, treated as adjective), case-insensitively. A tag that maps to no part of speech + * yields the unknown-word result.

+ * + *

Following {@code opennlp.tools.lemmatizer.DictionaryLemmatizer}, a token with no lemma + * yields {@link #UNKNOWN_LEMMA} from {@link #lemmatize(String[], String[])} and a singleton list + * of it from {@link #lemmatize(List, List)}. Both a lexicon and exception lists are required. + * Instances are immutable and safe for concurrent use.

+ */ +@ThreadSafe +public final class MorphyLemmatizer implements Lemmatizer { + + /** + * The output emitted for a token whose lemma is unknown, following the conventional + * {@link Lemmatizer} unknown marker also used by + * {@code opennlp.tools.lemmatizer.DictionaryLemmatizer}. + */ + public static final String UNKNOWN_LEMMA = "O"; + + private static final String[][] NOUN_RULES = { + {"s", ""}, {"ses", "s"}, {"xes", "x"}, {"zes", "z"}, + {"ches", "ch"}, {"shes", "sh"}, {"men", "man"}, {"ies", "y"}, + }; + + private static final String[][] VERB_RULES = { + {"s", ""}, {"ies", "y"}, {"es", "e"}, {"es", ""}, + {"ed", "e"}, {"ed", ""}, {"ing", "e"}, {"ing", ""}, + }; + + private static final String[][] ADJECTIVE_RULES = { + {"er", ""}, {"est", ""}, {"er", "e"}, {"est", "e"}, + }; + + private static final String[][] NO_RULES = {}; + + private final LexicalKnowledgeBase lexicon; + private final MorphyExceptions exceptions; + + /** + * Creates a Morphy lemmatizer over a loaded lexicon and exception lists. + * + * @param lexicon The lexicon rule candidates are validated against. Must not be + * {@code null}. + * @param exceptions The irregular-form exception lists. Must not be {@code null}. + * @throws IllegalArgumentException Thrown if {@code lexicon} or {@code exceptions} is + * {@code null}. + */ + public MorphyLemmatizer(LexicalKnowledgeBase lexicon, MorphyExceptions exceptions) { + if (lexicon == null) { + throw new IllegalArgumentException("Lexicon must not be null"); + } + if (exceptions == null) { + throw new IllegalArgumentException("Exceptions must not be null"); + } + this.lexicon = lexicon; + this.exceptions = exceptions; + } + + /** + * {@inheritDoc} + * + * @throws IllegalArgumentException Thrown if {@code toks} or {@code tags} is {@code null}, + * contains a {@code null} element, or the two differ in length. + */ + @Override + public String[] lemmatize(String[] toks, String[] tags) { + if (toks == null || tags == null) { + throw new IllegalArgumentException("Toks and tags must not be null"); + } + if (toks.length != tags.length) { + throw new IllegalArgumentException("Toks and tags must have the same length, got " + + toks.length + " and " + tags.length); + } + final String[] lemmas = new String[toks.length]; + for (int i = 0; i < toks.length; i++) { + final List candidates = lemmasOf(toks[i], tags[i]); + lemmas[i] = candidates.isEmpty() ? UNKNOWN_LEMMA : candidates.get(0); + } + return lemmas; + } + + /** + * {@inheritDoc} + * + * @throws IllegalArgumentException Thrown if {@code toks} or {@code tags} is {@code null}, + * contains a {@code null} element, or the two differ in size. + */ + @Override + public List> lemmatize(List toks, List tags) { + if (toks == null || tags == null) { + throw new IllegalArgumentException("Toks and tags must not be null"); + } + if (toks.size() != tags.size()) { + throw new IllegalArgumentException("Toks and tags must have the same size, got " + + toks.size() + " and " + tags.size()); + } + final List> lemmas = new ArrayList<>(toks.size()); + for (int i = 0; i < toks.size(); i++) { + final List candidates = lemmasOf(toks.get(i), tags.get(i)); + lemmas.add(candidates.isEmpty() ? List.of(UNKNOWN_LEMMA) : candidates); + } + return lemmas; + } + + /** + * Finds all lemmas of one token, most preferred first. + * + * @param token The token to lemmatize. + * @param tag The part-of-speech tag. + * @return The candidate lemmas, empty when the word is unknown or the tag maps to no part of + * speech. + */ + private List lemmasOf(String token, String tag) { + if (token == null || tag == null) { + throw new IllegalArgumentException("Tokens and tags must not contain null elements"); + } + final WordNetPOS pos = posFromTag(tag); + if (pos == null) { + return List.of(); + } + final String folded = LemmaFolding.fold(token); + final List irregular = exceptions.lookup(folded, pos); + if (!irregular.isEmpty()) { + return irregular; + } + final List candidates = new ArrayList<>(2); + if (lexicon.contains(folded, pos)) { + candidates.add(folded); + } + for (final String[] rule : rulesFor(pos)) { + final String suffix = rule[0]; + if (folded.length() > suffix.length() && folded.endsWith(suffix)) { + final String candidate = + folded.substring(0, folded.length() - suffix.length()) + rule[1]; + if (!candidates.contains(candidate) && lexicon.contains(candidate, pos)) { + candidates.add(candidate); + } + } + } + return candidates; + } + + /** + * Selects the detachment-rule table for a part of speech. + * + * @param pos The part of speech. + * @return The suffix-substitution rules, empty for adverbs. + */ + private static String[][] rulesFor(WordNetPOS pos) { + return switch (pos) { + case NOUN -> NOUN_RULES; + case VERB -> VERB_RULES; + case ADJECTIVE -> ADJECTIVE_RULES; + case ADVERB -> NO_RULES; + }; + } + + /** + * Maps a part-of-speech tag to a {@link WordNetPOS}. Package-private so tests can pin the + * mapping directly. + * + * @param tag The tag to map. Must not be {@code null}. + * @return The part of speech, or {@code null} when the tag names none. + * @throws IllegalArgumentException Thrown if {@code tag} is {@code null}. + */ + static WordNetPOS posFromTag(String tag) { + if (tag == null) { + throw new IllegalArgumentException("Tag must not be null"); + } + if (tag.isEmpty()) { + return null; + } + final String upper = tag.toUpperCase(Locale.ROOT); + if (upper.startsWith("ADJ")) { + return WordNetPOS.ADJECTIVE; + } + if (upper.startsWith("ADV")) { + return WordNetPOS.ADVERB; + } + return switch (upper.charAt(0)) { + case 'N' -> WordNetPOS.NOUN; + case 'V' -> WordNetPOS.VERB; + case 'J' -> WordNetPOS.ADJECTIVE; + // Codes a and s mean adjective only as one-letter tags; AUX, ADP and the like do not. + case 'A', 'S' -> tag.length() == 1 ? WordNetPOS.ADJECTIVE : null; + case 'R' -> WordNetPOS.ADVERB; + default -> null; + }; + } +} diff --git a/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/SynsetSimilarity.java b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/SynsetSimilarity.java new file mode 100644 index 0000000000..050f85a16c --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/SynsetSimilarity.java @@ -0,0 +1,241 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package opennlp.wordnet; + +import java.util.ArrayDeque; +import java.util.ArrayList; +import java.util.Deque; +import java.util.HashMap; +import java.util.List; +import java.util.Map; + +import opennlp.tools.commons.ThreadSafe; +import opennlp.tools.wordnet.LexicalKnowledgeBase; +import opennlp.tools.wordnet.WordNetRelation; + +/** + * Taxonomy-based similarity between synsets: measures over the hypernym graph of a + * {@link LexicalKnowledgeBase}, computed on demand with no precomputed tables. + * + *

Path similarity is {@code 1 / (1 + d)} for the shortest hypernym-graph distance + * {@code d} through a common ancestor. Wu-Palmer similarity relates the depth of the + * deepest common ancestor to the depths of both synsets. Leacock-Chodorow scales the + * shortest path against a caller-supplied taxonomy depth, since the knowledge base + * interface does not enumerate the taxonomy. Unrelated synsets, those sharing no + * ancestor, score zero everywhere. Information-content measures need corpus counts and + * are not provided here.

+ * + *

Both plain and instance hypernyms count as taxonomy edges. The measures read only + * the knowledge base and hold no mutable state, so instances are as thread-safe as + * their knowledge base.

+ */ +@ThreadSafe +public class SynsetSimilarity { + + private final LexicalKnowledgeBase knowledgeBase; + + /** + * Initializes the measures. + * + * @param knowledgeBase The knowledge base to walk. Must not be {@code null}. + * @throws IllegalArgumentException Thrown if {@code knowledgeBase} is {@code null}. + */ + public SynsetSimilarity(LexicalKnowledgeBase knowledgeBase) { + if (knowledgeBase == null) { + throw new IllegalArgumentException("knowledgeBase must not be null"); + } + this.knowledgeBase = knowledgeBase; + } + + /** + * Computes path similarity: {@code 1 / (1 + d)} over the shortest hypernym-graph + * distance. + * + * @param synsetId The first synset identifier. Must not be {@code null}. + * @param otherSynsetId The second synset identifier. Must not be {@code null}. + * @return The similarity in {@code (0, 1]}, or {@code 0} when the synsets share no + * ancestor. + * @throws IllegalArgumentException Thrown if {@code synsetId} or + * {@code otherSynsetId} is {@code null}. + */ + public double path(String synsetId, String otherSynsetId) { + final int distance = shortestDistance(synsetId, otherSynsetId); + return distance < 0 ? 0.0 : 1.0 / (1.0 + distance); + } + + /** + * Computes Wu-Palmer similarity: {@code 2 * depth(lcs) / (depth(a) + depth(b))}, + * with depths counted from the taxonomy root and the deepest common ancestor as the + * lcs. + * + * @param synsetId The first synset identifier. Must not be {@code null}. + * @param otherSynsetId The second synset identifier. Must not be {@code null}. + * @return The similarity in {@code (0, 1]}, or {@code 0} when the synsets share no + * ancestor. + * @throws IllegalArgumentException Thrown if {@code synsetId} or + * {@code otherSynsetId} is {@code null}. + */ + public double wuPalmer(String synsetId, String otherSynsetId) { + validateIds(synsetId, otherSynsetId); + final Map up = depthsAbove(synsetId); + final Map otherUp = depthsAbove(otherSynsetId); + double best = 0.0; + for (final Map.Entry common : up.entrySet()) { + final Integer otherDistance = otherUp.get(common.getKey()); + if (otherDistance == null) { + continue; + } + final int rootDepth = depthFromRoot(common.getKey()); + final int depthA = rootDepth + common.getValue(); + final int depthB = rootDepth + otherDistance; + if (depthA + depthB == 0) { + continue; + } + final double score = 2.0 * rootDepth / (depthA + depthB); + best = Math.max(best, score); + } + return best; + } + + /** + * Computes Leacock-Chodorow similarity: + * {@code -log((d + 1) / (2 * taxonomyDepth))} over the shortest hypernym-graph + * distance {@code d}. + * + * @param synsetId The first synset identifier. Must not be {@code null}. + * @param otherSynsetId The second synset identifier. Must not be {@code null}. + * @param taxonomyDepth The maximum depth of the taxonomy the synsets live in. Must + * be positive. + * @return The similarity, higher for closer synsets, or {@code 0} when the synsets + * share no ancestor. + * @throws IllegalArgumentException Thrown if {@code synsetId} or + * {@code otherSynsetId} is {@code null}, or {@code taxonomyDepth} is not + * positive. + */ + public double leacockChodorow(String synsetId, String otherSynsetId, + int taxonomyDepth) { + validateIds(synsetId, otherSynsetId); + if (taxonomyDepth <= 0) { + throw new IllegalArgumentException( + "taxonomyDepth must be positive: " + taxonomyDepth); + } + final int distance = shortestDistance(synsetId, otherSynsetId); + if (distance < 0) { + return 0.0; + } + return -Math.log((distance + 1.0) / (2.0 * taxonomyDepth)); + } + + /** + * Finds the shortest distance between two synsets through a common ancestor. + * + * @param synsetId The first synset identifier. Must not be {@code null}. + * @param otherSynsetId The second synset identifier. Must not be {@code null}. + * @return The edge count of the shortest connecting path, or {@code -1} when no + * common ancestor exists. + * @throws IllegalArgumentException Thrown if {@code synsetId} or + * {@code otherSynsetId} is {@code null}. + */ + public int shortestDistance(String synsetId, String otherSynsetId) { + validateIds(synsetId, otherSynsetId); + final Map up = depthsAbove(synsetId); + final Map otherUp = depthsAbove(otherSynsetId); + int best = -1; + for (final Map.Entry common : up.entrySet()) { + final Integer otherDistance = otherUp.get(common.getKey()); + if (otherDistance != null) { + final int total = common.getValue() + otherDistance; + if (best < 0 || total < best) { + best = total; + } + } + } + return best; + } + + /** + * Validates the identifier arguments of the public measures. + * + * @param synsetId The first synset identifier. + * @param otherSynsetId The second synset identifier. + * @throws IllegalArgumentException Thrown if {@code synsetId} or + * {@code otherSynsetId} is {@code null}. + */ + private void validateIds(String synsetId, String otherSynsetId) { + if (synsetId == null) { + throw new IllegalArgumentException("synsetId must not be null"); + } + if (otherSynsetId == null) { + throw new IllegalArgumentException("otherSynsetId must not be null"); + } + } + + /** + * Collects every ancestor with its minimal upward distance, the synset itself included at + * distance zero. + * + * @param synsetId The synset to walk up from. + * @return The upward distance to each reachable ancestor, keyed by synset identifier. + */ + private Map depthsAbove(String synsetId) { + final Map depths = new HashMap<>(); + final Deque queue = new ArrayDeque<>(); + depths.put(synsetId, 0); + queue.add(synsetId); + while (!queue.isEmpty()) { + final String current = queue.remove(); + final int depth = depths.get(current); + for (final String parent : hypernyms(current)) { + if (!depths.containsKey(parent) || depths.get(parent) > depth + 1) { + depths.put(parent, depth + 1); + queue.add(parent); + } + } + } + return depths; + } + + /** + * Measures a synset's depth as the distance to its farthest ancestor, which is the taxonomy + * root reached the long way round when several paths lead up. + * + * @param synsetId The synset to measure. + * @return The edge count to the farthest ancestor, {@code 0} for a root. + */ + private int depthFromRoot(String synsetId) { + final Map above = depthsAbove(synsetId); + int deepest = 0; + for (final int distance : above.values()) { + deepest = Math.max(deepest, distance); + } + return deepest; + } + + /** + * Collects the synsets one taxonomy edge above a synset. + * + * @param synsetId The synset whose parents are collected. + * @return The plain hypernyms followed by the instance hypernyms. + */ + private Iterable hypernyms(String synsetId) { + final List parents = new ArrayList<>( + knowledgeBase.related(synsetId, WordNetRelation.HYPERNYM)); + parents.addAll(knowledgeBase.related(synsetId, WordNetRelation.INSTANCE_HYPERNYM)); + return parents; + } +} diff --git a/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/WnLmfReader.java b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/WnLmfReader.java new file mode 100644 index 0000000000..1222cfbfd1 --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/WnLmfReader.java @@ -0,0 +1,600 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package opennlp.wordnet; + +import java.io.BufferedInputStream; +import java.io.IOException; +import java.io.InputStream; +import java.nio.file.Files; +import java.nio.file.Path; +import java.util.ArrayList; +import java.util.HashMap; +import java.util.HashSet; +import java.util.LinkedHashMap; +import java.util.LinkedHashSet; +import java.util.List; +import java.util.Map; +import java.util.Set; +import javax.xml.XMLConstants; +import javax.xml.stream.Location; +import javax.xml.stream.XMLInputFactory; +import javax.xml.stream.XMLStreamConstants; +import javax.xml.stream.XMLStreamException; +import javax.xml.stream.XMLStreamReader; + +import opennlp.tools.util.InvalidFormatException; +import opennlp.tools.wordnet.LexicalKnowledgeBase; +import opennlp.tools.wordnet.Synset; +import opennlp.tools.wordnet.WordNetPOS; +import opennlp.tools.wordnet.WordNetRelation; + +/** + * Reads a WN-LMF XML document (the Global WordNet Association + * interchange format, used by + * Open English WordNet and many + * other language wordnets) into a {@link LexicalKnowledgeBase} using the JDK StAX parser. + * + *

It reads the subset of the format the contract serves: lexical entries, synsets with their + * definitions and typed relations, and sense relations, which are lifted to the synset level as + * documented on {@link WordNetRelation}. Elements outside that subset are skipped, as are + * relations of type {@code other} (the format's untyped escape hatch); any other unknown + * relation type fails loud.

+ * + *

The parser is hardened against XXE: DTD processing and external entities are disabled, so a + * DOCTYPE is skipped but nothing it names is fetched or resolved.

+ * + *

Malformed structure fails loud with an {@link InvalidFormatException} naming the resource + * and, where the parser provides one, the line; I/O failures propagate as {@link IOException}. + * Part-of-speech code {@code s} normalizes to {@link WordNetPOS#ADJECTIVE}, and a {@code similar} + * relation on a verb synset maps to {@link WordNetRelation#VERB_GROUP} rather than + * {@link WordNetRelation#SIMILAR_TO}. The returned lexicon is immutable and safe for concurrent + * lookups.

+ */ +public final class WnLmfReader { + + private static final Map RELATION_NAMES = relationNames(); + + /** The format's escape-hatch relation type; carries no type the contract can express. */ + private static final String OTHER_RELATION = "other"; + + /** The element declaring a lexical entry; opened and closed by the same handlers. */ + private static final String LEXICAL_ENTRY_ELEMENT = "LexicalEntry"; + + /** The element declaring a sense; opened and closed by the same handlers. */ + private static final String SENSE_ELEMENT = "Sense"; + + /** The element declaring a synset; opened and closed by the same handlers. */ + private static final String SYNSET_ELEMENT = "Synset"; + + /** Not instantiable. */ + private WnLmfReader() { + } + + /** + * Reads a WN-LMF XML file. + * + * @param file The XML file. Must not be {@code null} and must exist. + * @return The loaded lexicon. + * @throws IllegalArgumentException Thrown if {@code file} is {@code null} or missing. + * @throws InvalidFormatException Thrown if the document is malformed; the message names the + * file and, where available, the line. + * @throws IOException Thrown if reading the file fails. + */ + public static LexicalKnowledgeBase read(Path file) throws IOException { + if (file == null) { + throw new IllegalArgumentException("File must not be null"); + } + if (!Files.isRegularFile(file)) { + throw new IllegalArgumentException("File does not exist or is not a regular file: " + file); + } + try (InputStream in = new BufferedInputStream(Files.newInputStream(file))) { + return read(in, file.toString()); + } + } + + /** + * Reads a WN-LMF XML document from a stream. The stream is not closed. + * + * @param in The document stream. Must not be {@code null}. + * @param resourceName The name used in error messages. Must not be {@code null}. + * @return The loaded lexicon. + * @throws IllegalArgumentException Thrown if an argument is {@code null}. + * @throws InvalidFormatException Thrown if the document is malformed; the message names the + * resource and, where available, the line. + * @throws IOException Thrown if reading the stream fails. + */ + public static LexicalKnowledgeBase read(InputStream in, String resourceName) throws IOException { + if (in == null) { + throw new IllegalArgumentException("In must not be null"); + } + if (resourceName == null) { + throw new IllegalArgumentException("ResourceName must not be null"); + } + final Parser parser = new Parser(resourceName); + try { + final XMLStreamReader reader = hardenedFactory().createXMLStreamReader(in); + try { + parser.parse(reader); + } finally { + reader.close(); + } + } catch (XMLStreamException e) { + // StAX wraps a failing stream read in an XMLStreamException; surface it as the I/O failure. + final Throwable nested = e.getNestedException() == null ? e.getCause() + : e.getNestedException(); + if (nested instanceof IOException io) { + throw io; + } + throw parser.malformed(e.getLocation(), "XML error: " + e.getMessage(), e); + } + return parser.build(); + } + + /** + * Builds an XXE-hardened StAX factory: the DTD internal subset is not processed and external + * entities and the external DTD subset are denied, so a DOCTYPE is skipped but never resolved. + * + * @return The hardened factory. + */ + private static XMLInputFactory hardenedFactory() { + final XMLInputFactory factory = XMLInputFactory.newFactory(); + factory.setProperty(XMLInputFactory.SUPPORT_DTD, Boolean.FALSE); + factory.setProperty(XMLInputFactory.IS_SUPPORTING_EXTERNAL_ENTITIES, Boolean.FALSE); + factory.setProperty(XMLConstants.ACCESS_EXTERNAL_DTD, ""); + factory.setProperty(XMLInputFactory.IS_COALESCING, Boolean.TRUE); + factory.setXMLResolver((publicId, systemId, baseUri, namespace) -> { + throw new XMLStreamException("External entity resolution is disabled, refusing " + systemId); + }); + return factory; + } + + /** Holds the streaming parse state and performs post-parse resolution. */ + private static final class Parser { + + private final String resourceName; + + // Entry state. + private final Set entryIds = new HashSet<>(); + private final Map lemmaByEntryId = new HashMap<>(); + private final Map posByEntryId = new HashMap<>(); + private final Map synsetBySenseId = new HashMap<>(); + private final Map> senseOrder = + new LinkedHashMap<>(); + private final List senseRelations = new ArrayList<>(); + private final Map rawSynsets = new LinkedHashMap<>(); + // Fallback membership (entry ids per synset in document order) when members is absent. + private final Map> entryIdsBySynset = new HashMap<>(); + + // Cursor state. + private String currentEntryId; + private String currentEntryLemma; + private WordNetPOS currentEntryPos; + private String currentSenseId; + private RawSynset currentSynset; + + /** + * Creates a parser. + * + * @param resourceName The name used in error messages. + */ + Parser(String resourceName) { + this.resourceName = resourceName; + } + + /** + * Streams the document, dispatching start and end elements. + * + * @param reader The StAX reader. + * @throws XMLStreamException Thrown if the stream read fails. + * @throws InvalidFormatException Thrown if the document is malformed. + */ + void parse(XMLStreamReader reader) throws XMLStreamException, InvalidFormatException { + while (reader.hasNext()) { + final int event = reader.next(); + // A DTD event carries nothing that can affect parsing once the factory is hardened. + if (event == XMLStreamConstants.START_ELEMENT) { + startElement(reader); + } else if (event == XMLStreamConstants.END_ELEMENT) { + endElement(reader.getLocalName()); + } + } + } + + /** + * Handles one start element, updating cursor state and collecting raw entries, senses, and + * synsets. + * + * @param reader The StAX reader positioned on the start element. + * @throws XMLStreamException Thrown if reading element text fails. + * @throws InvalidFormatException Thrown if the element violates the format. + */ + private void startElement(XMLStreamReader reader) + throws XMLStreamException, InvalidFormatException { + final String name = reader.getLocalName(); + switch (name) { + case LEXICAL_ENTRY_ELEMENT -> { + currentEntryId = requireAttribute(reader, "id"); + if (!entryIds.add(currentEntryId)) { + throw malformed(reader.getLocation(), + "Duplicate lexical entry id " + currentEntryId, null); + } + currentEntryLemma = null; + currentEntryPos = null; + } + case "Lemma" -> { + if (currentEntryId == null) { + throw malformed(reader.getLocation(), "Lemma outside a LexicalEntry", null); + } + currentEntryLemma = requireAttribute(reader, "writtenForm"); + currentEntryPos = parsePos(requireAttribute(reader, "partOfSpeech"), + reader.getLocation()); + lemmaByEntryId.put(currentEntryId, currentEntryLemma); + posByEntryId.put(currentEntryId, currentEntryPos); + } + case SENSE_ELEMENT -> { + if (currentEntryLemma == null) { + throw malformed(reader.getLocation(), + "Sense before its entry's Lemma in LexicalEntry " + currentEntryId, null); + } + currentSenseId = requireAttribute(reader, "id"); + final String synsetId = requireAttribute(reader, "synset"); + if (synsetBySenseId.putIfAbsent(currentSenseId, synsetId) != null) { + throw malformed(reader.getLocation(), "Duplicate sense id " + currentSenseId, null); + } + entryIdsBySynset.computeIfAbsent(synsetId, unused -> new ArrayList<>(2)) + .add(currentEntryId); + final List order = senseOrder.computeIfAbsent( + InMemoryWordNetLexicon.LemmaKey.of(currentEntryLemma, currentEntryPos), + unused -> new ArrayList<>(2)); + if (!order.contains(synsetId)) { + order.add(synsetId); + } + } + case "SenseRelation" -> { + if (currentSenseId == null) { + throw malformed(reader.getLocation(), "SenseRelation outside a Sense", null); + } + senseRelations.add(new RawSenseRelation(currentSenseId, + requireAttribute(reader, "relType"), requireAttribute(reader, "target"), + line(reader.getLocation()))); + } + case SYNSET_ELEMENT -> { + final String id = requireAttribute(reader, "id"); + final WordNetPOS pos = parsePos(requireAttribute(reader, "partOfSpeech"), + reader.getLocation()); + currentSynset = new RawSynset(id, pos, reader.getAttributeValue(null, "members"), + line(reader.getLocation())); + if (rawSynsets.putIfAbsent(id, currentSynset) != null) { + throw malformed(reader.getLocation(), "Duplicate synset id " + id, null); + } + } + case "Definition" -> { + if (currentSynset != null && currentSynset.gloss == null) { + currentSynset.gloss = reader.getElementText(); + } + } + case "SynsetRelation" -> { + if (currentSynset == null) { + throw malformed(reader.getLocation(), "SynsetRelation outside a Synset", null); + } + final String relType = requireAttribute(reader, "relType"); + final String target = requireAttribute(reader, "target"); + // The escape-hatch type is a documented skip, not a rejection. + if (!OTHER_RELATION.equals(relType)) { + currentSynset.relations.add( + new RawRelation(relType, target, line(reader.getLocation()))); + } + } + default -> { + // Pronunciation, Form, Example, SyntacticBehaviour, ILIDefinition, and other + // elements outside the contract subset are skipped. + } + } + } + + /** + * Clears cursor state when a tracked element closes. + * + * @param name The local name of the closing element. + */ + private void endElement(String name) { + switch (name) { + case LEXICAL_ENTRY_ELEMENT -> { + currentEntryId = null; + currentEntryLemma = null; + currentEntryPos = null; + } + case SENSE_ELEMENT -> currentSenseId = null; + case SYNSET_ELEMENT -> currentSynset = null; + default -> { + // Nothing to close for skipped elements. + } + } + } + + /** + * Resolves the collected raw state into an immutable lexicon: validates sense targets, lifts + * sense relations to the synset level, and materializes the contract synsets. + * + * @return The loaded lexicon. + * @throws InvalidFormatException Thrown if a sense or relation references an undeclared + * target, or a synset has no members. + */ + LexicalKnowledgeBase build() throws InvalidFormatException { + // Every sense must point to a declared synset, with a consistent part of speech. + for (final Map.Entry sense : synsetBySenseId.entrySet()) { + final RawSynset target = rawSynsets.get(sense.getValue()); + if (target == null) { + throw malformed(null, + "Sense " + sense.getKey() + " references undeclared synset " + sense.getValue(), + null); + } + } + // Lift sense relations to the synset level. + for (final RawSenseRelation relation : senseRelations) { + if (OTHER_RELATION.equals(relation.relType)) { + continue; + } + final String sourceSynsetId = synsetBySenseId.get(relation.sourceSenseId); + final String targetSynsetId = synsetBySenseId.get(relation.targetSenseId); + if (targetSynsetId == null) { + throw malformed(null, "SenseRelation at line " + relation.line + " from sense " + + relation.sourceSenseId + " references undeclared sense " + relation.targetSenseId, + null); + } + final RawSynset source = rawSynsets.get(sourceSynsetId); + source.relations.add(new RawRelation(relation.relType, targetSynsetId, relation.line)); + } + // Resolve raw synsets into contract synsets. + final Map synsetsById = new LinkedHashMap<>(rawSynsets.size() * 2); + for (final RawSynset raw : rawSynsets.values()) { + final Map> relations = resolveRelations(raw); + synsetsById.put(raw.id, + new Synset(raw.id, raw.pos, memberLemmas(raw), raw.gloss == null ? "" : raw.gloss, + relations)); + } + return new InMemoryWordNetLexicon(synsetsById, senseOrder); + } + + /** + * Resolves a raw synset's relations into typed target-id lists, deduplicated in source order. + * + * @param raw The raw synset. + * @return The typed relations for the contract synset. + * @throws InvalidFormatException Thrown if a relation type is unknown or its target is + * undeclared. + */ + private Map> resolveRelations(RawSynset raw) + throws InvalidFormatException { + final Map> typed = new LinkedHashMap<>(); + for (final RawRelation relation : raw.relations) { + final WordNetRelation type = parseRelation(relation.relType, raw.pos, relation.line); + final RawSynset target = rawSynsets.get(relation.target); + if (target == null) { + throw malformed(null, "Relation " + relation.relType + " at line " + relation.line + + " on synset " + raw.id + " references undeclared synset " + relation.target, null); + } + // Share the synset table's id instance so only one copy of each id is retained. + typed.computeIfAbsent(type, unused -> new LinkedHashSet<>()).add(target.id); + } + final Map> relations = new LinkedHashMap<>(typed.size() * 2); + for (final Map.Entry> entry : typed.entrySet()) { + relations.put(entry.getKey(), List.copyOf(entry.getValue())); + } + return relations; + } + + /** + * Resolves a synset's member entry ids to their lemmas, from the {@code members} attribute + * when present and otherwise from the senses that pointed at the synset. + * + * @param raw The raw synset. + * @return The member lemmas in source order, deduplicated. + * @throws InvalidFormatException Thrown if the synset has no members, names an undeclared + * entry, or a member's part of speech disagrees with the synset's. + */ + private List memberLemmas(RawSynset raw) throws InvalidFormatException { + final List entryIds; + if (raw.members != null && !raw.members.isEmpty()) { + entryIds = LemmaFolding.splitOnSpaces(raw.members); + } else { + final List fromSenses = entryIdsBySynset.get(raw.id); + entryIds = fromSenses == null ? List.of() : fromSenses; + } + if (entryIds.isEmpty()) { + throw malformed(null, "Synset " + raw.id + " at line " + raw.line + + " has no member entries", null); + } + final List lemmas = new ArrayList<>(entryIds.size()); + for (final String entryId : entryIds) { + final String lemma = lemmaByEntryId.get(entryId); + if (lemma == null) { + throw malformed(null, "Synset " + raw.id + " at line " + raw.line + + " lists undeclared member entry " + entryId, null); + } + if (raw.pos != posByEntryId.get(entryId)) { + throw malformed(null, "Synset " + raw.id + " at line " + raw.line + + " has part of speech " + raw.pos + " but member entry " + entryId + + " has " + posByEntryId.get(entryId), null); + } + if (!lemmas.contains(lemma)) { + lemmas.add(lemma); + } + } + return lemmas; + } + + /** + * Maps a WN-LMF part-of-speech code to a {@link WordNetPOS}; code {@code s} normalizes to + * {@link WordNetPOS#ADJECTIVE}. + * + * @param code The part-of-speech code. + * @param location The parser location, for error reporting. + * @return The part of speech. + * @throws InvalidFormatException Thrown if the code is unknown. + */ + private WordNetPOS parsePos(String code, Location location) throws InvalidFormatException { + return switch (code) { + case "n" -> WordNetPOS.NOUN; + case "v" -> WordNetPOS.VERB; + case "a", "s" -> WordNetPOS.ADJECTIVE; + case "r" -> WordNetPOS.ADVERB; + default -> throw malformed(location, "Unknown part-of-speech code: " + code, null); + }; + } + + /** + * Maps a WN-LMF relation name to a {@link WordNetRelation}. A {@code similar} relation on a + * verb synset maps to {@link WordNetRelation#VERB_GROUP}, otherwise to + * {@link WordNetRelation#SIMILAR_TO}. + * + * @param relType The relation name. + * @param sourcePos The part of speech of the source synset. + * @param line The document line, for error reporting. + * @return The mapped relation. + * @throws InvalidFormatException Thrown if the relation name is unknown. + */ + private WordNetRelation parseRelation(String relType, WordNetPOS sourcePos, int line) + throws InvalidFormatException { + if ("similar".equals(relType)) { + return sourcePos == WordNetPOS.VERB ? WordNetRelation.VERB_GROUP + : WordNetRelation.SIMILAR_TO; + } + final WordNetRelation relation = RELATION_NAMES.get(relType); + if (relation == null) { + throw malformed(null, "Unknown relation type " + relType + " at line " + line, null); + } + return relation; + } + + /** + * Reads a required attribute from the current element. + * + * @param reader The StAX reader. + * @param attribute The attribute name. + * @return The non-empty attribute value. + * @throws InvalidFormatException Thrown if the attribute is absent or empty. + */ + private String requireAttribute(XMLStreamReader reader, String attribute) + throws InvalidFormatException { + final String value = reader.getAttributeValue(null, attribute); + if (value == null || value.isEmpty()) { + throw malformed(reader.getLocation(), "Element " + reader.getLocalName() + + " is missing required attribute " + attribute, null); + } + return value; + } + + /** + * Builds a malformed-document exception naming the resource and, when known, the line. + * + * @param location The parser location, or {@code null} when unavailable. + * @param message The failure detail. + * @param cause The underlying cause, or {@code null}. + * @return The exception to throw. + */ + InvalidFormatException malformed(Location location, String message, Throwable cause) { + final int line = line(location); + final String prefix = line < 0 ? "Malformed WN-LMF document " + resourceName + ": " + : "Malformed WN-LMF document " + resourceName + " at line " + line + ": "; + return cause == null ? new InvalidFormatException(prefix + message) + : new InvalidFormatException(prefix + message, cause); + } + + /** + * Extracts a line number from a parser location. + * + * @param location The location, or {@code null}. + * @return The line number, or {@code -1} when unknown. + */ + private static int line(Location location) { + return location == null ? -1 : location.getLineNumber(); + } + } + + private static final class RawSynset { + private final String id; + private final WordNetPOS pos; + private final String members; + private final int line; + private final List relations = new ArrayList<>(4); + private String gloss; + + /** + * Creates a raw synset gathered during parsing. + * + * @param id The synset id. + * @param pos The part of speech. + * @param members The {@code members} attribute value, or {@code null} when absent. + * @param line The document line. + */ + RawSynset(String id, WordNetPOS pos, String members, int line) { + this.id = id; + this.pos = pos; + this.members = members; + this.line = line; + } + } + + /** A parsed synset relation, kept until the target synset is known. */ + private record RawRelation(String relType, String target, int line) { + } + + /** A parsed sense relation, kept until both sense ids are known. */ + private record RawSenseRelation(String sourceSenseId, String relType, String targetSenseId, + int line) { + } + + /** + * Builds the WN-LMF relation-name to {@link WordNetRelation} table. + * + * @return The immutable name table. + */ + private static Map relationNames() { + final Map names = new HashMap<>(); + names.put("antonym", WordNetRelation.ANTONYM); + names.put("hypernym", WordNetRelation.HYPERNYM); + names.put("instance_hypernym", WordNetRelation.INSTANCE_HYPERNYM); + names.put("hyponym", WordNetRelation.HYPONYM); + names.put("instance_hyponym", WordNetRelation.INSTANCE_HYPONYM); + names.put("holo_member", WordNetRelation.MEMBER_HOLONYM); + names.put("holo_substance", WordNetRelation.SUBSTANCE_HOLONYM); + names.put("holo_part", WordNetRelation.PART_HOLONYM); + names.put("mero_member", WordNetRelation.MEMBER_MERONYM); + names.put("mero_substance", WordNetRelation.SUBSTANCE_MERONYM); + names.put("mero_part", WordNetRelation.PART_MERONYM); + names.put("attribute", WordNetRelation.ATTRIBUTE); + names.put("derivation", WordNetRelation.DERIVATIONALLY_RELATED); + names.put("entails", WordNetRelation.ENTAILMENT); + names.put("is_entailed_by", WordNetRelation.ENTAILED_BY); + names.put("causes", WordNetRelation.CAUSE); + names.put("is_caused_by", WordNetRelation.CAUSED_BY); + names.put("also", WordNetRelation.ALSO_SEE); + names.put("participle", WordNetRelation.PARTICIPLE); + names.put("pertainym", WordNetRelation.PERTAINYM); + names.put("domain_topic", WordNetRelation.DOMAIN_TOPIC); + names.put("has_domain_topic", WordNetRelation.MEMBER_OF_DOMAIN_TOPIC); + names.put("domain_region", WordNetRelation.DOMAIN_REGION); + names.put("has_domain_region", WordNetRelation.MEMBER_OF_DOMAIN_REGION); + // The usage domain carries both its current WN-LMF name and the legacy alias. + names.put("exemplifies", WordNetRelation.DOMAIN_USAGE); + names.put("domain_usage", WordNetRelation.DOMAIN_USAGE); + names.put("is_exemplified_by", WordNetRelation.MEMBER_OF_DOMAIN_USAGE); + names.put("has_domain_usage", WordNetRelation.MEMBER_OF_DOMAIN_USAGE); + return Map.copyOf(names); + } +} diff --git a/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/WndbReader.java b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/WndbReader.java new file mode 100644 index 0000000000..ed6b18ea8b --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/main/java/opennlp/wordnet/WndbReader.java @@ -0,0 +1,627 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package opennlp.wordnet; + +import java.io.IOException; +import java.nio.charset.StandardCharsets; +import java.nio.file.Files; +import java.nio.file.Path; +import java.util.ArrayList; +import java.util.HashMap; +import java.util.LinkedHashMap; +import java.util.LinkedHashSet; +import java.util.List; +import java.util.Map; + +import opennlp.tools.util.InvalidFormatException; +import opennlp.tools.wordnet.LexicalKnowledgeBase; +import opennlp.tools.wordnet.Synset; +import opennlp.tools.wordnet.WordNetPOS; +import opennlp.tools.wordnet.WordNetRelation; + +/** + * Reads a Princeton WordNet database directory in the + * WNDB format + * ({@code index.noun}, {@code data.noun}, and the corresponding pairs for verbs, adjectives, and + * adverbs) into a {@link LexicalKnowledgeBase}. + * + *

All eight index and data files must be present. License preamble lines (which begin with a + * space in the released files) are skipped. {@code index.sense} is not read, and the + * {@code *.exc} exception lists are the {@link MorphyLemmatizer} companion input, read + * separately.

+ * + *

Synset ids are minted as {@code wndb-}offset{@code -}pos from the data file's + * 8-digit byte offset and part-of-speech letter, for example {@code wndb-00001740-n}; the id is + * opaque to consumers. Adjective satellite lines normalize to {@link WordNetPOS#ADJECTIVE}, the + * syntactic markers the adjective files append ({@code (p)}, {@code (a)}, {@code (ip)}) are + * stripped, and underscores in lemmas become spaces. Sense order per lemma follows the index + * file's offset order.

+ * + *

Malformed content fails loud with an {@link InvalidFormatException} naming the file and + * line; I/O failures propagate as {@link IOException}. The returned lexicon is immutable and safe + * for concurrent lookups.

+ */ +public final class WndbReader { + + private static final Map POINTER_SYMBOLS = pointerSymbols(); + + /** The prefix of every synset id this reader mints. */ + private static final String SYNSET_ID_PREFIX = "wndb-"; + + /** Not instantiable. */ + private WndbReader() { + } + + /** + * Mints a synset id in this reader's scheme: the {@code wndb-} prefix, the 8-digit data-file + * byte offset, a hyphen, and the part-of-speech letter, for example {@code wndb-00001740-n}. + * + * @param offset The 8-digit synset offset field. + * @param posChar The WNDB part-of-speech letter. + * @return The minted synset id. + */ + private static String synsetId(String offset, char posChar) { + return SYNSET_ID_PREFIX + offset + '-' + posChar; + } + + /** + * Reads a WNDB database directory. + * + * @param directory The directory containing the eight index and data files. Must not be + * {@code null} and must exist. + * @return The loaded lexicon. + * @throws IllegalArgumentException Thrown if {@code directory} is {@code null} or not a + * directory. + * @throws InvalidFormatException Thrown if a database file is missing or any file is + * malformed; the message names the file and line. + * @throws IOException Thrown if reading a file fails. + */ + public static LexicalKnowledgeBase read(Path directory) throws IOException { + if (directory == null) { + throw new IllegalArgumentException("Directory must not be null"); + } + if (!Files.isDirectory(directory)) { + throw new IllegalArgumentException( + "Directory does not exist or is not a directory: " + directory); + } + final Map rawSynsets = new LinkedHashMap<>(); + for (final FilePos filePos : FilePos.values()) { + parseDataFile(directory, filePos, rawSynsets); + } + final Map synsetsById = resolve(rawSynsets); + final Map> senseOrder = new LinkedHashMap<>(); + for (final FilePos filePos : FilePos.values()) { + parseIndexFile(directory, filePos, rawSynsets, senseOrder); + } + return new InMemoryWordNetLexicon(synsetsById, senseOrder); + } + + /** The four part-of-speech file pairs of a WNDB directory. */ + private enum FilePos { + NOUN("noun", 'n', WordNetPOS.NOUN), + VERB("verb", 'v', WordNetPOS.VERB), + ADJECTIVE("adj", 'a', WordNetPOS.ADJECTIVE), + ADVERB("adv", 'r', WordNetPOS.ADVERB); + + private final String suffix; + private final char posChar; + private final WordNetPOS pos; + + /** + * Binds a part of speech to its file suffix and WNDB letter. + * + * @param suffix The file suffix, for example {@code noun}. + * @param posChar The WNDB part-of-speech letter. + * @param pos The mapped part of speech. + */ + FilePos(String suffix, char posChar, WordNetPOS pos) { + this.suffix = suffix; + this.posChar = posChar; + this.pos = pos; + } + } + + /** + * Parses one {@code data.*} file, collecting its synsets keyed by minted id. + * + * @param directory The database directory. + * @param filePos The part-of-speech file pair. + * @param rawSynsets The accumulating synset table. + * @throws IOException Thrown if the file is missing, malformed, or unreadable. + */ + private static void parseDataFile(Path directory, FilePos filePos, + Map rawSynsets) throws IOException { + final String fileName = "data." + filePos.suffix; + final byte[] bytes = readAll(directory.resolve(fileName), fileName); + int lineStart = 0; + int lineNumber = 0; + while (lineStart < bytes.length) { + lineNumber++; + int lineEnd = lineStart; + while (lineEnd < bytes.length && bytes[lineEnd] != '\n') { + lineEnd++; + } + // ISO-8859-1 decodes bytes one-to-one, keeping offsets exact for any released file. + final String line = + new String(bytes, lineStart, lineEnd - lineStart, StandardCharsets.ISO_8859_1); + if (!line.isEmpty() && line.charAt(0) != ' ') { + parseDataLine(line, lineStart, fileName, lineNumber, filePos, rawSynsets); + } + lineStart = lineEnd + 1; + } + } + + /** + * Parses one data-file synset line into a raw synset. + * + * @param line The decoded line, without its trailing newline. + * @param byteOffset The line's byte offset, matched against the line's own offset field. + * @param fileName The data file name, for error reporting. + * @param lineNumber The 1-based line number. + * @param filePos The part-of-speech file pair. + * @param rawSynsets The accumulating synset table. + * @throws InvalidFormatException Thrown if the line is malformed or its offset field disagrees + * with its byte position. + */ + private static void parseDataLine(String line, int byteOffset, String fileName, int lineNumber, + FilePos filePos, Map rawSynsets) + throws InvalidFormatException { + final Tokenizer tokens = new Tokenizer(line, fileName, lineNumber); + final String offsetField = tokens.next("synset_offset"); + if (parseOffset(offsetField, tokens) != byteOffset) { + throw malformed(fileName, lineNumber, "Synset offset field " + offsetField + + " disagrees with the actual byte position " + byteOffset); + } + tokens.next("lex_filenum (lexicographer file number)"); + final String ssType = tokens.next("ss_type (synset type)"); + final boolean validType = switch (filePos) { + case ADJECTIVE -> "a".equals(ssType) || "s".equals(ssType); + default -> ssType.length() == 1 && ssType.charAt(0) == filePos.posChar; + }; + if (!validType) { + throw malformed(fileName, lineNumber, + "Synset type " + ssType + " does not belong in " + fileName); + } + final int wordCount = tokens.nextInt("w_cnt (word count)", 16); + if (wordCount < 1) { + throw malformed(fileName, lineNumber, "Word count must be at least 1, got: " + wordCount); + } + final List lemmas = new ArrayList<>(wordCount); + for (int i = 0; i < wordCount; i++) { + final String lemma = cleanLemma(tokens.next("word"), fileName, lineNumber); + tokens.nextInt("lex_id (sense id within the lexicographer file)", 16); + if (!lemmas.contains(lemma)) { + lemmas.add(lemma); + } + } + final int pointerCount = tokens.nextInt("p_cnt (pointer count)", 10); + final List pointers = new ArrayList<>(pointerCount); + for (int i = 0; i < pointerCount; i++) { + final String symbol = tokens.next("pointer_symbol"); + final WordNetRelation relation = POINTER_SYMBOLS.get(symbol); + if (relation == null) { + throw malformed(fileName, lineNumber, "Undeclared pointer symbol: " + symbol); + } + final String targetOffset = tokens.next("pointer synset_offset"); + parseOffset(targetOffset, tokens); + final char targetPos = posChar(tokens.next("pointer pos"), tokens); + tokens.next("pointer source/target"); + pointers.add(new RawPointer(relation, synsetId(targetOffset, targetPos), lineNumber)); + } + if (filePos == FilePos.VERB) { + final int frameCount = tokens.nextInt("f_cnt (verb frame count)", 10); + for (int i = 0; i < frameCount; i++) { + tokens.next("frame marker"); + tokens.next("f_num (verb frame number)"); + tokens.next("w_num (word number)"); + } + } + final String gloss = tokens.gloss(); + final String id = synsetId(offsetField, filePos.posChar); + rawSynsets.put(id, new RawSynset(id, filePos.pos, lemmas, gloss, pointers, + fileName, lineNumber)); + } + + /** + * Resolves raw synsets into contract synsets, validating every pointer target. + * + * @param rawSynsets The parsed synsets keyed by id. + * @return The contract synsets keyed by id. + * @throws InvalidFormatException Thrown if a pointer targets a nonexistent synset. + */ + private static Map resolve(Map rawSynsets) + throws InvalidFormatException { + final Map synsetsById = new LinkedHashMap<>(rawSynsets.size() * 2); + for (final RawSynset raw : rawSynsets.values()) { + final Map> typed = new LinkedHashMap<>(); + for (final RawPointer pointer : raw.pointers) { + final RawSynset target = rawSynsets.get(pointer.targetId); + if (target == null) { + throw malformed(raw.fileName, pointer.lineNumber, "Synset " + raw.id + " has a " + + pointer.relation + " pointer to nonexistent synset " + pointer.targetId); + } + // Share the synset table's id instance so only one copy of each id is retained. + typed.computeIfAbsent(pointer.relation, unused -> new LinkedHashSet<>()) + .add(target.id); + } + final Map> relations = new LinkedHashMap<>(typed.size() * 2); + for (final Map.Entry> entry : typed.entrySet()) { + relations.put(entry.getKey(), List.copyOf(entry.getValue())); + } + synsetsById.put(raw.id, new Synset(raw.id, raw.pos, raw.lemmas, raw.gloss, relations)); + } + return synsetsById; + } + + /** + * Parses one {@code index.*} file, building the sense order per folded lemma key. + * + * @param directory The database directory. + * @param filePos The part-of-speech file pair. + * @param rawSynsets The resolved synset table, for offset validation. + * @param senses The accumulating sense-order map. + * @throws IOException Thrown if the file is missing, malformed, or unreadable. + */ + private static void parseIndexFile(Path directory, FilePos filePos, + Map rawSynsets, + Map> senses) + throws IOException { + final String fileName = "index." + filePos.suffix; + final byte[] bytes = readAll(directory.resolve(fileName), fileName); + final String content = new String(bytes, StandardCharsets.ISO_8859_1); + int lineNumber = 0; + int lineStart = 0; + while (lineStart < content.length()) { + lineNumber++; + int lineEnd = content.indexOf('\n', lineStart); + if (lineEnd < 0) { + lineEnd = content.length(); + } + final String line = content.substring(lineStart, lineEnd); + if (!line.isEmpty() && line.charAt(0) != ' ') { + parseIndexLine(line, fileName, lineNumber, filePos, rawSynsets, senses); + } + lineStart = lineEnd + 1; + } + } + + /** + * Parses one index-file line into a lemma's sense order. + * + * @param line The line to parse. + * @param fileName The index file name, for error reporting. + * @param lineNumber The 1-based line number. + * @param filePos The part-of-speech file pair. + * @param rawSynsets The resolved synset table, for offset validation. + * @param senses The accumulating sense-order map. + * @throws InvalidFormatException Thrown if the line is malformed or references an unknown + * offset. + */ + private static void parseIndexLine(String line, String fileName, int lineNumber, + FilePos filePos, Map rawSynsets, + Map> senses) + throws InvalidFormatException { + final Tokenizer tokens = new Tokenizer(line, fileName, lineNumber); + final String lemma = tokens.next("lemma"); + final String pos = tokens.next("pos"); + if (pos.length() != 1 || pos.charAt(0) != filePos.posChar) { + throw malformed(fileName, lineNumber, "Index pos " + pos + " does not belong in " + + fileName); + } + final int synsetCount = tokens.nextInt("synset_cnt (synset count)", 10); + if (synsetCount < 1) { + throw malformed(fileName, lineNumber, + "Synset count must be at least 1, got: " + synsetCount); + } + final int pointerTypeCount = tokens.nextInt("p_cnt (pointer count)", 10); + for (int i = 0; i < pointerTypeCount; i++) { + // The summary symbols are informational; the data file's pointers are authoritative. + tokens.next("ptr_symbol (pointer symbol)"); + } + tokens.next("sense_cnt (sense count)"); + tokens.next("tagsense_cnt (tagged-sense count)"); + final List order = new ArrayList<>(synsetCount); + for (int i = 0; i < synsetCount; i++) { + final String offset = tokens.next("synset_offset"); + parseOffset(offset, tokens); + final String synsetId = synsetId(offset, filePos.posChar); + if (!rawSynsets.containsKey(synsetId)) { + throw malformed(fileName, lineNumber, "Lemma " + lemma + " references offset " + offset + + " with no data." + filePos.suffix + " line"); + } + if (!order.contains(synsetId)) { + order.add(synsetId); + } + } + final InMemoryWordNetLexicon.LemmaKey key = + InMemoryWordNetLexicon.LemmaKey.of(lemma, filePos.pos); + final List existing = senses.get(key); + if (existing == null) { + senses.put(key, order); + } else { + // Two index lemmas can fold to one key; keep first-listed order and append the rest. + for (final String synsetId : order) { + if (!existing.contains(synsetId)) { + existing.add(synsetId); + } + } + } + } + + /** + * Strips the adjective syntactic markers ({@code (p)}, {@code (a)}, {@code (ip)}) and turns + * underscores into spaces. + * + * @param word The raw word field. + * @param fileName The data file name, for error reporting. + * @param lineNumber The 1-based line number. + * @return The cleaned lemma. + * @throws InvalidFormatException Thrown if the word carries an unknown marker or is empty. + */ + private static String cleanLemma(String word, String fileName, int lineNumber) + throws InvalidFormatException { + String cleaned = word; + if (cleaned.endsWith(")")) { + final int open = cleaned.lastIndexOf('('); + final String marker = open < 0 ? "" : cleaned.substring(open); + if (!"(p)".equals(marker) && !"(a)".equals(marker) && !"(ip)".equals(marker)) { + throw malformed(fileName, lineNumber, "Unknown syntactic marker on word: " + word); + } + cleaned = cleaned.substring(0, open); + } + if (cleaned.isEmpty()) { + throw malformed(fileName, lineNumber, "Empty word field"); + } + return cleaned.replace('_', ' '); + } + + /** + * Parses an 8-digit synset offset. + * + * @param offset The offset field. + * @param tokens The tokenizer, for error reporting. + * @return The offset as an integer. + * @throws InvalidFormatException Thrown if the field is not 8 digits. + */ + private static int parseOffset(String offset, Tokenizer tokens) throws InvalidFormatException { + if (offset.length() != 8) { + throw tokens.malformedToken("Synset offset must be 8 digits, got: " + offset); + } + int value = 0; + for (int i = 0; i < 8; i++) { + final char c = offset.charAt(i); + if (c < '0' || c > '9') { + throw tokens.malformedToken("Synset offset must be 8 digits, got: " + offset); + } + value = value * 10 + (c - '0'); + } + return value; + } + + /** + * Parses a pointer's one-letter part-of-speech code. + * + * @param pos The code field. + * @param tokens The tokenizer, for error reporting. + * @return One of {@code n}, {@code v}, {@code a}, {@code r}. + * @throws InvalidFormatException Thrown if the code is not one of those letters. + */ + private static char posChar(String pos, Tokenizer tokens) throws InvalidFormatException { + if (pos.length() == 1) { + final char c = pos.charAt(0); + if (c == 'n' || c == 'v' || c == 'a' || c == 'r') { + return c; + } + } + throw tokens.malformedToken("Pointer pos must be one of n, v, a, r, got: " + pos); + } + + /** + * Reads a required database file in full. + * + * @param file The file path. + * @param fileName The file name, for error reporting. + * @return The file bytes. + * @throws InvalidFormatException Thrown if the file is missing. + * @throws IOException Thrown if reading fails. + */ + private static byte[] readAll(Path file, String fileName) throws IOException { + if (!Files.isRegularFile(file)) { + throw new InvalidFormatException("Missing WNDB database file: " + file); + } + return Files.readAllBytes(file); + } + + /** + * Builds a malformed-file exception naming the file and line. + * + * @param fileName The file name. + * @param lineNumber The 1-based line number. + * @param message The failure detail. + * @return The exception to throw. + */ + private static InvalidFormatException malformed(String fileName, int lineNumber, + String message) { + return new InvalidFormatException( + "Malformed WNDB file " + fileName + " at line " + lineNumber + ": " + message); + } + + /** A cursor over one line's space-separated fields. */ + private static final class Tokenizer { + + private final String line; + private final String fileName; + private final int lineNumber; + private int position; + + /** + * Creates a tokenizer over one line. + * + * @param line The line to tokenize. + * @param fileName The file name, for error reporting. + * @param lineNumber The 1-based line number. + */ + Tokenizer(String line, String fileName, int lineNumber) { + this.line = line; + this.fileName = fileName; + this.lineNumber = lineNumber; + } + + /** + * Reads the next space-separated field. + * + * @param field The field name, for error reporting. + * @return The field value. + * @throws InvalidFormatException Thrown if the line is truncated before the field. + */ + String next(String field) throws InvalidFormatException { + while (position < line.length() && line.charAt(position) == ' ') { + position++; + } + if (position >= line.length()) { + throw malformed(fileName, lineNumber, "Truncated line, missing field: " + field); + } + final int start = position; + while (position < line.length() && line.charAt(position) != ' ') { + position++; + } + return line.substring(start, position); + } + + /** + * Reads the next field as an integer in the given radix. + * + * @param field The field name, for error reporting. + * @param radix The numeric radix. + * @return The parsed value. + * @throws InvalidFormatException Thrown if the field is missing or not a valid integer. + */ + int nextInt(String field, int radix) throws InvalidFormatException { + final String token = next(field); + try { + return Integer.parseInt(token, radix); + } catch (NumberFormatException e) { + throw new InvalidFormatException(malformed(fileName, lineNumber, + "Field " + field + " is not a base-" + radix + " integer: " + token).getMessage(), e); + } + } + + /** + * Reads the gloss: the remainder after the pipe separator, trimmed of surrounding spaces. + * + * @return The gloss text. + * @throws InvalidFormatException Thrown if the pipe separator is missing. + */ + String gloss() throws InvalidFormatException { + final String separator = next("gloss separator"); + if (!"|".equals(separator)) { + throw malformed(fileName, lineNumber, "Expected the | gloss separator, got: " + separator); + } + int start = position; + while (start < line.length() && line.charAt(start) == ' ') { + start++; + } + int end = line.length(); + while (end > start && line.charAt(end - 1) == ' ') { + end--; + } + return line.substring(start, end); + } + + /** + * Builds a malformed-file exception at this tokenizer's line. + * + * @param message The failure detail. + * @return The exception to throw. + */ + InvalidFormatException malformedToken(String message) { + return malformed(fileName, lineNumber, message); + } + } + + /** A parsed pointer line, kept until the target synset is known. */ + private record RawPointer(WordNetRelation relation, String targetId, int lineNumber) { + } + + private static final class RawSynset { + private final String id; + private final WordNetPOS pos; + private final List lemmas; + private final String gloss; + private final List pointers; + private final String fileName; + private final int lineNumber; + + /** + * Creates a raw synset gathered while parsing a data file. + * + * @param id The minted synset id. + * @param pos The part of speech. + * @param lemmas The member lemmas. + * @param gloss The gloss text. + * @param pointers The raw pointers to resolve. + * @param fileName The source file name. + * @param lineNumber The source line number. + */ + RawSynset(String id, WordNetPOS pos, List lemmas, String gloss, + List pointers, String fileName, int lineNumber) { + this.id = id; + this.pos = pos; + this.lemmas = lemmas; + this.gloss = gloss; + this.pointers = pointers; + this.fileName = fileName; + this.lineNumber = lineNumber; + } + } + + /** + * Builds the WNDB pointer-symbol to {@link WordNetRelation} table. + * + * @return The immutable symbol table. + */ + private static Map pointerSymbols() { + final Map symbols = new HashMap<>(); + symbols.put("!", WordNetRelation.ANTONYM); + symbols.put("@", WordNetRelation.HYPERNYM); + symbols.put("@i", WordNetRelation.INSTANCE_HYPERNYM); + symbols.put("~", WordNetRelation.HYPONYM); + symbols.put("~i", WordNetRelation.INSTANCE_HYPONYM); + symbols.put("#m", WordNetRelation.MEMBER_HOLONYM); + symbols.put("#s", WordNetRelation.SUBSTANCE_HOLONYM); + symbols.put("#p", WordNetRelation.PART_HOLONYM); + symbols.put("%m", WordNetRelation.MEMBER_MERONYM); + symbols.put("%s", WordNetRelation.SUBSTANCE_MERONYM); + symbols.put("%p", WordNetRelation.PART_MERONYM); + symbols.put("=", WordNetRelation.ATTRIBUTE); + symbols.put("+", WordNetRelation.DERIVATIONALLY_RELATED); + symbols.put("*", WordNetRelation.ENTAILMENT); + symbols.put(">", WordNetRelation.CAUSE); + symbols.put("^", WordNetRelation.ALSO_SEE); + symbols.put("$", WordNetRelation.VERB_GROUP); + symbols.put("&", WordNetRelation.SIMILAR_TO); + symbols.put("<", WordNetRelation.PARTICIPLE); + symbols.put("\\", WordNetRelation.PERTAINYM); + symbols.put(";c", WordNetRelation.DOMAIN_TOPIC); + symbols.put("-c", WordNetRelation.MEMBER_OF_DOMAIN_TOPIC); + symbols.put(";r", WordNetRelation.DOMAIN_REGION); + symbols.put("-r", WordNetRelation.MEMBER_OF_DOMAIN_REGION); + symbols.put(";u", WordNetRelation.DOMAIN_USAGE); + symbols.put("-u", WordNetRelation.MEMBER_OF_DOMAIN_USAGE); + return Map.copyOf(symbols); + } +} diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/ExpansionAssertions.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/ExpansionAssertions.java new file mode 100644 index 0000000000..c67d7c425e --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/ExpansionAssertions.java @@ -0,0 +1,42 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package opennlp.wordnet; + +import java.util.List; + +import opennlp.wordnet.LexicalExpander.Expansion; + +/** + * Shared lookup helpers for tests that assert on {@link LexicalExpander} output. + */ +final class ExpansionAssertions { + + /** Not instantiable. */ + private ExpansionAssertions() { + } + + /** + * Finds the first expansion of a term in an expansion list. + * + * @param expansions The expansions to search. Must not be {@code null}. + * @param term The exact term to find. Must not be {@code null}. + * @return The first expansion whose term equals {@code term}, or {@code null} when absent. + */ + static Expansion find(List expansions, String term) { + return expansions.stream().filter(e -> e.term().equals(term)).findFirst().orElse(null); + } +} diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/HypernymTyperTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/HypernymTyperTest.java new file mode 100644 index 0000000000..cba6828fc5 --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/HypernymTyperTest.java @@ -0,0 +1,118 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package opennlp.wordnet; + +import java.util.LinkedHashMap; +import java.util.Map; +import java.util.Optional; + +import org.junit.jupiter.api.Assertions; +import org.junit.jupiter.api.Test; + +/** + * Tests that {@link HypernymTyper} labels a word by its nearest anchored hypernym over + * the fixture taxonomy of {@link SynsetSimilarityTest}, follows instance hypernymy, + * prefers the closer of two anchors, and validates its arguments. + */ +public class HypernymTyperTest { + + /** + * @return A typer over the shared fixture taxonomy with person and location anchors. + * Never {@code null}. + */ + private static HypernymTyper typer() { + return new HypernymTyper(taxonomy(), + Map.of("person", "person", "location", "location")); + } + + /** + * @return The taxonomy shared with {@link SynsetSimilarityTest}. Never {@code null}. + */ + private static SynsetSimilarityTest.FixtureKnowledgeBase taxonomy() { + return SynsetSimilarityTest.taxonomy(); + } + + /** + * Verifies that a noun whose hypernym chain reaches an anchor receives that anchor's + * label, both through plain and instance hypernymy, and that the anchor lemma itself + * is typed with its own label at distance zero. + */ + @Test + void testTypesThroughHypernymAndInstanceChains() { + final HypernymTyper typer = typer(); + Assertions.assertEquals(Optional.of("person"), typer.type("chemist")); + Assertions.assertEquals(Optional.of("location"), typer.type("paris")); + Assertions.assertEquals(Optional.of("person"), typer.type("person")); + Assertions.assertEquals(Optional.of("location"), typer.typeSynset("n8")); + } + + /** + * Verifies that no label is produced when no sense of the word, or no ancestor of + * the synset, reaches an anchor. + */ + @Test + void testUnreachableAnchorsYieldEmpty() { + final HypernymTyper typer = typer(); + Assertions.assertEquals(Optional.empty(), typer.type("abstract")); + Assertions.assertEquals(Optional.empty(), typer.type("unknownword")); + Assertions.assertEquals(Optional.empty(), typer.typeSynset("n12")); + Assertions.assertEquals(Optional.empty(), typer.typeSynset("missing")); + } + + /** + * Verifies that the nearest anchor wins: with scientist registered as its own + * anchor, a chemist is a scientist rather than the more distant person. + */ + @Test + void testNearestAnchorWins() { + final Map anchors = new LinkedHashMap<>(); + anchors.put("person", "person"); + anchors.put("scientist", "scientist"); + final HypernymTyper typer = new HypernymTyper(taxonomy(), anchors); + Assertions.assertEquals(Optional.of("scientist"), typer.type("chemist")); + Assertions.assertEquals(Optional.of("scientist"), typer.type("scientist")); + // the walk is upward only, so an ancestor of an anchor is never typed by it + Assertions.assertEquals(Optional.empty(), typer.type("organism")); + } + + /** + * Verifies that invalid construction and query arguments are rejected: null or + * empty inputs, blank anchor entries, and an anchor lemma the knowledge base does + * not know. + */ + @Test + void testInvalidArguments() { + final Map anchors = Map.of("person", "person"); + Assertions.assertThrows(IllegalArgumentException.class, + () -> new HypernymTyper(null, anchors)); + Assertions.assertThrows(IllegalArgumentException.class, + () -> new HypernymTyper(taxonomy(), null)); + Assertions.assertThrows(IllegalArgumentException.class, + () -> new HypernymTyper(taxonomy(), Map.of())); + Assertions.assertThrows(IllegalArgumentException.class, + () -> new HypernymTyper(taxonomy(), Map.of(" ", "person"))); + Assertions.assertThrows(IllegalArgumentException.class, + () -> new HypernymTyper(taxonomy(), Map.of("person", "\u00A0"))); + Assertions.assertThrows(IllegalArgumentException.class, + () -> new HypernymTyper(taxonomy(), Map.of("notaword", "label"))); + final HypernymTyper typer = typer(); + Assertions.assertThrows(IllegalArgumentException.class, () -> typer.type(null)); + Assertions.assertThrows(IllegalArgumentException.class, () -> typer.type(" ")); + Assertions.assertThrows(IllegalArgumentException.class, () -> typer.typeSynset(null)); + } +} diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/InMemoryWordNetLexiconTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/InMemoryWordNetLexiconTest.java new file mode 100644 index 0000000000..8a238832fb --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/InMemoryWordNetLexiconTest.java @@ -0,0 +1,89 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package opennlp.wordnet; + +import java.util.List; +import java.util.Map; + +import org.junit.jupiter.api.Test; + +import opennlp.tools.wordnet.Synset; +import opennlp.tools.wordnet.WordNetPOS; +import opennlp.tools.wordnet.WordNetRelation; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertThrows; +import static org.junit.jupiter.api.Assertions.assertTrue; + +/** + * Exercises the constructor's referential-integrity validation directly, with deliberately + * inconsistent maps a reader would never produce: any future reader relies on these checks, + * so they are pinned independently of both existing readers. + */ +public class InMemoryWordNetLexiconTest { + + private static Synset synset(String id, Map> relations) { + return new Synset(id, WordNetPOS.NOUN, List.of("lemma"), "a gloss", relations); + } + + @Test + void testAcceptsConsistentMaps() { + final Synset a = synset("a", Map.of(WordNetRelation.HYPERNYM, List.of("b"))); + final Synset b = synset("b", Map.of()); + final InMemoryWordNetLexicon lexicon = new InMemoryWordNetLexicon( + Map.of("a", a, "b", b), + Map.of(InMemoryWordNetLexicon.LemmaKey.of("lemma", WordNetPOS.NOUN), List.of("a", "b"))); + assertEquals(2, lexicon.size()); + assertEquals(List.of(a, b), lexicon.lookup("lemma", WordNetPOS.NOUN)); + } + + @Test + void testRejectsKeyThatDoesNotMatchSynsetId() { + final Map table = Map.of("wrong-key", synset("real-id", Map.of())); + final IllegalArgumentException e = assertThrows(IllegalArgumentException.class, + () -> new InMemoryWordNetLexicon(table, Map.of())); + assertTrue(e.getMessage().contains("wrong-key")); + } + + @Test + void testRejectsDanglingRelationTarget() { + final Map table = + Map.of("a", synset("a", Map.of(WordNetRelation.HYPERNYM, List.of("nope")))); + final IllegalArgumentException e = assertThrows(IllegalArgumentException.class, + () -> new InMemoryWordNetLexicon(table, Map.of())); + assertTrue(e.getMessage().contains("nope")); + assertTrue(e.getMessage().contains("HYPERNYM")); + } + + @Test + void testRejectsSenseOrderEntryWithUnknownSynset() { + final Map table = Map.of("a", synset("a", Map.of())); + final Map> senseOrder = + Map.of(InMemoryWordNetLexicon.LemmaKey.of("lemma", WordNetPOS.NOUN), List.of("missing")); + final IllegalArgumentException e = assertThrows(IllegalArgumentException.class, + () -> new InMemoryWordNetLexicon(table, senseOrder)); + assertTrue(e.getMessage().contains("missing")); + assertTrue(e.getMessage().contains("lemma")); + } + + @Test + void testRejectsNullMaps() { + assertThrows(IllegalArgumentException.class, () -> new InMemoryWordNetLexicon(null, Map.of())); + assertThrows(IllegalArgumentException.class, + () -> new InMemoryWordNetLexicon(Map.of(), null)); + } +} diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LemmaFoldingTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LemmaFoldingTest.java new file mode 100644 index 0000000000..3d32eb4431 --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LemmaFoldingTest.java @@ -0,0 +1,66 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package opennlp.wordnet; + +import java.util.List; + +import org.junit.jupiter.api.Test; + +import opennlp.tools.wordnet.WordNetPOS; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertThrows; + +/** + * Pins the shared fold and split behavior every user of {@link LemmaFolding} depends on: + * the exception lists, the sense index keys, and the WN-LMF members parsing all fold and + * split through this one implementation. + */ +public class LemmaFoldingTest { + + @Test + void testFoldLowercasesWithRootLocaleAndTreatsUnderscoreAsSpace() { + assertEquals("mice", LemmaFolding.fold("MICE")); + assertEquals("domestic dog", LemmaFolding.fold("Domestic_Dog")); + assertEquals("attorney general", LemmaFolding.fold("attorney_general")); + assertEquals("dog", LemmaFolding.fold("dog")); + assertEquals("", LemmaFolding.fold("")); + } + + @Test + void testSplitOnSpacesCollapsesRunsAndIgnoresEdges() { + assertEquals(List.of("a", "b", "c"), LemmaFolding.splitOnSpaces("a b c")); + assertEquals(List.of("a", "b"), LemmaFolding.splitOnSpaces("a b")); + assertEquals(List.of("a"), LemmaFolding.splitOnSpaces("a")); + assertEquals(List.of("a"), LemmaFolding.splitOnSpaces(" a ")); + assertEquals(List.of(), LemmaFolding.splitOnSpaces("")); + assertEquals(List.of(), LemmaFolding.splitOnSpaces(" ")); + } + + @Test + void testLemmaKeyAndExceptionLookupAgreeOnTheFold() { + // The agreement that makes Morphy correct: a key built from a stored written form and a + // query folded at lookup time land on the same canonical shape. + assertEquals(InMemoryWordNetLexicon.LemmaKey.of("Domestic_Dog", WordNetPOS.NOUN), + InMemoryWordNetLexicon.LemmaKey.of(LemmaFolding.fold("DOMESTIC_DOG"), WordNetPOS.NOUN)); + } + + @Test + void testFoldRejectsNull() { + assertThrows(IllegalArgumentException.class, () -> LemmaFolding.fold(null)); + } +} diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LexicalExpanderLexiconTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LexicalExpanderLexiconTest.java new file mode 100644 index 0000000000..176c1142d7 --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LexicalExpanderLexiconTest.java @@ -0,0 +1,106 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package opennlp.wordnet; + +import java.util.List; + +import org.junit.jupiter.api.Test; + +import opennlp.tools.wordnet.LexicalKnowledgeBase; +import opennlp.tools.wordnet.WordNetPOS; +import opennlp.wordnet.LexicalExpander.Expansion; +import opennlp.wordnet.LexicalExpander.Kind; + +import static opennlp.wordnet.ExpansionAssertions.find; +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertNotNull; + +/** + * End-to-end expansion over the miniature lexicon fixtures: the WN-LMF and WNDB readers each + * feed the expander, and the Morphy lemmatizer bridges inflected input, exercising the whole + * stack the way a consumer wires it. + */ +public class LexicalExpanderLexiconTest { + + @Test + void testExpansionOverTheWnLmfLexicon() { + final LexicalExpander expander = + LexicalExpander.builder(WnLmfReaderTest.fixture()).build(); + + final List expansions = expander.expand("dog", WordNetPOS.NOUN); + final Expansion domesticDog = find(expansions, "domestic dog"); + assertEquals(Kind.SYNONYM, domesticDog.kind()); + assertEquals(1.0, domesticDog.weight()); + final Expansion canid = find(expansions, "canid"); + assertEquals(Kind.HYPERNYM, canid.kind()); + assertEquals(0.5, canid.weight()); + } + + @Test + void testExpansionOverTheWndbLexicon() { + final LexicalExpander expander = + LexicalExpander.builder(WndbReaderTest.fixture()).build(); + + final List expansions = expander.expand("dog", WordNetPOS.NOUN); + assertNotNull(find(expansions, "domestic dog"), "got " + expansions); + final Expansion canid = find(expansions, "canid"); + assertEquals(Kind.HYPERNYM, canid.kind()); + } + + @Test + void testUnderscoreQueryFoldsToTheMultiwordLexiconEntry() { + // The WNDB index stores the entry as "domestic_dog"; the reader folds it to "domestic dog" + // at load time. The expander must fold the underscore query the same way, so the entry's + // own synset expands and the query never surfaces as its own synonym. + final List expansions = LexicalExpander.builder(WndbReaderTest.fixture()) + .build().expand("domestic_dog", WordNetPOS.NOUN); + + assertEquals(List.of( + new Expansion("dog", Kind.SYNONYM, 0, 0, 1.0), + new Expansion("canid", Kind.HYPERNYM, 1, 0, 0.5)), expansions); + } + + @Test + void testReadersAgreeOnExpansions() { + final List lmf = LexicalExpander.builder(WnLmfReaderTest.fixture()) + .hypernymDepth(2).build().expand("mouse", WordNetPOS.NOUN); + final List wndb = LexicalExpander.builder(WndbReaderTest.fixture()) + .hypernymDepth(2).build().expand("mouse", WordNetPOS.NOUN); + + assertEquals( + lmf.stream().map(e -> e.term() + "|" + e.kind() + "|" + e.weight()).toList(), + wndb.stream().map(e -> e.term() + "|" + e.kind() + "|" + e.weight()).toList()); + assertNotNull(find(lmf, "rodent"), "got " + lmf); + } + + @Test + void testMorphyBridgesInflectedInput() { + final LexicalKnowledgeBase lexicon = WnLmfReaderTest.fixture(); + final LexicalExpander expander = LexicalExpander.builder(lexicon) + .lemmatizer(new MorphyLemmatizer(lexicon, MorphyExceptionsTest.fixture())) + .build(); + + // A regular inflection resolves by rule, an irregular one by the exception list. + final List dogs = expander.expand("dogs", WordNetPOS.NOUN); + assertEquals(Kind.SYNONYM, find(dogs, "dog").kind()); + assertNotNull(find(dogs, "canid"), "got " + dogs); + + final List mice = expander.expand("mice", WordNetPOS.NOUN); + assertEquals(Kind.SYNONYM, find(mice, "mouse").kind()); + assertNotNull(find(mice, "rodent"), "got " + mice); + } +} diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LexicalExpanderTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LexicalExpanderTest.java new file mode 100644 index 0000000000..85f8c700bb --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LexicalExpanderTest.java @@ -0,0 +1,349 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package opennlp.wordnet; + +import java.util.HashMap; +import java.util.List; +import java.util.Map; +import java.util.Optional; +import java.util.stream.Stream; + +import org.junit.jupiter.api.Test; +import org.junit.jupiter.params.ParameterizedTest; +import org.junit.jupiter.params.provider.Arguments; +import org.junit.jupiter.params.provider.MethodSource; + +import opennlp.tools.lemmatizer.Lemmatizer; +import opennlp.tools.wordnet.LexicalKnowledgeBase; +import opennlp.tools.wordnet.Synset; +import opennlp.tools.wordnet.WordNetPOS; +import opennlp.tools.wordnet.WordNetRelation; +import opennlp.wordnet.LexicalExpander.Expansion; +import opennlp.wordnet.LexicalExpander.Kind; + +import static opennlp.wordnet.ExpansionAssertions.find; +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertFalse; +import static org.junit.jupiter.api.Assertions.assertNotNull; +import static org.junit.jupiter.api.Assertions.assertNull; +import static org.junit.jupiter.api.Assertions.assertThrows; +import static org.junit.jupiter.api.Assertions.assertTrue; + +/** + * Behavioral tests over a hand-built lexicon whose graph shape is fully controlled: sense + * ranking, hypernym depth and decay, hyponym opt-in, deduplication, exclusion of the input, + * cycle termination, and configuration validation. + */ +public class LexicalExpanderTest { + + // dog: sense 1 = {dog, domestic dog} -> canid -> carnivore, with hyponym puppy; + // sense 2 = {dog, frank, hot dog} -> sausage. The verb sense = {dog, chase}. + // hot dog: the standalone multiword sense {hot dog, red hot}. + // alpha <-> beta form a malformed hypernym cycle. + // Lookups fold through LemmaFolding, exactly as the readers fold their keys at load time. + private static LexicalKnowledgeBase lexicon() { + final Map synsets = new HashMap<>(); + final Map> senses = new HashMap<>(); + + final Synset n1 = new Synset("n1", WordNetPOS.NOUN, List.of("dog", "domestic dog"), "canine", + Map.of(WordNetRelation.HYPERNYM, List.of("n2"), + WordNetRelation.HYPONYM, List.of("n4"))); + final Synset n2 = new Synset("n2", WordNetPOS.NOUN, List.of("canid"), "canid family", + Map.of(WordNetRelation.HYPERNYM, List.of("n3"))); + final Synset n3 = new Synset("n3", WordNetPOS.NOUN, List.of("carnivore"), "meat eater", + Map.of()); + final Synset n4 = new Synset("n4", WordNetPOS.NOUN, List.of("puppy"), "young dog", Map.of()); + final Synset n5 = new Synset("n5", WordNetPOS.NOUN, List.of("dog", "frank", "hot dog"), + "sausage in a bun", Map.of(WordNetRelation.HYPERNYM, List.of("n6"))); + final Synset n6 = new Synset("n6", WordNetPOS.NOUN, List.of("sausage"), "ground meat", + Map.of()); + final Synset v1 = new Synset("v1", WordNetPOS.VERB, List.of("dog", "chase"), "follow", + Map.of()); + final Synset m1 = new Synset("m1", WordNetPOS.NOUN, List.of("hot dog", "red hot"), + "grilled sausage", Map.of()); + final Synset c1 = new Synset("c1", WordNetPOS.NOUN, List.of("alpha"), "cycle start", + Map.of(WordNetRelation.HYPERNYM, List.of("c2"))); + final Synset c2 = new Synset("c2", WordNetPOS.NOUN, List.of("beta"), "cycle end", + Map.of(WordNetRelation.HYPERNYM, List.of("c1"))); + + for (final Synset synset : List.of(n1, n2, n3, n4, n5, n6, v1, m1, c1, c2)) { + synsets.put(synset.id(), synset); + } + senses.put("dog|NOUN", List.of(n1, n5)); + senses.put("dog|VERB", List.of(v1)); + senses.put("domestic dog|NOUN", List.of(n1)); + senses.put("hot dog|NOUN", List.of(m1)); + senses.put("alpha|NOUN", List.of(c1)); + + return new LexicalKnowledgeBase() { + @Override + public List lookup(String lemma, WordNetPOS pos) { + if (lemma == null || pos == null) { + throw new IllegalArgumentException("null"); + } + return senses.getOrDefault(LemmaFolding.fold(lemma) + "|" + pos, List.of()); + } + + @Override + public Optional synset(String synsetId) { + return Optional.ofNullable(synsets.get(synsetId)); + } + }; + } + + @Test + void testSynonymsAndHypernymsWithDefaultConfiguration() { + final List expansions = + LexicalExpander.builder(lexicon()).build().expand("dog", WordNetPOS.NOUN); + + final Expansion domesticDog = find(expansions, "domestic dog"); + assertEquals(Kind.SYNONYM, domesticDog.kind()); + assertEquals(1.0, domesticDog.weight()); + assertEquals(0, domesticDog.senseRank()); + + final Expansion canid = find(expansions, "canid"); + assertEquals(Kind.HYPERNYM, canid.kind()); + assertEquals(1, canid.depth()); + assertEquals(0.5, canid.weight()); + + final Expansion frank = find(expansions, "frank"); + assertEquals(Kind.SYNONYM, frank.kind()); + assertEquals(1, frank.senseRank()); + assertEquals(0.5, frank.weight()); + + // Depth 1 by default: the grandparent stays out, and so do hyponyms. + assertNull(find(expansions, "carnivore")); + assertNull(find(expansions, "puppy")); + } + + @Test + void testUnderscoreInputReachesTheSpaceFoldedLexiconEntry() { + // The lexicon keys "hot_dog" under its folded form "hot dog". The expander must fold the + // input the same way, so "hot dog" is excluded as the input itself and cannot displace the + // other member "red hot" from the capped result. + final List expansions = LexicalExpander.builder(lexicon()) + .maxExpansions(1).build().expand("hot_dog", WordNetPOS.NOUN); + + assertEquals(List.of(new Expansion("red hot", Kind.SYNONYM, 0, 0, 1.0)), expansions); + } + + @Test + void testTheInputTermIsNeverAnExpansion() { + for (final Expansion expansion : + LexicalExpander.builder(lexicon()).build().expand("dog", WordNetPOS.NOUN)) { + assertFalse(expansion.term().equalsIgnoreCase("dog"), "got " + expansion); + } + } + + @Test + void testDeeperHypernymWalkDecaysPerStep() { + final List expansions = LexicalExpander.builder(lexicon()) + .hypernymDepth(2).build().expand("dog", WordNetPOS.NOUN); + + final Expansion carnivore = find(expansions, "carnivore"); + assertEquals(2, carnivore.depth()); + assertEquals(0.25, carnivore.weight()); + } + + @Test + void testHyponymsAreOptIn() { + final List expansions = LexicalExpander.builder(lexicon()) + .includeHyponyms(true).build().expand("dog", WordNetPOS.NOUN); + + final Expansion puppy = find(expansions, "puppy"); + assertEquals(Kind.HYPONYM, puppy.kind()); + assertEquals(0.5, puppy.weight()); + } + + @Test + void testMaxSensesLimitsToTheMostSalient() { + final List expansions = LexicalExpander.builder(lexicon()) + .maxSenses(1).build().expand("dog", WordNetPOS.NOUN); + + assertNull(find(expansions, "frank")); + assertNotNull(find(expansions, "domestic dog")); + } + + @Test + void testAllPosExpansionIncludesVerbSynonyms() { + final List expansions = + LexicalExpander.builder(lexicon()).build().expand("dog"); + + assertNotNull(find(expansions, "chase")); + assertNotNull(find(expansions, "domestic dog")); + } + + @Test + void testCyclicHypernymDataTerminates() { + final List expansions = LexicalExpander.builder(lexicon()) + .hypernymDepth(10).build().expand("alpha", WordNetPOS.NOUN); + + final Expansion beta = find(expansions, "beta"); + assertEquals(1, beta.depth()); + // The cycle leads back to alpha's own synset, which is visited and the input besides. + assertEquals(1, expansions.size()); + } + + @Test + void testDeduplicationKeepsTheHighestWeight() { + // "domestic dog" reaches n1 at rank 0; its synonym "dog" is the only other member and + // must appear once with the rank-0 weight even though deeper paths could yield it again. + final List expansions = LexicalExpander.builder(lexicon()) + .hypernymDepth(2).build().expand("domestic dog", WordNetPOS.NOUN); + + final Expansion dog = find(expansions, "dog"); + assertEquals(1.0, dog.weight()); + assertEquals(1, expansions.stream().filter(e -> e.term().equals("dog")).count()); + } + + @Test + void testOrderingIsWeightDescendingAndStable() { + final List expansions = LexicalExpander.builder(lexicon()) + .hypernymDepth(2).includeHyponyms(true).build().expand("dog", WordNetPOS.NOUN); + + for (int i = 1; i < expansions.size(); i++) { + assertTrue(expansions.get(i - 1).weight() >= expansions.get(i).weight(), + "weights must not increase: " + expansions); + } + assertEquals(expansions, + LexicalExpander.builder(lexicon()).hypernymDepth(2).includeHyponyms(true).build() + .expand("dog", WordNetPOS.NOUN)); + } + + @Test + void testMaxExpansionsCapsAfterRanking() { + final List expansions = LexicalExpander.builder(lexicon()) + .hypernymDepth(2).includeHyponyms(true).maxExpansions(2).build() + .expand("dog", WordNetPOS.NOUN); + + assertEquals(2, expansions.size()); + assertEquals(1.0, expansions.get(0).weight()); + } + + @Test + void testUnknownTermExpandsToNothing() { + assertEquals(List.of(), + LexicalExpander.builder(lexicon()).build().expand("xyzzy", WordNetPOS.NOUN)); + } + + @Test + void testLemmatizerFallbackExpandsInflectedInput() { + final LexicalExpander expander = LexicalExpander.builder(lexicon()) + .lemmatizer(new Lemmatizer() { + @Override + public String[] lemmatize(String[] tokens, String[] tags) { + final String[] lemmas = new String[tokens.length]; + for (int i = 0; i < tokens.length; i++) { + lemmas[i] = "dogs".equals(tokens[i]) ? "dog" : "O"; + } + return lemmas; + } + + @Override + public List> lemmatize(List tokens, List tags) { + throw new UnsupportedOperationException(); + } + }) + .build(); + + final List expansions = expander.expand("dogs", WordNetPOS.NOUN); + // The lemma itself surfaces as a synonym, along with the rest of its synsets. + final Expansion dog = find(expansions, "dog"); + assertEquals(Kind.SYNONYM, dog.kind()); + assertEquals(1.0, dog.weight()); + assertNotNull(find(expansions, "domestic dog")); + + assertEquals(List.of(), expander.expand("cats", WordNetPOS.NOUN)); + } + + @Test + void testValidationFailsLoudly() { + assertThrows(IllegalArgumentException.class, () -> LexicalExpander.builder(null)); + final LexicalExpander.Builder builder = LexicalExpander.builder(lexicon()); + assertThrows(IllegalArgumentException.class, () -> builder.lemmatizer(null)); + assertThrows(IllegalArgumentException.class, () -> builder.maxSenses(0)); + assertThrows(IllegalArgumentException.class, () -> builder.hypernymDepth(-1)); + assertThrows(IllegalArgumentException.class, () -> builder.maxExpansions(0)); + assertThrows(IllegalArgumentException.class, () -> builder.senseDecay(0)); + assertThrows(IllegalArgumentException.class, () -> builder.senseDecay(1.5)); + assertThrows(IllegalArgumentException.class, () -> builder.depthDecay(0)); + assertThrows(IllegalArgumentException.class, () -> builder.depthDecay(1.5)); + + final LexicalExpander expander = builder.build(); + assertThrows(IllegalArgumentException.class, () -> expander.expand(null)); + assertThrows(IllegalArgumentException.class, () -> expander.expand(" ")); + assertThrows(IllegalArgumentException.class, () -> expander.expand("dog", null)); + } + + /** + * Verifies that decay products which underflow to zero are dropped instead of emitted: + * with the smallest positive depth decay, the first hypernym level keeps the smallest + * positive weight while the second level underflows to zero and never appears, and no + * reported expansion carries a weight outside the documented range. + */ + @Test + void testUnderflowedWeightsAreDropped() { + final List expansions = LexicalExpander.builder(lexicon()) + .depthDecay(Double.MIN_VALUE).hypernymDepth(2).maxSenses(1).build() + .expand("domestic dog", WordNetPOS.NOUN); + + final Expansion canid = find(expansions, "canid"); + assertNotNull(canid); + assertEquals(Double.MIN_VALUE, canid.weight(), 0.0); + assertNull(find(expansions, "carnivore")); + for (final Expansion expansion : expansions) { + assertTrue(expansion.weight() > 0.0 && expansion.weight() <= 1.0, + "weight out of (0, 1]: " + expansion); + } + } + + private static Stream invalidExpansions() { + return Stream.of( + Arguments.of(null, Kind.SYNONYM, 0, 0, 1.0), + Arguments.of(" ", Kind.SYNONYM, 0, 0, 1.0), + Arguments.of("dog", null, 0, 0, 1.0), + Arguments.of("dog", Kind.SYNONYM, -1, 0, 1.0), + Arguments.of("dog", Kind.SYNONYM, 0, -1, 1.0), + Arguments.of("dog", Kind.SYNONYM, 0, 0, 0.0), + Arguments.of("dog", Kind.SYNONYM, 0, 0, 1.5), + Arguments.of("dog", Kind.SYNONYM, 0, 0, Double.NaN)); + } + + /** + * Verifies that the {@link Expansion} record rejects every component + * outside its documented range with a loud exception. + */ + @ParameterizedTest + @MethodSource("invalidExpansions") + void testExpansionValidatesItsComponents(String term, Kind kind, int depth, int senseRank, + double weight) { + assertThrows(IllegalArgumentException.class, + () -> new Expansion(term, kind, depth, senseRank, weight)); + } + + /** Verifies that a fully valid component set is accepted. */ + @Test + void testExpansionAcceptsValidComponents() { + final Expansion expansion = new Expansion("dog", Kind.HYPERNYM, 2, 1, 0.25); + + assertEquals("dog", expansion.term()); + assertEquals(Kind.HYPERNYM, expansion.kind()); + assertEquals(2, expansion.depth()); + assertEquals(1, expansion.senseRank()); + assertEquals(0.25, expansion.weight()); + } +} diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LexicalExpansionUsageExampleTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LexicalExpansionUsageExampleTest.java new file mode 100644 index 0000000000..c89a62499a --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LexicalExpansionUsageExampleTest.java @@ -0,0 +1,154 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package opennlp.wordnet; + +import java.util.HashMap; +import java.util.List; +import java.util.Map; +import java.util.Optional; + +import org.junit.jupiter.api.Assertions; +import org.junit.jupiter.api.Test; + +import opennlp.tools.wordnet.LexicalKnowledgeBase; +import opennlp.tools.wordnet.Synset; +import opennlp.tools.wordnet.WordNetPOS; +import opennlp.tools.wordnet.WordNetRelation; +import opennlp.wordnet.LexicalExpander.Expansion; +import opennlp.wordnet.LexicalExpander.Kind; + +import static opennlp.wordnet.ExpansionAssertions.find; + +/** + * Runs the manual's lexical expansion and synset similarity examples (docbkx + * {@code wordnet.xml}) verbatim: every value the chapter states is asserted here, so a + * change breaking this test breaks the manual. The taxonomy is a hand-built miniature + * matching the shapes used elsewhere in this module's tests. + */ +public class LexicalExpansionUsageExampleTest { + + /** + * dog sense 1 = {dog, domestic dog} -> canid; sense 2 = {dog, frank, hot dog} -> sausage. + */ + private static LexicalKnowledgeBase dogTaxonomy() { + final Map synsets = new HashMap<>(); + final Map> senses = new HashMap<>(); + + final Synset n1 = new Synset("n1", WordNetPOS.NOUN, List.of("dog", "domestic dog"), "canine", + Map.of(WordNetRelation.HYPERNYM, List.of("n2"))); + final Synset n2 = new Synset("n2", WordNetPOS.NOUN, List.of("canid"), "canid family", Map.of()); + final Synset n5 = new Synset("n5", WordNetPOS.NOUN, List.of("dog", "frank", "hot dog"), + "sausage in a bun", Map.of(WordNetRelation.HYPERNYM, List.of("n6"))); + final Synset n6 = new Synset("n6", WordNetPOS.NOUN, List.of("sausage"), "ground meat", + Map.of()); + + for (final Synset synset : List.of(n1, n2, n5, n6)) { + synsets.put(synset.id(), synset); + } + senses.put("dog|NOUN", List.of(n1, n5)); + senses.put("domestic dog|NOUN", List.of(n1)); + + return new LexicalKnowledgeBase() { + @Override + public List lookup(String lemma, WordNetPOS pos) { + if (lemma == null) { + throw new IllegalArgumentException("lemma must not be null"); + } + if (pos == null) { + throw new IllegalArgumentException("pos must not be null"); + } + return senses.getOrDefault(LemmaFolding.fold(lemma) + "|" + pos, List.of()); + } + + @Override + public Optional synset(String synsetId) { + return Optional.ofNullable(synsets.get(synsetId)); + } + }; + } + + /** + * chemist -> scientist -> person -> organism -> physical -> entity; city -> location -> + * physical. + */ + private static LexicalKnowledgeBase similarityTaxonomy() { + final Map byId = new HashMap<>(); + add(byId, "n1", "entity"); + add(byId, "n2", "physical", "n1"); + add(byId, "n3", "organism", "n2"); + add(byId, "n4", "person", "n3"); + add(byId, "n5", "scientist", "n4"); + add(byId, "n6", "chemist", "n5"); + add(byId, "n7", "location", "n2"); + add(byId, "n8", "city", "n7"); + return new LexicalKnowledgeBase() { + @Override + public List lookup(String lemma, WordNetPOS pos) { + return List.of(); + } + + @Override + public Optional synset(String synsetId) { + return Optional.ofNullable(byId.get(synsetId)); + } + }; + } + + private static void add(Map byId, String id, String lemma, String... parents) { + final Map> relations = parents.length == 0 + ? Map.of() : Map.of(WordNetRelation.HYPERNYM, List.of(parents)); + byId.put(id, new Synset(id, WordNetPOS.NOUN, List.of(lemma), "fixture", relations)); + } + + /** + * Default expansion of noun {@code dog}: synonym and depth-1 hypernym weights. + */ + @Test + void testExpandDogNoun() { + final List expansions = + LexicalExpander.builder(dogTaxonomy()).build().expand("dog", WordNetPOS.NOUN); + + final Expansion domesticDog = find(expansions, "domestic dog"); + Assertions.assertNotNull(domesticDog); + Assertions.assertEquals(Kind.SYNONYM, domesticDog.kind()); + Assertions.assertEquals(1.0, domesticDog.weight()); + Assertions.assertEquals(0, domesticDog.senseRank()); + + final Expansion canid = find(expansions, "canid"); + Assertions.assertNotNull(canid); + Assertions.assertEquals(Kind.HYPERNYM, canid.kind()); + Assertions.assertEquals(1, canid.depth()); + Assertions.assertEquals(0.5, canid.weight()); + + final Expansion frank = find(expansions, "frank"); + Assertions.assertNotNull(frank); + Assertions.assertEquals(Kind.SYNONYM, frank.kind()); + Assertions.assertEquals(1, frank.senseRank()); + Assertions.assertEquals(0.5, frank.weight()); + } + + /** + * Path and Wu-Palmer scores on the miniature scientist/city taxonomy. + */ + @Test + void testSynsetSimilarityScores() { + final SynsetSimilarity similarity = new SynsetSimilarity(similarityTaxonomy()); + Assertions.assertEquals(0.5, similarity.path("n6", "n5"), 1e-9); + Assertions.assertEquals(8.0 / 9.0, similarity.wuPalmer("n5", "n6"), 1e-9); + } +} diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LexiconConcurrencyTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LexiconConcurrencyTest.java new file mode 100644 index 0000000000..0292f7ffe2 --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/LexiconConcurrencyTest.java @@ -0,0 +1,91 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package opennlp.wordnet; + +import java.util.List; +import java.util.Queue; +import java.util.concurrent.ConcurrentLinkedQueue; +import java.util.concurrent.CountDownLatch; +import java.util.concurrent.TimeUnit; + +import org.junit.jupiter.api.Test; + +import opennlp.tools.wordnet.LexicalKnowledgeBase; +import opennlp.tools.wordnet.WordNetPOS; +import opennlp.tools.wordnet.WordNetRelation; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertTrue; + +/** + * Exercises the immutable-after-load contract: one loaded lexicon serves many threads issuing + * concurrent lookups, and every thread observes exactly the single-threaded results. + */ +public class LexiconConcurrencyTest { + + private static final int THREADS = 8; + private static final int ITERATIONS = 500; + + @Test + void testConcurrentLookupsSeeConsistentResults() throws InterruptedException { + final LexicalKnowledgeBase lexicon = WndbReaderTest.fixture(); + final CountDownLatch start = new CountDownLatch(1); + final CountDownLatch done = new CountDownLatch(THREADS); + final Queue problems = new ConcurrentLinkedQueue<>(); + for (int t = 0; t < THREADS; t++) { + final Thread thread = new Thread(() -> { + try { + start.await(); + for (int i = 0; i < ITERATIONS; i++) { + verifyOnce(lexicon, problems); + } + } catch (InterruptedException e) { + Thread.currentThread().interrupt(); + problems.add("Interrupted: " + e); + } catch (RuntimeException e) { + problems.add("Unexpected exception: " + e); + } finally { + done.countDown(); + } + }); + thread.setDaemon(true); + thread.start(); + } + start.countDown(); + assertTrue(done.await(60, TimeUnit.SECONDS), "Worker threads must finish in time"); + assertEquals(List.of(), List.copyOf(problems)); + } + + private static void verifyOnce(LexicalKnowledgeBase lexicon, Queue problems) { + if (!"wndb-00001075-n".equals(lexicon.lookup("dog", WordNetPOS.NOUN).get(0).id())) { + problems.add("Wrong dog lookup"); + } + if (lexicon.lookup("run", WordNetPOS.NOUN).size() != 2) { + problems.add("Wrong run sense count"); + } + if (!List.of("wndb-00001160-n") + .equals(lexicon.related("wndb-00001075-n", WordNetRelation.HYPERNYM))) { + problems.add("Wrong dog hypernym"); + } + if (lexicon.contains("zebra", WordNetPOS.NOUN)) { + problems.add("Phantom zebra"); + } + if (!lexicon.contains("walk", WordNetPOS.VERB)) { + problems.add("Missing walk verb"); + } + } +} diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/MorphyExceptionsTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/MorphyExceptionsTest.java new file mode 100644 index 0000000000..b3aaf78ba2 --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/MorphyExceptionsTest.java @@ -0,0 +1,124 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package opennlp.wordnet; + +import java.io.IOException; +import java.nio.file.Files; +import java.nio.file.Path; +import java.util.List; + +import org.junit.jupiter.api.Test; +import org.junit.jupiter.api.io.TempDir; +import org.junit.jupiter.params.ParameterizedTest; +import org.junit.jupiter.params.provider.CsvSource; + +import opennlp.tools.util.InvalidFormatException; +import opennlp.tools.wordnet.WordNetPOS; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertThrows; +import static org.junit.jupiter.api.Assertions.assertTrue; + +public class MorphyExceptionsTest { + + static MorphyExceptions fixture() { + try { + return MorphyExceptions.load(WndbReaderTest.fixtureDirectory()); + } catch (IOException e) { + throw new IllegalStateException("Unexpected IOException reading the fixture lists", e); + } + } + + /** + * Writes the standard one-entry exception list for each part of speech into + * {@code directory}. Tests that need a variation overwrite or delete individual files + * afterwards. + * + * @param directory The directory to receive {@code noun.exc}, {@code verb.exc}, + * {@code adj.exc}, and {@code adv.exc}. + * @throws IOException Thrown if writing a file fails. + */ + private static void writeStandardLists(Path directory) throws IOException { + Files.writeString(directory.resolve("noun.exc"), "mice mouse\n"); + Files.writeString(directory.resolve("verb.exc"), "went go\n"); + Files.writeString(directory.resolve("adj.exc"), "better good\n"); + Files.writeString(directory.resolve("adv.exc"), "best well\n"); + } + + @ParameterizedTest + @CsvSource(nullValues = "unknown", value = { + "mice, NOUN, mouse", + "went, VERB, go", + "better, ADJECTIVE, good", + "best, ADVERB, well", + // Entries are part-of-speech scoped: went is only a verb exception. + "went, NOUN, unknown", + "dog, NOUN, unknown", + }) + void testLookupPerPartOfSpeech(String form, WordNetPOS pos, String lemma) { + final List expected = lemma == null ? List.of() : List.of(lemma); + assertEquals(expected, fixture().lookup(form, pos)); + } + + @Test + void testLookupFoldsCase() { + assertEquals(List.of("mouse"), fixture().lookup("Mice", WordNetPOS.NOUN)); + assertEquals(List.of("mouse"), fixture().lookup("MICE", WordNetPOS.NOUN)); + } + + @Test + void testLookupRejectsNulls() { + final MorphyExceptions exceptions = fixture(); + assertThrows(IllegalArgumentException.class, + () -> exceptions.lookup(null, WordNetPOS.NOUN)); + assertThrows(IllegalArgumentException.class, () -> exceptions.lookup("mice", null)); + } + + @Test + void testLoadRejectsNullAndMissingDirectory(@TempDir Path tempDir) { + assertThrows(IllegalArgumentException.class, () -> MorphyExceptions.load(null)); + assertThrows(IllegalArgumentException.class, + () -> MorphyExceptions.load(tempDir.resolve("absent"))); + } + + @Test + void testLoadRejectsMissingFile(@TempDir Path tempDir) throws IOException { + writeStandardLists(tempDir); + Files.delete(tempDir.resolve("adv.exc")); + final InvalidFormatException e = assertThrows(InvalidFormatException.class, + () -> MorphyExceptions.load(tempDir)); + assertTrue(e.getMessage().contains("adv.exc")); + } + + @Test + void testLoadRejectsMalformedLine(@TempDir Path tempDir) throws IOException { + writeStandardLists(tempDir); + Files.writeString(tempDir.resolve("noun.exc"), "mice mouse\nlonely\n"); + final InvalidFormatException e = assertThrows(InvalidFormatException.class, + () -> MorphyExceptions.load(tempDir)); + assertTrue(e.getMessage().contains("noun.exc")); + assertTrue(e.getMessage().contains("line 2")); + } + + @Test + void testMultipleBaseFormsKeepFileOrder(@TempDir Path tempDir) throws IOException { + writeStandardLists(tempDir); + Files.writeString(tempDir.resolve("noun.exc"), "axes axis ax\n"); + assertEquals(List.of("axis", "ax"), + MorphyExceptions.load(tempDir).lookup("axes", WordNetPOS.NOUN)); + } +} diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/MorphyLemmatizerTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/MorphyLemmatizerTest.java new file mode 100644 index 0000000000..24df815548 --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/MorphyLemmatizerTest.java @@ -0,0 +1,188 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package opennlp.wordnet; + +import java.util.List; + +import org.junit.jupiter.api.Test; +import org.junit.jupiter.params.ParameterizedTest; +import org.junit.jupiter.params.provider.CsvSource; + +import opennlp.tools.wordnet.WordNetPOS; + +import static org.junit.jupiter.api.Assertions.assertArrayEquals; +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertThrows; + +public class MorphyLemmatizerTest { + + private static MorphyLemmatizer morphy() { + return new MorphyLemmatizer(WndbReaderTest.fixture(), MorphyExceptionsTest.fixture()); + } + + private static String one(String token, String tag) { + return morphy().lemmatize(new String[] {token}, new String[] {tag})[0]; + } + + @ParameterizedTest + @CsvSource({ + // Irregular forms resolve through the exception lists. + "mice, NN, mouse", + "Mice, NNS, mouse", + "men, NNS, man", + "ran, VBD, run", + "running, VBG, run", + "went, VBD, go", + "gone, VBN, go", + "best, RBS, well", + // Regular detachments, validated against the lexicon. + "dogs, NNS, dog", + "boxes, NNS, box", + "berries, NNS, berry", + "runs, NNS, run", + "runs, VBZ, run", + "walked, VBD, walk", + "walking, VBG, walk", + "walks, VBZ, walk", + "moved, VBD, move", + "taller, JJR, tall", + "tallest, JJS, tall", + "larger, JJR, large", + // A word that is already a lemma comes back as itself. + "dog, NN, dog", + "quickly, RB, quickly", + // WordNet letter tags are accepted alongside Penn tags. + "dogs, n, dog", + "walked, v, walk", + "taller, a, tall", + "best, r, well", + }) + void testLemmatizesToken(String token, String tag, String lemma) { + assertEquals(lemma, one(token, tag)); + } + + @ParameterizedTest + @CsvSource({ + // Rule candidates not in the lexicon are rejected, not returned. + "dogged, VBD", + "boxes, VBZ", + "glarbs, NNS", + // A known word under the wrong part of speech is unknown. + "walk, NN", + // Tags outside the mapping yield the unknown marker. + "dog, DT", + "dog, XYZ", + "dogs, ''", + // Multi-letter closed-class tags that merely begin with a WordNet letter code are not + // adjective lookups: AUX was must be unknown, and AUX taller must not detach to tall. + "was, AUX", + "taller, AUX", + }) + void testUnknownYieldsMarker(String token, String tag) { + assertEquals("O", one(token, tag)); + } + + @Test + void testExceptionHitsAreReturnedWithoutLexiconValidation() { + // oxen maps to ox, which the miniature lexicon does not contain; the exception list is + // authoritative for irregulars, so the lemma is returned anyway. + assertEquals("ox", one("oxen", "NNS")); + // better maps to good, also absent from the miniature lexicon. + assertEquals("good", one("better", "JJR")); + } + + @Test + void testArrayFormKeepsPositions() { + final String[] lemmas = morphy().lemmatize( + new String[] {"The", "mice", "ran", "quickly"}, + new String[] {"DT", "NNS", "VBD", "RB"}); + assertArrayEquals(new String[] {"O", "mouse", "run", "quickly"}, lemmas); + } + + @Test + void testListFormReturnsAllCandidates() { + final List> lemmas = morphy().lemmatize( + List.of("glarbs", "berries"), List.of("NNS", "NNS")); + assertEquals(List.of("O"), lemmas.get(0)); + assertEquals(List.of("berry"), lemmas.get(1)); + } + + @Test + void testWorksIdenticallyOverTheWnLmfLexicon() { + final MorphyLemmatizer lmfMorphy = + new MorphyLemmatizer(WnLmfReaderTest.fixture(), MorphyExceptionsTest.fixture()); + assertArrayEquals(new String[] {"mouse", "box", "walk", "large", "O"}, + lmfMorphy.lemmatize( + new String[] {"mice", "boxes", "walking", "larger", "dogged"}, + new String[] {"NNS", "NNS", "VBG", "JJR", "VBD"})); + } + + @ParameterizedTest + @CsvSource(nullValues = "none", value = { + "NNP, NOUN", + "VBZ, VERB", + "JJ, ADJECTIVE", + "RBR, ADVERB", + "a, ADJECTIVE", + "s, ADJECTIVE", + "ADJ, ADJECTIVE", + "ADV, ADVERB", + "r, ADVERB", + "DT, none", + "'', none", + // The letter codes a and s match only as one-letter tags: multi-letter tags beginning + // with those letters are closed-class or symbol tags, never adjectives. + "AUX, none", + "ADP, none", + "SCONJ, none", + "SYM, none", + }) + void testPosFromTagMapping(String tag, WordNetPOS pos) { + assertEquals(pos, MorphyLemmatizer.posFromTag(tag)); + } + + @Test + void testPosFromTagRejectsNull() { + assertThrows(IllegalArgumentException.class, () -> MorphyLemmatizer.posFromTag(null)); + } + + @Test + void testConstructorFailsLoudWithoutInputs() { + final MorphyExceptions exceptions = MorphyExceptionsTest.fixture(); + assertThrows(IllegalArgumentException.class, + () -> new MorphyLemmatizer(null, exceptions)); + assertThrows(IllegalArgumentException.class, + () -> new MorphyLemmatizer(WndbReaderTest.fixture(), null)); + } + + @Test + void testRejectsNullOrMismatchedSequences() { + final MorphyLemmatizer morphy = morphy(); + assertThrows(IllegalArgumentException.class, + () -> morphy.lemmatize((String[]) null, new String[0])); + assertThrows(IllegalArgumentException.class, + () -> morphy.lemmatize(new String[0], (String[]) null)); + assertThrows(IllegalArgumentException.class, + () -> morphy.lemmatize(new String[] {"a", "b"}, new String[] {"NN"})); + assertThrows(IllegalArgumentException.class, + () -> morphy.lemmatize(List.of("a"), List.of("NN", "NN"))); + assertThrows(IllegalArgumentException.class, + () -> morphy.lemmatize(new String[] {null}, new String[] {"NN"})); + assertThrows(IllegalArgumentException.class, + () -> morphy.lemmatize(new String[] {"dog"}, new String[] {null})); + } +} diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/ReaderEquivalenceTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/ReaderEquivalenceTest.java new file mode 100644 index 0000000000..b9f3523cbc --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/ReaderEquivalenceTest.java @@ -0,0 +1,121 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package opennlp.wordnet; + +import java.util.HashMap; +import java.util.HashSet; +import java.util.List; +import java.util.Map; +import java.util.Set; + +import org.junit.jupiter.api.Test; + +import opennlp.tools.wordnet.LexicalKnowledgeBase; +import opennlp.tools.wordnet.Synset; +import opennlp.tools.wordnet.WordNetPOS; +import opennlp.tools.wordnet.WordNetRelation; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertNotNull; + +/** + * Asserts that the WN-LMF fixture and the WNDB fixture, which encode the same miniature + * wordnet, load into equivalent lexicon views. Synset ids are reader-minted and intentionally + * differ, so the comparison is structural, joining synsets on their glosses (unique within the + * fixtures) and comparing everything else through that join. + */ +public class ReaderEquivalenceTest { + + @Test + void testBothReadersProduceEquivalentViews() { + final InMemoryWordNetLexicon lmf = (InMemoryWordNetLexicon) WnLmfReaderTest.fixture(); + final InMemoryWordNetLexicon wndb = (InMemoryWordNetLexicon) WndbReaderTest.fixture(); + assertEquals(lmf.size(), wndb.size(), "Both fixtures encode the same synsets"); + + final Map wndbByGloss = byGloss(wndb); + assertEquals(byGloss(lmf).keySet(), wndbByGloss.keySet(), "Same glosses on both sides"); + + for (final Synset expected : lmf.synsets()) { + final Synset actual = wndbByGloss.get(expected.gloss()); + assertNotNull(actual, "WNDB view has a synset for gloss: " + expected.gloss()); + assertEquals(expected.pos(), actual.pos(), "Part of speech for: " + expected.gloss()); + assertEquals(expected.lemmas(), actual.lemmas(), "Lemmas for: " + expected.gloss()); + assertEquals(relationsByGloss(expected, lmf), relationsByGloss(actual, wndb), + "Relations for: " + expected.gloss()); + } + } + + @Test + void testLookupAgreesForEveryLemmaAndPos() { + final LexicalKnowledgeBase lmf = WnLmfReaderTest.fixture(); + final InMemoryWordNetLexicon wndb = (InMemoryWordNetLexicon) WndbReaderTest.fixture(); + final Set checked = new HashSet<>(); + for (final Synset synset : wndb.synsets()) { + for (final String lemma : synset.lemmas()) { + if (!checked.add(lemma + "/" + synset.pos())) { + continue; + } + assertEquals( + glosses(lmf.lookup(lemma, synset.pos())), + glosses(wndb.lookup(lemma, synset.pos())), + "Sense sequence for " + lemma + " as " + synset.pos()); + } + } + for (final WordNetPOS pos : WordNetPOS.values()) { + assertEquals(lmf.contains("dog", pos), wndb.contains("dog", pos)); + } + } + + @Test + void testSenseOrderAgreesForMultiSenseLemma() { + final LexicalKnowledgeBase lmf = WnLmfReaderTest.fixture(); + final LexicalKnowledgeBase wndb = WndbReaderTest.fixture(); + final List lmfOrder = glosses(lmf.lookup("run", WordNetPOS.NOUN)); + final List wndbOrder = glosses(wndb.lookup("run", WordNetPOS.NOUN)); + assertEquals(2, lmfOrder.size()); + assertEquals(lmfOrder, wndbOrder); + } + + private static Map byGloss(InMemoryWordNetLexicon lexicon) { + final Map byGloss = new HashMap<>(); + for (final Synset synset : lexicon.synsets()) { + final Synset previous = byGloss.put(synset.gloss(), synset); + assertEquals(null, previous, "Fixture glosses must be unique, duplicated: " + + synset.gloss()); + } + return byGloss; + } + + // A synset's relations with targets replaced by their glosses, id-scheme independent. + private static Map> relationsByGloss(Synset synset, + LexicalKnowledgeBase lexicon) { + final Map> result = new HashMap<>(); + for (final Map.Entry> relation : + synset.relations().entrySet()) { + final Set targetGlosses = new HashSet<>(); + for (final String targetId : relation.getValue()) { + targetGlosses.add(lexicon.synset(targetId).orElseThrow().gloss()); + } + result.put(relation.getKey(), targetGlosses); + } + return result; + } + + private static List glosses(List synsets) { + return synsets.stream().map(Synset::gloss).toList(); + } +} diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/SynsetSimilarityTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/SynsetSimilarityTest.java new file mode 100644 index 0000000000..48accbc230 --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/SynsetSimilarityTest.java @@ -0,0 +1,145 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package opennlp.wordnet; + +import java.util.ArrayList; +import java.util.HashMap; +import java.util.List; +import java.util.Map; +import java.util.Optional; + +import org.junit.jupiter.api.Assertions; +import org.junit.jupiter.api.Test; + +import opennlp.tools.wordnet.LexicalKnowledgeBase; +import opennlp.tools.wordnet.Synset; +import opennlp.tools.wordnet.WordNetPOS; +import opennlp.tools.wordnet.WordNetRelation; + +/** + * Tests the taxonomy measures against a project-authored miniature taxonomy; no external + * lexicon data is involved. {@link HypernymTyperTest} shares the same taxonomy. + */ +public class SynsetSimilarityTest { + + /** A tiny in-memory knowledge base over a hand-built noun taxonomy. */ + static final class FixtureKnowledgeBase implements LexicalKnowledgeBase { + private final Map byId = new HashMap<>(); + private final Map> byLemma = new HashMap<>(); + + void add(String id, String lemma, WordNetRelation relation, String... parents) { + final Map> relations = parents.length == 0 + ? Map.of() : Map.of(relation, List.of(parents)); + final Synset synset = + new Synset(id, WordNetPOS.NOUN, List.of(lemma), "fixture", relations); + byId.put(id, synset); + byLemma.computeIfAbsent(lemma, key -> new ArrayList<>()).add(synset); + } + + @Override + public List lookup(String lemma, WordNetPOS pos) { + return byLemma.getOrDefault(lemma, List.of()); + } + + @Override + public Optional synset(String synsetId) { + return Optional.ofNullable(byId.get(synsetId)); + } + } + + /** {@return the taxonomy both this test and {@link HypernymTyperTest} assert against} */ + static FixtureKnowledgeBase taxonomy() { + final FixtureKnowledgeBase kb = new FixtureKnowledgeBase(); + kb.add("n1", "entity", WordNetRelation.HYPERNYM); + kb.add("n2", "physical", WordNetRelation.HYPERNYM, "n1"); + kb.add("n3", "organism", WordNetRelation.HYPERNYM, "n2"); + kb.add("n4", "person", WordNetRelation.HYPERNYM, "n3"); + kb.add("n5", "scientist", WordNetRelation.HYPERNYM, "n4"); + kb.add("n6", "chemist", WordNetRelation.HYPERNYM, "n5"); + kb.add("n7", "location", WordNetRelation.HYPERNYM, "n2"); + kb.add("n8", "city", WordNetRelation.HYPERNYM, "n7"); + kb.add("n9", "organization", WordNetRelation.HYPERNYM, "n1"); + kb.add("n10", "company", WordNetRelation.HYPERNYM, "n9"); + kb.add("n11", "paris", WordNetRelation.INSTANCE_HYPERNYM, "n8"); + kb.add("n12", "abstract", WordNetRelation.HYPERNYM); + return kb; + } + + @Test + void testPathSimilarity() { + final SynsetSimilarity similarity = new SynsetSimilarity(taxonomy()); + Assertions.assertEquals(1.0, similarity.path("n5", "n5"), 1e-9); + Assertions.assertEquals(0.5, similarity.path("n6", "n5"), 1e-9); + // chemist up four to physical, city up two: six edges apart + Assertions.assertEquals(1.0 / 7.0, similarity.path("n6", "n8"), 1e-9); + Assertions.assertEquals(0.0, similarity.path("n6", "n12"), 1e-9); + } + + @Test + void testWuPalmerRewardsDeepSharedAncestry() { + final SynsetSimilarity similarity = new SynsetSimilarity(taxonomy()); + // scientist and chemist share scientist itself at depth four + Assertions.assertEquals(8.0 / 9.0, similarity.wuPalmer("n5", "n6"), 1e-9); + final double siblingBranches = similarity.wuPalmer("n6", "n8"); + Assertions.assertTrue(siblingBranches < similarity.wuPalmer("n5", "n6")); + Assertions.assertTrue(siblingBranches > 0.0); + Assertions.assertEquals(0.0, similarity.wuPalmer("n6", "n12"), 1e-9); + } + + @Test + void testLeacockChodorow() { + final SynsetSimilarity similarity = new SynsetSimilarity(taxonomy()); + Assertions.assertEquals(Math.log(10.0), + similarity.leacockChodorow("n5", "n6", 10), 1e-9); + Assertions.assertEquals(0.0, similarity.leacockChodorow("n6", "n12", 10), 1e-9); + Assertions.assertThrows(IllegalArgumentException.class, + () -> similarity.leacockChodorow("n5", "n6", 0)); + } + + @Test + void testInstanceHypernymsCountAsEdges() { + final SynsetSimilarity similarity = new SynsetSimilarity(taxonomy()); + Assertions.assertEquals(0.5, similarity.path("n11", "n8"), 1e-9); + } + + @Test + void testInvalidArguments() { + Assertions.assertThrows(IllegalArgumentException.class, + () -> new SynsetSimilarity(null)); + final SynsetSimilarity similarity = new SynsetSimilarity(taxonomy()); + Assertions.assertThrows(IllegalArgumentException.class, + () -> similarity.path(null, "n1")); + Assertions.assertThrows(IllegalArgumentException.class, + () -> similarity.path("n1", null)); + Assertions.assertThrows(IllegalArgumentException.class, + () -> similarity.wuPalmer(null, "n1")); + Assertions.assertThrows(IllegalArgumentException.class, + () -> similarity.shortestDistance("n1", null)); + Assertions.assertThrows(IllegalArgumentException.class, + () -> similarity.leacockChodorow(null, "n1", 10)); + } + + @Test + void testUnknownSynsetsAreUnrelatedRatherThanFatal() { + final SynsetSimilarity similarity = new SynsetSimilarity(taxonomy()); + Assertions.assertEquals(-1, similarity.shortestDistance("n5", "missing")); + Assertions.assertEquals(0.0, similarity.path("n5", "missing"), 1e-9); + Assertions.assertEquals(0.0, similarity.wuPalmer("n5", "missing"), 1e-9); + Assertions.assertEquals(0.0, similarity.leacockChodorow("n5", "missing", 10), 1e-9); + } +} diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/WnLmfReaderTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/WnLmfReaderTest.java new file mode 100644 index 0000000000..5d48d17969 --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/WnLmfReaderTest.java @@ -0,0 +1,405 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package opennlp.wordnet; + +import java.io.ByteArrayInputStream; +import java.io.IOException; +import java.io.InputStream; +import java.nio.charset.StandardCharsets; +import java.nio.file.Files; +import java.nio.file.Path; +import java.util.List; + +import org.junit.jupiter.api.Test; +import org.junit.jupiter.api.io.TempDir; + +import opennlp.tools.util.InvalidFormatException; +import opennlp.tools.wordnet.LexicalKnowledgeBase; +import opennlp.tools.wordnet.Synset; +import opennlp.tools.wordnet.WordNetPOS; +import opennlp.tools.wordnet.WordNetRelation; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertFalse; +import static org.junit.jupiter.api.Assertions.assertNotNull; +import static org.junit.jupiter.api.Assertions.assertSame; +import static org.junit.jupiter.api.Assertions.assertThrows; +import static org.junit.jupiter.api.Assertions.assertTrue; + +public class WnLmfReaderTest { + + static LexicalKnowledgeBase fixture() { + try (InputStream in = WnLmfReaderTest.class.getResourceAsStream("mini-wn-lmf.xml")) { + assertNotNull(in, "Fixture mini-wn-lmf.xml must be on the test classpath"); + return WnLmfReader.read(in, "mini-wn-lmf.xml"); + } catch (IOException e) { + throw new IllegalStateException("Unexpected IOException from a classpath stream", e); + } + } + + private static LexicalKnowledgeBase parse(String document) throws IOException { + return WnLmfReader.read( + new ByteArrayInputStream(document.getBytes(StandardCharsets.UTF_8)), "inline.xml"); + } + + private static String wrap(String body) { + return "\n\n" + + "\n" + + body + "\n\n\n"; + } + + @Test + void testLookupReturnsSynsetWithAllComponents() { + final List senses = fixture().lookup("dog", WordNetPOS.NOUN); + assertEquals(1, senses.size()); + final Synset dog = senses.get(0); + assertEquals("mini-n1", dog.id()); + assertEquals(WordNetPOS.NOUN, dog.pos()); + assertEquals(List.of("dog", "domestic dog"), dog.lemmas()); + assertEquals("a domesticated canid", dog.gloss()); + assertEquals(List.of("mini-n2"), dog.related(WordNetRelation.HYPERNYM)); + } + + @Test + void testLookupFoldsCaseAndUnderscore() { + final LexicalKnowledgeBase lexicon = fixture(); + assertEquals("mini-n1", lexicon.lookup("Domestic_Dog", WordNetPOS.NOUN).get(0).id()); + assertEquals("mini-n1", lexicon.lookup("DOG", WordNetPOS.NOUN).get(0).id()); + } + + @Test + void testLookupKeepsSenseOrder() { + final List runSenses = fixture().lookup("run", WordNetPOS.NOUN); + assertEquals(List.of("mini-n5", "mini-n9"), + runSenses.stream().map(Synset::id).toList()); + } + + @Test + void testLookupIsPosScoped() { + final LexicalKnowledgeBase lexicon = fixture(); + assertEquals(1, lexicon.lookup("run", WordNetPOS.VERB).size()); + assertTrue(lexicon.lookup("dog", WordNetPOS.VERB).isEmpty()); + assertFalse(lexicon.contains("walk", WordNetPOS.NOUN)); + assertTrue(lexicon.contains("walk", WordNetPOS.VERB)); + } + + @Test + void testRelationNavigation() { + final LexicalKnowledgeBase lexicon = fixture(); + assertEquals(List.of("mini-n1"), lexicon.related("mini-n2", WordNetRelation.HYPONYM)); + assertEquals(List.of("mini-v1", "mini-v2"), + lexicon.related("mini-v4", WordNetRelation.HYPONYM)); + assertEquals(List.of("mini-v4"), lexicon.related("mini-v1", WordNetRelation.HYPERNYM)); + } + + @Test + void testRelationTargetSharesCanonicalIdInstance() { + final LexicalKnowledgeBase lexicon = fixture(); + final String target = lexicon.synset("mini-n1").orElseThrow() + .related(WordNetRelation.HYPERNYM).get(0); + // Not just equal: the identical instance from the synset table, so a loaded lexicon keeps + // one copy of each id no matter how many relations point at it. + assertSame(lexicon.synset("mini-n2").orElseThrow().id(), target); + } + + @Test + void testSenseRelationsAreLiftedToSynsetLevel() { + final LexicalKnowledgeBase lexicon = fixture(); + assertEquals(List.of("mini-a2"), lexicon.related("mini-a1", WordNetRelation.ANTONYM)); + assertEquals(List.of("mini-a1"), lexicon.related("mini-a2", WordNetRelation.ANTONYM)); + assertEquals(List.of("mini-v1"), + lexicon.related("mini-n5", WordNetRelation.DERIVATIONALLY_RELATED)); + assertEquals(List.of("mini-n5"), + lexicon.related("mini-v1", WordNetRelation.DERIVATIONALLY_RELATED)); + } + + @Test + void testSatelliteNormalizesToAdjective() { + final List senses = fixture().lookup("large", WordNetPOS.ADJECTIVE); + assertEquals(1, senses.size()); + assertEquals(WordNetPOS.ADJECTIVE, senses.get(0).pos()); + assertEquals(List.of("mini-a4"), fixture().related("mini-a3", WordNetRelation.SIMILAR_TO)); + assertEquals(List.of("mini-a3"), fixture().related("mini-a4", WordNetRelation.SIMILAR_TO)); + } + + @Test + void testSimilarOnVerbSynsetMapsToVerbGroup() throws IOException { + // Documents derived from Princeton data express verb groups as similar on verb synsets; + // the fixture only carries similar on adjectives, so this pins the verb branch directly. + final LexicalKnowledgeBase lexicon = parse(wrap( + "" + + "" + + "" + + "" + + "" + + "produce musical tones" + + "" + + "" + + "sing monotonously")); + assertEquals(List.of("t-v2"), lexicon.related("t-v1", WordNetRelation.VERB_GROUP)); + assertTrue(lexicon.related("t-v1", WordNetRelation.SIMILAR_TO).isEmpty()); + } + + @Test + void testUnknownLemmaOrSynsetIsEmpty() { + final LexicalKnowledgeBase lexicon = fixture(); + assertTrue(lexicon.lookup("zebra", WordNetPOS.NOUN).isEmpty()); + assertTrue(lexicon.synset("mini-n99").isEmpty()); + } + + @Test + void testReadPath(@TempDir Path tempDir) throws IOException { + final Path file = tempDir.resolve("tiny.xml"); + Files.writeString(file, wrap( + "" + + "" + + "a feline")); + final LexicalKnowledgeBase lexicon = WnLmfReader.read(file); + assertEquals("a feline", lexicon.lookup("cat", WordNetPOS.NOUN).get(0).gloss()); + } + + @Test + void testReadPathRejectsNullAndMissing(@TempDir Path tempDir) { + assertThrows(IllegalArgumentException.class, () -> WnLmfReader.read((Path) null)); + assertThrows(IllegalArgumentException.class, + () -> WnLmfReader.read(tempDir.resolve("absent.xml"))); + } + + @Test + void testReadStreamRejectsNulls() { + assertThrows(IllegalArgumentException.class, () -> WnLmfReader.read(null, "x")); + assertThrows(IllegalArgumentException.class, + () -> WnLmfReader.read(new ByteArrayInputStream(new byte[0]), null)); + } + + @Test + void testStreamReadFailurePropagatesAsIOException() { + final InputStream failing = new InputStream() { + @Override + public int read() throws IOException { + throw new IOException("Simulated stream failure"); + } + }; + final IOException e = + assertThrows(IOException.class, () -> WnLmfReader.read(failing, "failing.xml")); + // The I/O failure must surface as itself, not be misreported as a malformed document. + assertFalse(e instanceof InvalidFormatException); + } + + @Test + void testSkipsDoctypeDeclaration() throws IOException { + // Real Open English WordNet releases ship exactly this shape: a DOCTYPE naming the schema + // DTD by an unreachable SYSTEM identifier (example.invalid is the RFC 2606 reserved domain + // that must never resolve). The reader must parse past it without attempting to fetch it. + final String document = "\n" + + "\n" + + "" + + "" + + "" + + "a feline" + + ""; + final LexicalKnowledgeBase lexicon = parse(document); + assertEquals("a feline", lexicon.lookup("cat", WordNetPOS.NOUN).get(0).gloss()); + } + + @Test + void testInternalSubsetEntityIsNeverExpanded(@TempDir Path tempDir) throws IOException { + // A DOCTYPE-declared internal-subset entity is the classic XXE payload: if the parser ever + // honored it, the entity reference below would be replaced by the target file's content. + // With SUPPORT_DTD disabled the declaration itself is never registered, so the reference is + // undefined and parsing must fail loud rather than silently expand it. + final Path secret = tempDir.resolve("secret.txt"); + Files.writeString(secret, "xxe-marker-should-never-appear"); + final String document = "\n" + + "]>\n" + + "" + + "" + + "" + + "a feline" + + ""; + final InvalidFormatException e = + assertThrows(InvalidFormatException.class, () -> parse(document)); + assertFalse(e.getMessage().contains("xxe-marker-should-never-appear")); + } + + @Test + void testRejectsTruncatedDocument() { + final InvalidFormatException e = assertThrows(InvalidFormatException.class, + () -> parse("\n parse( + wrap("" + + ""))); + assertTrue(e.getMessage().contains("synset")); + } + + @Test + void testRejectsSenseToUndeclaredSynset() { + final InvalidFormatException e = assertThrows(InvalidFormatException.class, () -> parse( + wrap("" + + ""))); + assertTrue(e.getMessage().contains("t-9")); + } + + @Test + void testRejectsRelationToUndeclaredSynset() { + final InvalidFormatException e = assertThrows(InvalidFormatException.class, () -> parse( + wrap("" + + "" + + "a feline" + + ""))); + assertTrue(e.getMessage().contains("t-9")); + } + + @Test + void testRejectsUnknownRelationType() { + final InvalidFormatException e = assertThrows(InvalidFormatException.class, () -> parse( + wrap("" + + "" + + "a feline" + + ""))); + assertTrue(e.getMessage().contains("quasi_synonym")); + } + + @Test + void testSkipsOtherRelationTypeOnSenseRelation() throws IOException { + final LexicalKnowledgeBase lexicon = parse( + wrap("" + + "" + + "" + + "a feline")); + assertTrue(lexicon.synset("t-1").orElseThrow().relations().isEmpty()); + } + + @Test + void testSkipsOtherRelationTypeOnSynsetRelation() throws IOException { + // The DTD permits relType="other" on SynsetRelation too, and several OMW-family wordnets + // emit it; it is skipped exactly like the SenseRelation case, not rejected. + final LexicalKnowledgeBase lexicon = parse( + wrap("" + + "" + + "a feline" + + "")); + assertTrue(lexicon.synset("t-1").orElseThrow().relations().isEmpty()); + } + + @Test + void testRejectsUnknownPartOfSpeech() { + final InvalidFormatException e = assertThrows(InvalidFormatException.class, () -> parse( + wrap("" + + "" + + "a feline"))); + assertTrue(e.getMessage().contains("x")); + } + + @Test + void testRejectsSynsetWithoutMembers() { + final InvalidFormatException e = assertThrows(InvalidFormatException.class, () -> parse( + wrap("orphan"))); + assertTrue(e.getMessage().contains("t-1")); + } + + @Test + void testRejectsDuplicateSynsetId() { + final InvalidFormatException e = assertThrows(InvalidFormatException.class, () -> parse( + wrap("" + + "" + + "a feline" + + "a repeat"))); + assertTrue(e.getMessage().contains("Duplicate synset id t-1")); + } + + @Test + void testRejectsDuplicateLexicalEntryId() { + final InvalidFormatException e = assertThrows(InvalidFormatException.class, () -> parse( + wrap("" + + "" + + "" + + "" + + "a feline"))); + assertTrue(e.getMessage().contains("Duplicate lexical entry id t-cat-n")); + } + + @Test + void testRejectsDuplicateSenseId() { + final InvalidFormatException e = assertThrows(InvalidFormatException.class, () -> parse( + wrap("" + + "" + + "" + + "a feline" + + "a second"))); + assertTrue(e.getMessage().contains("Duplicate sense id t-cat-n-1")); + } + + @Test + void testRejectsSynsetMemberPosMismatch() { + final InvalidFormatException e = assertThrows(InvalidFormatException.class, () -> parse( + wrap("" + + "" + + "a feline"))); + assertTrue(e.getMessage().contains("t-cat-n")); + assertTrue(e.getMessage().contains("VERB")); + assertTrue(e.getMessage().contains("NOUN")); + } + + @Test + void testRejectsSenseRelationToUndeclaredSense() { + final InvalidFormatException e = assertThrows(InvalidFormatException.class, () -> parse( + wrap("" + + "" + + "" + + "a feline"))); + assertTrue(e.getMessage().contains("t-ghost-1")); + } + + @Test + void testRejectsLemmaOutsideLexicalEntry() { + final InvalidFormatException e = assertThrows(InvalidFormatException.class, + () -> parse(wrap(""))); + assertTrue(e.getMessage().contains("Lemma outside a LexicalEntry")); + } + + @Test + void testRejectsSenseBeforeLemma() { + final InvalidFormatException e = assertThrows(InvalidFormatException.class, () -> parse( + wrap("" + + "" + + "a feline"))); + assertTrue(e.getMessage().contains("Sense before its entry's Lemma")); + } + + @Test + void testRejectsSenseRelationOutsideSense() { + final InvalidFormatException e = assertThrows(InvalidFormatException.class, () -> parse( + wrap("" + + "" + + "" + + "a feline"))); + assertTrue(e.getMessage().contains("SenseRelation outside a Sense")); + } + + @Test + void testRejectsSynsetRelationOutsideSynset() { + final InvalidFormatException e = assertThrows(InvalidFormatException.class, + () -> parse(wrap(""))); + assertTrue(e.getMessage().contains("SynsetRelation outside a Synset")); + } +} diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/WndbReaderTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/WndbReaderTest.java new file mode 100644 index 0000000000..5931957e3b --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/WndbReaderTest.java @@ -0,0 +1,273 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package opennlp.wordnet; + +import java.io.IOException; +import java.net.URISyntaxException; +import java.net.URL; +import java.nio.charset.StandardCharsets; +import java.nio.file.Files; +import java.nio.file.Path; +import java.util.List; +import java.util.Locale; +import java.util.function.UnaryOperator; + +import org.junit.jupiter.api.Test; +import org.junit.jupiter.api.io.TempDir; + +import opennlp.tools.util.InvalidFormatException; +import opennlp.tools.wordnet.LexicalKnowledgeBase; +import opennlp.tools.wordnet.Synset; +import opennlp.tools.wordnet.WordNetPOS; +import opennlp.tools.wordnet.WordNetRelation; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertNotNull; +import static org.junit.jupiter.api.Assertions.assertSame; +import static org.junit.jupiter.api.Assertions.assertThrows; +import static org.junit.jupiter.api.Assertions.assertTrue; + +public class WndbReaderTest { + + private static final String DOG_ID = "wndb-00001075-n"; + private static final String CANID_ID = "wndb-00001160-n"; + + static Path fixtureDirectory() { + final URL url = WndbReaderTest.class.getResource("mini-wndb"); + assertNotNull(url, "Fixture directory mini-wndb must be on the test classpath"); + try { + return Path.of(url.toURI()); + } catch (URISyntaxException e) { + throw new IllegalStateException("Unexpected fixture URI: " + url, e); + } + } + + static LexicalKnowledgeBase fixture() { + try { + return WndbReader.read(fixtureDirectory()); + } catch (IOException e) { + throw new IllegalStateException("Unexpected IOException reading the WNDB fixture", e); + } + } + + @Test + void testLookupReturnsSynsetWithAllComponents() { + final List senses = fixture().lookup("dog", WordNetPOS.NOUN); + assertEquals(1, senses.size()); + final Synset dog = senses.get(0); + assertEquals(DOG_ID, dog.id()); + assertEquals(WordNetPOS.NOUN, dog.pos()); + assertEquals(List.of("dog", "domestic dog"), dog.lemmas()); + assertEquals("a domesticated canid", dog.gloss()); + assertEquals(List.of(CANID_ID), dog.related(WordNetRelation.HYPERNYM)); + } + + @Test + void testLookupFoldsCaseAndUnderscore() { + final LexicalKnowledgeBase lexicon = fixture(); + assertEquals(DOG_ID, lexicon.lookup("Domestic_Dog", WordNetPOS.NOUN).get(0).id()); + assertEquals(DOG_ID, lexicon.lookup("DOG", WordNetPOS.NOUN).get(0).id()); + } + + @Test + void testLookupKeepsIndexSenseOrder() { + assertEquals(List.of("wndb-00001427-n", "wndb-00001669-n"), + fixture().lookup("run", WordNetPOS.NOUN).stream().map(Synset::id).toList()); + } + + @Test + void testRelationNavigation() { + final LexicalKnowledgeBase lexicon = fixture(); + assertEquals(List.of(DOG_ID), lexicon.related(CANID_ID, WordNetRelation.HYPONYM)); + assertEquals(List.of("wndb-00001075-v", "wndb-00001171-v"), + lexicon.related("wndb-00001324-v", WordNetRelation.HYPONYM)); + assertEquals(List.of("wndb-00001075-v"), + lexicon.related("wndb-00001427-n", WordNetRelation.DERIVATIONALLY_RELATED)); + } + + @Test + void testRelationTargetSharesCanonicalIdInstance() { + final LexicalKnowledgeBase lexicon = fixture(); + final String target = lexicon.synset(DOG_ID).orElseThrow() + .related(WordNetRelation.HYPERNYM).get(0); + // Not just equal: the identical instance from the synset table, so a loaded lexicon keeps + // one copy of each id no matter how many pointers reference it. + assertSame(lexicon.synset(CANID_ID).orElseThrow().id(), target); + } + + @Test + void testLexicalPointersSurfaceAtSynsetLevel() { + final LexicalKnowledgeBase lexicon = fixture(); + assertEquals(List.of("wndb-00001141-a"), + lexicon.related("wndb-00001075-a", WordNetRelation.ANTONYM)); + assertEquals(List.of("wndb-00001075-a"), + lexicon.related("wndb-00001141-a", WordNetRelation.ANTONYM)); + } + + @Test + void testSatelliteNormalizesToAdjectiveAndMarkerIsStripped() { + final LexicalKnowledgeBase lexicon = fixture(); + final Synset large = lexicon.lookup("large", WordNetPOS.ADJECTIVE).get(0); + assertEquals(WordNetPOS.ADJECTIVE, large.pos()); + assertEquals(List.of("wndb-00001211-a"), large.related(WordNetRelation.SIMILAR_TO)); + // short is stored as short(p); the syntactic marker is not part of the lemma. + assertEquals(List.of("short"), + lexicon.lookup("short", WordNetPOS.ADJECTIVE).get(0).lemmas()); + } + + @Test + void testVerbGroupPointerMapsToVerbGroup(@TempDir Path tempDir) throws IOException { + // The fixture has no $ pointer, so the VERB_GROUP mapping is pinned against a minimal + // constructed database whose byte offsets are computed, not hard-coded: every offset field + // is exactly eight digits, so the second line's position is independent of the digit values. + writeEmptyDb(tempDir, "noun", "adj", "adv"); + final String template = + "00000000 29 v 01 sing 0 001 $ XXXXXXXX v 0000 00 | produce musical tones"; + final String off2 = String.format(Locale.ROOT, "%08d", template.length() + 1); + final String line1 = template.replace("XXXXXXXX", off2); + final String line2 = off2 + " 29 v 01 chant 0 001 $ 00000000 v 0000 00 | sing monotonously"; + Files.writeString(tempDir.resolve("data.verb"), line1 + "\n" + line2 + "\n", + StandardCharsets.ISO_8859_1); + Files.writeString(tempDir.resolve("index.verb"), + "chant v 1 1 $ 1 0 " + off2 + "\nsing v 1 1 $ 1 0 00000000\n", + StandardCharsets.ISO_8859_1); + final LexicalKnowledgeBase lexicon = WndbReader.read(tempDir); + assertEquals(List.of("wndb-" + off2 + "-v"), + lexicon.related("wndb-00000000-v", WordNetRelation.VERB_GROUP)); + assertEquals(List.of("wndb-00000000-v"), + lexicon.related("wndb-" + off2 + "-v", WordNetRelation.VERB_GROUP)); + } + + @Test + void testUnknownLemmaOrSynsetIsEmpty() { + final LexicalKnowledgeBase lexicon = fixture(); + assertTrue(lexicon.lookup("zebra", WordNetPOS.NOUN).isEmpty()); + assertTrue(lexicon.synset("wndb-99999999-n").isEmpty()); + } + + @Test + void testRejectsNullAndMissingDirectory(@TempDir Path tempDir) { + assertThrows(IllegalArgumentException.class, () -> WndbReader.read(null)); + assertThrows(IllegalArgumentException.class, + () -> WndbReader.read(tempDir.resolve("absent"))); + } + + @Test + void testRejectsMissingDatabaseFile(@TempDir Path tempDir) throws IOException { + copyFixture(tempDir); + Files.delete(tempDir.resolve("data.verb")); + final InvalidFormatException e = + assertThrows(InvalidFormatException.class, () -> WndbReader.read(tempDir)); + assertTrue(e.getMessage().contains("data.verb")); + } + + @Test + void testRejectsIndexOffsetWithoutDataLine(@TempDir Path tempDir) throws IOException { + copyFixture(tempDir); + mutate(tempDir, "index.noun", line -> line.startsWith("berry ") + ? line.replace("00001564", "00001565") : line); + final InvalidFormatException e = + assertThrows(InvalidFormatException.class, () -> WndbReader.read(tempDir)); + assertTrue(e.getMessage().contains("berry")); + assertTrue(e.getMessage().contains("00001565")); + } + + @Test + void testRejectsDataOffsetFieldMismatch(@TempDir Path tempDir) throws IOException { + copyFixture(tempDir); + mutate(tempDir, "data.noun", + line -> line.replace("00001503 03 n 01 box", "00001504 03 n 01 box")); + final InvalidFormatException e = + assertThrows(InvalidFormatException.class, () -> WndbReader.read(tempDir)); + assertTrue(e.getMessage().contains("disagrees")); + } + + @Test + void testRejectsTruncatedDataLine(@TempDir Path tempDir) throws IOException { + copyFixture(tempDir); + mutate(tempDir, "data.noun", line -> line.startsWith("00001564") + ? line.substring(0, line.indexOf(" 000 |")) : line); + final InvalidFormatException e = + assertThrows(InvalidFormatException.class, () -> WndbReader.read(tempDir)); + assertTrue(e.getMessage().contains("data.noun")); + assertTrue(e.getMessage().contains("Truncated")); + } + + @Test + void testRejectsUndeclaredPointerSymbol(@TempDir Path tempDir) throws IOException { + copyFixture(tempDir); + mutate(tempDir, "data.noun", line -> line.replace("001 @ 00001160 n 0000", + "001 ? 00001160 n 0000")); + final InvalidFormatException e = + assertThrows(InvalidFormatException.class, () -> WndbReader.read(tempDir)); + assertTrue(e.getMessage().contains("Undeclared pointer symbol: ?")); + } + + @Test + void testRejectsPointerToNonexistentSynset(@TempDir Path tempDir) throws IOException { + copyFixture(tempDir); + mutate(tempDir, "data.noun", line -> line.replace("001 @ 00001160 n 0000", + "001 @ 00009999 n 0000")); + final InvalidFormatException e = + assertThrows(InvalidFormatException.class, () -> WndbReader.read(tempDir)); + assertTrue(e.getMessage().contains("wndb-00009999-n")); + } + + @Test + void testDanglingPointerErrorNamesPointerLine(@TempDir Path tempDir) throws IOException { + // A constructed database with no preamble, so the dangling pointer sits on a known line + // and the error message can be pinned to name it. + writeEmptyDb(tempDir, "noun", "adj", "adv"); + Files.writeString(tempDir.resolve("data.verb"), + "00000000 29 v 01 sing 0 001 $ 00009999 v 0000 00 | produce musical tones\n", + StandardCharsets.ISO_8859_1); + Files.writeString(tempDir.resolve("index.verb"), "sing v 1 1 $ 1 0 00000000\n", + StandardCharsets.ISO_8859_1); + final InvalidFormatException e = + assertThrows(InvalidFormatException.class, () -> WndbReader.read(tempDir)); + assertTrue(e.getMessage().contains("wndb-00009999-v")); + assertTrue(e.getMessage().contains("line 1")); + } + + private static void writeEmptyDb(Path directory, String... suffixes) throws IOException { + for (final String suffix : suffixes) { + Files.writeString(directory.resolve("data." + suffix), ""); + Files.writeString(directory.resolve("index." + suffix), ""); + } + } + + private static void copyFixture(Path target) throws IOException { + try (var files = Files.list(fixtureDirectory())) { + for (final Path file : files.toList()) { + Files.copy(file, target.resolve(file.getFileName().toString())); + } + } + } + + // Applies a line transformation to one fixture file. The mutations only ever keep or shrink + // line lengths of the affected line's own fields, so surrounding offsets stay valid. + private static void mutate(Path directory, String fileName, UnaryOperator edit) + throws IOException { + final Path file = directory.resolve(fileName); + final List lines = Files.readAllLines(file, StandardCharsets.ISO_8859_1); + final StringBuilder out = new StringBuilder(); + for (final String line : lines) { + out.append(edit.apply(line)).append('\n'); + } + Files.writeString(file, out.toString(), StandardCharsets.ISO_8859_1); + } +} diff --git a/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/WordNetUsageExampleTest.java b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/WordNetUsageExampleTest.java new file mode 100644 index 0000000000..e4f7a31916 --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/java/opennlp/wordnet/WordNetUsageExampleTest.java @@ -0,0 +1,78 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package opennlp.wordnet; + +import java.io.IOException; +import java.io.InputStream; +import java.net.URISyntaxException; +import java.net.URL; +import java.nio.file.Path; +import java.util.List; + +import org.junit.jupiter.api.Assertions; +import org.junit.jupiter.api.Test; + +import opennlp.tools.wordnet.LexicalKnowledgeBase; +import opennlp.tools.wordnet.Synset; +import opennlp.tools.wordnet.WordNetPOS; + +/** + * Runs the manual's WordNet load and lookup examples (docbkx {@code wordnet.xml}) + * verbatim: every value the chapter states is asserted here, so a change breaking this + * test breaks the manual. The lexicon is the classpath fixture {@code mini-wn-lmf.xml}; + * exception lists come from the sibling {@code mini-wndb} directory. + */ +public class WordNetUsageExampleTest { + + private static LexicalKnowledgeBase loadMiniWnLmf() throws IOException { + try (InputStream in = WordNetUsageExampleTest.class.getResourceAsStream("mini-wn-lmf.xml")) { + Assertions.assertNotNull(in, "Fixture mini-wn-lmf.xml must be on the test classpath"); + return WnLmfReader.read(in, "mini-wn-lmf.xml"); + } + } + + private static Path miniWndbDirectory() { + final URL url = WordNetUsageExampleTest.class.getResource("mini-wndb"); + Assertions.assertNotNull(url, "Fixture directory mini-wndb must be on the test classpath"); + try { + return Path.of(url.toURI()); + } catch (URISyntaxException e) { + throw new IllegalStateException("Unexpected fixture URI: " + url, e); + } + } + + /** + * Load, lookup, and Morphy lemmatize as the chapter shows. + */ + @Test + void testLoadLookupAndLemmatize() throws IOException { + final LexicalKnowledgeBase lexicon = loadMiniWnLmf(); + final List senses = lexicon.lookup("dog", WordNetPOS.NOUN); + Assertions.assertEquals(1, senses.size()); + Assertions.assertEquals("mini-n1", senses.get(0).id()); + Assertions.assertEquals(List.of("dog", "domestic dog"), senses.get(0).lemmas()); + Assertions.assertEquals("a domesticated canid", senses.get(0).gloss()); + + final MorphyLemmatizer lemmatizer = + new MorphyLemmatizer(lexicon, MorphyExceptions.load(miniWndbDirectory())); + Assertions.assertEquals("mouse", + lemmatizer.lemmatize(new String[] {"mice"}, new String[] {"NNS"})[0]); + Assertions.assertEquals("dog", + lemmatizer.lemmatize(new String[] {"dogs"}, new String[] {"NNS"})[0]); + } +} diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wn-lmf.xml b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wn-lmf.xml new file mode 100644 index 0000000000..10e039214b --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wn-lmf.xml @@ -0,0 +1,183 @@ + + + + + + + dog + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + a domesticated canid + + the dog barked + + + a carnivorous mammal with nonretractile claws + + + + a small rodent with a long tail + + + + a gnawing mammal with chisel teeth + + + + an act of running at speed + + + a rigid rectangular container + + + a small juicy fruit + + + an adult male person + + + a score made in baseball + + + move fast on foot + + + + move at a regular pace + + + + change location or position + + + change position in space + + + + + of great height + + + of small height + + + of great size + + + + above average in size + + + + with speed + + + in a good or proper manner + + + diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/.gitattributes b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/.gitattributes new file mode 100644 index 0000000000..3868b4d103 --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/.gitattributes @@ -0,0 +1,7 @@ +# WNDB is a byte-offset format: each data line embeds its own byte position in the +# file, and WndbReader validates that offset against the actual position. Line-ending +# normalization on checkout (the repo root's `* text=auto`) would insert a CR before +# every LF on Windows, shifting every offset after the first line and breaking every +# fixture that has more than one line. Disable it here so checkout is byte-identical +# on every platform. +* -text diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/adj.exc b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/adj.exc new file mode 100644 index 0000000000..404a2e4ddb --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/adj.exc @@ -0,0 +1 @@ +better good diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/adv.exc b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/adv.exc new file mode 100644 index 0000000000..c43a2cd529 --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/adv.exc @@ -0,0 +1 @@ +best well diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/data.adj b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/data.adj new file mode 100644 index 0000000000..1340d16a45 --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/data.adj @@ -0,0 +1,22 @@ + 1 Licensed to the Apache Software Foundation (ASF) under one or more + 2 contributor license agreements. See the NOTICE file distributed with + 3 this work for additional information regarding copyright ownership. + 4 The ASF licenses this file to You under the Apache License, Version 2.0 + 5 (the "License"); you may not use this file except in compliance with + 6 the License. You may obtain a copy of the License at + 7 + 8 http://www.apache.org/licenses/LICENSE-2.0 + 9 + 10 Unless required by applicable law or agreed to in writing, software + 11 distributed under the License is distributed on an "AS IS" BASIS, + 12 WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + 13 See the License for the specific language governing permissions and + 14 limitations under the License. + 15 + 16 Project-authored miniature WNDB fixture mirroring mini-wn-lmf.xml. + 17 License preamble lines begin with two spaces, as in released WNDB files, + 18 so readers skip them; data line offsets include this preamble. +00001075 00 a 01 tall 0 001 ! 00001141 a 0101 | of great height +00001141 00 a 01 short(p) 0 001 ! 00001075 a 0101 | of small height +00001211 00 a 01 big 0 001 & 00001274 a 0000 | of great size +00001274 00 s 01 large 0 001 & 00001211 a 0000 | above average in size diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/data.adv b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/data.adv new file mode 100644 index 0000000000..732c21dfd5 --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/data.adv @@ -0,0 +1,20 @@ + 1 Licensed to the Apache Software Foundation (ASF) under one or more + 2 contributor license agreements. See the NOTICE file distributed with + 3 this work for additional information regarding copyright ownership. + 4 The ASF licenses this file to You under the Apache License, Version 2.0 + 5 (the "License"); you may not use this file except in compliance with + 6 the License. You may obtain a copy of the License at + 7 + 8 http://www.apache.org/licenses/LICENSE-2.0 + 9 + 10 Unless required by applicable law or agreed to in writing, software + 11 distributed under the License is distributed on an "AS IS" BASIS, + 12 WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + 13 See the License for the specific language governing permissions and + 14 limitations under the License. + 15 + 16 Project-authored miniature WNDB fixture mirroring mini-wn-lmf.xml. + 17 License preamble lines begin with two spaces, as in released WNDB files, + 18 so readers skip them; data line offsets include this preamble. +00001075 02 r 01 quickly 0 000 | with speed +00001121 02 r 01 well 0 000 | in a good or proper manner diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/data.noun b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/data.noun new file mode 100644 index 0000000000..0598111bf1 --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/data.noun @@ -0,0 +1,27 @@ + 1 Licensed to the Apache Software Foundation (ASF) under one or more + 2 contributor license agreements. See the NOTICE file distributed with + 3 this work for additional information regarding copyright ownership. + 4 The ASF licenses this file to You under the Apache License, Version 2.0 + 5 (the "License"); you may not use this file except in compliance with + 6 the License. You may obtain a copy of the License at + 7 + 8 http://www.apache.org/licenses/LICENSE-2.0 + 9 + 10 Unless required by applicable law or agreed to in writing, software + 11 distributed under the License is distributed on an "AS IS" BASIS, + 12 WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + 13 See the License for the specific language governing permissions and + 14 limitations under the License. + 15 + 16 Project-authored miniature WNDB fixture mirroring mini-wn-lmf.xml. + 17 License preamble lines begin with two spaces, as in released WNDB files, + 18 so readers skip them; data line offsets include this preamble. +00001075 03 n 02 dog 0 domestic_dog 0 001 @ 00001160 n 0000 | a domesticated canid +00001160 03 n 01 canid 0 001 ~ 00001075 n 0000 | a carnivorous mammal with nonretractile claws +00001257 03 n 01 mouse 0 001 @ 00001340 n 0000 | a small rodent with a long tail +00001340 03 n 01 rodent 0 001 ~ 00001257 n 0000 | a gnawing mammal with chisel teeth +00001427 03 n 01 run 0 001 + 00001075 v 0101 | an act of running at speed +00001503 03 n 01 box 0 000 | a rigid rectangular container +00001564 03 n 01 berry 0 000 | a small juicy fruit +00001617 03 n 01 man 0 000 | an adult male person +00001669 03 n 01 run 0 000 | a score made in baseball diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/data.verb b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/data.verb new file mode 100644 index 0000000000..048546ed71 --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/data.verb @@ -0,0 +1,22 @@ + 1 Licensed to the Apache Software Foundation (ASF) under one or more + 2 contributor license agreements. See the NOTICE file distributed with + 3 this work for additional information regarding copyright ownership. + 4 The ASF licenses this file to You under the Apache License, Version 2.0 + 5 (the "License"); you may not use this file except in compliance with + 6 the License. You may obtain a copy of the License at + 7 + 8 http://www.apache.org/licenses/LICENSE-2.0 + 9 + 10 Unless required by applicable law or agreed to in writing, software + 11 distributed under the License is distributed on an "AS IS" BASIS, + 12 WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + 13 See the License for the specific language governing permissions and + 14 limitations under the License. + 15 + 16 Project-authored miniature WNDB fixture mirroring mini-wn-lmf.xml. + 17 License preamble lines begin with two spaces, as in released WNDB files, + 18 so readers skip them; data line offsets include this preamble. +00001075 29 v 01 run 0 002 @ 00001324 v 0000 + 00001427 n 0101 01 + 02 00 | move fast on foot +00001171 29 v 01 walk 0 001 @ 00001324 v 0000 01 + 02 00 | move at a regular pace +00001255 29 v 01 go 0 000 01 + 02 00 | change location or position +00001324 29 v 01 move 0 002 ~ 00001075 v 0000 ~ 00001171 v 0000 01 + 02 00 | change position in space diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/index.adj b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/index.adj new file mode 100644 index 0000000000..827a988a7d --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/index.adj @@ -0,0 +1,22 @@ + 1 Licensed to the Apache Software Foundation (ASF) under one or more + 2 contributor license agreements. See the NOTICE file distributed with + 3 this work for additional information regarding copyright ownership. + 4 The ASF licenses this file to You under the Apache License, Version 2.0 + 5 (the "License"); you may not use this file except in compliance with + 6 the License. You may obtain a copy of the License at + 7 + 8 http://www.apache.org/licenses/LICENSE-2.0 + 9 + 10 Unless required by applicable law or agreed to in writing, software + 11 distributed under the License is distributed on an "AS IS" BASIS, + 12 WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + 13 See the License for the specific language governing permissions and + 14 limitations under the License. + 15 + 16 Project-authored miniature WNDB fixture mirroring mini-wn-lmf.xml. + 17 License preamble lines begin with two spaces, as in released WNDB files, + 18 so readers skip them; data line offsets include this preamble. +big a 1 1 & 1 0 00001211 +large a 1 1 & 1 0 00001274 +short a 1 1 ! 1 0 00001141 +tall a 1 1 ! 1 0 00001075 diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/index.adv b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/index.adv new file mode 100644 index 0000000000..da20fe1193 --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/index.adv @@ -0,0 +1,20 @@ + 1 Licensed to the Apache Software Foundation (ASF) under one or more + 2 contributor license agreements. See the NOTICE file distributed with + 3 this work for additional information regarding copyright ownership. + 4 The ASF licenses this file to You under the Apache License, Version 2.0 + 5 (the "License"); you may not use this file except in compliance with + 6 the License. You may obtain a copy of the License at + 7 + 8 http://www.apache.org/licenses/LICENSE-2.0 + 9 + 10 Unless required by applicable law or agreed to in writing, software + 11 distributed under the License is distributed on an "AS IS" BASIS, + 12 WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + 13 See the License for the specific language governing permissions and + 14 limitations under the License. + 15 + 16 Project-authored miniature WNDB fixture mirroring mini-wn-lmf.xml. + 17 License preamble lines begin with two spaces, as in released WNDB files, + 18 so readers skip them; data line offsets include this preamble. +quickly r 1 0 1 0 00001075 +well r 1 0 1 0 00001121 diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/index.noun b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/index.noun new file mode 100644 index 0000000000..41a8a4317b --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/index.noun @@ -0,0 +1,27 @@ + 1 Licensed to the Apache Software Foundation (ASF) under one or more + 2 contributor license agreements. See the NOTICE file distributed with + 3 this work for additional information regarding copyright ownership. + 4 The ASF licenses this file to You under the Apache License, Version 2.0 + 5 (the "License"); you may not use this file except in compliance with + 6 the License. You may obtain a copy of the License at + 7 + 8 http://www.apache.org/licenses/LICENSE-2.0 + 9 + 10 Unless required by applicable law or agreed to in writing, software + 11 distributed under the License is distributed on an "AS IS" BASIS, + 12 WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + 13 See the License for the specific language governing permissions and + 14 limitations under the License. + 15 + 16 Project-authored miniature WNDB fixture mirroring mini-wn-lmf.xml. + 17 License preamble lines begin with two spaces, as in released WNDB files, + 18 so readers skip them; data line offsets include this preamble. +berry n 1 0 1 0 00001564 +box n 1 0 1 0 00001503 +canid n 1 1 ~ 1 0 00001160 +dog n 1 1 @ 1 0 00001075 +domestic_dog n 1 1 @ 1 0 00001075 +man n 1 0 1 0 00001617 +mouse n 1 1 @ 1 0 00001257 +rodent n 1 1 ~ 1 0 00001340 +run n 2 1 + 2 1 00001427 00001669 diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/index.verb b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/index.verb new file mode 100644 index 0000000000..2b380478de --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/index.verb @@ -0,0 +1,22 @@ + 1 Licensed to the Apache Software Foundation (ASF) under one or more + 2 contributor license agreements. See the NOTICE file distributed with + 3 this work for additional information regarding copyright ownership. + 4 The ASF licenses this file to You under the Apache License, Version 2.0 + 5 (the "License"); you may not use this file except in compliance with + 6 the License. You may obtain a copy of the License at + 7 + 8 http://www.apache.org/licenses/LICENSE-2.0 + 9 + 10 Unless required by applicable law or agreed to in writing, software + 11 distributed under the License is distributed on an "AS IS" BASIS, + 12 WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + 13 See the License for the specific language governing permissions and + 14 limitations under the License. + 15 + 16 Project-authored miniature WNDB fixture mirroring mini-wn-lmf.xml. + 17 License preamble lines begin with two spaces, as in released WNDB files, + 18 so readers skip them; data line offsets include this preamble. +go v 1 0 1 0 00001255 +move v 1 1 ~ 1 0 00001324 +run v 1 2 @ + 1 1 00001075 +walk v 1 1 @ 1 0 00001171 diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/noun.exc b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/noun.exc new file mode 100644 index 0000000000..e5b3080a88 --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/noun.exc @@ -0,0 +1,3 @@ +men man +mice mouse +oxen ox diff --git a/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/verb.exc b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/verb.exc new file mode 100644 index 0000000000..486d0c7851 --- /dev/null +++ b/opennlp-extensions/opennlp-wordnet/src/test/resources/opennlp/wordnet/mini-wndb/verb.exc @@ -0,0 +1,4 @@ +gone go +ran run +running run +went go diff --git a/opennlp-extensions/pom.xml b/opennlp-extensions/pom.xml index 9afcd3fe3c..8c0e432a58 100644 --- a/opennlp-extensions/pom.xml +++ b/opennlp-extensions/pom.xml @@ -41,6 +41,7 @@ opennlp-morfologik opennlp-spellcheck opennlp-uima + opennlp-wordnet \ No newline at end of file diff --git a/opennlp-tools/src/test/java/opennlp/tools/util/StringUtilTest.java b/opennlp-tools/src/test/java/opennlp/tools/util/StringUtilTest.java index a47306d3d4..73f81c209e 100644 --- a/opennlp-tools/src/test/java/opennlp/tools/util/StringUtilTest.java +++ b/opennlp-tools/src/test/java/opennlp/tools/util/StringUtilTest.java @@ -679,4 +679,23 @@ void testLowercaseBeyondBMP() { String lc = StringUtil.toLowerCase(input); Assertions.assertArrayEquals(expectedCodePoints, lc.codePoints().toArray()); } + + /** + * Verifies the blank check against the toolkit's whitespace definition: the + * no-break space is blank here although the JDK's own check does not cover it, + * whitespace-only and empty values are blank, and any non-whitespace code point, + * supplementary ones included, makes a value non-blank. + */ + @Test + void testIsBlankFollowsTheToolkitWhitespaceDefinition() { + Assertions.assertTrue(StringUtil.isBlank("")); + Assertions.assertTrue(StringUtil.isBlank(" \t\n")); + // U+00A0 no-break space and U+2007 figure space: JDK String.isBlank says false + Assertions.assertTrue(StringUtil.isBlank("\u00A0")); + Assertions.assertTrue(StringUtil.isBlank(" \u00A0\u2007 ")); + Assertions.assertFalse(StringUtil.isBlank("a")); + Assertions.assertFalse(StringUtil.isBlank(" a ")); + // U+10428, a supplementary-plane letter read as one code point, not two chars + Assertions.assertFalse(StringUtil.isBlank("\uD801\uDC28")); + } } diff --git a/pom.xml b/pom.xml index 168f44b845..9e30dd9fad 100644 --- a/pom.xml +++ b/pom.xml @@ -216,6 +216,12 @@ ${project.version} + + opennlp-wordnet + ${project.groupId} + ${project.version} + + opennlp-uima ${project.groupId} diff --git a/rat-excludes b/rat-excludes index 5a5d86b90c..561869bd46 100644 --- a/rat-excludes +++ b/rat-excludes @@ -70,3 +70,12 @@ src/main/resources/opennlp/tools/tokenize/uax29/WordBreakProperty.txt src/main/resources/opennlp/tools/tokenize/uax29/ExtendedPictographic.txt src/main/resources/opennlp/tools/util/normalizer/confusables.txt src/test/resources/opennlp/tools/tokenize/uax29/WordBreakTest.txt + + +src/test/resources/opennlp/wordnet/mini-wndb/noun.exc +src/test/resources/opennlp/wordnet/mini-wndb/verb.exc +src/test/resources/opennlp/wordnet/mini-wndb/adj.exc +src/test/resources/opennlp/wordnet/mini-wndb/adv.exc