feat: hunspell benchmarks
This commit is contained in:
@@ -442,7 +442,7 @@ jmh {
|
|||||||
|
|
||||||
tasks.named('jmh') {
|
tasks.named('jmh') {
|
||||||
group = 'verification'
|
group = 'verification'
|
||||||
description = 'Runs JMH benchmarks for the Radixor algorithmic core and Snowball comparison suite.'
|
description = 'Runs JMH benchmarks for the Radixor algorithmic core and external stemmer comparison suites.'
|
||||||
}
|
}
|
||||||
|
|
||||||
apply from: 'gradle/lucene-benchmarks.gradle'
|
apply from: 'gradle/lucene-benchmarks.gradle'
|
||||||
@@ -518,6 +518,7 @@ javadoc {
|
|||||||
apply from: 'gradle/snowball-benchmarks.gradle'
|
apply from: 'gradle/snowball-benchmarks.gradle'
|
||||||
apply from: 'gradle/paicehusk-benchmarks.gradle'
|
apply from: 'gradle/paicehusk-benchmarks.gradle'
|
||||||
apply from: 'gradle/opennlp-benchmarks.gradle'
|
apply from: 'gradle/opennlp-benchmarks.gradle'
|
||||||
|
apply from: 'gradle/hunspell-benchmarks.gradle'
|
||||||
|
|
||||||
gradle.taskGraph.whenReady { taskGraph ->
|
gradle.taskGraph.whenReady { taskGraph ->
|
||||||
def banner = """
|
def banner = """
|
||||||
|
|||||||
121
gradle/hunspell-benchmarks.gradle
Normal file
121
gradle/hunspell-benchmarks.gradle
Normal file
@@ -0,0 +1,121 @@
|
|||||||
|
import org.gradle.plugins.ide.eclipse.model.SourceFolder
|
||||||
|
|
||||||
|
|
||||||
|
def hunspellDictionaryBaseUrl = 'https://raw.githubusercontent.com/wooorm/dictionaries/main/dictionaries'
|
||||||
|
def hunspellDictionaryLanguages = [
|
||||||
|
en: 'English',
|
||||||
|
cs: 'Czech',
|
||||||
|
de: 'German',
|
||||||
|
es: 'Spanish',
|
||||||
|
fr: 'French',
|
||||||
|
nl: 'Dutch',
|
||||||
|
pl: 'Polish',
|
||||||
|
uk: 'Ukrainian'
|
||||||
|
]
|
||||||
|
def hunspellDownloadDirectory = layout.buildDirectory.dir('third-party/hunspell')
|
||||||
|
def hunspellGeneratedResourcesDirectory = layout.buildDirectory.dir('generated/resources/hunspell')
|
||||||
|
def hunspellGeneratedResourcesPath = provider {
|
||||||
|
project.relativePath(hunspellGeneratedResourcesDirectory.get().asFile)
|
||||||
|
}
|
||||||
|
def hunspellEclipseClasspathAttributes = [
|
||||||
|
gradle_scope : 'jmh',
|
||||||
|
gradle_used_by_scope: 'jmh',
|
||||||
|
test : 'true'
|
||||||
|
]
|
||||||
|
def hunspellIsAbsolutePath = { String path ->
|
||||||
|
path.startsWith('/') || path ==~ /^[A-Za-z]:[\\\/].*/
|
||||||
|
}
|
||||||
|
|
||||||
|
def hunspellDownloadedFiles = hunspellDictionaryLanguages.keySet().collectMany { String code ->
|
||||||
|
[
|
||||||
|
hunspellDownloadDirectory.map { it.file("${code}/index.aff") },
|
||||||
|
hunspellDownloadDirectory.map { it.file("${code}/index.dic") },
|
||||||
|
hunspellDownloadDirectory.map { it.file("${code}/license") }
|
||||||
|
]
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.register('downloadHunspellBenchmarkDictionaries') {
|
||||||
|
group = 'build setup'
|
||||||
|
description = 'Downloads benchmark-only Hunspell dictionaries from wooorm/dictionaries.'
|
||||||
|
|
||||||
|
outputs.files(hunspellDownloadedFiles)
|
||||||
|
|
||||||
|
doLast {
|
||||||
|
hunspellDictionaryLanguages.each { String code, String displayName ->
|
||||||
|
['index.aff', 'index.dic', 'license'].each { String fileName ->
|
||||||
|
final File targetFile = hunspellDownloadDirectory.get().file("${code}/${fileName}").asFile
|
||||||
|
targetFile.parentFile.mkdirs()
|
||||||
|
|
||||||
|
if (!targetFile.exists()) {
|
||||||
|
final URL sourceUrl = new URL("${hunspellDictionaryBaseUrl}/${code}/${fileName}")
|
||||||
|
try {
|
||||||
|
sourceUrl.withInputStream { inputStream ->
|
||||||
|
targetFile.withOutputStream { outputStream ->
|
||||||
|
outputStream << inputStream
|
||||||
|
}
|
||||||
|
}
|
||||||
|
} catch (FileNotFoundException exception) {
|
||||||
|
throw new GradleException(
|
||||||
|
"Unable to download Hunspell ${fileName} file for ${displayName} (${code}) from ${sourceUrl}.",
|
||||||
|
exception)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (targetFile.length() <= 0L) {
|
||||||
|
throw new GradleException("Downloaded Hunspell ${fileName} file for ${displayName} was empty.")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.register('prepareHunspellBenchmarkResources', Copy) {
|
||||||
|
group = 'build setup'
|
||||||
|
description = 'Copies benchmark-only Hunspell dictionaries into the JMH resource output.'
|
||||||
|
|
||||||
|
dependsOn(tasks.named('downloadHunspellBenchmarkDictionaries'))
|
||||||
|
|
||||||
|
from(hunspellDownloadDirectory) {
|
||||||
|
include '**/index.aff'
|
||||||
|
include '**/index.dic'
|
||||||
|
include '**/license'
|
||||||
|
into 'hunspell'
|
||||||
|
}
|
||||||
|
into(hunspellGeneratedResourcesDirectory)
|
||||||
|
}
|
||||||
|
|
||||||
|
sourceSets {
|
||||||
|
jmh {
|
||||||
|
resources {
|
||||||
|
srcDir(hunspellGeneratedResourcesDirectory)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.named('processJmhResources') {
|
||||||
|
dependsOn(tasks.named('prepareHunspellBenchmarkResources'))
|
||||||
|
}
|
||||||
|
|
||||||
|
eclipse {
|
||||||
|
classpath {
|
||||||
|
file {
|
||||||
|
whenMerged { classpath ->
|
||||||
|
String generatedPath = hunspellGeneratedResourcesPath.get()
|
||||||
|
|
||||||
|
classpath.entries.removeAll { entry ->
|
||||||
|
entry.hasProperty('path') && (
|
||||||
|
entry.path == generatedPath ||
|
||||||
|
hunspellIsAbsolutePath(entry.path)
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
SourceFolder hunspellEntry = new SourceFolder(generatedPath, null)
|
||||||
|
hunspellEntry.output = 'bin/jmh'
|
||||||
|
hunspellEclipseClasspathAttributes.each { String name, String value ->
|
||||||
|
hunspellEntry.entryAttributes[name] = value
|
||||||
|
}
|
||||||
|
classpath.entries.add(hunspellEntry)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,280 @@
|
|||||||
|
/*******************************************************************************
|
||||||
|
* Copyright (C) 2026, Leo Galambos
|
||||||
|
* All rights reserved.
|
||||||
|
*
|
||||||
|
* Redistribution and use in source and binary forms, with or without
|
||||||
|
* modification, are permitted provided that the following conditions are met:
|
||||||
|
*
|
||||||
|
* 1. Redistributions of source code must retain the above copyright notice,
|
||||||
|
* this list of conditions and the following disclaimer.
|
||||||
|
*
|
||||||
|
* 2. Redistributions in binary form must reproduce the above copyright notice,
|
||||||
|
* this list of conditions and the following disclaimer in the documentation
|
||||||
|
* and/or other materials provided with the distribution.
|
||||||
|
*
|
||||||
|
* 3. Neither the name of the copyright holder nor the names of its contributors
|
||||||
|
* may be used to endorse or promote products derived from this software
|
||||||
|
* without specific prior written permission.
|
||||||
|
*
|
||||||
|
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||||
|
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||||
|
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
||||||
|
* ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE
|
||||||
|
* LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
|
||||||
|
* CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
|
||||||
|
* SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
|
||||||
|
* INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
|
||||||
|
* CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
|
||||||
|
* ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||||
|
* POSSIBILITY OF SUCH DAMAGE.
|
||||||
|
******************************************************************************/
|
||||||
|
package org.egothor.stemmer.benchmark;
|
||||||
|
|
||||||
|
import java.io.IOException;
|
||||||
|
import java.io.InputStream;
|
||||||
|
import java.text.ParseException;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Objects;
|
||||||
|
import java.util.concurrent.TimeUnit;
|
||||||
|
|
||||||
|
import org.apache.lucene.analysis.LowerCaseFilter;
|
||||||
|
import org.apache.lucene.analysis.TokenStream;
|
||||||
|
import org.apache.lucene.analysis.hunspell.Dictionary;
|
||||||
|
import org.apache.lucene.analysis.hunspell.HunspellStemFilter;
|
||||||
|
import org.apache.lucene.analysis.hunspell.SortingStrategy;
|
||||||
|
import org.apache.lucene.analysis.tokenattributes.CharTermAttribute;
|
||||||
|
import org.apache.lucene.analysis.tokenattributes.PositionIncrementAttribute;
|
||||||
|
import org.egothor.stemmer.StemmerPatchTrieLoader;
|
||||||
|
import org.openjdk.jmh.annotations.AuxCounters;
|
||||||
|
import org.openjdk.jmh.annotations.Benchmark;
|
||||||
|
import org.openjdk.jmh.annotations.BenchmarkMode;
|
||||||
|
import org.openjdk.jmh.annotations.Fork;
|
||||||
|
import org.openjdk.jmh.annotations.Level;
|
||||||
|
import org.openjdk.jmh.annotations.Measurement;
|
||||||
|
import org.openjdk.jmh.annotations.Mode;
|
||||||
|
import org.openjdk.jmh.annotations.OutputTimeUnit;
|
||||||
|
import org.openjdk.jmh.annotations.Scope;
|
||||||
|
import org.openjdk.jmh.annotations.Setup;
|
||||||
|
import org.openjdk.jmh.annotations.State;
|
||||||
|
import org.openjdk.jmh.annotations.Warmup;
|
||||||
|
import org.openjdk.jmh.infra.Blackhole;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Emits exact-root agreement metrics for the benchmark-only English Hunspell
|
||||||
|
* comparison.
|
||||||
|
*/
|
||||||
|
@BenchmarkMode(Mode.AverageTime)
|
||||||
|
@OutputTimeUnit(TimeUnit.NANOSECONDS)
|
||||||
|
@Warmup(iterations = 0)
|
||||||
|
@Measurement(iterations = 1, time = 1, timeUnit = TimeUnit.MILLISECONDS)
|
||||||
|
@Fork(0)
|
||||||
|
public class EnglishHunspellStemmerComparisonBenchmarkQuality {
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Shared English quality corpus and Hunspell dictionary.
|
||||||
|
*/
|
||||||
|
@State(Scope.Benchmark)
|
||||||
|
public static class SharedState {
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Complete English resource-derived corpus.
|
||||||
|
*/
|
||||||
|
private LanguageBenchmarkCorpus.Corpus corpus;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Parsed benchmark-only Hunspell dictionary.
|
||||||
|
*/
|
||||||
|
private Dictionary dictionary;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Initializes quality resources.
|
||||||
|
*
|
||||||
|
* @throws IOException if corpus or dictionary loading fails
|
||||||
|
* @throws ParseException if the Hunspell dictionary cannot be parsed
|
||||||
|
*/
|
||||||
|
@Setup(Level.Trial)
|
||||||
|
public void setUp() throws IOException, ParseException {
|
||||||
|
this.corpus = LanguageBenchmarkCorpus.createFullCorpus(StemmerPatchTrieLoader.Language.US_UK);
|
||||||
|
this.dictionary = loadEnglishDictionary();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* JMH auxiliary counters for exact-root agreement.
|
||||||
|
*/
|
||||||
|
@State(Scope.Thread)
|
||||||
|
@AuxCounters(AuxCounters.Type.EVENTS)
|
||||||
|
public static class AccuracyCounters {
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Number of exact-root matches.
|
||||||
|
*/
|
||||||
|
public long correctMatches;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Number of evaluated tokens.
|
||||||
|
*/
|
||||||
|
public long evaluatedTokens;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Number of exact-root matches where the input token differs from the
|
||||||
|
* expected root.
|
||||||
|
*/
|
||||||
|
public long changedCorrectMatches;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Number of evaluated tokens where the input token differs from the expected
|
||||||
|
* root.
|
||||||
|
*/
|
||||||
|
public long changedEvaluatedTokens;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Number of exact-root matches where the input token is already the expected
|
||||||
|
* root.
|
||||||
|
*/
|
||||||
|
public long rootPreservedMatches;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Number of evaluated tokens where the input token is already the expected
|
||||||
|
* root.
|
||||||
|
*/
|
||||||
|
public long rootEvaluatedTokens;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Resets counters before the measured iteration.
|
||||||
|
*/
|
||||||
|
@Setup(Level.Iteration)
|
||||||
|
public void reset() {
|
||||||
|
this.correctMatches = 0L;
|
||||||
|
this.evaluatedTokens = 0L;
|
||||||
|
this.changedCorrectMatches = 0L;
|
||||||
|
this.changedEvaluatedTokens = 0L;
|
||||||
|
this.rootPreservedMatches = 0L;
|
||||||
|
this.rootEvaluatedTokens = 0L;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Evaluates exact-root agreement for English Hunspell.
|
||||||
|
*
|
||||||
|
* @param sharedState shared English quality state
|
||||||
|
* @param counters JMH auxiliary counters
|
||||||
|
* @param blackhole result sink
|
||||||
|
* @return exact-root match count
|
||||||
|
* @throws IOException if Lucene token streaming fails
|
||||||
|
*/
|
||||||
|
@Benchmark
|
||||||
|
public int luceneHunspellStemFilterAccuracy(final SharedState sharedState, final AccuracyCounters counters,
|
||||||
|
final Blackhole blackhole) throws IOException {
|
||||||
|
final String[] actualStems = firstHunspellOutputs(sharedState.corpus.tokens(), sharedState.dictionary,
|
||||||
|
blackhole);
|
||||||
|
final String[] tokens = sharedState.corpus.tokens();
|
||||||
|
final String[] expectedRoots = sharedState.corpus.expectedRoots();
|
||||||
|
|
||||||
|
int correct = 0;
|
||||||
|
int changedCorrect = 0;
|
||||||
|
int changedEvaluated = 0;
|
||||||
|
int rootPreserved = 0;
|
||||||
|
int rootEvaluated = 0;
|
||||||
|
for (int index = 0; index < actualStems.length; index++) {
|
||||||
|
final String token = tokens[index];
|
||||||
|
final String expectedRoot = expectedRoots[index];
|
||||||
|
final boolean exact = Objects.equals(expectedRoot, actualStems[index]);
|
||||||
|
if (exact) {
|
||||||
|
correct++;
|
||||||
|
}
|
||||||
|
if (Objects.equals(token, expectedRoot)) {
|
||||||
|
rootEvaluated++;
|
||||||
|
if (exact) {
|
||||||
|
rootPreserved++;
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
changedEvaluated++;
|
||||||
|
if (exact) {
|
||||||
|
changedCorrect++;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
counters.correctMatches += correct;
|
||||||
|
counters.evaluatedTokens += actualStems.length;
|
||||||
|
counters.changedCorrectMatches += changedCorrect;
|
||||||
|
counters.changedEvaluatedTokens += changedEvaluated;
|
||||||
|
counters.rootPreservedMatches += rootPreserved;
|
||||||
|
counters.rootEvaluatedTokens += rootEvaluated;
|
||||||
|
return correct;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Extracts the first emitted Hunspell stem for each input token.
|
||||||
|
*
|
||||||
|
* @param tokens token corpus
|
||||||
|
* @param dictionary Hunspell dictionary
|
||||||
|
* @param blackhole result sink
|
||||||
|
* @return first emitted term per input token
|
||||||
|
* @throws IOException if Lucene streaming fails
|
||||||
|
*/
|
||||||
|
private static String[] firstHunspellOutputs(final String[] tokens, final Dictionary dictionary,
|
||||||
|
final Blackhole blackhole) throws IOException {
|
||||||
|
final String[] outputs = new String[tokens.length];
|
||||||
|
final BenchmarkTokenStream input = new BenchmarkTokenStream(tokens);
|
||||||
|
final TokenStream output = new HunspellStemFilter(new LowerCaseFilter(input), dictionary, true);
|
||||||
|
final CharTermAttribute termAttribute = output.addAttribute(CharTermAttribute.class);
|
||||||
|
final PositionIncrementAttribute positionAttribute = output.addAttribute(PositionIncrementAttribute.class);
|
||||||
|
int inputIndex = -1;
|
||||||
|
boolean recordedForPosition = false;
|
||||||
|
|
||||||
|
output.reset();
|
||||||
|
while (output.incrementToken()) {
|
||||||
|
final int positionIncrement = positionAttribute.getPositionIncrement();
|
||||||
|
if (positionIncrement > 0) {
|
||||||
|
inputIndex += positionIncrement;
|
||||||
|
recordedForPosition = false;
|
||||||
|
}
|
||||||
|
if (inputIndex >= 0 && inputIndex < outputs.length && !recordedForPosition) {
|
||||||
|
outputs[inputIndex] = termAttribute.toString();
|
||||||
|
recordedForPosition = true;
|
||||||
|
}
|
||||||
|
blackhole.consume(termAttribute);
|
||||||
|
}
|
||||||
|
output.end();
|
||||||
|
output.close();
|
||||||
|
|
||||||
|
for (int index = 0; index < outputs.length; index++) {
|
||||||
|
if (outputs[index] == null) {
|
||||||
|
outputs[index] = tokens[index];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return outputs;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Loads the benchmark-only English Hunspell dictionary.
|
||||||
|
*
|
||||||
|
* @return parsed Hunspell dictionary
|
||||||
|
* @throws IOException if dictionary resources cannot be read
|
||||||
|
* @throws ParseException if the Hunspell dictionary cannot be parsed
|
||||||
|
*/
|
||||||
|
private static Dictionary loadEnglishDictionary() throws IOException, ParseException {
|
||||||
|
final ClassLoader classLoader = EnglishHunspellStemmerComparisonBenchmarkQuality.class.getClassLoader();
|
||||||
|
try (InputStream affixStream = openRequiredResource(classLoader, "hunspell/en/index.aff");
|
||||||
|
InputStream dictionaryStream = openRequiredResource(classLoader, "hunspell/en/index.dic")) {
|
||||||
|
return new Dictionary(affixStream, List.of(dictionaryStream), true, SortingStrategy.inMemory());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Opens a required classpath resource.
|
||||||
|
*
|
||||||
|
* @param classLoader class loader
|
||||||
|
* @param path resource path
|
||||||
|
* @return resource stream
|
||||||
|
*/
|
||||||
|
private static InputStream openRequiredResource(final ClassLoader classLoader, final String path) {
|
||||||
|
final InputStream stream = classLoader.getResourceAsStream(path);
|
||||||
|
if (stream == null) {
|
||||||
|
throw new IllegalStateException("Missing benchmark-only Hunspell resource: " + path);
|
||||||
|
}
|
||||||
|
return stream;
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,315 @@
|
|||||||
|
/*******************************************************************************
|
||||||
|
* Copyright (C) 2026, Leo Galambos
|
||||||
|
* All rights reserved.
|
||||||
|
*
|
||||||
|
* Redistribution and use in source and binary forms, with or without
|
||||||
|
* modification, are permitted provided that the following conditions are met:
|
||||||
|
*
|
||||||
|
* 1. Redistributions of source code must retain the above copyright notice,
|
||||||
|
* this list of conditions and the following disclaimer.
|
||||||
|
*
|
||||||
|
* 2. Redistributions in binary form must reproduce the above copyright notice,
|
||||||
|
* this list of conditions and the following disclaimer in the documentation
|
||||||
|
* and/or other materials provided with the distribution.
|
||||||
|
*
|
||||||
|
* 3. Neither the name of the copyright holder nor the names of its contributors
|
||||||
|
* may be used to endorse or promote products derived from this software
|
||||||
|
* without specific prior written permission.
|
||||||
|
*
|
||||||
|
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||||
|
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||||
|
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
||||||
|
* ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE
|
||||||
|
* LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
|
||||||
|
* CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
|
||||||
|
* SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
|
||||||
|
* INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
|
||||||
|
* CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
|
||||||
|
* ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||||
|
* POSSIBILITY OF SUCH DAMAGE.
|
||||||
|
******************************************************************************/
|
||||||
|
package org.egothor.stemmer.benchmark;
|
||||||
|
|
||||||
|
import java.io.IOException;
|
||||||
|
import java.io.InputStream;
|
||||||
|
import java.text.ParseException;
|
||||||
|
import java.util.List;
|
||||||
|
import java.util.Locale;
|
||||||
|
import java.util.concurrent.TimeUnit;
|
||||||
|
|
||||||
|
import org.apache.lucene.analysis.LowerCaseFilter;
|
||||||
|
import org.apache.lucene.analysis.TokenStream;
|
||||||
|
import org.apache.lucene.analysis.hunspell.Dictionary;
|
||||||
|
import org.apache.lucene.analysis.hunspell.HunspellStemFilter;
|
||||||
|
import org.apache.lucene.analysis.hunspell.SortingStrategy;
|
||||||
|
import org.apache.lucene.analysis.tokenattributes.CharTermAttribute;
|
||||||
|
import org.egothor.stemmer.CompiledPatchCommand;
|
||||||
|
import org.egothor.stemmer.FrequencyTrie;
|
||||||
|
import org.egothor.stemmer.ReductionMode;
|
||||||
|
import org.egothor.stemmer.StemmerPatchTrieLoader;
|
||||||
|
import org.openjdk.jmh.annotations.Benchmark;
|
||||||
|
import org.openjdk.jmh.annotations.BenchmarkMode;
|
||||||
|
import org.openjdk.jmh.annotations.Level;
|
||||||
|
import org.openjdk.jmh.annotations.Measurement;
|
||||||
|
import org.openjdk.jmh.annotations.Mode;
|
||||||
|
import org.openjdk.jmh.annotations.OutputTimeUnit;
|
||||||
|
import org.openjdk.jmh.annotations.Param;
|
||||||
|
import org.openjdk.jmh.annotations.Scope;
|
||||||
|
import org.openjdk.jmh.annotations.Setup;
|
||||||
|
import org.openjdk.jmh.annotations.State;
|
||||||
|
import org.openjdk.jmh.annotations.Warmup;
|
||||||
|
import org.openjdk.jmh.infra.Blackhole;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Compares Radixor with Lucene's Hunspell integration over selected
|
||||||
|
* benchmark-only Hunspell dictionaries.
|
||||||
|
*/
|
||||||
|
@BenchmarkMode(Mode.AverageTime)
|
||||||
|
@OutputTimeUnit(TimeUnit.NANOSECONDS)
|
||||||
|
@Warmup(iterations = 3, time = 1, timeUnit = TimeUnit.SECONDS)
|
||||||
|
@Measurement(iterations = 5, time = 1, timeUnit = TimeUnit.SECONDS)
|
||||||
|
public class HunspellStemmerComparisonBenchmark {
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Parameterized benchmark case.
|
||||||
|
*/
|
||||||
|
@State(Scope.Benchmark)
|
||||||
|
public static class SharedState {
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Selected language case.
|
||||||
|
*/
|
||||||
|
@Param({ "ENGLISH", "CZECH", "GERMAN", "SPANISH", "FRENCH", "DUTCH", "POLISH", "UKRAINIAN" })
|
||||||
|
public String languageCaseName;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Selected language case descriptor.
|
||||||
|
*/
|
||||||
|
private HunspellLanguageCase languageCase;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Shared deterministic changed-token corpus.
|
||||||
|
*/
|
||||||
|
private String[] tokens;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Radixor benchmark adapter.
|
||||||
|
*/
|
||||||
|
private RadixorBenchmarkStemmer radixorStemmer;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Initializes the selected language corpus and Radixor stemmer.
|
||||||
|
*
|
||||||
|
* @throws IOException if the Radixor corpus or trie cannot be loaded
|
||||||
|
*/
|
||||||
|
@Setup(Level.Trial)
|
||||||
|
public void setUp() throws IOException {
|
||||||
|
this.languageCase = HunspellLanguageCase.valueOf(this.languageCaseName);
|
||||||
|
this.tokens = LanguageBenchmarkCorpus.createTokens(this.languageCase.radixorLanguage());
|
||||||
|
final FrequencyTrie<CompiledPatchCommand> trie = StemmerPatchTrieLoader.loadCompiled(
|
||||||
|
this.languageCase.radixorLanguage(), true,
|
||||||
|
ReductionMode.MERGE_SUBTREES_WITH_EQUIVALENT_RANKED_GET_ALL_RESULTS);
|
||||||
|
this.radixorStemmer = new RadixorBenchmarkStemmer(trie);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Reusable Hunspell filter state.
|
||||||
|
*/
|
||||||
|
@State(Scope.Thread)
|
||||||
|
public static class HunspellState {
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Reusable benchmark input stream.
|
||||||
|
*/
|
||||||
|
private BenchmarkTokenStream input;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Reusable Hunspell filter output stream.
|
||||||
|
*/
|
||||||
|
private TokenStream output;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Output term attribute.
|
||||||
|
*/
|
||||||
|
private CharTermAttribute termAttribute;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Initializes the Hunspell dictionary and filter for the selected language.
|
||||||
|
*
|
||||||
|
* @param sharedState selected language state
|
||||||
|
* @throws IOException if dictionary resources cannot be read
|
||||||
|
* @throws ParseException if the Hunspell dictionary cannot be parsed
|
||||||
|
*/
|
||||||
|
@Setup(Level.Trial)
|
||||||
|
public void setUp(final SharedState sharedState) throws IOException, ParseException {
|
||||||
|
this.input = new BenchmarkTokenStream(new String[0]);
|
||||||
|
final Dictionary dictionary = loadDictionary(sharedState.languageCase);
|
||||||
|
this.output = new HunspellStemFilter(new LowerCaseFilter(this.input), dictionary, true);
|
||||||
|
this.termAttribute = this.output.addAttribute(CharTermAttribute.class);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Runs Hunspell over one token corpus.
|
||||||
|
*
|
||||||
|
* @param tokens token corpus
|
||||||
|
* @param blackhole result sink
|
||||||
|
* @throws IOException if Lucene token streaming fails
|
||||||
|
*/
|
||||||
|
private void run(final String[] tokens, final Blackhole blackhole) throws IOException {
|
||||||
|
this.input.setTokens(tokens);
|
||||||
|
this.output.reset();
|
||||||
|
while (this.output.incrementToken()) {
|
||||||
|
blackhole.consume(this.termAttribute.toString());
|
||||||
|
}
|
||||||
|
this.output.end();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Runs Radixor direct lookup and patch application.
|
||||||
|
*
|
||||||
|
* @param sharedState selected language state
|
||||||
|
* @param blackhole result sink
|
||||||
|
*/
|
||||||
|
@Benchmark
|
||||||
|
public void radixor(final SharedState sharedState, final Blackhole blackhole) {
|
||||||
|
final String[] tokens = sharedState.tokens;
|
||||||
|
final RadixorBenchmarkStemmer stemmer = sharedState.radixorStemmer;
|
||||||
|
for (String token : tokens) {
|
||||||
|
blackhole.consume(stemmer.stem(token));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Runs Lucene HunspellStemFilter over the selected language corpus.
|
||||||
|
*
|
||||||
|
* @param sharedState selected language state
|
||||||
|
* @param hunspellState reusable Hunspell state
|
||||||
|
* @param blackhole result sink
|
||||||
|
* @throws IOException if Lucene token streaming fails
|
||||||
|
*/
|
||||||
|
@Benchmark
|
||||||
|
public void luceneHunspellStemFilter(final SharedState sharedState, final HunspellState hunspellState,
|
||||||
|
final Blackhole blackhole) throws IOException {
|
||||||
|
hunspellState.run(sharedState.tokens, blackhole);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Loads a benchmark-only Hunspell dictionary from generated JMH resources.
|
||||||
|
*
|
||||||
|
* @param languageCase selected language case
|
||||||
|
* @return parsed Hunspell dictionary
|
||||||
|
* @throws IOException if dictionary resources cannot be read
|
||||||
|
* @throws ParseException if the Hunspell dictionary cannot be parsed
|
||||||
|
*/
|
||||||
|
private static Dictionary loadDictionary(final HunspellLanguageCase languageCase) throws IOException,
|
||||||
|
ParseException {
|
||||||
|
final ClassLoader classLoader = HunspellStemmerComparisonBenchmark.class.getClassLoader();
|
||||||
|
final String basePath = "hunspell/" + languageCase.hunspellResourceCode() + "/index.";
|
||||||
|
try (InputStream affixStream = openRequiredResource(classLoader, basePath + "aff");
|
||||||
|
InputStream dictionaryStream = openRequiredResource(classLoader, basePath + "dic")) {
|
||||||
|
return new Dictionary(affixStream, List.of(dictionaryStream), true, SortingStrategy.inMemory());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Opens a classpath resource or fails with a descriptive exception.
|
||||||
|
*
|
||||||
|
* @param classLoader class loader
|
||||||
|
* @param path resource path
|
||||||
|
* @return resource stream
|
||||||
|
*/
|
||||||
|
private static InputStream openRequiredResource(final ClassLoader classLoader, final String path) {
|
||||||
|
final InputStream stream = classLoader.getResourceAsStream(path);
|
||||||
|
if (stream == null) {
|
||||||
|
throw new IllegalStateException("Missing benchmark-only Hunspell resource: " + path);
|
||||||
|
}
|
||||||
|
return stream;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Benchmark language mapping.
|
||||||
|
*/
|
||||||
|
private enum HunspellLanguageCase {
|
||||||
|
|
||||||
|
/**
|
||||||
|
* English Hunspell dictionary over the Radixor English corpus.
|
||||||
|
*/
|
||||||
|
ENGLISH("en", StemmerPatchTrieLoader.Language.US_UK),
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Czech Hunspell dictionary over the Radixor Czech corpus.
|
||||||
|
*/
|
||||||
|
CZECH("cs", StemmerPatchTrieLoader.Language.CS_CZ),
|
||||||
|
|
||||||
|
/**
|
||||||
|
* German Hunspell dictionary over the Radixor German corpus.
|
||||||
|
*/
|
||||||
|
GERMAN("de", StemmerPatchTrieLoader.Language.DE_DE),
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Spanish Hunspell dictionary over the Radixor Spanish corpus.
|
||||||
|
*/
|
||||||
|
SPANISH("es", StemmerPatchTrieLoader.Language.ES_ES),
|
||||||
|
|
||||||
|
/**
|
||||||
|
* French Hunspell dictionary over the Radixor French corpus.
|
||||||
|
*/
|
||||||
|
FRENCH("fr", StemmerPatchTrieLoader.Language.FR_FR),
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Dutch Hunspell dictionary over the Radixor Dutch corpus.
|
||||||
|
*/
|
||||||
|
DUTCH("nl", StemmerPatchTrieLoader.Language.NL_NL),
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Polish Hunspell dictionary over the Radixor Polish corpus.
|
||||||
|
*/
|
||||||
|
POLISH("pl", StemmerPatchTrieLoader.Language.PL_PL),
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Ukrainian Hunspell dictionary over the Radixor Ukrainian corpus.
|
||||||
|
*/
|
||||||
|
UKRAINIAN("uk", StemmerPatchTrieLoader.Language.UK_UA);
|
||||||
|
|
||||||
|
/**
|
||||||
|
* wooorm/dictionaries resource code.
|
||||||
|
*/
|
||||||
|
private final String hunspellResourceCode;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Matching Radixor language.
|
||||||
|
*/
|
||||||
|
private final StemmerPatchTrieLoader.Language radixorLanguage;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Creates a language mapping.
|
||||||
|
*
|
||||||
|
* @param hunspellResourceCode Hunspell resource code
|
||||||
|
* @param radixorLanguage Radixor language
|
||||||
|
*/
|
||||||
|
HunspellLanguageCase(final String hunspellResourceCode, final StemmerPatchTrieLoader.Language radixorLanguage) {
|
||||||
|
this.hunspellResourceCode = hunspellResourceCode.toLowerCase(Locale.ROOT);
|
||||||
|
this.radixorLanguage = radixorLanguage;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Returns the Hunspell resource code.
|
||||||
|
*
|
||||||
|
* @return resource code
|
||||||
|
*/
|
||||||
|
String hunspellResourceCode() {
|
||||||
|
return this.hunspellResourceCode;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Returns the matching Radixor language.
|
||||||
|
*
|
||||||
|
* @return Radixor language
|
||||||
|
*/
|
||||||
|
StemmerPatchTrieLoader.Language radixorLanguage() {
|
||||||
|
return this.radixorLanguage;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user