feat(benchmarks): expand multilingual stemming quality evaluation

* cover all Radixor dictionary languages
* add PRIMARY_OUTPUT, ANY_CANDIDATE, and ALL_CANDIDATES policies
* measure pairwise over-stemming and under-stemming
* add balanced accuracy and complementary quality metrics
* compare single-output and multi-output stemmers fairly
* improve result validation, reporting, and documentation
* move stemming quality tests into the standard test source set
* preserve the existing JMH benchmark structure and badge output
This commit is contained in:
2026-07-20 23:20:17 +02:00
parent 6d35f01303
commit 05f3855b99
62 changed files with 11472 additions and 75 deletions

1
.gitignore vendored
View File

@@ -37,6 +37,7 @@ local.properties
.settings/ .settings/
.loadpath .loadpath
.recommenders .recommenders
.classpath
# External tool builders # External tool builders
.externalToolBuilders/ .externalToolBuilders/

View File

@@ -30,6 +30,11 @@ apply from: 'gradle/maven-pom.gradle'
configurations { configurations {
mockitoAgent mockitoAgent
stemmingQualityJmhRuntime {
canBeConsumed = false
canBeResolved = true
extendsFrom(jmhImplementation, jmhRuntimeOnly)
}
} }
java { java {
@@ -78,6 +83,16 @@ dependencies {
} }
} }
sourceSets.jmh.compileClasspath = sourceSets.jmh.compileClasspath - sourceSets.test.output
sourceSets.jmh.runtimeClasspath = sourceSets.jmh.runtimeClasspath - sourceSets.test.output
sourceSets.test.compileClasspath += sourceSets.jmh.output + configurations.jmhCompileClasspath
sourceSets.test.runtimeClasspath += sourceSets.jmh.output + configurations.jmhCompileClasspath
tasks.named('compileJmhJava', JavaCompile) {
classpath = classpath - sourceSets.test.output
setDependsOn([tasks.named('classes')])
}
dependencyCheck { dependencyCheck {
failBuildOnCVSS = 7.0 failBuildOnCVSS = 7.0
failOnError = true failOnError = true
@@ -426,6 +441,7 @@ tasks.named('distTar') {
jmh { jmh {
jmhVersion = '1.37' jmhVersion = '1.37'
includeTests = false
warmupIterations = 3 warmupIterations = 3
iterations = 5 iterations = 5
fork = 1 fork = 1
@@ -468,6 +484,85 @@ tasks.register('regressionArtifactGenerator', JavaExec) {
} }
} }
tasks.register('stemmingQuality', JavaExec) {
group = 'verification'
description = 'Evaluates pairwise over-stemming and under-stemming against bundled dictionary groups.'
dependsOn(tasks.named('testClasses'))
dependsOn(tasks.named('jmhClasses'))
classpath = files(sourceSets.test.runtimeClasspath, configurations.stemmingQualityJmhRuntime)
mainClass = 'org.egothor.stemmer.benchmark.quality.StemmingQualityApplication'
args layout.buildDirectory.dir('reports/stemming-quality').get().asFile.absolutePath,
layout.projectDirectory.dir('src/main/resources').asFile.absolutePath,
providers.gradleProperty('stemmingQualityLanguage').getOrElse(''),
providers.gradleProperty('stemmingQualityStemmer').getOrElse(''),
providers.gradleProperty('stemmingQualityMode').getOrElse(''),
providers.gradleProperty('stemmingQualityOutputPolicy').getOrElse(''),
providers.gradleProperty('stemmingQualityRankMetric').getOrElse('PAIRWISE_F05'),
providers.gradleProperty('stemmingQualityAudit').getOrElse('false'),
providers.gradleProperty('stemmingQualityAuditLimit').getOrElse('25')
maxHeapSize = '6g'
}
tasks.register('publishStemmingQualityDocumentation', JavaExec) {
group = 'documentation'
description = 'Publishes validated complete stemming-quality results on the language benchmark pages.'
dependsOn(tasks.named('testClasses'))
classpath = sourceSets.test.runtimeClasspath
mainClass = 'org.egothor.stemmer.benchmark.quality.StemmingQualityDocumentationPublisher'
args layout.buildDirectory.file('reports/stemming-quality/stemming-quality.csv').get().asFile.absolutePath,
layout.projectDirectory.dir('docs').asFile.absolutePath,
'update'
doFirst {
if (!file("$buildDir/reports/stemming-quality/stemming-quality.csv").isFile()) {
throw new GradleException('A complete stemming-quality CSV is required. Run stemmingQuality only when no validated complete report is available.')
}
}
}
tasks.register('verifyStemmingQualityDocumentation', JavaExec) {
group = 'verification'
description = 'Verifies published language-page quality tables against the checked-in authoritative CSV.'
dependsOn(tasks.named('testClasses'))
classpath = sourceSets.test.runtimeClasspath
mainClass = 'org.egothor.stemmer.benchmark.quality.StemmingQualityDocumentationPublisher'
args layout.projectDirectory.file('docs/benchmarks/data/stemming-quality.csv').asFile.absolutePath,
layout.projectDirectory.dir('docs').asFile.absolutePath,
'verify'
}
tasks.named('check') {
dependsOn(tasks.named('verifyStemmingQualityDocumentation'))
}
tasks.register('verifyStemmingQualitySourceSets') {
group = 'verification'
description = 'Verifies the production, JMH, and standard-test ownership of stemming-quality infrastructure.'
doLast {
if (sourceSets.findByName('stemmingQualityTest') != null || file('src/stemmingQualityTest').exists()) {
throw new GradleException('The obsolete stemmingQualityTest source set or directory still exists.')
}
if (!file('src/jmh/java/org/egothor/stemmer/benchmark/QualityStemmerMatrix.java').isFile()) {
throw new GradleException('The authoritative JMH stemmer matrix is not in src/jmh.')
}
if (!file('src/test/java/org/egothor/stemmer/benchmark/quality/StemmingQualityApplication.java').isFile()) {
throw new GradleException('The stemming-quality evaluator is not in the standard test source set.')
}
}
}
tasks.register('verifyProductionJarExcludesStemmingQuality') {
group = 'verification'
description = 'Verifies that analytical stemming-quality classes are absent from the production JAR.'
dependsOn(tasks.named('jar'))
doLast {
final File archive = tasks.named('jar').get().archiveFile.get().asFile
final def forbidden = zipTree(archive).matching { include '**/benchmark/**' }.files
if (!forbidden.isEmpty()) {
throw new GradleException("Production JAR contains analytical stemming-quality classes: ${forbidden}")
}
}
}
tasks.register('printDependencyCheckNvdConfig') { tasks.register('printDependencyCheckNvdConfig') {
doLast { doLast {
System.out.println("NVD API key present: " + (nvdApiKey != null && !nvdApiKey.isBlank())) System.out.println("NVD API key present: " + (nvdApiKey != null && !nvdApiKey.isBlank()))

View File

@@ -67,6 +67,104 @@
padding: 0.45rem 0.7rem; padding: 0.45rem 0.7rem;
} }
/* Publication-quality benchmark tables retain identity columns while scrolling. */
.quality-table {
max-width: 100%;
overflow-x: auto;
margin: 0.65rem 0 1rem;
border: 1px solid var(--md-default-fg-color--lightest);
border-radius: 0.2rem;
scrollbar-gutter: stable;
}
.quality-table:focus {
outline: 0.15rem solid var(--md-accent-fg-color);
outline-offset: 0.1rem;
}
.quality-table::before {
content: "Scrollable table: Rank, Stemmer, and Output policy remain visible.";
display: block;
padding: 0.35rem 0.55rem;
color: var(--md-default-fg-color--light);
font-size: 0.68rem;
}
.quality-table .md-typeset__table,
.quality-table table {
margin: 0;
}
.quality-table table th:nth-child(1),
.quality-table table td:nth-child(1),
.quality-table table th:nth-child(2),
.quality-table table td:nth-child(2),
.quality-table table th:nth-child(3),
.quality-table table td:nth-child(3) {
position: sticky;
z-index: 2;
background: var(--md-default-bg-color);
background-clip: padding-box;
}
.quality-table table th:nth-child(1),
.quality-table table td:nth-child(1) {
left: 0;
min-width: 2.8rem;
}
.quality-table table th:nth-child(2),
.quality-table table td:nth-child(2) {
left: 2.8rem;
min-width: 13rem;
white-space: normal;
}
.quality-table table th:nth-child(3),
.quality-table table td:nth-child(3) {
left: 15.8rem;
min-width: 8.5rem;
box-shadow: 0.2rem 0 0.25rem rgb(0 0 0 / 8%);
}
.quality-details > summary {
font-weight: 600;
}
@media screen and (max-width: 44.99em) {
.quality-table table th:nth-child(2),
.quality-table table td:nth-child(2) {
min-width: 10rem;
}
.quality-table table th:nth-child(3),
.quality-table table td:nth-child(3) {
position: static;
min-width: 7.5rem;
box-shadow: none;
}
}
@media print {
.quality-table {
overflow: visible;
border: 0;
}
.quality-table::before {
display: none;
}
.quality-table table th,
.quality-table table td {
position: static !important;
}
.quality-details:not([open]) > *:not(summary) {
display: block;
}
}
/* Code blocks */ /* Code blocks */
.md-typeset pre > code { .md-typeset pre > code {
font-size: 0.72rem; font-size: 0.72rem;

View File

@@ -18,6 +18,9 @@ This page is the entry point for benchmark interpretation. Detailed tables and l
| Page | Purpose | | Page | Purpose |
| --- | --- | | --- | --- |
| [Benchmark methodology](benchmarks/reference/methodology.md) | Workload design, speed pass, quality pass, normalization policy, and exact-root metrics. | | [Benchmark methodology](benchmarks/reference/methodology.md) | Workload design, speed pass, quality pass, normalization policy, and exact-root metrics. |
| [Linguistic quality methodology](benchmarks/reference/linguistic-quality.md) | Pairwise gold standard, over/under-stemming, candidate policies, metrics, and ranking rules. |
| [Tested stemmers](benchmarks/reference/tested-stemmers.md) | Upstream attribution, tested versions, language coverage, adapter behaviour, and limitations. |
| [Reproducibility and raw data](benchmarks/reference/reproducibility.md) | Versioned quality snapshot, checksum, commands, reports, and provenance limitations. |
| [Benchmark corpora](benchmarks/reference/corpora.md) | Dictionary row counts, complete quality tokens, already-root tokens, changed speed tokens, and timing token counts. | | [Benchmark corpora](benchmarks/reference/corpora.md) | Dictionary row counts, complete quality tokens, already-root tokens, changed speed tokens, and timing token counts. |
| [Benchmark environment and reports](benchmarks/reference/environment.md) | Hardware, OS, JVM, JMH settings, report files, and current badge/report policy. | | [Benchmark environment and reports](benchmarks/reference/environment.md) | Hardware, OS, JVM, JMH settings, report files, and current badge/report policy. |
| [English dictionary coverage benchmark](benchmarks/reference/english-coverage.md) | The quality/speed operating curve for contracted Radixor tries built from 100% down to 10% of English dictionary rows. | | [English dictionary coverage benchmark](benchmarks/reference/english-coverage.md) | The quality/speed operating curve for contracted Radixor tries built from 100% down to 10% of English dictionary rows. |

View File

@@ -0,0 +1,309 @@
Stemmer,Language,Dictionary mode,Output policy,Applied dictionary rows,Processed word forms,Singleton dictionary rows,Forms with one candidate,Forms with multiple candidates,Maximum candidates for one form,Total candidate assignments,Distinct output stems,True-positive pairs,False-positive pairs,False-negative pairs,True-negative pairs,Over-stemming error pairs,Over-stemming possible pairs,Over-stemming percentage,Under-stemming error pairs,Under-stemming possible pairs,Under-stemming percentage,Pairwise precision,Pairwise recall,Pairwise specificity,Pairwise accuracy,Balanced accuracy,Pairwise F0.5,Pairwise F1,Pairwise F2,Jaccard index,Fowlkes-Mallows index,Matthews correlation coefficient,Pairwise error rate,Adjusted Rand Index,Homogeneity,Completeness,V-measure,Normalized mutual information
"CZECH_LUCENE_CZECH_STEM_FILTER","CS_CZ","ALL_WORDS","PRIMARY_OUTPUT","5113","51676","2","51676","0","1","51676","9647","177249","14480","124586","1334862335","14480","1334876815","0.001085","124586","301835","41.276194","0.924476735392","0.587238060530","0.999989152557","0.999895844650","0.793613606543","0.829234311828","0.718241200736","0.633453389361","0.560355974266","0.736809286788","0.736765291417","0.000104155350","0.718191706079","0.993800637348","0.944976928457","0.968774025802","0.968774025802"
"CZECH_LUCENE_CZECH_STEM_FILTER","CS_CZ","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","5038","50968","2","50968","0","1","50968","9558","174387","13950","124426","1298530265","13950","1298544215","0.001074","124426","298813","41.640089","0.925930645598","0.583599107134","0.999989257201","0.999893462107","0.791794182167","0.828708724235","0.715947860002","0.630197985095","0.557569149804","0.735100195918","0.735055396892","0.000106537893","0.715897321649","0.993897397445","0.944297456221","0.968462775946","0.968462775946"
"CZECH_RADIXOR","CS_CZ","ALL_WORDS","PRIMARY_OUTPUT","5113","51676","2","51676","0","1","51676","5162","299762","3867","2073","1334872948","3867","1334876815","0.000290","2073","301835","0.686799","0.987264062392","0.993132009210","0.999997103103","0.999995551157","0.996564556157","0.988432097845","0.990189342389","0.991952846154","0.980569312599","0.990193689085","0.990191466141","0.000004448843","0.990187117482","0.998733220675","0.998685552738","0.998709386137","0.998709386137"
"CZECH_RADIXOR","CS_CZ","ALL_WORDS","ANY_CANDIDATE","5113","51676","2","51080","596","4","52319","5166","301835","0","0","1334876815","0","1334876815","0.000000","0","301835","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"CZECH_RADIXOR","CS_CZ","ALL_WORDS","ALL_CANDIDATES","5113","51676","2","51080","596","4","52319","5166","301835","5850","0","1334870965","5850","1334876815","0.000438","0","301835","0.000000","0.980987048442","1.000000000000","0.999995617573","0.999995618564","0.999997808787","0.984731579205","0.990402283764","0.996138677580","0.980987048442","0.990447902942","0.990445732657","0.000004381436","","","","",""
"CZECH_RADIXOR","CS_CZ","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","5038","50968","2","50968","0","1","50968","5037","297104","3863","1709","1298540352","3863","1298544215","0.000297","1709","298813","0.571930","0.987164705765","0.994280703985","0.999997025130","0.999995710028","0.997138864558","0.988579745136","0.990709926973","0.992849308824","0.981590876052","0.990716315904","0.990714173387","0.000004289972","0.990707781520","0.998726091764","0.999029907266","0.998877976413","0.998877976413"
"CZECH_RADIXOR","CS_CZ","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","5038","50968","2","50428","540","4","51543","5040","298813","0","0","1298544215","0","1298544215","0.000000","0","298813","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"CZECH_RADIXOR","CS_CZ","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","5038","50968","2","50428","540","4","51543","5040","298813","5782","0","1298538433","5782","1298544215","0.000445","0","298813","0.000000","0.981017416570","1.000000000000","0.999995547321","0.999995548346","0.999997773661","0.984756059381","0.990417760454","0.996144940117","0.981017416570","0.990463233325","0.990461028216","0.000004451654","","","","",""
"DA_DK_RADIXOR","DA_DK","ALL_WORDS","PRIMARY_OUTPUT","4179","28079","32","28079","0","1","28079","4184","89188","1165","707","394110021","1165","394111186","0.000296","707","89895","0.786473","0.987106128186","0.992135268925","0.999997043981","0.999995251155","0.996066156453","0.988107873355","0.989614309174","0.991125345329","0.979442126071","0.989617503860","0.989615130363","0.000004748845","0.989611934224","0.998465862775","0.998718664384","0.998592247580","0.998592247580"
"DA_DK_RADIXOR","DA_DK","ALL_WORDS","ANY_CANDIDATE","4179","28079","32","27756","323","3","28405","4187","89895","0","0","394111186","0","394111186","0.000000","0","89895","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"DA_DK_RADIXOR","DA_DK","ALL_WORDS","ALL_CANDIDATES","4179","28079","32","27756","323","3","28405","4187","89895","1849","0","394109337","1849","394111186","0.000469","0","89895","0.000000","0.979846093478","1.000000000000","0.999995308431","0.999995309500","0.999997654215","0.983811622975","0.989820468071","0.995903164910","0.979846093478","0.989871756076","0.989869434047","0.000004690500","","","","",""
"DA_DK_RADIXOR","DA_DK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4173","28033","32","28033","0","1","28033","4170","89077","1165","663","392819623","1165","392820788","0.000297","663","89740","0.738801","0.987090268389","0.992611990194","0.999997034271","0.999995347541","0.996304512232","0.988189692661","0.989843428787","0.991502709249","0.979891095099","0.989847279032","0.989844954043","0.000004652459","0.989841102043","0.998463063294","0.998811876590","0.998637439483","0.998637439483"
"DA_DK_RADIXOR","DA_DK","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","4173","28033","32","27718","315","3","28351","4173","89740","0","0","392820788","0","392820788","0.000000","0","89740","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"DA_DK_RADIXOR","DA_DK","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","4173","28033","32","27718","315","3","28351","4173","89740","1849","0","392818939","1849","392820788","0.000471","0","89740","0.000000","0.979811986156","1.000000000000","0.999995293019","0.999995294094","0.999997646509","0.983784115625","0.989803065147","0.995896117847","0.979811986156","0.989854527774","0.989852198158","0.000004705906","","","","",""
"ENGLISH_LUCENE_KSTEM_FILTER","US_UK","ALL_WORDS","PRIMARY_OUTPUT","396939","607439","250964","607439","0","1","607439","371125","237565","1368501","76305","184489083270","1368501","184490451771","0.000742","76305","313870","24.311020","0.147917333410","0.756889795138","0.999992582267","0.999992168681","0.878441188702","0.176283968232","0.247471790726","0.415099040868","0.141208449266","0.334599940499","0.334597833111","0.000007831319","0.247469648794","0.980686838187","0.992107972963","0.986364345289","0.986364345289"
"ENGLISH_LUCENE_KSTEM_FILTER","US_UK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","374384","583910","228735","583910","0","1","583910","347624","237551","1367069","74340","170473473135","1367069","170474840204","0.000802","74340","311891","23.835250","0.148041904002","0.761647498645","0.999991980817","0.999991544756","0.880819739731","0.176476898525","0.247899438093","0.416437018089","0.141486991947","0.335791223646","0.335788961354","0.000008455244","0.247897133948","0.979822223835","0.991990504792","0.985868818390","0.985868818390"
"ENGLISH_LUCENE_MINIMAL_FILTER","US_UK","ALL_WORDS","PRIMARY_OUTPUT","396939","607439","250964","607439","0","1","607439","453328","137225","1122264","176645","184489329507","1122264","184490451771","0.000608","176645","313870","56.279670","0.108952916619","0.437203300730","0.999993916953","0.999992959490","0.718598608842","0.128203906480","0.174435713655","0.272816484020","0.095551668577","0.218253464509","0.218250987161","0.000007040510","0.174433464995","0.995202198233","0.981173943304","0.988138284715","0.988138284715"
"ENGLISH_LUCENE_MINIMAL_FILTER","US_UK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","374384","583910","228735","583910","0","1","583910","430129","136932","1120871","174959","170473719333","1120871","170474840204","0.000657","174959","311891","56.096200","0.108866014789","0.439037997249","0.999993425006","0.999992398716","0.719515711128","0.128139023335","0.174469673707","0.273277328232","0.095572048952","0.218623688336","0.218621020778","0.000007601284","0.174467253215","0.994993790771","0.980519680106","0.987703711280","0.987703711280"
"ENGLISH_LUCENE_PORTER_COPIED","US_UK","ALL_WORDS","PRIMARY_OUTPUT","396939","607439","250964","607439","0","1","607439","319968","285390","1557406","28480","184488894365","1557406","184490451771","0.000844","28480","313870","9.073820","0.154867928951","0.909261796285","0.999991558338","0.999991403982","0.954626677312","0.185678591198","0.264658505304","0.460562583837","0.152510906996","0.375253902399","0.375251973425","0.000008596018","0.264656367392","0.969648379409","0.997199419831","0.983230936080","0.983230936080"
"ENGLISH_LUCENE_PORTER_COPIED","US_UK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","374384","583910","228735","583910","0","1","583910","298779","283761","1552702","28130","170473287502","1552702","170474840204","0.000911","28130","311891","9.019177","0.154514956196","0.909808234287","0.999990891899","0.999990726907","0.954899563093","0.185277176317","0.264165961476","0.460049474275","0.152183881415","0.374938634269","0.374936557303","0.000009273093","0.264163659878","0.968644600847","0.997108656044","0.982670549062","0.982670549062"
"ENGLISH_LUCENE_PORTER_FILTER","US_UK","ALL_WORDS","PRIMARY_OUTPUT","396939","607439","250964","607439","0","1","607439","319968","285390","1557406","28480","184488894365","1557406","184490451771","0.000844","28480","313870","9.073820","0.154867928951","0.909261796285","0.999991558338","0.999991403982","0.954626677312","0.185678591198","0.264658505304","0.460562583837","0.152510906996","0.375253902399","0.375251973425","0.000008596018","0.264656367392","0.969648379409","0.997199419831","0.983230936080","0.983230936080"
"ENGLISH_LUCENE_PORTER_FILTER","US_UK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","374384","583910","228735","583910","0","1","583910","298779","283761","1552702","28130","170473287502","1552702","170474840204","0.000911","28130","311891","9.019177","0.154514956196","0.909808234287","0.999990891899","0.999990726907","0.954899563093","0.185277176317","0.264165961476","0.460049474275","0.152183881415","0.374938634269","0.374936557303","0.000009273093","0.264163659878","0.968644600847","0.997108656044","0.982670549062","0.982670549062"
"ENGLISH_LUCENE_POSSESSIVE_FILTER","US_UK","ALL_WORDS","PRIMARY_OUTPUT","396939","607439","250964","607439","0","1","607439","591899","7","1115154","313863","184489336617","1115154","184490451771","0.000604","313863","313870","99.997770","0.000006277121","0.000022302227","0.999993955492","0.999992254263","0.500008128860","0.000007330589","0.000009796848","0.000014763939","0.000004898448","0.000011831896","0.000008625150","0.000007745737","0.000007141644","0.995789196698","0.958018540631","0.976538780935","0.976538780935"
"ENGLISH_LUCENE_POSSESSIVE_FILTER","US_UK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","374384","583910","228735","583910","0","1","583910","568400","5","1113773","311886","170473726431","1113773","170474840204","0.000653","311886","311891","99.998397","0.000004489225","0.000016031242","0.999993466643","0.999991637145","0.500004748942","0.000005244385","0.000007014251","0.000010587200","0.000003507138","0.000008483387","0.000005026087","0.000008362855","0.000004155674","0.995605378040","0.956423154691","0.975621022465","0.975621022465"
"ENGLISH_OPENNLP_PORTER","US_UK","ALL_WORDS","PRIMARY_OUTPUT","396939","607439","250964","607439","0","1","607439","319968","285390","1557406","28480","184488894365","1557406","184490451771","0.000844","28480","313870","9.073820","0.154867928951","0.909261796285","0.999991558338","0.999991403982","0.954626677312","0.185678591198","0.264658505304","0.460562583837","0.152510906996","0.375253902399","0.375251973425","0.000008596018","0.264656367392","0.969648379409","0.997199419831","0.983230936080","0.983230936080"
"ENGLISH_OPENNLP_PORTER","US_UK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","374384","583910","228735","583910","0","1","583910","298779","283761","1552702","28130","170473287502","1552702","170474840204","0.000911","28130","311891","9.019177","0.154514956196","0.909808234287","0.999990891899","0.999990726907","0.954899563093","0.185277176317","0.264165961476","0.460049474275","0.152183881415","0.374938634269","0.374936557303","0.000009273093","0.264163659878","0.968644600847","0.997108656044","0.982670549062","0.982670549062"
"ENGLISH_PAICE_HUSK_LANCASTER","US_UK","ALL_WORDS","PRIMARY_OUTPUT","396939","607439","250964","607439","0","1","607439","268169","283991","3062661","29879","184487389110","3062661","184490451771","0.001660","29879","313870","9.519546","0.084858240415","0.904804536910","0.999983399352","0.999983237427","0.952393968131","0.103642734217","0.155164208820","0.308542866654","0.084107327905","0.277092260667","0.277089454298","0.000016762573","0.155161580693","0.937768073854","0.996599815184","0.966289292109","0.966289292109"
"ENGLISH_PAICE_HUSK_LANCASTER","US_UK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","374384","583910","228735","583910","0","1","583910","249411","282398","3045870","29493","170471794334","3045870","170474840204","0.001787","29493","311891","9.456188","0.084848335531","0.905438117804","0.999982133023","0.999981960051","0.952710125414","0.103632575002","0.155156958803","0.308575577075","0.084103067491","0.277173081705","0.277170064389","0.000018039949","0.155154132315","0.936076835754","0.996486960554","0.965337716341","0.965337716341"
"ENGLISH_RADIXOR","US_UK","ALL_WORDS","PRIMARY_OUTPUT","396939","607439","250964","607439","0","1","607439","390361","292001","1149886","21869","184489301885","1149886","184490451771","0.000623","21869","313870","6.967534","0.202513095686","0.930324656705","0.999993767233","0.999993648707","0.965159211969","0.240076409811","0.332621199859","0.541270431499","0.199487482886","0.434054059102","0.434052478080","0.000006351293","0.332619335001","0.994214506865","0.997769723414","0.995988942533","0.995988942533"
"ENGLISH_RADIXOR","US_UK","ALL_WORDS","ANY_CANDIDATE","396939","607439","250964","578231","29208","1355","2838145","397392","313855","12","15","184490451759","12","184490451771","0.000000","15","313870","0.004779","0.999961767245","0.999952209513","0.999999999935","0.999999999854","0.999976104724","0.999959855684","0.999956988357","0.999954121045","0.999913980413","0.999956988368","0.999956988295","0.000000000146","","","","",""
"ENGLISH_RADIXOR","US_UK","ALL_WORDS","ALL_CANDIDATES","396939","607439","250964","578231","29208","1355","2838145","397392","313855","11482166","15","184478969605","11482166","184490451771","0.006224","15","313870","0.004779","0.026606853277","0.999952209513","0.999937762817","0.999937762842","0.999944986165","0.033038791524","0.051834488023","0.120237128281","0.026606819443","0.163112175274","0.163107098882","0.000062237158","","","","",""
"ENGLISH_RADIXOR","US_UK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","374384","583910","228735","583910","0","1","583910","367590","290572","1148489","21319","170473691715","1148489","170474840204","0.000674","21319","311891","6.835401","0.201917778329","0.931645991709","0.999993263000","0.999993137956","0.965819627354","0.239424468968","0.331901731173","0.540775136091","0.198970131062","0.433723286019","0.433721583515","0.000006862044","0.331899721995","0.993959181482","0.997731171071","0.995841604460","0.995841604460"
"ENGLISH_RADIXOR","US_UK","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","374384","583910","228735","555084","28826","1355","2812871","374506","311891","0","0","170474840204","0","170474840204","0.000000","0","311891","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"ENGLISH_RADIXOR","US_UK","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","374384","583910","228735","555084","28826","1355","2812871","374506","311891","11470018","0","170463370186","11470018","170474840204","0.006728","0","311891","0.000000","0.026472025883","1.000000000000","0.999932717239","0.999932717362","0.999966358619","0.032872482055","0.051578660140","0.119686728696","0.026472025883","0.162702261457","0.162696787836","0.000067282638","","","","",""
"ENGLISH_SNOWBALL_ORIGINAL_PORTER","US_UK","ALL_WORDS","PRIMARY_OUTPUT","396939","607439","250964","607439","0","1","607439","321092","285304","1555293","28566","184488896478","1555293","184490451771","0.000843","28566","313870","9.101220","0.155006228957","0.908987797496","0.999991569791","0.999991414969","0.954489683644","0.185835337999","0.264848800190","0.460750814660","0.152637303435","0.375364850057","0.375362921954","0.000008585031","0.264846663203","0.969891477221","0.997192899073","0.983352728141","0.983352728141"
"ENGLISH_SNOWBALL_ORIGINAL_PORTER","US_UK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","374384","583910","228735","583910","0","1","583910","299877","283675","1550615","28216","170473289589","1550615","170474840204","0.000910","28216","311891","9.046750","0.154651118416","0.909532496930","0.999990904142","0.999990738644","0.954761700536","0.185431499934","0.264353286139","0.460234326480","0.152308234175","0.375046954242","0.375044878196","0.000009261356","0.264350985524","0.968893806180","0.997101844661","0.982795461434","0.982795461434"
"ENGLISH_SNOWBALL_PORTER2","US_UK","ALL_WORDS","PRIMARY_OUTPUT","396939","607439","250964","607439","0","1","607439","318385","285334","1566711","28536","184488885060","1566711","184490451771","0.000849","28536","313870","9.091662","0.154064291094","0.909083378469","0.999991507902","0.999991353242","0.954537443185","0.184752753479","0.263476636895","0.459101696688","0.151726514306","0.374242282819","0.374240346981","0.000008646758","0.263474493989","0.969037354042","0.997181597682","0.982908049045","0.982908049045"
"ENGLISH_SNOWBALL_PORTER2","US_UK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","374384","583910","228735","583910","0","1","583910","297220","283730","1561891","28161","170473278313","1561891","170474840204","0.000916","28161","311891","9.029116","0.153731454074","0.909708840589","0.999990837997","0.999990672823","0.954849839293","0.184374949232","0.263015918336","0.458637294569","0.151421029768","0.373966392672","0.373964308569","0.000009327177","0.263013611479","0.968019617024","0.997095706553","0.982342555079","0.982342555079"
"FINNISH_LUCENE_FINNISH_LIGHT_STEM_FILTER","FI_FI","ALL_WORDS","PRIMARY_OUTPUT","57027","1811717","292","1811717","0","1","1811717","439975","12355389","2223150","19168306","1641124591341","2223150","1641126814491","0.000135","19168306","31523695","60.806025","0.847505295284","0.391939745642","0.999998645351","0.999986965635","0.695969195497","0.687649407375","0.535999578676","0.439151826652","0.366119825424","0.576342788507","0.576337821084","0.000013034365","0.535993941880","0.988126027331","0.886473473160","0.934543630400","0.934543630400"
"FINNISH_LUCENE_FINNISH_LIGHT_STEM_FILTER","FI_FI","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","54762","1757055","274","1757055","0","1","1757055","431848","11988389","1806392","18825444","1543587637760","1806392","1543589444152","0.000117","18825444","30813833","61.094133","0.869052506162","0.389058673746","0.999998829746","0.999986634125","0.694528751746","0.697056446146","0.537492108587","0.437372459518","0.367513988637","0.581474346349","0.581469391800","0.000013365875","0.537486398327","0.989268269625","0.885293761089","0.934397488899","0.934397488899"
"FINNISH_RADIXOR","FI_FI","ALL_WORDS","PRIMARY_OUTPUT","57027","1811717","292","1811717","0","1","1811717","69091","30552427","731279","971268","1641126083212","731279","1641126814491","0.000045","971268","31523695","3.081073","0.976624284859","0.969189271753","0.999999554404","0.999998962594","0.984594413078","0.975128170336","0.972892573600","0.970667204156","0.947215985975","0.972899675927","0.972899157490","0.000001037406","0.972892054895","0.996084757586","0.993746341306","0.994914175412","0.994914175412"
"FINNISH_RADIXOR","FI_FI","ALL_WORDS","ANY_CANDIDATE","57027","1811717","292","1754389","57328","6","1876272","69769","31523695","0","0","1641126814491","0","1641126814491","0.000000","0","31523695","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"FINNISH_RADIXOR","FI_FI","ALL_WORDS","ALL_CANDIDATES","57027","1811717","292","1754389","57328","6","1876272","69769","31523695","1683575","0","1641125130916","1683575","1641126814491","0.000103","0","31523695","0.000000","0.949301011495","1.000000000000","0.999998974135","0.999998974154","0.999999487067","0.959025334376","0.973991195713","0.989431554710","0.949301011495","0.974320794962","0.974320295201","0.000001025846","","","","",""
"FINNISH_RADIXOR","FI_FI","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","54762","1757055","274","1757055","0","1","1757055","54633","30078528","730145","735305","1543588714007","730145","1543589444152","0.000047","735305","30813833","2.386282","0.976300667023","0.976137178390","0.999999526982","0.999999050641","0.988068352686","0.976267964916","0.976218915862","0.976169871736","0.953542638154","0.976218919284","0.976218444595","0.000000949359","0.976218441173","0.996000407428","0.996068984852","0.996034694959","0.996034694959"
"FINNISH_RADIXOR","FI_FI","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","54762","1757055","274","1712724","44331","6","1805864","54984","30813833","0","0","1543589444152","0","1543589444152","0.000000","0","30813833","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"FINNISH_RADIXOR","FI_FI","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","54762","1757055","274","1712724","44331","6","1805864","54984","30813833","1653320","0","1543587790832","1653320","1543589444152","0.000107","0","30813833","0.000000","0.949077148834","1.000000000000","0.999998928912","0.999998928933","0.999999464456","0.958842548108","0.973873352732","0.989382907677","0.949077148834","0.974205906795","0.974205385065","0.000001071067","","","","",""
"FRENCH_LUCENE_FRENCH_LIGHT_STEM_FILTER","FR_FR","ALL_WORDS","PRIMARY_OUTPUT","59240","425210","2301","425210","0","1","425210","245918","202782","276403","5251833","90395828427","276403","90396104830","0.000306","5251833","5454615","96.282377","0.423181026117","0.037176226003","0.999996942313","0.999938848002","0.518586584158","0.137547303040","0.068348107452","0.045471618191","0.035383242558","0.125428359900","0.125414592230","0.000061151998","0.068339028277","0.974109647704","0.812375827422","0.885921707253","0.885921707253"
"FRENCH_LUCENE_FRENCH_LIGHT_STEM_FILTER","FR_FR","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","57698","421231","2133","421231","0","1","421231","245182","200690","262689","5239869","88711863817","262689","88712126506","0.000296","5239869","5440559","96.311225","0.433101197939","0.036887753630","0.999997038860","0.999937976681","0.518442396245","0.137570562409","0.067985131280","0.045148356975","0.035188720533","0.126396717862","0.126383026150","0.000062023319","0.067976159358","0.975085555241","0.811143708698","0.885591261484","0.885591261484"
"FRENCH_LUCENE_FRENCH_MINIMAL_STEM_FILTER","FR_FR","ALL_WORDS","PRIMARY_OUTPUT","59240","425210","2301","425210","0","1","425210","269236","183612","160438","5271003","90395944392","160438","90396104830","0.000177","5271003","5454615","96.633823","0.533678244441","0.033661770812","0.999998225167","0.999939918724","0.516829997990","0.134399775137","0.063329059361","0.041424008382","0.032699958487","0.134031916915","0.134021061615","0.000060081276","0.063322352769","0.984019125555","0.810978546011","0.889158144694","0.889158144694"
"FRENCH_LUCENE_FRENCH_MINIMAL_STEM_FILTER","FR_FR","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","57698","421231","2133","421231","0","1","421231","268411","181686","147476","5258873","88711979030","147476","88712126506","0.000166","5258873","5440559","96.660527","0.551965293685","0.033394730211","0.999998337589","0.999939061122","0.516696533900","0.134438681544","0.062979128454","0.041121435592","0.032513396928","0.135767198057","0.135756528540","0.000060938878","0.062972571968","0.985086367216","0.809773733549","0.888868235590","0.888868235590"
"FRENCH_RADIXOR","FR_FR","ALL_WORDS","PRIMARY_OUTPUT","59240","425210","2301","425210","0","1","425210","60225","4985455","318767","469160","90395786063","318767","90396104830","0.000353","469160","5454615","8.601157","0.939903156391","0.913988429981","0.999996473664","0.999991284144","0.956992451823","0.934603310507","0.926764667966","0.919056419273","0.863524187383","0.926855226151","0.926850879168","0.000008715856","0.926760310630","0.988772235003","0.985214034569","0.986989927876","0.986989927876"
"FRENCH_RADIXOR","FR_FR","ALL_WORDS","ANY_CANDIDATE","59240","425210","2301","382170","43040","56","477024","60383","5454383","12","232","90396104818","12","90396104830","0.000000","232","5454615","0.004253","0.999997799939","0.999957467209","0.999999999867","0.999999997301","0.999978733538","0.999989733133","0.999977633167","0.999965533495","0.999955267335","0.999977633371","0.999977632021","0.000000002699","","","","",""
"FRENCH_RADIXOR","FR_FR","ALL_WORDS","ALL_CANDIDATES","59240","425210","2301","382170","43040","56","477024","60383","5454383","1056255","232","90395048575","1056255","90396104830","0.001168","232","5454615","0.004253","0.837764747479","0.999957467209","0.999988315260","0.999988313399","0.999972891234","0.865852951156","0.911703747510","0.962682080453","0.837734895644","0.915275431226","0.915270082203","0.000011686601","","","","",""
"FRENCH_RADIXOR","FR_FR","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","57698","421231","2133","421231","0","1","421231","58069","4975123","315266","465436","88711811240","315266","88712126506","0.000355","465436","5440559","8.554930","0.940407784758","0.914450702584","0.999996446190","0.999991200142","0.957223574387","0.935099145312","0.927247620620","0.919526848134","0.864363145162","0.927338427699","0.927334038919","0.000008799858","0.927243221287","0.988915897225","0.985549842615","0.987230000708","0.987230000708"
"FRENCH_RADIXOR","FR_FR","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","57698","421231","2133","380101","41130","56","468574","58208","5440559","0","0","88712126506","0","88712126506","0.000000","0","5440559","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"FRENCH_RADIXOR","FR_FR","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","57698","421231","2133","380101","41130","56","468574","58208","5440559","938985","0","88711187521","938985","88712126506","0.001058","0","5440559","0.000000","0.852813147774","1.000000000000","0.999989415370","0.999989416019","0.999994707685","0.878679151458","0.920560336911","0.966633773699","0.852813147774","0.923478829088","0.923473941734","0.000010583981","","","","",""
"GERMAN_CISTEM","DE_DE","ALL_WORDS","PRIMARY_OUTPUT","54092","296974","1474","296974","0","1","296974","59097","1053889","477122","329983","44094768857","477122","44095245979","0.001082","329983","1383872","23.844908","0.688361481400","0.761550923785","0.999989179741","0.999981696901","0.880770051763","0.701851885397","0.723108954973","0.745693871888","0.566304351331","0.724031989665","0.724022910459","0.000018303099","0.723099826442","0.974048119240","0.975147027686","0.974597263694","0.974597263694"
"GERMAN_CISTEM","DE_DE","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","16007","150098","228","150098","0","1","150098","23023","725447","156784","147964","11263599558","156784","11263756342","0.001392","147964","873411","16.940936","0.822286906717","0.830590638313","0.999986080665","0.999972946470","0.915288359489","0.823934343933","0.826417914358","0.828916502414","0.704184159310","0.826428343371","0.826414817348","0.000027053530","0.826404386881","0.985935685912","0.973569821618","0.979713735095","0.979713735095"
"GERMAN_LUCENE_GERMAN_LIGHT_STEM_FILTER","DE_DE","ALL_WORDS","PRIMARY_OUTPUT","54092","296974","1474","296974","0","1","296974","98357","709263","205740","674609","44095040239","205740","44095245979","0.000467","674609","1383872","48.747933","0.775148278202","0.512520666651","0.999995334191","0.999980035912","0.756258000421","0.703092101246","0.617052253820","0.549774428024","0.446186239158","0.630301128270","0.630292039259","0.000019964088","0.617042686770","0.980753120457","0.936533167951","0.958133203614","0.958133203614"
"GERMAN_LUCENE_GERMAN_LIGHT_STEM_FILTER","DE_DE","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","16007","150098","228","150098","0","1","150098","50335","471565","55477","401846","11263700865","55477","11263756342","0.000493","401846","873411","46.008809","0.894738939212","0.539911908597","0.999995074734","0.999959401861","0.769953491666","0.790797426464","0.673446377708","0.586423560557","0.507666155661","0.695039717114","0.695022690564","0.000040598139","0.673427319227","0.991320177896","0.915069562866","0.951669957566","0.951669957566"
"GERMAN_LUCENE_GERMAN_MINIMAL_STEM_FILTER","DE_DE","ALL_WORDS","PRIMARY_OUTPUT","54092","296974","1474","296974","0","1","296974","140505","271626","110840","1112246","44095135139","110840","44095245979","0.000251","1112246","1383872","80.372029","0.710196461908","0.196279713731","0.999997486350","0.999972263504","0.598138600041","0.466112921692","0.307558349534","0.229493166050","0.181724639931","0.373359288402","0.373350267608","0.000027736496","0.307548938689","0.983615403456","0.896263607272","0.937910029995","0.937910029995"
"GERMAN_LUCENE_GERMAN_MINIMAL_STEM_FILTER","DE_DE","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","16007","150098","228","150098","0","1","150098","80363","132221","21214","741190","11263735128","21214","11263756342","0.000188","741190","873411","84.861537","0.861739498811","0.151384628772","0.999998116614","0.999932318770","0.575691372693","0.444544636019","0.257528392768","0.181269722976","0.147794886125","0.361184321539","0.361168285320","0.000067681230","0.257511188319","0.992642524078","0.854402700840","0.918349417869","0.918349417869"
"GERMAN_LUCENE_GERMAN_STEM_FILTER","DE_DE","ALL_WORDS","PRIMARY_OUTPUT","54092","296974","1474","296974","0","1","296974","81085","619354","331871","764518","44094914108","331871","44095245979","0.000753","764518","1383872","55.244849","0.651111987174","0.447551507654","0.999992473769","0.999975136671","0.723771990712","0.596821367368","0.530473894660","0.477402037056","0.360982967729","0.539820480819","0.539808754751","0.000024863329","0.530461889452","0.975549631706","0.942889706548","0.958941664321","0.958941664321"
"GERMAN_LUCENE_GERMAN_STEM_FILTER","DE_DE","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","16007","150098","228","150098","0","1","150098","41574","378734","78723","494677","11263677619","78723","11263756342","0.000699","494677","873411","56.637368","0.827911694432","0.433626322545","0.999993010946","0.999949097306","0.716809666745","0.700518896035","0.569153364571","0.479276535831","0.397773842757","0.599169678345","0.599148958371","0.000050902694","0.569130398173","0.988583531594","0.918717729459","0.952371013508","0.952371013508"
"GERMAN_RADIXOR","DE_DE","ALL_WORDS","PRIMARY_OUTPUT","54092","296974","1474","296974","0","1","296974","68104","1128969","98192","254903","44095147787","98192","44095245979","0.000223","254903","1383872","18.419550","0.919984419322","0.815804496370","0.999997773184","0.999991992699","0.907901134777","0.897072808397","0.864768082211","0.834709150216","0.761754553110","0.866329859738","0.866325955582","0.000008007301","0.864764092865","0.989946248415","0.975085217969","0.982459538105","0.982459538105"
"GERMAN_RADIXOR","DE_DE","ALL_WORDS","ANY_CANDIDATE","54092","296974","1474","248400","48574","8","361016","70717","1272705","1375","111167","44095244604","1375","44095245979","0.000003","111167","1383872","8.033041","0.998920789903","0.919669593720","0.999999968818","0.999997447832","0.959834781269","0.981996366774","0.957658377578","0.934497606897","0.918756727140","0.958476435291","0.958475209548","0.000002552168","","","","",""
"GERMAN_RADIXOR","DE_DE","ALL_WORDS","ALL_CANDIDATES","54092","296974","1474","248400","48574","8","361016","70717","1272705","244817","111167","44095001162","244817","44095245979","0.000555","111167","1383872","8.033041","0.838673179038","0.919669593720","0.999994447996","0.999991927184","0.959832020858","0.853710645080","0.877305874349","0.902242446842","0.781429112618","0.878238135035","0.878234164088","0.000008072816","","","","",""
"GERMAN_RADIXOR","DE_DE","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","16007","150098","228","150098","0","1","150098","17264","814297","47898","59114","11263708444","47898","11263756342","0.000425","59114","873411","6.768177","0.944446441930","0.932318232768","0.999995747600","0.999990500176","0.966156990184","0.941995622128","0.938343149309","0.934718891125","0.883847872972","0.938362743125","0.938357995965","0.000009499824","0.938338399230","0.994062310308","0.990664418294","0.992360455671","0.992360455671"
"GERMAN_RADIXOR","DE_DE","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","16007","150098","228","135120","14978","8","167157","18366","873411","0","0","11263756342","0","11263756342","0.000000","0","873411","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"GERMAN_RADIXOR","DE_DE","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","16007","150098","228","135120","14978","8","167157","18366","873411","97544","0","11263658798","97544","11263756342","0.000866","0","873411","0.000000","0.899538083639","1.000000000000","0.999991340012","0.999991340683","0.999995670006","0.917982540684","0.947112449481","0.978151677228","0.899538083639","0.948439815507","0.948435708759","0.000008659317","","","","",""
"HE_IL_RADIXOR","HE_IL","ALL_WORDS","PRIMARY_OUTPUT","2358","58714","0","58714","0","1","58714","2358","688361","25234","19916","1722904030","25234","1722929264","0.001465","19916","708277","2.811894","0.964638205144","0.971881057835","0.999985354013","0.999973805398","0.985933205924","0.966078126522","0.968246086849","0.970423799230","0.938446730860","0.968252859146","0.968239762122","0.000026194602","0.968232984327","0.993166390361","0.993627509527","0.993396896433","0.993396896433"
"HE_IL_RADIXOR","HE_IL","ALL_WORDS","ANY_CANDIDATE","2358","58714","0","56674","2040","40","62376","2358","708277","0","0","1722929264","0","1722929264","0.000000","0","708277","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"HE_IL_RADIXOR","HE_IL","ALL_WORDS","ALL_CANDIDATES","2358","58714","0","56674","2040","40","62376","2358","708277","86628","0","1722842636","86628","1722929264","0.005028","0","708277","0.000000","0.891020939609","1.000000000000","0.999949720513","0.999949741174","0.999974860256","0.910874182109","0.942370251906","0.976122467036","0.891020939609","0.943939055029","0.943915324345","0.000050258826","","","","",""
"HE_IL_RADIXOR","HE_IL","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","2358","58714","0","58714","0","1","58714","2358","688361","25234","19916","1722904030","25234","1722929264","0.001465","19916","708277","2.811894","0.964638205144","0.971881057835","0.999985354013","0.999973805398","0.985933205924","0.966078126522","0.968246086849","0.970423799230","0.938446730860","0.968252859146","0.968239762122","0.000026194602","0.968232984327","0.993166390361","0.993627509527","0.993396896433","0.993396896433"
"HE_IL_RADIXOR","HE_IL","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","2358","58714","0","56674","2040","40","62376","2358","708277","0","0","1722929264","0","1722929264","0.000000","0","708277","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"HE_IL_RADIXOR","HE_IL","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","2358","58714","0","56674","2040","40","62376","2358","708277","86628","0","1722842636","86628","1722929264","0.005028","0","708277","0.000000","0.891020939609","1.000000000000","0.999949720513","0.999949741174","0.999974860256","0.910874182109","0.942370251906","0.976122467036","0.891020939609","0.943939055029","0.943915324345","0.000050258826","","","","",""
"HUNGARIAN_LUCENE_HUNGARIAN_LIGHT_STEM_FILTER","HU_HU","ALL_WORDS","PRIMARY_OUTPUT","19406","916344","1","916344","0","1","916344","94328","14036270","4132555","8125833","419816410338","4132555","419820542893","0.000984","8125833","22162103","36.665442","0.772546931351","0.633345580968","0.999990156377","0.999970802427","0.816667868673","0.740017627855","0.696054898613","0.657022705053","0.533806904809","0.699492090778","0.699477892426","0.000029197573","0.696040442258","0.982615378770","0.926771756762","0.953876941828","0.953876941828"
"HUNGARIAN_LUCENE_HUNGARIAN_LIGHT_STEM_FILTER","HU_HU","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","18360","878513","1","878513","0","1","878513","91516","13492703","3639046","7918708","385867055871","3639046","385870694917","0.000943","7918708","21411411","36.983588","0.787584676848","0.630164121365","0.999990569261","0.999970049260","0.815077345313","0.750107959995","0.700134758022","0.656404225003","0.538621031944","0.704491026122","0.704476576404","0.000029950740","0.700119966552","0.983686881315","0.925487153321","0.953699929950","0.953699929950"
"HUNGARIAN_RADIXOR","HU_HU","ALL_WORDS","PRIMARY_OUTPUT","19406","916344","1","916344","0","1","916344","20535","21962266","272900","199837","419820269993","272900","419820542893","0.000065","199837","22162103","0.901706","0.987726648859","0.990982940563","0.999999349960","0.999998874014","0.995491145262","0.988376194087","0.989352115329","0.990329965723","0.978928596533","0.989353455019","0.989352892139","0.000001125986","0.989351552308","0.998036093538","0.997808712909","0.997922390271","0.997922390271"
"HUNGARIAN_RADIXOR","HU_HU","ALL_WORDS","ANY_CANDIDATE","19406","916344","1","904024","12320","5","929326","20567","22162103","0","0","419820542893","0","419820542893","0.000000","0","22162103","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"HUNGARIAN_RADIXOR","HU_HU","ALL_WORDS","ALL_CANDIDATES","19406","916344","1","904024","12320","5","929326","20567","22162103","460158","0","419820082735","460158","419820542893","0.000110","0","22162103","0.000000","0.979659062372","1.000000000000","0.999998903917","0.999998903975","0.999999451959","0.983660778882","0.989725029923","0.995864516790","0.979659062372","0.989777279176","0.989776736737","0.000001096025","","","","",""
"HUNGARIAN_RADIXOR","HU_HU","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","18360","878513","1","878513","0","1","878513","18363","21247134","272775","164277","385870422142","272775","385870694917","0.000071","164277","21411411","0.767240","0.987324528185","0.992327595785","0.999999293092","0.999998867424","0.996163444439","0.988321101757","0.989819739994","0.991322930046","0.979844666523","0.989822900984","0.989822335019","0.000001132576","0.989819173678","0.997945135090","0.998273386381","0.998109233747","0.998109233747"
"HUNGARIAN_RADIXOR","HU_HU","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","18360","878513","1","867360","11153","5","890245","18375","21411411","0","0","385870694917","0","385870694917","0.000000","0","21411411","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"HUNGARIAN_RADIXOR","HU_HU","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","18360","878513","1","867360","11153","5","890245","18375","21411411","458462","0","385870236455","458462","385870694917","0.000119","0","21411411","0.000000","0.979036823854","1.000000000000","0.999998811877","0.999998811943","0.999999405938","0.983158850285","0.989407384494","0.995735852714","0.979036823854","0.989462896653","0.989462308851","0.000001188057","","","","",""
"HUNSPELL_CZECH_LUCENE_FILTER","CS_CZ","ALL_WORDS","PRIMARY_OUTPUT","5113","51676","2","51676","0","1","51676","10920","213552","11408","88283","1334865407","11408","1334876815","0.000855","88283","301835","29.248762","0.949288762447","0.707512382593","0.999991453893","0.999925335085","0.853751918243","0.888559718726","0.810759403563","0.745486280807","0.681745481942","0.819532521678","0.819499025505","0.000074664915","0.810722859062","0.995776551361","0.952852006662","0.973841506371","0.973841506371"
"HUNSPELL_CZECH_LUCENE_FILTER","CS_CZ","ALL_WORDS","ANY_CANDIDATE","5113","51676","2","48359","3317","5","55596","11359","224312","10102","77523","1334866713","10102","1334876815","0.000757","77523","301835","25.683900","0.956905304291","0.743160998559","0.999992432261","0.999934372078","0.871576715410","0.904855299474","0.836596431881","0.777913569166","0.719093919606","0.843288029954","0.843258147533","0.000065627922","","","","",""
"HUNSPELL_CZECH_LUCENE_FILTER","CS_CZ","ALL_WORDS","ALL_CANDIDATES","5113","51676","2","48359","3317","5","55596","11359","224312","13917","77523","1334862898","13917","1334876815","0.001043","77523","301835","25.683900","0.941581419558","0.743160998559","0.999989574319","0.999931514783","0.871575286439","0.893850652440","0.830686733424","0.775860578084","0.710405634802","0.836508570179","0.836476906392","0.000068485217","","","","",""
"HUNSPELL_CZECH_LUCENE_FILTER","CS_CZ","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","5038","50968","2","50968","0","1","50968","10816","210827","11239","87986","1298532976","11239","1298544215","0.000866","87986","298813","29.445171","0.949388920411","0.705548286052","0.999991344923","0.999923605087","0.852769815487","0.888008949714","0.809504702628","0.743753342581","0.679973036781","0.818437368155","0.818403143837","0.000076394913","0.809467327086","0.995812473772","0.952393750423","0.973619286126","0.973619286126"
"HUNSPELL_CZECH_LUCENE_FILTER","CS_CZ","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","5038","50968","2","47731","3237","5","54804","11240","221382","10028","77431","1298534187","10028","1298544215","0.000772","77431","298813","25.912862","0.956665658355","0.740871381098","0.999992277506","0.999932663918","0.870431829302","0.904003665310","0.835052421340","0.775874033233","0.716815448726","0.841882537861","0.841851913927","0.000067336082","","","","",""
"HUNSPELL_CZECH_LUCENE_FILTER","CS_CZ","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","5038","50968","2","47731","3237","5","54804","11240","221382","13601","77431","1298530614","13601","1298544215","0.001047","77431","298813","25.912862","0.942119217135","0.740871381098","0.999989525963","0.999929913009","0.870430453530","0.893573737936","0.829462940899","0.773935751817","0.708617411512","0.835457458856","0.835425115185","0.000070086991","","","","",""
"HUNSPELL_DUTCH_LUCENE_FILTER","NL_NL","ALL_WORDS","PRIMARY_OUTPUT","4992","26477","85","26477","0","1","26477","15909","18482","1333","46084","350436627","1333","350437960","0.000380","46084","64566","71.375027","0.932727731517","0.286249728960","0.999996196188","0.999864717095","0.643122962574","0.642512480358","0.438060700869","0.332315636923","0.280459491039","0.516713712165","0.516673857221","0.000135282905","0.438012080403","0.996931617211","0.889026094124","0.939891935467","0.939891935467"
"HUNSPELL_DUTCH_LUCENE_FILTER","NL_NL","ALL_WORDS","ANY_CANDIDATE","4992","26477","85","25223","1254","3","27763","16027","21374","1164","43192","350436796","1164","350437960","0.000332","43192","64566","66.895889","0.948353891206","0.331041105226","0.999996678442","0.999873450270","0.665518891834","0.690740573172","0.490769654666","0.380588457347","0.325178761600","0.560307166017","0.560267948638","0.000126549730","","","","",""
"HUNSPELL_DUTCH_LUCENE_FILTER","NL_NL","ALL_WORDS","ALL_CANDIDATES","4992","26477","85","25223","1254","3","27763","16027","21374","1738","43192","350436222","1738","350437960","0.000496","43192","64566","66.895889","0.924800969193","0.331041105226","0.999995040492","0.999871812621","0.665518072859","0.680639942935","0.487556741714","0.379812066416","0.322363658301","0.553305643343","0.553264631551","0.000128187379","","","","",""
"HUNSPELL_DUTCH_LUCENE_FILTER","NL_NL","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4796","25678","84","25678","0","1","25678","15258","18333","1310","44814","329602546","1310","329603856","0.000397","44814","63147","70.967742","0.933309575930","0.290322580645","0.999996025532","0.999860089122","0.645159303089","0.646808120294","0.442879574828","0.336717714000","0.284422172921","0.520538994337","0.520497519470","0.000139910878","0.442828931093","0.996884174988","0.889060613994","0.939890140871","0.939890140871"
"HUNSPELL_DUTCH_LUCENE_FILTER","NL_NL","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","4796","25678","84","24492","1186","3","26896","15323","21212","1141","41935","329602715","1141","329603856","0.000346","41935","63147","66.408539","0.948955397486","0.335914611937","0.999996538269","0.999869334815","0.667955575103","0.695206444720","0.496187134503","0.385755489360","0.329952712792","0.564595416287","0.564554662442","0.000130665185","","","","",""
"HUNSPELL_DUTCH_LUCENE_FILTER","NL_NL","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","4796","25678","84","24492","1186","3","26896","15323","21212","1712","41935","329602144","1712","329603856","0.000519","41935","63147","66.408539","0.925318443553","0.335914611937","0.999994805886","0.999867602764","0.667954708912","0.684951854459","0.492895400309","0.384956009176","0.327047903915","0.557519493726","0.557476858392","0.000132397236","","","","",""
"HUNSPELL_ENGLISH_LUCENE_FILTER","US_UK","ALL_WORDS","PRIMARY_OUTPUT","396939","607439","250964","607439","0","1","607439","557518","46002","1981986","267868","184488469785","1981986","184490451771","0.001074","267868","313870","85.343614","0.022683566175","0.146563864020","0.999989256972","0.999987805059","0.573276560496","0.027298226808","0.039286754363","0.070050933952","0.020036970960","0.057659267324","0.057655308782","0.000012194941","0.039283923590","0.993096189204","0.963677172906","0.978165531652","0.978165531652"
"HUNSPELL_ENGLISH_LUCENE_FILTER","US_UK","ALL_WORDS","ANY_CANDIDATE","396939","607439","250964","600602","6837","4","614296","557638","51229","1978852","262641","184488472919","1978852","184490451771","0.001073","262641","313870","83.678274","0.025234953679","0.163217255552","0.999989273960","0.999987850378","0.581603264756","0.030369825498","0.043711664621","0.077960810954","0.022344183028","0.064177721084","0.064173802046","0.000012149622","","","","",""
"HUNSPELL_ENGLISH_LUCENE_FILTER","US_UK","ALL_WORDS","ALL_CANDIDATES","396939","607439","250964","600602","6837","4","614296","557638","51229","2008917","262641","184488442854","2008917","184490451771","0.001089","262641","313870","83.678274","0.024866684206","0.163217255552","0.999989110997","0.999987687416","0.581603183275","0.029942881217","0.043158091605","0.077253888104","0.022054971033","0.063707707153","0.063703758400","0.000012312584","","","","",""
"HUNSPELL_ENGLISH_LUCENE_FILTER","US_UK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","374384","583910","228735","583910","0","1","583910","535362","45926","1978041","265965","170472862163","1978041","170474840204","0.001160","265965","311891","85.274984","0.022691081426","0.147250161114","0.999988396874","0.999986836756","0.573619278994","0.027311677226","0.039322595808","0.070190378755","0.020055617372","0.057803679777","0.057799415161","0.000013163244","0.039319549964","0.993066314983","0.962316521867","0.977449637185","0.977449637185"
"HUNSPELL_ENGLISH_LUCENE_FILTER","US_UK","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","374384","583910","228735","577124","6786","4","590716","535485","51150","1974950","260741","170472865254","1974950","170474840204","0.001158","260741","311891","83.600040","0.025245545630","0.163999602425","0.999988415006","0.999986885532","0.581994008716","0.030387494919","0.043755514884","0.078123472659","0.022367099418","0.064344847861","0.064340626006","0.000013114468","","","","",""
"HUNSPELL_ENGLISH_LUCENE_FILTER","US_UK","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","374384","583910","228735","577124","6786","4","590716","535485","51150","2004598","260741","170472835606","2004598","170474840204","0.001176","260741","311891","83.600040","0.024881454342","0.163999602425","0.999988241092","0.999986711618","0.581993921758","0.029965261387","0.043207600483","0.077422296168","0.022080830084","0.063879172034","0.063874918547","0.000013288382","","","","",""
"HUNSPELL_FRENCH_LUCENE_FILTER","FR_FR","ALL_WORDS","PRIMARY_OUTPUT","59240","425210","2301","425210","0","1","425210","154336","3422734","776728","2031881","90395328102","776728","90396104830","0.000859","2031881","5454615","37.250677","0.815041069547","0.627493232795","0.999991407506","0.999968931852","0.813742320150","0.769068574566","0.709075347131","0.657764674673","0.549277098051","0.715145268872","0.715130511338","0.000031068148","0.709060074832","0.978337247291","0.913705954219","0.944917713687","0.944917713687"
"HUNSPELL_FRENCH_LUCENE_FILTER","FR_FR","ALL_WORDS","ANY_CANDIDATE","59240","425210","2301","411699","13511","4","439015","154718","3610612","745831","1844003","90395358999","745831","90396104830","0.000825","1844003","5454615","33.806291","0.828798173189","0.661937093635","0.999991749302","0.999971351888","0.830964421468","0.789018996925","0.736029080656","0.689708764155","0.582314885091","0.740683639600","0.740669908387","0.000028648112","","","","",""
"HUNSPELL_FRENCH_LUCENE_FILTER","FR_FR","ALL_WORDS","ALL_CANDIDATES","59240","425210","2301","411699","13511","4","439015","154718","3610612","1043199","1844003","90395061631","1043199","90396104830","0.001154","1844003","5454615","33.806291","0.775839843947","0.661937093635","0.999988459691","0.999968062476","0.830962776663","0.750027659074","0.714376699201","0.681961135862","0.555665643861","0.716629033342","0.716613365354","0.000031937524","","","","",""
"HUNSPELL_FRENCH_LUCENE_FILTER","FR_FR","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","57698","421231","2133","421231","0","1","421231","153822","3412548","763305","2028011","88711363201","763305","88712126506","0.000860","2028011","5440559","37.275784","0.817209801207","0.627242163903","0.999991395708","0.999968537054","0.813616779806","0.770536594362","0.709734150326","0.657825640123","0.550068151075","0.715952822518","0.715937898033","0.000031462946","0.709718690125","0.979328164393","0.913161860024","0.945088340370","0.945088340370"
"HUNSPELL_FRENCH_LUCENE_FILTER","FR_FR","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","57698","421231","2133","407794","13437","4","434961","154205","3600083","733584","1840476","88711392922","733584","88712126506","0.000827","1840476","5440559","33.828803","0.830724418835","0.661711967465","0.999991730736","0.999970985904","0.830851849101","0.790350629656","0.736648201095","0.689779349655","0.583090317150","0.741417756470","0.741403865778","0.000029014096","","","","",""
"HUNSPELL_FRENCH_LUCENE_FILTER","FR_FR","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","57698","421231","2133","407794","13437","4","434961","154205","3600083","1027635","1840476","88711098871","1027635","88712126506","0.001158","1840476","5440559","33.828803","0.777939148410","0.661711967465","0.999988416071","0.999967671442","0.830850191768","0.751538185756","0.715133880405","0.682093458746","0.556582409247","0.717475884237","0.717460037184","0.000032328558","","","","",""
"HUNSPELL_GERMAN_LUCENE_FILTER","DE_DE","ALL_WORDS","PRIMARY_OUTPUT","54092","296974","1474","296974","0","1","296974","182774","391862","203883","992010","44095042096","203883","44095245979","0.000462","992010","1383872","71.683653","0.657768004767","0.283163471766","0.999995376304","0.999972880172","0.641579424035","0.520145203475","0.395896782054","0.319562150060","0.246802560848","0.431573715426","0.431562811676","0.000027119828","0.395885371175","0.980462638581","0.886872825924","0.931322397630","0.931322397630"
"HUNSPELL_GERMAN_LUCENE_FILTER","DE_DE","ALL_WORDS","ANY_CANDIDATE","54092","296974","1474","289083","7891","3","305052","183111","408175","158403","975697","44095087576","158403","44095245979","0.000359","975697","1383872","70.504859","0.720421548313","0.294951411691","0.999996407708","0.999974281481","0.647473909700","0.559115650060","0.418544438463","0.334456395588","0.264657729653","0.460965674088","0.460955788036","0.000025718519","","","","",""
"HUNSPELL_GERMAN_LUCENE_FILTER","DE_DE","ALL_WORDS","ALL_CANDIDATES","54092","296974","1474","289083","7891","3","305052","183111","408175","242551","975697","44095003428","242551","44095245979","0.000550","975697","1383872","70.504859","0.627260936247","0.294951411691","0.999994499384","0.999972373218","0.647472955538","0.511911128190","0.401234052132","0.329906951166","0.250964847398","0.430129630047","0.430118032816","0.000027626782","","","","",""
"HUNSPELL_GERMAN_LUCENE_FILTER","DE_DE","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","16007","150098","228","150098","0","1","150098","86983","278093","84679","595318","11263671663","84679","11263756342","0.000752","595318","873411","68.160122","0.766577905682","0.318398783620","0.999992482170","0.999939634323","0.659195632895","0.598178360154","0.449922058465","0.360558871242","0.290257700216","0.494041974653","0.494019111671","0.000060365677","0.449897024669","0.988041339480","0.865580709092","0.922765807515","0.922765807515"
"HUNSPELL_GERMAN_LUCENE_FILTER","DE_DE","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","16007","150098","228","145109","4989","3","155207","87393","288864","60996","584547","11263695346","60996","11263756342","0.000542","584547","873411","66.926911","0.825655976676","0.330730893016","0.999994584755","0.999942692923","0.665362738886","0.635466205220","0.472281285177","0.375782098835","0.309141519702","0.522560942370","0.522540242219","0.000057307077","","","","",""
"HUNSPELL_GERMAN_LUCENE_FILTER","DE_DE","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","16007","150098","228","145109","4989","3","155207","87393","288864","96545","584547","11263659797","96545","11263756342","0.000857","584547","873411","66.926911","0.749499881944","0.330730893016","0.999991428703","0.999939537116","0.665361160860","0.598050472724","0.458944090497","0.372338300095","0.297811447117","0.497878263505","0.497854575726","0.000060462884","","","","",""
"HUNSPELL_POLISH_LUCENE_FILTER","PL_PL","ALL_WORDS","PRIMARY_OUTPUT","9990","122341","1","122341","0","1","122341","18419","971262","52652","149705","7482425351","52652","7482478003","0.000704","149705","1120967","13.354987","0.948577712581","0.866450127435","0.999992963294","0.999972959935","0.933221545364","0.930929837176","0.905655838249","0.881717903868","0.827578626454","0.906584403102","0.906571161039","0.000027040065","0.905642343969","0.994545991966","0.970520439141","0.982386343372","0.982386343372"
"HUNSPELL_POLISH_LUCENE_FILTER","PL_PL","ALL_WORDS","ANY_CANDIDATE","9990","122341","1","110894","11447","6","135231","19068","1040224","42213","80743","7482435790","42213","7482478003","0.000564","80743","1120967","7.202977","0.961001887408","0.927970225707","0.999994358420","0.999983569937","0.963982292063","0.954208759768","0.944197251162","0.934393641743","0.894293230626","0.944341642819","0.944333470354","0.000016430063","","","","",""
"HUNSPELL_POLISH_LUCENE_FILTER","PL_PL","ALL_WORDS","ALL_CANDIDATES","9990","122341","1","110894","11447","6","135231","19068","1040224","82745","80743","7482395258","82745","7482478003","0.001106","80743","1120967","7.202977","0.926315864463","0.927970225707","0.999988941498","0.999978153827","0.963979583602","0.926646264647","0.927142307089","0.927638880888","0.864180136112","0.927142676087","0.927131751477","0.000021846173","","","","",""
"HUNSPELL_POLISH_LUCENE_FILTER","PL_PL","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","9846","120925","1","120925","0","1","120925","18149","965984","51950","148667","7310200749","51950","7310252699","0.000711","148667","1114651","13.337538","0.948965257080","0.866624620621","0.999992893543","0.999972560946","0.933308757082","0.931268723294","0.905927782480","0.881929423296","0.828032892137","0.906860880124","0.906847444801","0.000027439054","0.905914089179","0.994583905165","0.970514203019","0.982401644006","0.982401644006"
"HUNSPELL_POLISH_LUCENE_FILTER","PL_PL","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","9846","120925","1","109660","11265","6","133595","18789","1034283","41671","80368","7310211028","41671","7310252699","0.000570","80368","1114651","7.210149","0.961270649117","0.927898508143","0.999994299650","0.999983308321","0.963946403896","0.954405554191","0.944289819479","0.934386268967","0.894459328803","0.944437187555","0.944428885928","0.000016691679","","","","",""
"HUNSPELL_POLISH_LUCENE_FILTER","PL_PL","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","9846","120925","1","109660","11265","6","133595","18789","1034283","81865","80368","7310170834","81865","7310252699","0.001120","80368","1114651","7.210149","0.926653992123","0.927898508143","0.999988801345","0.999977810854","0.963943654744","0.926902628188","0.927275832560","0.927649337585","0.864412176686","0.927276041347","0.927264945147","0.000022189146","","","","",""
"HUNSPELL_SPANISH_LUCENE_FILTER","ES_ES","ALL_WORDS","PRIMARY_OUTPUT","65059","871332","3589","871332","0","1","871332","495840","9662476","536192","32310860","379566781918","536192","379567318110","0.000141","32310860","41973336","76.979490","0.947425291224","0.230205099733","0.999998587360","0.999913471422","0.615101843546","0.583708381625","0.370408466579","0.271277635967","0.227301418167","0.467014061518","0.466991649518","0.000086528578","0.370381248953","0.993314263125","0.790558492734","0.880413722434","0.880413722434"
"HUNSPELL_SPANISH_LUCENE_FILTER","ES_ES","ALL_WORDS","ANY_CANDIDATE","65059","871332","3589","853455","17877","5","890999","496361","10079118","416345","31894218","379566901765","416345","379567318110","0.000110","31894218","41973336","75.986855","0.960330954432","0.240131449166","0.999998903106","0.999914884689","0.620065176136","0.600267728541","0.384194728757","0.282504215637","0.237772914592","0.480214185303","0.480192080762","0.000085115311","","","","",""
"HUNSPELL_SPANISH_LUCENE_FILTER","ES_ES","ALL_WORDS","ALL_CANDIDATES","65059","871332","3589","853455","17877","5","890999","496361","10079118","888077","31894218","379566430033","888077","379567318110","0.000234","31894218","41973336","75.986855","0.919024235459","0.240131449166","0.999997660291","0.999913642011","0.620064554728","0.587073016700","0.380771322449","0.281759130783","0.235155989841","0.469772946730","0.469749183448","0.000086357989","","","","",""
"HUNSPELL_SPANISH_LUCENE_FILTER","ES_ES","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","64918","869371","3525","869371","0","1","869371","495045","9628515","531181","32234855","377860138584","531181","377860669765","0.000141","32234855","41863370","77.000144","0.947716841134","0.229998564377","0.999998594241","0.999913295008","0.614998579309","0.583531128169","0.370163304101","0.271052948234","0.227116805648","0.466876335765","0.466853897372","0.000086704992","0.370136051001","0.993362468962","0.790499503024","0.880396073652","0.880396073652"
"HUNSPELL_SPANISH_LUCENE_FILTER","ES_ES","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","64918","869371","3525","851564","17807","5","888962","495572","10041262","412198","31822108","377860257567","412198","377860669765","0.000109","31822108","41863370","76.014205","0.960568271175","0.239857947413","0.999998909127","0.999914702064","0.619928428270","0.599999808789","0.383863548307","0.282205460900","0.237519268813","0.479999931119","0.479977799265","0.000085297936","","","","",""
"HUNSPELL_SPANISH_LUCENE_FILTER","ES_ES","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","64918","869371","3525","851564","17807","5","888962","495572","10041262","878949","31822108","377859790816","878949","377860669765","0.000233","31822108","41863370","76.014205","0.919511720057","0.239857947413","0.999997673881","0.999913466955","0.619927810647","0.586904802235","0.380469146267","0.281467012980","0.234925531298","0.469629847641","0.469606065497","0.000086533045","","","","",""
"HUNSPELL_UKRAINIAN_LUCENE_FILTER","UK_UA","ALL_WORDS","PRIMARY_OUTPUT","1493","14245","4","14245","0","1","14245","3137","50416","794","14924","101386756","794","101387550","0.000783","14924","65340","22.840526","0.984495215778","0.771594735231","0.999992168664","0.999845070949","0.885793451947","0.933007624547","0.865139425139","0.806475349522","0.762331024889","0.871568313648","0.871498740996","0.000154929051","0.865063055969","0.998114340300","0.949803904722","0.973360047526","0.973360047526"
"HUNSPELL_UKRAINIAN_LUCENE_FILTER","UK_UA","ALL_WORDS","ANY_CANDIDATE","1493","14245","4","12923","1322","6","15740","3311","55875","326","9465","101387224","326","101387550","0.000322","9465","65340","14.485767","0.994199391470","0.855142332415","0.999996784615","0.999903492153","0.927569558515","0.962883947281","0.919442821764","0.879752236578","0.850896963421","0.922053136488","0.922008115880","0.000096507847","","","","",""
"HUNSPELL_UKRAINIAN_LUCENE_FILTER","UK_UA","ALL_WORDS","ALL_CANDIDATES","1493","14245","4","12923","1322","6","15740","3311","55875","1271","9465","101386279","1271","101387550","0.001254","9465","65340","14.485767","0.977758723270","0.855142332415","0.999987463944","0.999894177485","0.927564898180","0.950500809733","0.912349166435","0.877142031861","0.838825419225","0.914397547654","0.914347195556","0.000105822515","","","","",""
"HUNSPELL_UKRAINIAN_LUCENE_FILTER","UK_UA","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","1491","14236","4","14236","0","1","14236","3134","50404","794","14920","101258612","794","101259406","0.000784","14920","65324","22.839998","0.984491581702","0.771600024493","0.999992158753","0.999844914465","0.885796091623","0.933006560145","0.865141346698","0.806479484406","0.762334008893","0.871569692311","0.871500048785","0.000155085535","0.865064900251","0.998112856419","0.949788155938","0.973351072054","0.973351072054"
"HUNSPELL_UKRAINIAN_LUCENE_FILTER","UK_UA","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","1491","14236","4","12915","1321","6","15730","3308","55859","326","9465","101259080","326","101259406","0.000322","9465","65324","14.489315","0.994197739610","0.855106851999","0.999996780546","0.999903370085","0.927551816273","0.962873710629","0.919421606630","0.879721936116","0.850860624524","0.922033242016","0.921988165267","0.000096629915","","","","",""
"HUNSPELL_UKRAINIAN_LUCENE_FILTER","UK_UA","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","1491","14236","4","12915","1321","6","15730","3308","55859","1271","9465","101258135","1271","101259406","0.001255","9465","65324","14.489315","0.977752494311","0.855106851999","0.999987448080","0.999894043636","0.927547150039","0.950487333415","0.912326261290","0.877111165546","0.838786695698","0.914375665383","0.914325250217","0.000105956364","","","","",""
"ITALIAN_LUCENE_ITALIAN_LIGHT_STEM_FILTER","IT_IT","ALL_WORDS","PRIMARY_OUTPUT","10009","327551","0","327551","0","1","327551","244870","109684","10589","6034130","53638510622","10589","53638521211","0.000020","6034130","6143814","98.214725","0.911958627456","0.017852754006","0.999999802586","0.999887319289","0.508926278296","0.082781551919","0.035019947839","0.022207258650","0.017822037328","0.127596916262","0.127588341500","0.000112680711","0.035015703871","0.997481424185","0.737537113266","0.848036553212","0.848036553212"
"ITALIAN_LUCENE_ITALIAN_LIGHT_STEM_FILTER","IT_IT","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","10007","327469","0","327469","0","1","327469","244808","109658","10588","6032516","53611656484","10588","53611667072","0.000020","6032516","6142174","98.214671","0.911947174958","0.017853287777","0.999999802506","0.999887292971","0.508926545141","0.082783771729","0.035020966336","0.022207918023","0.017822564890","0.127598022525","0.127589445488","0.000112707029","0.035016721203","0.997481197125","0.737534145120","0.848034509070","0.848034509070"
"ITALIAN_RADIXOR","IT_IT","ALL_WORDS","PRIMARY_OUTPUT","10009","327551","0","327551","0","1","327551","10010","6100906","124172","42908","53638397039","124172","53638521211","0.000231","42908","6143814","0.698394","0.980052940702","0.993016064614","0.999997685022","0.999996885431","0.996506874818","0.982618418699","0.986491918597","0.990396078172","0.973343909830","0.986513210398","0.986511657877","0.000003114569","0.986490361200","0.995780270704","0.997112550353","0.996445965204","0.996445965204"
"ITALIAN_RADIXOR","IT_IT","ALL_WORDS","ANY_CANDIDATE","10009","327551","0","321297","6254","4","334175","10012","6143734","0","80","53638521211","0","53638521211","0.000000","80","6143814","0.001302","1.000000000000","0.999986978772","1.000000000000","0.999999998509","0.999993489386","0.999997395727","0.999993489344","0.999989582991","0.999986978772","0.999993489365","0.999993488619","0.000000001491","","","","",""
"ITALIAN_RADIXOR","IT_IT","ALL_WORDS","ALL_CANDIDATES","10009","327551","0","321297","6254","4","334175","10012","6143734","170950","80","53638350261","170950","53638521211","0.000319","80","6143814","0.001302","0.972928178195","0.999986978772","0.999996812925","0.999996811799","0.999991895849","0.978222150749","0.986272020913","0.994455476443","0.972915852437","0.986364795335","0.986363222748","0.000003188201","","","","",""
"ITALIAN_RADIXOR","IT_IT","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","10007","327469","0","327469","0","1","327469","10007","6099346","124171","42828","53611542901","124171","53611667072","0.000232","42828","6142174","0.697278","0.980048098206","0.993027224563","0.999997683881","0.999996885382","0.996512454222","0.982616709845","0.986494972258","0.990403969991","0.973349855458","0.986516316590","0.986514764059","0.000003114618","0.986493414837","0.995779575755","0.997114909855","0.996446795437","0.996446795437"
"ITALIAN_RADIXOR","IT_IT","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","10007","327469","0","321217","6252","4","334089","10007","6142174","0","0","53611667072","0","53611667072","0.000000","0","6142174","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"ITALIAN_RADIXOR","IT_IT","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","10007","327469","0","321217","6252","4","334089","10007","6142174","170949","0","53611496123","170949","53611667072","0.000319","0","6142174","0.000000","0.972921642743","1.000000000000","0.999996811347","0.999996811712","0.999998405674","0.978219357390","0.986274996092","0.994464412864","0.972921642743","0.986367904356","0.986366331762","0.000003188288","","","","",""
"NL_NL_RADIXOR","NL_NL","ALL_WORDS","PRIMARY_OUTPUT","4992","26477","85","26477","0","1","26477","5015","63102","1214","1464","350436746","1214","350437960","0.000346","1464","64566","2.267447","0.981124448038","0.977325527367","0.999996535763","0.999992359542","0.988661031565","0.980362303079","0.979221303208","0.978082956166","0.959288537549","0.979223145453","0.979219325206","0.000007640458","0.979217482290","0.997464133435","0.997003025118","0.997233525974","0.997233525974"
"NL_NL_RADIXOR","NL_NL","ALL_WORDS","ANY_CANDIDATE","4992","26477","85","25905","572","3","27061","5016","64566","0","0","350437960","0","350437960","0.000000","0","64566","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"NL_NL_RADIXOR","NL_NL","ALL_WORDS","ALL_CANDIDATES","4992","26477","85","25905","572","3","27061","5016","64566","2651","0","350435309","2651","350437960","0.000756","0","64566","0.000000","0.960560572474","1.000000000000","0.999992435180","0.999992436574","0.999996217590","0.968197604323","0.979883596519","0.991855131329","0.960560572474","0.980081921308","0.980078214229","0.000007563426","","","","",""
"NL_NL_RADIXOR","NL_NL","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4796","25678","84","25678","0","1","25678","4797","61763","1214","1384","329602642","1214","329603856","0.000368","1384","63147","2.191711","0.980723121139","0.978082885964","0.999996316791","0.999992119320","0.989039601378","0.980193934392","0.979401224192","0.978609795129","0.959633939808","0.979402113872","0.979398173122","0.000007880680","0.979397283106","0.997373193672","0.997139403762","0.997256285015","0.997256285015"
"NL_NL_RADIXOR","NL_NL","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","4796","25678","84","25129","549","3","26239","4797","63147","0","0","329603856","0","329603856","0.000000","0","63147","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"NL_NL_RADIXOR","NL_NL","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","4796","25678","84","25129","549","3","26239","4797","63147","2651","0","329601205","2651","329603856","0.000804","0","63147","0.000000","0.959710021581","1.000000000000","0.999991957012","0.999991958552","0.999995978506","0.967506182222","0.979440846873","0.991673628866","0.959710021581","0.979647906945","0.979643967288","0.000008041448","","","","",""
"NN_NO_RADIXOR","NN_NO","ALL_WORDS","PRIMARY_OUTPUT","4688","18250","23","18250","0","1","18250","4680","26716","6230","3936","166485243","6230","166491473","0.003742","3936","30652","12.840924","0.810902689249","0.871590760799","0.999962580666","0.999938951055","0.935776670732","0.822354650447","0.840152206044","0.858737158800","0.724364188493","0.840699287413","0.840668985911","0.000061048945","0.840121715471","0.983845117159","0.986801676045","0.985321178741","0.985321178741"
"NN_NO_RADIXOR","NN_NO","ALL_WORDS","ANY_CANDIDATE","4688","18250","23","15846","2404","5","21513","4693","30652","0","0","166491473","0","166491473","0.000000","0","30652","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"NN_NO_RADIXOR","NN_NO","ALL_WORDS","ALL_CANDIDATES","4688","18250","23","15846","2404","5","21513","4693","30652","13214","0","166478259","13214","166491473","0.007937","0","30652","0.000000","0.698764418912","1.000000000000","0.999920632572","0.999920647181","0.999960316286","0.743561877778","0.822673716418","0.920624241623","0.698764418912","0.835921299473","0.835888126353","0.000079352819","","","","",""
"NN_NO_RADIXOR","NN_NO","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4681","18219","23","18219","0","1","18219","4668","26671","6230","3924","165920046","6230","165926276","0.003755","3924","30595","12.825625","0.810644053372","0.871743748979","0.999962453204","0.999938815429","0.935853101091","0.822169063928","0.840084414766","0.858797921188","0.724263408011","0.840638974932","0.840608609146","0.000061184571","0.840053856992","0.983814668550","0.986841730061","0.985325874420","0.985325874420"
"NN_NO_RADIXOR","NN_NO","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","4681","18219","23","15820","2399","5","21477","4681","30595","0","0","165926276","0","165926276","0.000000","0","30595","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"NN_NO_RADIXOR","NN_NO","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","4681","18219","23","15820","2399","5","21477","4681","30595","13214","0","165913062","13214","165926276","0.007964","0","30595","0.000000","0.698372480541","1.000000000000","0.999920362222","0.999920376903","0.999960181111","0.743206805583","0.822402021397","0.920488118949","0.698372480541","0.835686831618","0.835653554835","0.000079623097","","","","",""
"NORWEGIAN_BOKMAL_LUCENE_NORWEGIAN_LIGHT_STEM_FILTER","NB_NO","ALL_WORDS","PRIMARY_OUTPUT","17929","75310","252","75310","0","1","75310","25999","99529","25171","42651","2835593044","25171","2835618215","0.000888","42651","142180","29.997890","0.798147554130","0.700021100014","0.999991123276","0.999976083311","0.850006111645","0.776381478361","0.745870803357","0.717667503101","0.594732030284","0.747475838282","0.747464055956","0.000023916689","0.745858895755","0.987774378220","0.965622291196","0.976572729149","0.976572729149"
"NORWEGIAN_BOKMAL_LUCENE_NORWEGIAN_LIGHT_STEM_FILTER","NB_NO","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","17914","75251","252","75251","0","1","75251","25985","99450","25118","42641","2831151666","25118","2831176784","0.000887","42641","142091","30.009642","0.798359129150","0.699903582915","0.999991128071","0.999976068044","0.849947355493","0.776512696705","0.745896444523","0.717602881668","0.594764635875","0.747512150366","0.747500361706","0.000023931956","0.745884529658","0.987789184092","0.965602788874","0.976569991255","0.976569991255"
"NORWEGIAN_BOKMAL_LUCENE_NORWEGIAN_MINIMAL_STEM_FILTER","NB_NO","ALL_WORDS","PRIMARY_OUTPUT","17929","75310","252","75310","0","1","75310","27457","94526","14772","47654","2835603443","14772","2835618215","0.000521","47654","142180","33.516669","0.864846566268","0.664833309889","0.999994790554","0.999977986151","0.832414050221","0.815762584315","0.751763573752","0.697075888841","0.602260563739","0.758273568838","0.758263230800","0.000022013849","0.751752754540","0.992088894987","0.962515968891","0.977078714611","0.977078714611"
"NORWEGIAN_BOKMAL_LUCENE_NORWEGIAN_MINIMAL_STEM_FILTER","NB_NO","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","17914","75251","252","75251","0","1","75251","27443","94447","14719","47644","2831162065","14719","2831176784","0.000520","47644","142091","33.530625","0.865168642251","0.664693752595","0.999994801102","0.999977973869","0.832344276848","0.815949754214","0.751795969864","0.696994967013","0.602302149098","0.758335144541","0.758324803793","0.000022026131","0.751785145440","0.992107474043","0.962494026490","0.977076419047","0.977076419047"
"NORWEGIAN_BOKMAL_RADIXOR","NB_NO","ALL_WORDS","PRIMARY_OUTPUT","17929","75310","252","75310","0","1","75310","17886","135010","11482","7170","2835606733","11482","2835618215","0.000405","7170","142180","5.042903","0.921620293258","0.949570966381","0.999995950795","0.999993422575","0.974783458588","0.927078011613","0.935386875069","0.943846020481","0.878616704195","0.935491246621","0.935487968734","0.000006577425","0.935383586923","0.993354053349","0.994615320153","0.993984286646","0.993984286646"
"NORWEGIAN_BOKMAL_RADIXOR","NB_NO","ALL_WORDS","ANY_CANDIDATE","17929","75310","252","71073","4237","9","79825","17962","142180","0","0","2835618215","0","2835618215","0.000000","0","142180","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"NORWEGIAN_BOKMAL_RADIXOR","NB_NO","ALL_WORDS","ALL_CANDIDATES","17929","75310","252","71073","4237","9","79825","17962","142180","20161","0","2835598054","20161","2835618215","0.000711","0","142180","0.000000","0.875810793330","1.000000000000","0.999992890087","0.999992890443","0.999996445043","0.898118108406","0.933794385280","0.972422273928","0.875810793330","0.935847633608","0.935844306704","0.000007109557","","","","",""
"NORWEGIAN_BOKMAL_RADIXOR","NB_NO","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","17914","75251","252","75251","0","1","75251","17838","134987","11482","7104","2831165302","11482","2831176784","0.000406","7104","142091","4.999613","0.921607985307","0.950003870759","0.999995944443","0.999993435568","0.974999907601","0.927150543912","0.935590518436","0.944185565020","0.878976122105","0.935698217036","0.935694946007","0.000006564432","0.935587236809","0.993348241255","0.994693619298","0.994020475044","0.994020475044"
"NORWEGIAN_BOKMAL_RADIXOR","NB_NO","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","17914","75251","252","71047","4204","9","79733","17914","142091","0","0","2831176784","0","2831176784","0.000000","0","142091","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"NORWEGIAN_BOKMAL_RADIXOR","NB_NO","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","17914","75251","252","71047","4204","9","79733","17914","142091","20161","0","2831156623","20161","2831176784","0.000712","0","142091","0.000000","0.875742671893","1.000000000000","0.999992878933","0.999992879290","0.999996439466","0.898060798964","0.933755663840","0.972405477022","0.875742671893","0.935811237319","0.935807905326","0.000007120710","","","","",""
"PERSIAN_LUCENE_PERSIAN_STEM_FILTER","FA_IR","ALL_WORDS","PRIMARY_OUTPUT","69","3701","0","3701","0","1","3701","3190","430","179","98018","6748223","179","6748402","0.002652","98018","98448","99.563221","0.706075533662","0.004367788071","0.999973475202","0.985658076342","0.502170631636","0.021311605408","0.008681870034","0.005451304637","0.004359860890","0.055533668104","0.054800599292","0.014341923658","0.008506575635","0.985685778881","0.520346904570","0.681125382676","0.681125382676"
"PERSIAN_LUCENE_PERSIAN_STEM_FILTER","FA_IR","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","69","3701","0","3701","0","1","3701","3190","430","179","98018","6748223","179","6748402","0.002652","98018","98448","99.563221","0.706075533662","0.004367788071","0.999973475202","0.985658076342","0.502170631636","0.021311605408","0.008681870034","0.005451304637","0.004359860890","0.055533668104","0.054800599292","0.014341923658","0.008506575635","0.985685778881","0.520346904570","0.681125382676","0.681125382676"
"PERSIAN_RADIXOR","FA_IR","ALL_WORDS","PRIMARY_OUTPUT","69","3701","0","3701","0","1","3701","69","93636","8621","4812","6739781","8621","6748402","0.127749","4812","98448","4.887860","0.915692813206","0.951121404193","0.998722512381","0.998038075904","0.974921958287","0.922565796215","0.933070924989","0.943818050233","0.874538848780","0.933239001706","0.932248664283","0.001961924096","0.932075735269","0.980342969277","0.984586447427","0.982460126226","0.982460126226"
"PERSIAN_RADIXOR","FA_IR","ALL_WORDS","ANY_CANDIDATE","69","3701","0","3387","314","2","4015","69","98448","0","0","6748402","0","6748402","0.000000","0","98448","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"PERSIAN_RADIXOR","FA_IR","ALL_WORDS","ALL_CANDIDATES","69","3701","0","3387","314","2","4015","69","98448","13433","0","6734969","13433","6748402","0.199055","0","98448","0.000000","0.879934930864","1.000000000000","0.998009454683","0.998038075904","0.999004727341","0.901584696651","0.936133391021","0.973435401930","0.879934930864","0.938048469358","0.937114390300","0.001961924096","","","","",""
"PERSIAN_RADIXOR","FA_IR","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","69","3701","0","3701","0","1","3701","69","93636","8621","4812","6739781","8621","6748402","0.127749","4812","98448","4.887860","0.915692813206","0.951121404193","0.998722512381","0.998038075904","0.974921958287","0.922565796215","0.933070924989","0.943818050233","0.874538848780","0.933239001706","0.932248664283","0.001961924096","0.932075735269","0.980342969277","0.984586447427","0.982460126226","0.982460126226"
"PERSIAN_RADIXOR","FA_IR","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","69","3701","0","3387","314","2","4015","69","98448","0","0","6748402","0","6748402","0.000000","0","98448","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"PERSIAN_RADIXOR","FA_IR","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","69","3701","0","3387","314","2","4015","69","98448","13433","0","6734969","13433","6748402","0.199055","0","98448","0.000000","0.879934930864","1.000000000000","0.998009454683","0.998038075904","0.999004727341","0.901584696651","0.936133391021","0.973435401930","0.879934930864","0.938048469358","0.937114390300","0.001961924096","","","","",""
"POLISH_LUCENE_MORFOLOGIK_FILTER","PL_PL","ALL_WORDS","PRIMARY_OUTPUT","9990","122341","1","122341","0","1","122341","15519","1004747","99228","116220","7482378775","99228","7482478003","0.001326","116220","1120967","10.367834","0.910117529835","0.896321657997","0.999986738618","0.999971210643","0.948154198307","0.907324485129","0.903166914014","0.899047271013","0.823431500703","0.903193253581","0.903178865015","0.000028789357","0.903152518035","0.990022261217","0.977053921984","0.983495343428","0.983495343428"
"POLISH_LUCENE_MORFOLOGIK_FILTER","PL_PL","ALL_WORDS","ANY_CANDIDATE","9990","122341","1","109468","12873","5","136636","16295","1093112","85532","27855","7482392471","85532","7482478003","0.001143","27855","1120967","2.484908","0.927431862377","0.975150918805","0.999988569028","0.999984848600","0.987569743916","0.936598359399","0.950692965028","0.965218263555","0.906019814355","0.950992130738","0.950984648194","0.000015151400","","","","",""
"POLISH_LUCENE_MORFOLOGIK_FILTER","PL_PL","ALL_WORDS","ALL_CANDIDATES","9990","122341","1","109468","12873","5","136636","16295","1093112","143096","27855","7482334907","143096","7482478003","0.001912","27855","1120967","2.484908","0.884246016852","0.975150918805","0.999980875854","0.999977156579","0.987565897330","0.901045352805","0.927476322293","0.955504786999","0.864760696263","0.928586730350","0.928575670129","0.000022843421","","","","",""
"POLISH_LUCENE_MORFOLOGIK_FILTER","PL_PL","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","9846","120925","1","120925","0","1","120925","15277","999138","99224","115513","7310153475","99224","7310252699","0.001357","115513","1114651","10.363154","0.909661841906","0.896368459724","0.999986426735","0.999970629707","0.948177443229","0.906971715650","0.902966227492","0.898995962905","0.823097930182","0.902990688822","0.902976009256","0.000029370293","0.902951540918","0.989889334726","0.977011745487","0.983408384376","0.983408384376"
"POLISH_LUCENE_MORFOLOGIK_FILTER","PL_PL","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","9846","120925","1","108162","12763","5","135105","16044","1087157","85532","27494","7310167167","85532","7310252699","0.001170","27494","1114651","2.466602","0.927063356099","0.975333983462","0.999988299720","0.999984541059","0.987661141591","0.936331423447","0.950586270515","0.965281863330","0.905826028197","0.950892420848","0.950884788442","0.000015458941","","","","",""
"POLISH_LUCENE_MORFOLOGIK_FILTER","PL_PL","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","9846","120925","1","108162","12763","5","135105","16044","1087157","143085","27494","7310109614","143085","7310252699","0.001957","27494","1114651","2.466602","0.883693614752","0.975333983462","0.999980426805","0.999976669344","0.987657205134","0.900617649988","0.927255102898","0.955516285728","0.864376148890","0.928383764096","0.928372472930","0.000023330656","","","","",""
"POLISH_LUCENE_STEMPEL_DIRECT","PL_PL","ALL_WORDS","PRIMARY_OUTPUT","9990","122341","1","122341","0","1","122341","31432","797573","66669","323394","7482411334","66669","7482478003","0.000891","323394","1120967","28.849556","0.922858412343","0.711504442147","0.999991089984","0.999947877619","0.855747766065","0.871105640425","0.803515398127","0.745658746735","0.671563509358","0.810319603524","0.810295555502","0.000052122381","0.803489769425","0.991766794523","0.931068514076","0.960459620808","0.960459620808"
"POLISH_LUCENE_STEMPEL_DIRECT","PL_PL","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","9846","120925","1","120925","0","1","120925","30830","794493","66274","320158","7310186425","66274","7310252699","0.000907","320158","1114651","28.722712","0.923005877316","0.712772876892","0.999990934103","0.999947146412","0.856381905497","0.871590591697","0.804379630033","0.746792242917","0.672771767894","0.811106376847","0.811081975971","0.000052853588","0.804353636298","0.991711086900","0.931317880193","0.960566151650","0.960566151650"
"POLISH_LUCENE_STEMPEL_FILTER","PL_PL","ALL_WORDS","PRIMARY_OUTPUT","9990","122341","1","122341","0","1","122341","31432","797573","66669","323394","7482411334","66669","7482478003","0.000891","323394","1120967","28.849556","0.922858412343","0.711504442147","0.999991089984","0.999947877619","0.855747766065","0.871105640425","0.803515398127","0.745658746735","0.671563509358","0.810319603524","0.810295555502","0.000052122381","0.803489769425","0.991766794523","0.931068514076","0.960459620808","0.960459620808"
"POLISH_LUCENE_STEMPEL_FILTER","PL_PL","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","9846","120925","1","120925","0","1","120925","30830","794493","66274","320158","7310186425","66274","7310252699","0.000907","320158","1114651","28.722712","0.923005877316","0.712772876892","0.999990934103","0.999947146412","0.856381905497","0.871590591697","0.804379630033","0.746792242917","0.672771767894","0.811106376847","0.811081975971","0.000052853588","0.804353636298","0.991711086900","0.931317880193","0.960566151650","0.960566151650"
"POLISH_RADIXOR","PL_PL","ALL_WORDS","PRIMARY_OUTPUT","9990","122341","1","122341","0","1","122341","10074","1099420","13669","21547","7482464334","13669","7482478003","0.000183","21547","1120967","1.922180","0.987719760055","0.980778203105","0.999998173199","0.999995294243","0.990388188152","0.986323599045","0.984236742499","0.982158698021","0.968962733423","0.984242862020","0.984240510632","0.000004705757","0.984234389298","0.996967243455","0.996469409869","0.996718264498","0.996718264498"
"POLISH_RADIXOR","PL_PL","ALL_WORDS","ANY_CANDIDATE","9990","122341","1","119475","2866","4","125778","10079","1120967","0","0","7482478003","0","7482478003","0.000000","0","1120967","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"POLISH_RADIXOR","PL_PL","ALL_WORDS","ALL_CANDIDATES","9990","122341","1","119475","2866","4","125778","10079","1120967","38073","0","7482439930","38073","7482478003","0.000509","0","1120967","0.000000","0.967151263114","1.000000000000","0.999994911712","0.999994912475","0.999997455856","0.973547222425","0.983301367057","0.993252946885","0.967151263114","0.983438489746","0.983435987734","0.000005087525","","","","",""
"POLISH_RADIXOR","PL_PL","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","9846","120925","1","120925","0","1","120925","9844","1093651","13669","21000","7310239030","13669","7310252699","0.000187","21000","1114651","1.883998","0.987655781527","0.981160022285","0.999998130160","0.999995258206","0.990579076223","0.986349757961","0.984397186102","0.982452329568","0.969273787578","0.984402543989","0.984400174373","0.000004741794","0.984394814870","0.996926141446","0.996646530259","0.996786316244","0.996786316244"
"POLISH_RADIXOR","PL_PL","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","9846","120925","1","118145","2780","4","124274","9847","1114651","0","0","7310252699","0","7310252699","0.000000","0","1114651","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"POLISH_RADIXOR","PL_PL","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","9846","120925","1","118145","2780","4","124274","9847","1114651","38073","0","7310214626","38073","7310252699","0.000521","0","1114651","0.000000","0.966971278467","1.000000000000","0.999994791835","0.999994792629","0.999997395918","0.973401318686","0.983208335630","0.993214975136","0.966971278467","0.983346977657","0.983344416937","0.000005207371","","","","",""
"PORTUGUESE_LUCENE_PORTUGUESE_LIGHT_STEM_FILTER","PT_PT","ALL_WORDS","PRIMARY_OUTPUT","4001","211489","0","211489","0","1","211489","112814","149830","2511","5339230","22358201245","2511","22358203756","0.000011","5339230","5489060","97.270389","0.983517240927","0.027296112631","0.999999887692","0.999761142266","0.513648000162","0.122843213263","0.053118010934","0.033885033146","0.027283631587","0.163848092400","0.163827867245","0.000238857734","0.053105458848","0.999226179883","0.720580123442","0.837329788424","0.837329788424"
"PORTUGUESE_LUCENE_PORTUGUESE_LIGHT_STEM_FILTER","PT_PT","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4001","211489","0","211489","0","1","211489","112814","149830","2511","5339230","22358201245","2511","22358203756","0.000011","5339230","5489060","97.270389","0.983517240927","0.027296112631","0.999999887692","0.999761142266","0.513648000162","0.122843213263","0.053118010934","0.033885033146","0.027283631587","0.163848092400","0.163827867245","0.000238857734","0.053105458848","0.999226179883","0.720580123442","0.837329788424","0.837329788424"
"PORTUGUESE_LUCENE_PORTUGUESE_MINIMAL_STEM_FILTER","PT_PT","ALL_WORDS","PRIMARY_OUTPUT","4001","211489","0","211489","0","1","211489","167745","43406","598","5445654","22358203158","598","22358203756","0.000003","5445654","5489060","99.209227","0.986410326334","0.007907729192","0.999999973254","0.999756469021","0.503953851223","0.038310165654","0.015689679353","0.009864890589","0.007906867787","0.088319113068","0.088308061848","0.000243530979","0.015685836581","0.999664059174","0.692382565086","0.818121623354","0.818121623354"
"PORTUGUESE_LUCENE_PORTUGUESE_MINIMAL_STEM_FILTER","PT_PT","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4001","211489","0","211489","0","1","211489","167745","43406","598","5445654","22358203158","598","22358203756","0.000003","5445654","5489060","99.209227","0.986410326334","0.007907729192","0.999999973254","0.999756469021","0.503953851223","0.038310165654","0.015689679353","0.009864890589","0.007906867787","0.088319113068","0.088308061848","0.000243530979","0.015685836581","0.999664059174","0.692382565086","0.818121623354","0.818121623354"
"PORTUGUESE_LUCENE_PORTUGUESE_STEM_FILTER","PT_PT","ALL_WORDS","PRIMARY_OUTPUT","4001","211489","0","211489","0","1","211489","27586","3803488","99075","1685572","22358104681","99075","22358203756","0.000443","1685572","5489060","30.707844","0.974612837768","0.692921556696","0.999995568741","0.999920198913","0.846458562719","0.901329863268","0.809974591186","0.735433886866","0.680636384053","0.821784792219","0.821750382444","0.000079801087","0.809935821347","0.996728545383","0.918475110538","0.956003146785","0.956003146785"
"PORTUGUESE_LUCENE_PORTUGUESE_STEM_FILTER","PT_PT","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4001","211489","0","211489","0","1","211489","27586","3803488","99075","1685572","22358104681","99075","22358203756","0.000443","1685572","5489060","30.707844","0.974612837768","0.692921556696","0.999995568741","0.999920198913","0.846458562719","0.901329863268","0.809974591186","0.735433886866","0.680636384053","0.821784792219","0.821750382444","0.000079801087","0.809935821347","0.996728545383","0.918475110538","0.956003146785","0.956003146785"
"PORTUGUESE_RADIXOR","PT_PT","ALL_WORDS","PRIMARY_OUTPUT","4001","211489","0","211489","0","1","211489","4001","5472616","20678","16444","22358183078","20678","22358203756","0.000092","16444","5489060","0.299578","0.996235774018","0.997004222945","0.999999075149","0.999998340077","0.998501649047","0.996389369023","0.996619850353","0.996850438335","0.993262474550","0.996619924417","0.996619094289","0.000001659923","0.996619020188","0.999299376330","0.999346803887","0.999323089546","0.999323089546"
"PORTUGUESE_RADIXOR","PT_PT","ALL_WORDS","ANY_CANDIDATE","4001","211489","0","210699","790","3","212297","4001","5489060","0","0","22358203756","0","22358203756","0.000000","0","5489060","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"PORTUGUESE_RADIXOR","PT_PT","ALL_WORDS","ALL_CANDIDATES","4001","211489","0","210699","790","3","212297","4001","5489060","38310","0","22358165446","38310","22358203756","0.000171","0","5489060","0.000000","0.993069036450","1.000000000000","0.999998286535","0.999998286956","0.999999143267","0.994447532369","0.996522466897","0.998606078314","0.993069036450","0.996528492543","0.996527638784","0.000001713044","","","","",""
"PORTUGUESE_RADIXOR","PT_PT","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4001","211489","0","211489","0","1","211489","4001","5472616","20678","16444","22358183078","20678","22358203756","0.000092","16444","5489060","0.299578","0.996235774018","0.997004222945","0.999999075149","0.999998340077","0.998501649047","0.996389369023","0.996619850353","0.996850438335","0.993262474550","0.996619924417","0.996619094289","0.000001659923","0.996619020188","0.999299376330","0.999346803887","0.999323089546","0.999323089546"
"PORTUGUESE_RADIXOR","PT_PT","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","4001","211489","0","210699","790","3","212297","4001","5489060","0","0","22358203756","0","22358203756","0.000000","0","5489060","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"PORTUGUESE_RADIXOR","PT_PT","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","4001","211489","0","210699","790","3","212297","4001","5489060","38310","0","22358165446","38310","22358203756","0.000171","0","5489060","0.000000","0.993069036450","1.000000000000","0.999998286535","0.999998286956","0.999999143267","0.994447532369","0.996522466897","0.998606078314","0.993069036450","0.996528492543","0.996527638784","0.000001713044","","","","",""
"RUSSIAN_LUCENE_RUSSIAN_LIGHT_STEM_FILTER","RU_RU","ALL_WORDS","PRIMARY_OUTPUT","37410","768882","10","768882","0","1","768882","232250","3081067","321183","10008438","295575969833","321183","295576291016","0.000109","10008438","13089505","76.461547","0.905596884415","0.235384531348","0.999998913367","0.999965054154","0.617691722357","0.577011147253","0.373649378129","0.276277984307","0.229747124085","0.461696326851","0.461686629842","0.000034945846","0.373637933830","0.994310930069","0.870888421754","0.928516167212","0.928516167212"
"RUSSIAN_LUCENE_RUSSIAN_LIGHT_STEM_FILTER","RU_RU","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","37297","768133","10","768133","0","1","768133","232143","3078888","318921","10008238","295000362731","318921","295000681652","0.000108","10008238","13087126","76.473918","0.906139220892","0.235260820443","0.999998918914","0.999964994315","0.617629869679","0.577038425373","0.373539598427","0.276151716079","0.229664120975","0.461713175622","0.461703471649","0.000035005685","0.373528142103","0.994349544377","0.870767393403","0.928464208705","0.928464208705"
"RUSSIAN_RADIXOR","RU_RU","ALL_WORDS","PRIMARY_OUTPUT","37410","768882","10","768882","0","1","768882","37561","12823203","155850","266302","295576135166","155850","295576291016","0.000053","266302","13089505","2.034470","0.987992190185","0.979655304001","0.999999472725","0.999998571830","0.989827388363","0.986313480705","0.983806085477","0.981311406466","0.968128298562","0.983814916245","0.983814202914","0.000001428170","0.983805371373","0.997699288696","0.997273959852","0.997486578934","0.997486578934"
"RUSSIAN_RADIXOR","RU_RU","ALL_WORDS","ANY_CANDIDATE","37410","768882","10","749720","19162","4","788492","37593","13089492","0","13","295576291016","0","295576291016","0.000000","13","13089505","0.000099","1.000000000000","0.999999006838","1.000000000000","0.999999999956","0.999999503419","0.999999801367","0.999999503419","0.999999205470","0.999999006838","0.999999503419","0.999999503397","0.000000000044","","","","",""
"RUSSIAN_RADIXOR","RU_RU","ALL_WORDS","ALL_CANDIDATES","37410","768882","10","749720","19162","4","788492","37593","13089492","434710","13","295575856306","434710","295576291016","0.000147","13","13089505","0.000099","0.967856883534","0.999999006838","0.999998529280","0.999998529301","0.999998768059","0.974118939969","0.983665447282","0.993400920813","0.967855953192","0.983796687479","0.983795964011","0.000001470699","","","","",""
"RUSSIAN_RADIXOR","RU_RU","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","37297","768133","10","768133","0","1","768133","37282","12821513","155850","265613","295000525802","155850","295000681652","0.000053","265613","13087126","2.029575","0.987990626447","0.979704252867","0.999999471696","0.999998571379","0.989851862281","0.986322156837","0.983829991833","0.981350389119","0.968174600634","0.983838715706","0.983838002141","0.000001428621","0.983829277503","0.997696524283","0.997321167437","0.997508810549","0.997508810549"
"RUSSIAN_RADIXOR","RU_RU","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","37297","768133","10","749142","18991","4","787549","37306","13087126","0","0","295000681652","0","295000681652","0.000000","0","13087126","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"RUSSIAN_RADIXOR","RU_RU","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","37297","768133","10","749142","18991","4","787549","37306","13087126","434710","0","295000246942","434710","295000681652","0.000147","0","13087126","0.000000","0.967851259252","1.000000000000","0.999998526410","0.999998526476","0.999999263205","0.974114570610","0.983663023007","0.993400519870","0.967851259252","0.983794317554","0.983793592699","0.000001473524","","","","",""
"SNOWBALL_DANISH_DIRECT","DA_DK","ALL_WORDS","PRIMARY_OUTPUT","4179","28079","32","28079","0","1","28079","5553","78732","6341","11163","394104845","6341","394111186","0.001609","11163","89895","12.417821","0.925464013259","0.875821792091","0.999983910632","0.999955596266","0.937902851361","0.915090414169","0.899958849618","0.885319563795","0.818113803566","0.900300811178","0.900278764621","0.000044403734","0.899936659693","0.994195946109","0.978579164615","0.986325742978","0.986325742978"
"SNOWBALL_DANISH_DIRECT","DA_DK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4173","28033","32","28033","0","1","28033","5539","78627","6341","11113","392814447","6341","392820788","0.001614","11113","89740","12.383552","0.925371904717","0.876164475150","0.999983857779","0.999955577673","0.938074166465","0.915093153823","0.900096160451","0.885582797210","0.818340774971","0.900432112497","0.900410054086","0.000044422327","0.900073960926","0.994185354918","0.978644331817","0.986353630937","0.986353630937"
"SNOWBALL_DANISH_LUCENE_FILTER","DA_DK","ALL_WORDS","PRIMARY_OUTPUT","4179","28079","32","28079","0","1","28079","5546","78744","6507","11151","394104679","6507","394111186","0.001651","11151","89895","12.404472","0.923672449590","0.875955281161","0.999983489431","0.999955205602","0.937969385296","0.913717599716","0.899181254496","0.885100184115","0.816829526358","0.899497504322","0.899475250557","0.000044794398","0.899158868074","0.994052746860","0.978603476314","0.986267614487","0.986267614487"
"SNOWBALL_DANISH_LUCENE_FILTER","DA_DK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4173","28033","32","28033","0","1","28033","5539","78627","6341","11113","392814447","6341","392820788","0.001614","11113","89740","12.383552","0.925371904717","0.876164475150","0.999983857779","0.999955577673","0.938074166465","0.915093153823","0.900096160451","0.885582797210","0.818340774971","0.900432112497","0.900410054086","0.000044422327","0.900073960926","0.994185354918","0.978644331817","0.986353630937","0.986353630937"
"SNOWBALL_DUTCH_DIRECT","NL_NL","ALL_WORDS","PRIMARY_OUTPUT","4992","26477","85","26477","0","1","26477","12051","29325","4382","35241","350433578","4382","350437960","0.001250","35241","64566","54.581359","0.869997329931","0.454186413902","0.999987495647","0.999886953739","0.727086954774","0.735353119953","0.596806854375","0.502190286022","0.425320531415","0.628602392126","0.628557411604","0.000113046261","0.596755898222","0.992814719235","0.917346281080","0.953589661124","0.953589661124"
"SNOWBALL_DUTCH_DIRECT","NL_NL","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4796","25678","84","25678","0","1","25678","11466","29111","4382","34036","329599474","4382","329603856","0.001329","34036","63147","53.899631","0.869166691547","0.461003689803","0.999986705253","0.999883464224","0.730495197528","0.738411822300","0.602462748344","0.508789468717","0.431088865524","0.633000040962","0.632953313739","0.000116535776","0.602409959779","0.992557044900","0.918058993802","0.953855618789","0.953855618789"
"SNOWBALL_DUTCH_LUCENE_FILTER","NL_NL","ALL_WORDS","PRIMARY_OUTPUT","4992","26477","85","26477","0","1","26477","14573","15302","1588","49264","350436372","1588","350437960","0.000453","49264","64566","76.300220","0.905979869745","0.236997800700","0.999995468527","0.999854916880","0.618496634614","0.579068464950","0.375712040856","0.278062466837","0.231308764398","0.463373754768","0.463333378452","0.000145083120","0.375664346452","0.995827666179","0.888410302915","0.939057139355","0.939057139355"
"SNOWBALL_DUTCH_LUCENE_FILTER","NL_NL","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4796","25678","84","25678","0","1","25678","14116","14972","1544","48175","329602312","1544","329603856","0.000468","48175","63147","76.290243","0.906514894648","0.237097565997","0.999995315589","0.999849184178","0.618546440793","0.579362438183","0.375883408860","0.278182412747","0.231438685443","0.463608105042","0.463566154833","0.000150815822","0.375833834664","0.995816517119","0.887491664267","0.938538755173","0.938538755173"
"SNOWBALL_FINNISH_DIRECT","FI_FI","ALL_WORDS","PRIMARY_OUTPUT","57027","1811717","292","1811717","0","1","1811717","381483","15114332","1544812","16409363","1641125269679","1544812","1641126814491","0.000094","16409363","31523695","52.054060","0.907269425128","0.479459403474","0.999999058688","0.999989060059","0.739729231081","0.769880311353","0.627374073993","0.529384116965","0.457061215373","0.659544431681","0.659540149918","0.000010939941","0.627369124557","0.991871857177","0.904138579582","0.945975396220","0.945975396220"
"SNOWBALL_FINNISH_DIRECT","FI_FI","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","54762","1757055","274","1757055","0","1","1757055","372232","14692070","1513705","16121763","1543587930447","1513705","1543589444152","0.000098","16121763","30813833","52.319888","0.906594717007","0.476801117213","0.999999019360","0.999988575255","0.738400068286","0.768116957494","0.624933751043","0.526744348874","0.454475376380","0.657468914800","0.657464451566","0.000011424745","0.624928589970","0.991731678095","0.902933406396","0.945251664438","0.945251664438"
"SNOWBALL_FINNISH_LUCENE_FILTER","FI_FI","ALL_WORDS","PRIMARY_OUTPUT","57027","1811717","292","1811717","0","1","1811717","377778","15153638","1922153","16370057","1641124892338","1922153","1641126814491","0.000117","16370057","31523695","51.929372","0.887434028678","0.480706275073","0.999998828760","0.999988854086","0.740352551917","0.758996033322","0.623613097472","0.529216231176","0.453079796332","0.653142485450","0.653138019077","0.000011145914","0.623608016975","0.990717710840","0.904385055188","0.945584912497","0.945584912497"
"SNOWBALL_FINNISH_LUCENE_FILTER","FI_FI","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","54762","1757055","274","1757055","0","1","1757055","372232","14692070","1513705","16121763","1543587930447","1513705","1543589444152","0.000098","16121763","30813833","52.319888","0.906594717007","0.476801117213","0.999999019360","0.999988575255","0.738400068286","0.768116957494","0.624933751043","0.526744348874","0.454475376380","0.657468914800","0.657464451566","0.000011424745","0.624928589970","0.991731678095","0.902933406396","0.945251664438","0.945251664438"
"SNOWBALL_FRENCH_DIRECT","FR_FR","ALL_WORDS","PRIMARY_OUTPUT","59240","425210","2301","425210","0","1","425210","85627","3766640","1654723","1687975","90394450107","1654723","90396104830","0.001831","1687975","5454615","30.945814","0.694777309691","0.690541862258","0.999981694753","0.999963023890","0.845261778506","0.693926068790","0.692653111288","0.691384815533","0.529815856272","0.692656348624","0.692637859933","0.000036976110","0.692634622294","0.959459328254","0.944947915186","0.952148333884","0.952148333884"
"SNOWBALL_FRENCH_DIRECT","FR_FR","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","57698","421231","2133","421231","0","1","421231","84526","3758589","1646111","1681970","88710480395","1646111","88712126506","0.001856","1681970","5440559","30.915389","0.695429718578","0.690846106071","0.999981444352","0.999962486787","0.845413775212","0.694508136723","0.693130334647","0.691757988461","0.530374491828","0.693134123475","0.693115366288","0.000037513213","0.693111577099","0.959520798119","0.944537159644","0.951970023370","0.951970023370"
"SNOWBALL_FRENCH_LUCENE_FILTER","FR_FR","ALL_WORDS","PRIMARY_OUTPUT","59240","425210","2301","425210","0","1","425210","85202","3763777","1661388","1690838","90394443442","1661388","90396104830","0.001838","1690838","5454615","30.998301","0.693762678186","0.690016985617","0.999981621022","0.999962918494","0.844999303320","0.693010289898","0.691884762376","0.690762884895","0.528917286853","0.691887297134","0.691868755638","0.000037081506","0.691866220643","0.958697792387","0.944714715363","0.951654891797","0.951654891797"
"SNOWBALL_FRENCH_LUCENE_FILTER","FR_FR","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","57698","421231","2133","421231","0","1","421231","84810","3755856","1641925","1684703","88710484581","1641925","88712126506","0.001851","1684703","5440559","30.965623","0.695814817237","0.690343767984","0.999981491538","0.999962503165","0.845162629761","0.694713680979","0.693068495729","0.691431084156","0.530302080457","0.693073894149","0.693055145391","0.000037496835","0.693049746458","0.959566165512","0.944384738154","0.951914926182","0.951914926182"
"SNOWBALL_GERMAN_DIRECT","DE_DE","ALL_WORDS","PRIMARY_OUTPUT","54092","296974","1474","296974","0","1","296974","81649","771138","190680","612734","44095055299","190680","44095245979","0.000432","612734","1383872","44.276783","0.801750435114","0.557232171762","0.999995675724","0.999981780603","0.778613923743","0.737064397386","0.657493530688","0.593429030432","0.489750735447","0.668401927114","0.668393541401","0.000018219397","0.657484715679","0.983724573695","0.949324273697","0.966218331938","0.966218331938"
"SNOWBALL_GERMAN_DIRECT","DE_DE","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","16007","150098","228","150098","0","1","150098","37843","516936","87697","356475","11263668645","87697","11263756342","0.000779","356475","873411","40.814118","0.854958297017","0.591858815609","0.999992214231","0.999960569321","0.795925514920","0.785153327381","0.699486618802","0.630674793334","0.537854226580","0.711347035607","0.711329191110","0.000039430679","0.699467554207","0.988417636496","0.932451723900","0.959619376607","0.959619376607"
"SNOWBALL_GERMAN_LUCENE_FILTER","DE_DE","ALL_WORDS","PRIMARY_OUTPUT","54092","296974","1474","296974","0","1","296974","86669","751056","295701","632816","44094950278","295701","44095245979","0.000671","632816","1383872","45.727929","0.717507501741","0.542720714054","0.999993294039","0.999978943584","0.771357004047","0.674088567377","0.617993120299","0.570516594262","0.447170798768","0.624024185176","0.624014089276","0.000021056416","0.617982794334","0.975844522648","0.942925157079","0.959102449371","0.959102449371"
"SNOWBALL_GERMAN_LUCENE_FILTER","DE_DE","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","16007","150098","228","150098","0","1","150098","46077","481501","77653","391910","11263678689","77653","11263756342","0.000689","391910","873411","44.871200","0.861124126806","0.551287996144","0.999993105941","0.999958315274","0.775640551042","0.774110642769","0.672222202832","0.594035281304","0.506276128631","0.689004640259","0.688986412764","0.000041684726","0.672202362239","0.989021274644","0.919542244735","0.953017108147","0.953017108147"
"SNOWBALL_HUNGARIAN_DIRECT","HU_HU","ALL_WORDS","PRIMARY_OUTPUT","19406","916344","1","916344","0","1","916344","116105","14287912","1506056","7874191","419819036837","1506056","419820542893","0.000359","7874191","22162103","35.529981","0.904643595580","0.644700189328","0.999996412620","0.999977657711","0.822348300974","0.837136808086","0.752865700984","0.684009307333","0.603676525918","0.763690969794","0.763680928295","0.000022342289","0.752854843818","0.991947513126","0.924304257644","0.956931989551","0.956931989551"
"SNOWBALL_HUNGARIAN_DIRECT","HU_HU","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","18360","878513","1","878513","0","1","878513","111379","13776526","1496670","7634885","385869198247","1496670","385870694917","0.000388","7634885","21411411","35.658019","0.902006757459","0.643419810119","0.999996121317","0.999976336507","0.821707965718","0.834898516372","0.751079383241","0.682554714263","0.601382804609","0.761819543337","0.761808891723","0.000023663493","0.751067882221","0.991609896137","0.923288102987","0.956230170303","0.956230170303"
"SNOWBALL_HUNGARIAN_LUCENE_FILTER","HU_HU","ALL_WORDS","PRIMARY_OUTPUT","19406","916344","1","916344","0","1","916344","114867","14299358","1792049","7862745","419818750844","1792049","419820542893","0.000427","7862745","22162103","35.478334","0.888633169244","0.645216656560","0.999995731393","0.999977003783","0.822606193976","0.826287586346","0.747610245439","0.682613266689","0.596946950992","0.757205997314","0.757195513225","0.000022996217","0.747599036407","0.990687085622","0.924490230693","0.956444632598","0.956444632598"
"SNOWBALL_HUNGARIAN_LUCENE_FILTER","HU_HU","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","18360","878513","1","878513","0","1","878513","111379","13776526","1496670","7634885","385869198247","1496670","385870694917","0.000388","7634885","21411411","35.658019","0.902006757459","0.643419810119","0.999996121317","0.999976336507","0.821707965718","0.834898516372","0.751079383241","0.682554714263","0.601382804609","0.761819543337","0.761808891723","0.000023663493","0.751067882221","0.991609896137","0.923288102987","0.956230170303","0.956230170303"
"SNOWBALL_ITALIAN_DIRECT","IT_IT","ALL_WORDS","PRIMARY_OUTPUT","10009","327551","0","327551","0","1","327551","46828","4499650","504775","1644164","53638016436","504775","53638521211","0.000941","1644164","6143814","26.761292","0.899134266174","0.732387080729","0.999990589319","0.999959941236","0.866188835024","0.859975076366","0.807239600802","0.760598128154","0.676782697802","0.811488952720","0.811469907061","0.000040058764","0.807219778599","0.987993752409","0.933407841859","0.959925420024","0.959925420024"
"SNOWBALL_ITALIAN_DIRECT","IT_IT","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","10007","327469","0","327469","0","1","327469","46814","4498652","504774","1643522","53611162298","504774","53611667072","0.000942","1643522","6142174","26.757985","0.899114326863","0.732420149608","0.999990584624","0.999959933163","0.866205367116","0.859969602244","0.807251650876","0.760623806435","0.676799637969","0.811498274672","0.811479224597","0.000040066837","0.807231824542","0.987990915845","0.933413003102","0.959926810498","0.959926810498"
"SNOWBALL_ITALIAN_LUCENE_FILTER","IT_IT","ALL_WORDS","PRIMARY_OUTPUT","10009","327551","0","327551","0","1","327551","46828","4499650","504775","1644164","53638016436","504775","53638521211","0.000941","1644164","6143814","26.761292","0.899134266174","0.732387080729","0.999990589319","0.999959941236","0.866188835024","0.859975076366","0.807239600802","0.760598128154","0.676782697802","0.811488952720","0.811469907061","0.000040058764","0.807219778599","0.987993752409","0.933407841859","0.959925420024","0.959925420024"
"SNOWBALL_ITALIAN_LUCENE_FILTER","IT_IT","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","10007","327469","0","327469","0","1","327469","46814","4498652","504774","1643522","53611162298","504774","53611667072","0.000942","1643522","6142174","26.757985","0.899114326863","0.732420149608","0.999990584624","0.999959933163","0.866205367116","0.859969602244","0.807251650876","0.760623806435","0.676799637969","0.811498274672","0.811479224597","0.000040066837","0.807231824542","0.987990915845","0.933413003102","0.959926810498","0.959926810498"
"SNOWBALL_NORWEGIAN_BOKMAL_DIRECT","NB_NO","ALL_WORDS","PRIMARY_OUTPUT","17929","75310","252","75310","0","1","75310","24394","106626","23997","35554","2835594218","23997","2835618215","0.000846","35554","142180","25.006330","0.816288096277","0.749936699958","0.999991537295","0.999978999989","0.874964118626","0.802094867845","0.781706946038","0.762329786671","0.641641141674","0.782409356499","0.782398932962","0.000021000011","0.781696464373","0.988328173631","0.971119668400","0.979648355684","0.979648355684"
"SNOWBALL_NORWEGIAN_BOKMAL_DIRECT","NB_NO","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","17914","75251","252","75251","0","1","75251","24367","106567","23997","35524","2831152787","23997","2831176784","0.000848","35524","142091","25.000880","0.816205079501","0.749991202821","0.999991524019","0.999978977642","0.874991363420","0.802043209347","0.781698483431","0.762360357576","0.641629738452","0.782397999310","0.782387564360","0.000021022358","0.781687990535","0.988317966243","0.971127035574","0.979647089734","0.979647089734"
"SNOWBALL_NORWEGIAN_BOKMAL_LUCENE_FILTER","NB_NO","ALL_WORDS","PRIMARY_OUTPUT","17929","75310","252","75310","0","1","75310","24396","106589","24046","35591","2835594169","24046","2835618215","0.000848","35591","142180","25.032353","0.815929880966","0.749676466451","0.999991520015","0.999978969662","0.874833993233","0.801758635215","0.781401315910","0.762052176648","0.641229410562","0.782101930719","0.782091491842","0.000021030338","0.781390819068","0.988295184140","0.971086244692","0.979615142710","0.979615142710"
"SNOWBALL_NORWEGIAN_BOKMAL_LUCENE_FILTER","NB_NO","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","17914","75251","252","75251","0","1","75251","24381","106512","23993","35579","2831152791","23993","2831176784","0.000847","35579","142091","25.039587","0.816152637830","0.749604126933","0.999991525432","0.999978959629","0.874797826182","0.801914137847","0.781464144742","0.762031224736","0.641314033862","0.782170943927","0.782160500766","0.000021040371","0.781453643056","0.988310445474","0.971073741466","0.979616277822","0.979616277822"
"SNOWBALL_NORWEGIAN_NYNORSK_DIRECT","NN_NO","ALL_WORDS","PRIMARY_OUTPUT","4688","18250","23","18250","0","1","18250","6138","22004","8274","8648","166483199","8274","166491473","0.004970","8648","30652","28.213493","0.726732280864","0.717865065901","0.999950303761","0.999898379870","0.858907684831","0.724941356316","0.722271459051","0.719621155632","0.565277706417","0.722285066089","0.722234252664","0.000101620130","0.722220641604","0.980542486408","0.964998187466","0.972708239744","0.972708239744"
"SNOWBALL_NORWEGIAN_NYNORSK_DIRECT","NN_NO","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4681","18219","23","18219","0","1","18219","6120","21971","8274","8624","165918002","8274","165926276","0.004987","8624","30595","28.187612","0.726434121342","0.718123876450","0.999950134480","0.999898178365","0.859037005465","0.724756721095","0.722255095332","0.719770679771","0.565257660346","0.722267047015","0.722216132089","0.000101821635","0.722204176866","0.980505813025","0.965064509051","0.972723884952","0.972723884952"
"SNOWBALL_NORWEGIAN_NYNORSK_LUCENE_FILTER","NN_NO","ALL_WORDS","PRIMARY_OUTPUT","4688","18250","23","18250","0","1","18250","6144","21978","8295","8674","166483178","8295","166491473","0.004982","8674","30652","28.298317","0.725993459518","0.717016834138","0.999950177629","0.999898097625","0.858483505883","0.724180198229","0.721477226098","0.718794356395","0.564305338023","0.721491186328","0.721440231913","0.000101902375","0.721426267560","0.980461058483","0.964862123312","0.972599049418","0.972599049418"
"SNOWBALL_NORWEGIAN_NYNORSK_LUCENE_FILTER","NN_NO","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4681","18219","23","18219","0","1","18219","6130","21948","8274","8647","165918002","8274","165926276","0.004987","8647","30595","28.262788","0.726225928132","0.717372119627","0.999950134480","0.999898039774","0.858661127054","0.724437725685","0.721771872996","0.719125568472","0.564665929147","0.721785448310","0.721734464790","0.000101960226","0.721720885459","0.980505813025","0.964944952705","0.972663150405","0.972663150405"
"SNOWBALL_PORTUGUESE_DIRECT","PT_PT","ALL_WORDS","PRIMARY_OUTPUT","4001","211489","0","211489","0","1","211489","11315","4817239","167230","671821","22358036526","167230","22358203756","0.000748","671821","5489060","12.239272","0.966449786326","0.877607277020","0.999992520419","0.999962481554","0.938799898719","0.947270839082","0.919888415834","0.894044585092","0.851660540743","0.920957852105","0.920939611009","0.000037518446","0.919869695779","0.996663145176","0.967923515462","0.982083116554","0.982083116554"
"SNOWBALL_PORTUGUESE_DIRECT","PT_PT","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4001","211489","0","211489","0","1","211489","11315","4817239","167230","671821","22358036526","167230","22358203756","0.000748","671821","5489060","12.239272","0.966449786326","0.877607277020","0.999992520419","0.999962481554","0.938799898719","0.947270839082","0.919888415834","0.894044585092","0.851660540743","0.920957852105","0.920939611009","0.000037518446","0.919869695779","0.996663145176","0.967923515462","0.982083116554","0.982083116554"
"SNOWBALL_PORTUGUESE_LUCENE_FILTER","PT_PT","ALL_WORDS","PRIMARY_OUTPUT","4001","211489","0","211489","0","1","211489","11315","4817239","167230","671821","22358036526","167230","22358203756","0.000748","671821","5489060","12.239272","0.966449786326","0.877607277020","0.999992520419","0.999962481554","0.938799898719","0.947270839082","0.919888415834","0.894044585092","0.851660540743","0.920957852105","0.920939611009","0.000037518446","0.919869695779","0.996663145176","0.967923515462","0.982083116554","0.982083116554"
"SNOWBALL_PORTUGUESE_LUCENE_FILTER","PT_PT","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4001","211489","0","211489","0","1","211489","11315","4817239","167230","671821","22358036526","167230","22358203756","0.000748","671821","5489060","12.239272","0.966449786326","0.877607277020","0.999992520419","0.999962481554","0.938799898719","0.947270839082","0.919888415834","0.894044585092","0.851660540743","0.920957852105","0.920939611009","0.000037518446","0.919869695779","0.996663145176","0.967923515462","0.982083116554","0.982083116554"
"SNOWBALL_RUSSIAN_DIRECT","RU_RU","ALL_WORDS","PRIMARY_OUTPUT","37410","768882","10","768882","0","1","768882","64358","8766656","3782908","4322849","295572508108","3782908","295576291016","0.001280","4322849","13089505","33.025305","0.698562595481","0.669746946122","0.999987201585","0.999972577645","0.834867073854","0.692602792505","0.683851352013","0.675318311031","0.519585195076","0.684003044583","0.683989349009","0.000027422355","0.683837646322","0.974179960240","0.953661001039","0.963811283954","0.963811283954"
"SNOWBALL_RUSSIAN_DIRECT","RU_RU","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","37297","768133","10","768133","0","1","768133","64159","8764719","3782908","4322407","294996898744","3782908","295000681652","0.001282","4322407","13087126","33.027931","0.698516062041","0.669720685810","0.999987176613","0.999972525638","0.834853931211","0.692560581516","0.683815365804","0.675288254087","0.519543647630","0.683966853085","0.683953131513","0.000027474362","0.683801634112","0.974148936252","0.953633561396","0.963782086975","0.963782086975"
"SNOWBALL_RUSSIAN_LUCENE_FILTER","RU_RU","ALL_WORDS","PRIMARY_OUTPUT","37410","768882","10","768882","0","1","768882","64266","8766889","3785790","4322616","295572505226","3785790","295576291016","0.001281","4322616","13089505","33.023525","0.698407806015","0.669764746642","0.999987191835","0.999972568683","0.834875969239","0.692484865100","0.683786451263","0.675303850911","0.519510266339","0.683936347366","0.683922647122","0.000027431317","0.683772741022","0.974131099393","0.953673855106","0.963793934411","0.963793934411"
"SNOWBALL_RUSSIAN_LUCENE_FILTER","RU_RU","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","37297","768133","10","768133","0","1","768133","64159","8764719","3782908","4322407","294996898744","3782908","295000681652","0.001282","4322407","13087126","33.027931","0.698516062041","0.669720685810","0.999987176613","0.999972525638","0.834853931211","0.692560581516","0.683815365804","0.675288254087","0.519543647630","0.683966853085","0.683953131513","0.000027474362","0.683801634112","0.974148936252","0.953633561396","0.963782086975","0.963782086975"
"SNOWBALL_SPANISH_DIRECT","ES_ES","ALL_WORDS","PRIMARY_OUTPUT","65059","871332","3589","871332","0","1","871332","195021","12811687","2228819","29161649","379565089291","2228819","379567318110","0.000587","29161649","41973336","69.476605","0.851812232913","0.305233946618","0.999994128001","0.999917308483","0.652614037309","0.627191552465","0.449423738186","0.350172671706","0.289843040458","0.509903921959","0.509876023351","0.000082691517","0.449391616998","0.981405614580","0.852462513401","0.912400934512","0.912400934512"
"SNOWBALL_SPANISH_DIRECT","ES_ES","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","64918","869371","3525","869371","0","1","869371","194444","12787018","2201196","29076352","377858468569","2201196","377860669765","0.000583","29076352","41863370","69.455354","0.853138205793","0.305446455935","0.999994174583","0.999917233823","0.652720315259","0.627945981812","0.449838583213","0.350441220963","0.290188220622","0.510478247707","0.510450359519","0.000082766177","0.449806446076","0.981468761133","0.852555702466","0.912481600659","0.912481600659"
"SNOWBALL_SPANISH_LUCENE_FILTER","ES_ES","ALL_WORDS","PRIMARY_OUTPUT","65059","871332","3589","871332","0","1","871332","194971","12811693","2230481","29161643","379565087629","2230481","379567318110","0.000588","29161643","41973336","69.476591","0.851718175843","0.305234089566","0.999994123622","0.999917304121","0.652614106594","0.627150877515","0.449410800675","0.350169642836","0.289832278511","0.509875888791","0.509847985527","0.000082695879","0.449378676109","0.981386049614","0.852460744495","0.912391466047","0.912391466047"
"SNOWBALL_SPANISH_LUCENE_FILTER","ES_ES","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","64918","869371","3525","869371","0","1","869371","194444","12787018","2201196","29076352","377858468569","2201196","377860669765","0.000583","29076352","41863370","69.455354","0.853138205793","0.305446455935","0.999994174583","0.999917233823","0.652720315259","0.627945981812","0.449838583213","0.350441220963","0.290188220622","0.510478247707","0.510450359519","0.000082766177","0.449806446076","0.981468761133","0.852555702466","0.912481600659","0.912481600659"
"SNOWBALL_SWEDISH_DIRECT","SV_SE","ALL_WORDS","PRIMARY_OUTPUT","12371","98108","68","98108","0","1","98108","25915","237017","67105","148325","4812088331","67105","4812155436","0.001394","148325","385342","38.491781","0.779348419384","0.615082186733","0.999986055105","0.999955235704","0.807534120919","0.739831942216","0.687539886056","0.642151948805","0.523855832838","0.692360693585","0.692339154006","0.000044764296","0.687517812951","0.984860422704","0.942685282143","0.963311451566","0.963311451566"
"SNOWBALL_SWEDISH_DIRECT","SV_SE","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","12342","97881","68","97881","0","1","97881","25840","236588","67105","147975","4789844472","67105","4789911577","0.001401","147975","384563","38.478741","0.779036724587","0.615212591955","0.999985990347","0.999955100897","0.807599291151","0.739644914918","0.687500000000","0.642223301999","0.523809523810","0.692295603454","0.692273994517","0.000044899103","0.687477858823","0.984821307273","0.942694565973","0.963297587020","0.963297587020"
"SNOWBALL_SWEDISH_LUCENE_FILTER","SV_SE","ALL_WORDS","PRIMARY_OUTPUT","12371","98108","68","98108","0","1","98108","26781","230676","64262","154666","4812091174","64262","4812155436","0.001335","154666","385342","40.137333","0.782116919488","0.598626674487","0.999986645901","0.999954508853","0.799306660194","0.736939762085","0.678179573117","0.628097931391","0.513064830384","0.684248529829","0.684226838572","0.000045491147","0.678157227687","0.985207247898","0.939659207875","0.961894327137","0.961894327137"
"SNOWBALL_SWEDISH_LUCENE_FILTER","SV_SE","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","12342","97881","68","97881","0","1","97881","26706","230247","64262","154316","4789847315","64262","4789911577","0.001342","154316","384563","40.127625","0.781799537535","0.598723746174","0.999986583886","0.999954370671","0.799355165030","0.736743719918","0.678122496584","0.628142458291","0.512999498691","0.684165146635","0.684143384784","0.000045629329","0.678100081584","0.985169028543","0.939660518299","0.961876797342","0.961876797342"
"SNOWBALL_YIDDISH_DIRECT","YI","ALL_WORDS","PRIMARY_OUTPUT","802","3578","0","3578","0","1","3578","1087","4962","1151","1382","6391758","1151","6392909","0.018004","1382","6344","21.784363","0.811712743334","0.782156368222","0.999819956768","0.999604172550","0.890988162495","0.805624107027","0.796660512162","0.787894185271","0.662041360907","0.796797522188","0.796599716782","0.000395827450","0.796462473806","0.982918530193","0.962014249878","0.972354049666","0.972354049666"
"SNOWBALL_YIDDISH_DIRECT","YI","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","802","3578","0","3578","0","1","3578","1087","4962","1151","1382","6391758","1151","6392909","0.018004","1382","6344","21.784363","0.811712743334","0.782156368222","0.999819956768","0.999604172550","0.890988162495","0.805624107027","0.796660512162","0.787894185271","0.662041360907","0.796797522188","0.796599716782","0.000395827450","0.796462473806","0.982918530193","0.962014249878","0.972354049666","0.972354049666"
"SNOWBALL_YIDDISH_LUCENE_FILTER","YI","ALL_WORDS","PRIMARY_OUTPUT","802","3578","0","3578","0","1","3578","1087","4962","1151","1382","6391758","1151","6392909","0.018004","1382","6344","21.784363","0.811712743334","0.782156368222","0.999819956768","0.999604172550","0.890988162495","0.805624107027","0.796660512162","0.787894185271","0.662041360907","0.796797522188","0.796599716782","0.000395827450","0.796462473806","0.982918530193","0.962014249878","0.972354049666","0.972354049666"
"SNOWBALL_YIDDISH_LUCENE_FILTER","YI","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","802","3578","0","3578","0","1","3578","1087","4962","1151","1382","6391758","1151","6392909","0.018004","1382","6344","21.784363","0.811712743334","0.782156368222","0.999819956768","0.999604172550","0.890988162495","0.805624107027","0.796660512162","0.787894185271","0.662041360907","0.796797522188","0.796599716782","0.000395827450","0.796462473806","0.982918530193","0.962014249878","0.972354049666","0.972354049666"
"SPANISH_LUCENE_SPANISH_LIGHT_STEM_FILTER","ES_ES","ALL_WORDS","PRIMARY_OUTPUT","65059","871332","3589","871332","0","1","871332","405552","1244317","147956","40729019","379567170154","147956","379567318110","0.000039","40729019","41973336","97.035458","0.893730611741","0.029645415842","0.999999610198","0.999892318297","0.514822513020","0.130863846499","0.057387272020","0.036752000024","0.029541282827","0.162772895888","0.162762055080","0.000107681703","0.057380579619","0.993823553768","0.756690454887","0.859195405761","0.859195405761"
"SPANISH_LUCENE_SPANISH_LIGHT_STEM_FILTER","ES_ES","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","64918","869371","3525","869371","0","1","869371","404617","1241848","146613","40621522","377860523152","146613","377860669765","0.000039","40621522","41863370","97.033569","0.894406108634","0.029664310351","0.999999611992","0.999892119974","0.514831961171","0.130949068412","0.057424066047","0.036775459718","0.029560783207","0.162886280533","0.162875426941","0.000107880026","0.057417362063","0.993866319748","0.756725425267","0.859233931168","0.859233931168"
"SPANISH_LUCENE_SPANISH_MINIMAL_STEM_FILTER","ES_ES","ALL_WORDS","PRIMARY_OUTPUT","65059","871332","3589","871332","0","1","871332","718633","148463","47859","41824873","379567270251","47859","379567318110","0.000013","41824873","41973336","99.646292","0.756221921130","0.003537078873","0.999999873912","0.999889695187","0.501768476392","0.017360591398","0.007041223811","0.004416184633","0.003533050405","0.051718628951","0.051713939443","0.000110304813","0.007040201537","0.995635307295","0.710609719902","0.829315972283","0.829315972283"
"SPANISH_LUCENE_SPANISH_MINIMAL_STEM_FILTER","ES_ES","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","64918","869371","3525","869371","0","1","869371","717093","148226","47148","41715144","377860622617","47148","377860669765","0.000012","41715144","41863370","99.645929","0.758678227400","0.003540708739","0.999999875224","0.999889489251","0.501770291981","0.017379114288","0.007048522419","0.004420728101","0.003536725554","0.051829129163","0.051824445330","0.000110510749","0.007047500484","0.995675746140","0.710626433954","0.829341382913","0.829341382913"
"SPANISH_LUCENE_SPANISH_PLURAL_STEM_FILTER","ES_ES","ALL_WORDS","PRIMARY_OUTPUT","65059","871332","3589","871332","0","1","871332","578805","325245","58578","41648091","379567259532","58578","379567318110","0.000015","41648091","41973336","99.225115","0.847382777999","0.007748847983","0.999999845672","0.999890132644","0.503874346827","0.037377069210","0.015357262275","0.009663967067","0.007738048760","0.081032341260","0.081026288458","0.000109867356","0.015355289170","0.995442321769","0.723731297627","0.838115191065","0.838115191065"
"SPANISH_LUCENE_SPANISH_PLURAL_STEM_FILTER","ES_ES","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","64918","869371","3525","869371","0","1","869371","577533","324656","57716","41538714","377860612049","57716","377860669765","0.000015","41538714","41863370","99.224487","0.849057985417","0.007755132948","0.999999847256","0.999889928152","0.503877490102","0.037408921072","0.015369880354","0.009671831022","0.007744455857","0.081145286724","0.081139234944","0.000110071848","0.015367905834","0.995484117647","0.723752971494","0.838144538380","0.838144538380"
"SPANISH_RADIXOR","ES_ES","ALL_WORDS","PRIMARY_OUTPUT","65059","871332","3589","871332","0","1","871332","64995","41074684","288483","898652","379567029627","288483","379567318110","0.000076","898652","41973336","2.141007","0.993025606574","0.978589931475","0.999999239969","0.999996872745","0.989294585722","0.990104500109","0.985754921826","0.981443392220","0.971909988067","0.985781345071","0.985779787115","0.000003127255","0.985753358111","0.995417814373","0.993266303762","0.994340895233","0.994340895233"
"SPANISH_RADIXOR","ES_ES","ALL_WORDS","ANY_CANDIDATE","65059","871332","3589","828695","42637","21","916797","65118","41972710","2","626","379567318108","2","379567318110","0.000000","626","41973336","0.001491","0.999999952350","0.999985085770","0.999999999995","0.999999998346","0.999992542882","0.999996978999","0.999992519005","0.999988059050","0.999985038121","0.999992519032","0.999992518205","0.000000001654","","","","",""
"SPANISH_RADIXOR","ES_ES","ALL_WORDS","ALL_CANDIDATES","65059","871332","3589","828695","42637","21","916797","65118","41972710","1349800","626","379565968310","1349800","379567318110","0.000356","626","41973336","0.001491","0.968842987168","0.999985085770","0.999996443846","0.999996442590","0.999990764808","0.974915259157","0.984167740127","0.993597526064","0.968828987818","0.984290880594","0.984289129583","0.000003557410","","","","",""
"SPANISH_RADIXOR","ES_ES","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","64918","869371","3525","869371","0","1","869371","64814","40978337","276044","885033","377860393721","276044","377860669765","0.000073","885033","41863370","2.114099","0.993308734895","0.978859012067","0.999999269456","0.999996927576","0.989429140761","0.990384762162","0.986030938205","0.981715226337","0.972446769193","0.986057405488","0.986055874970","0.000003072424","0.986029401906","0.995463637710","0.993323040564","0.994392187139","0.994392187139"
"SPANISH_RADIXOR","ES_ES","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","64918","869371","3525","826968","42403","21","914127","64933","41863370","0","0","377860669765","0","377860669765","0.000000","0","41863370","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"SPANISH_RADIXOR","ES_ES","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","64918","869371","3525","826968","42403","21","914127","64933","41863370","1255381","0","377859414384","1255381","377860669765","0.000332","0","41863370","0.000000","0.970885497124","1.000000000000","0.999996677662","0.999996678030","0.999998338831","0.976571978660","0.985227704543","0.994038240493","0.970885497124","0.985335220686","0.985333583876","0.000003321970","","","","",""
"SWEDISH_LUCENE_SWEDISH_LIGHT_STEM_FILTER","SV_SE","ALL_WORDS","PRIMARY_OUTPUT","12371","98108","68","98108","0","1","98108","22392","218635","45941","166707","4812109495","45941","4812155436","0.000955","166707","385342","43.262089","0.826359911708","0.567379107390","0.999990453135","0.999955813777","0.783684780262","0.757232036109","0.672807954234","0.605320541501","0.506940918144","0.684733049508","0.684712936280","0.000044186223","0.672786622564","0.986795482859","0.942302523776","0.964035907715","0.964035907715"
"SWEDISH_LUCENE_SWEDISH_LIGHT_STEM_FILTER","SV_SE","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","12342","97881","68","97881","0","1","97881","22338","218126","45941","166437","4789865636","45941","4789911577","0.000959","166437","384563","43.279515","0.826025213298","0.567204853301","0.999990408800","0.999955664954","0.783597631051","0.756945124029","0.672574503184","0.605125951621","0.506675896159","0.684489232882","0.684469049184","0.000044335046","0.672553099274","0.986761366955","0.942264620928","0.963999791833","0.963999791833"
"SWEDISH_LUCENE_SWEDISH_MINIMAL_STEM_FILTER","SV_SE","ALL_WORDS","PRIMARY_OUTPUT","12371","98108","68","98108","0","1","98108","23360","228181","40227","157161","4812115209","40227","4812155436","0.000836","157161","385342","40.784809","0.850127417961","0.592151906618","0.999991640544","0.999958984659","0.796071773581","0.781991317186","0.698068068834","0.630412272016","0.536178622033","0.709510092538","0.709491456160","0.000041015341","0.698048215965","0.988492665376","0.944581755622","0.966038479572","0.966038479572"
"SWEDISH_LUCENE_SWEDISH_MINIMAL_STEM_FILTER","SV_SE","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","12342","97881","68","97881","0","1","97881","23312","227624","40227","156939","4789871350","40227","4789911577","0.000840","156939","384563","40.809698","0.849815755775","0.591903017191","0.999991601724","0.999958840540","0.795947309457","0.781693541131","0.697790053555","0.630152322431","0.535850655618","0.709230928471","0.709212225121","0.000041159460","0.697770131116","0.988462934404","0.944527865197","0.965996098328","0.965996098328"
"SWEDISH_RADIXOR","SV_SE","ALL_WORDS","PRIMARY_OUTPUT","12371","98108","68","98108","0","1","98108","12330","365796","24473","19546","4812130963","24473","4812155436","0.000509","19546","385342","5.072377","0.937291970410","0.949276227351","0.999994914337","0.999990853272","0.974635570844","0.939664553041","0.943246034417","0.946854921499","0.892588119029","0.943265066457","0.943260495884","0.000009146728","0.943241460869","0.992630770222","0.993394969179","0.993012722673","0.993012722673"
"SWEDISH_RADIXOR","SV_SE","ALL_WORDS","ANY_CANDIDATE","12371","98108","68","92341","5767","5","104148","12371","385342","0","0","4812155436","0","4812155436","0.000000","0","385342","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"SWEDISH_RADIXOR","SV_SE","ALL_WORDS","ALL_CANDIDATES","12371","98108","68","92341","5767","5","104148","12371","385342","47848","0","4812107588","47848","4812155436","0.000994","0","385342","0.000000","0.889545003347","1.000000000000","0.999990056847","0.999990057643","0.999995028423","0.909639856815","0.941544130223","0.975767741439","0.889545003347","0.943156934634","0.943152245645","0.000009942357","","","","",""
"SWEDISH_RADIXOR","SV_SE","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","12342","97881","68","97881","0","1","97881","12301","365017","24473","19546","4789887104","24473","4789911577","0.000511","19546","384563","5.082652","0.937166551131","0.949173477428","0.999994890720","0.999990810798","0.974584184074","0.939543572972","0.943131801052","0.946747541943","0.892383555482","0.943150907472","0.943146315681","0.000009189202","0.943127206266","0.992611730682","0.993377890892","0.992994663001","0.992994663001"
"SWEDISH_RADIXOR","SV_SE","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","12342","97881","68","92114","5767","5","103921","12342","384563","0","0","4789911577","0","4789911577","0.000000","0","384563","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"SWEDISH_RADIXOR","SV_SE","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","12342","97881","68","92114","5767","5","103921","12342","384563","47848","0","4789863729","47848","4789911577","0.000999","0","384563","0.000000","0.889346015712","1.000000000000","0.999990010672","0.999990011473","0.999995005336","0.909473386475","0.941432652692","0.975719846569","0.889346015712","0.943051438529","0.943046728292","0.000009988527","","","","",""
"UKRAINIAN_LUCENE_MORFOLOGIK_FILTER","UK_UA","ALL_WORDS","PRIMARY_OUTPUT","1493","14245","4","14245","0","1","14245","2358","56032","828","9308","101386722","828","101387550","0.000817","9308","65340","14.245485","0.985437917693","0.857545148454","0.999991833317","0.999900091560","0.928768490886","0.956895962839","0.917054009820","0.880397209478","0.846814169992","0.919270093835","0.919222898475","0.000099908440","0.917004266345","0.997989675681","0.970999849348","0.984309781661","0.984309781661"
"UKRAINIAN_LUCENE_MORFOLOGIK_FILTER","UK_UA","ALL_WORDS","ANY_CANDIDATE","1493","14245","4","12038","2207","6","16937","2912","60394","122","4946","101387428","122","101387550","0.000120","4946","65340","7.569636","0.997984004230","0.924303642485","0.999998796696","0.999950045780","0.962151219591","0.982322936592","0.959731756929","0.938156308641","0.922581039382","0.960437530635","0.960413432420","0.000049954220","","","","",""
"UKRAINIAN_LUCENE_MORFOLOGIK_FILTER","UK_UA","ALL_WORDS","ALL_CANDIDATES","1493","14245","4","12038","2207","6","16937","2912","60394","1368","4946","101386182","1368","101387550","0.001349","4946","65340","7.569636","0.977850458211","0.924303642485","0.999986507219","0.999937764217","0.962145074852","0.966650447520","0.950323362339","0.934538657225","0.905348683816","0.950700131656","0.950669478973","0.000062235783","","","","",""
"UKRAINIAN_LUCENE_MORFOLOGIK_FILTER","UK_UA","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","1491","14236","4","14236","0","1","14236","2356","56016","828","9308","101258578","828","101259406","0.000818","9308","65324","14.248974","0.985433818873","0.857510256567","0.999991822982","0.999899965191","0.928751039775","0.956884181756","0.917032283413","0.880367133966","0.846777119361","0.919249480202","0.919202225823","0.000100034809","0.916982477117","0.997988093697","0.970977645923","0.984297603943","0.984297603943"
"UKRAINIAN_LUCENE_MORFOLOGIK_FILTER","UK_UA","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","1491","14236","4","12029","2207","6","16928","2910","60378","122","4946","101259284","122","101259406","0.000120","4946","65324","7.571490","0.997983471074","0.924285101953","0.999998795174","0.999949982596","0.962141948563","0.982318335047","0.959721515768","0.938140934008","0.922562112276","0.960427641371","0.960403512884","0.000050017404","","","","",""
"UKRAINIAN_LUCENE_MORFOLOGIK_FILTER","UK_UA","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","1491","14236","4","12029","2207","6","16928","2910","60378","1368","4946","101258038","1368","101259406","0.001351","4946","65324","7.571490","0.977844718686","0.924285101953","0.999986490144","0.999937685499","0.962135796049","0.966641904786","0.950310852286","0.934522445998","0.905325976129","0.950687806541","0.950657115187","0.000062314501","","","","",""
"UKRAINIAN_MORFOLOGIK_DIRECT","UK_UA","ALL_WORDS","PRIMARY_OUTPUT","1493","14245","4","14245","0","1","14245","2365","56016","828","9324","101386722","828","101387550","0.000817","9324","65340","14.269972","0.985433818873","0.857300275482","0.999991833317","0.999899933851","0.928646054399","0.956831877998","0.916912197996","0.880190066750","0.846572361262","0.919136923635","0.919089660097","0.000100066149","0.916862376978","0.997989675681","0.970875945953","0.984246115917","0.984246115917"
"UKRAINIAN_MORFOLOGIK_DIRECT","UK_UA","ALL_WORDS","ANY_CANDIDATE","1493","14245","4","12038","2207","6","16937","2919","60378","122","4962","101387428","122","101387550","0.000120","4962","65340","7.594123","0.997983471074","0.924058769513","0.999998796696","0.999949888071","0.962028783105","0.982267195939","0.959599491418","0.937954390108","0.922336622774","0.960310042786","0.960285871670","0.000050111929","","","","",""
"UKRAINIAN_MORFOLOGIK_DIRECT","UK_UA","ALL_WORDS","ALL_CANDIDATES","1493","14245","4","12038","2207","6","16937","2919","60378","1368","4962","101386182","1368","101387550","0.001349","4962","65340","7.594123","0.977844718686","0.924058769513","0.999986507219","0.999937606509","0.962022638366","0.966592384831","0.950191209102","0.934337338211","0.905108832524","0.950571400540","0.950540673331","0.000062393491","","","","",""
"UKRAINIAN_MORFOLOGIK_DIRECT","UK_UA","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","1491","14236","4","14236","0","1","14236","2356","56016","828","9308","101258578","828","101259406","0.000818","9308","65324","14.248974","0.985433818873","0.857510256567","0.999991822982","0.999899965191","0.928751039775","0.956884181756","0.917032283413","0.880367133966","0.846777119361","0.919249480202","0.919202225823","0.000100034809","0.916982477117","0.997988093697","0.970977645923","0.984297603943","0.984297603943"
"UKRAINIAN_MORFOLOGIK_DIRECT","UK_UA","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","1491","14236","4","12029","2207","6","16928","2910","60378","122","4946","101259284","122","101259406","0.000120","4946","65324","7.571490","0.997983471074","0.924285101953","0.999998795174","0.999949982596","0.962141948563","0.982318335047","0.959721515768","0.938140934008","0.922562112276","0.960427641371","0.960403512884","0.000050017404","","","","",""
"UKRAINIAN_MORFOLOGIK_DIRECT","UK_UA","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","1491","14236","4","12029","2207","6","16928","2910","60378","1368","4946","101258038","1368","101259406","0.001351","4946","65324","7.571490","0.977844718686","0.924285101953","0.999986490144","0.999937685499","0.962135796049","0.966641904786","0.950310852286","0.934522445998","0.905325976129","0.950687806541","0.950657115187","0.000062314501","","","","",""
"UKRAINIAN_RADIXOR","UK_UA","ALL_WORDS","PRIMARY_OUTPUT","1493","14245","4","14245","0","1","14245","1493","64732","880","608","101386670","880","101387550","0.000868","608","65340","0.930517","0.986587819301","0.990694827058","0.999991320433","0.999985333094","0.995343073746","0.987406494442","0.988637057853","0.989870692292","0.977529447297","0.988639190514","0.988631855097","0.000014666906","0.988629719696","0.997993591453","0.998265712624","0.998129633491","0.998129633491"
"UKRAINIAN_RADIXOR","UK_UA","ALL_WORDS","ANY_CANDIDATE","1493","14245","4","14055","190","2","14435","1493","65340","0","0","101387550","0","101387550","0.000000","0","65340","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"UKRAINIAN_RADIXOR","UK_UA","ALL_WORDS","ALL_CANDIDATES","1493","14245","4","14055","190","2","14435","1493","65340","1490","0","101386060","1490","101387550","0.001470","0","65340","0.000000","0.977704623672","1.000000000000","0.999985303916","0.999985313380","0.999992651958","0.982083809295","0.988726639933","0.995459946982","0.977704623672","0.988789473888","0.988782208195","0.000014686620","","","","",""
"UKRAINIAN_RADIXOR","UK_UA","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","1491","14236","4","14236","0","1","14236","1491","64716","880","608","101258526","880","101259406","0.000869","608","65324","0.930745","0.986584547838","0.990692547915","0.999991309449","0.999985314543","0.995341928682","0.987403420118","0.988634280477","0.989868213355","0.977524016676","0.988636414174","0.988629069474","0.000014685457","0.988626933033","0.997992012551","0.998264347489","0.998128161444","0.998128161444"
"UKRAINIAN_RADIXOR","UK_UA","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","1491","14236","4","14046","190","2","14426","1491","65324","0","0","101259406","0","101259406","0.000000","0","65324","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"UKRAINIAN_RADIXOR","UK_UA","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","1491","14236","4","14046","190","2","14426","1491","65324","1490","0","101257916","1490","101259406","0.001471","0","65324","0.000000","0.977699284581","1.000000000000","0.999985285318","0.999985294804","0.999992642659","0.982079499669","0.988723909852","0.995458840023","0.977699284581","0.988786774073","0.988779499204","0.000014705196","","","","",""
"YI_RADIXOR","YI","ALL_WORDS","PRIMARY_OUTPUT","802","3578","0","3578","0","1","3578","802","6195","195","149","6392714","195","6392909","0.003050","149","6344","2.348676","0.969483568075","0.976513240858","0.999969497454","0.999946243726","0.988241369156","0.970881394183","0.972985707555","0.975099162627","0.947392567671","0.972992055990","0.972965163911","0.000053756274","0.972958803000","0.995691103897","0.996142223728","0.995916612726","0.995916612726"
"YI_RADIXOR","YI","ALL_WORDS","ANY_CANDIDATE","802","3578","0","3489","89","3","3676","802","6344","0","0","6392909","0","6392909","0.000000","0","6344","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"YI_RADIXOR","YI","ALL_WORDS","ALL_CANDIDATES","802","3578","0","3489","89","3","3676","802","6344","389","0","6392520","389","6392909","0.006085","0","6344","0.000000","0.942224862617","1.000000000000","0.999939151332","0.999939211655","0.999969575666","0.953239572064","0.970253116158","0.987885016662","0.942224862617","0.970682678643","0.970653145819","0.000060788345","","","","",""
"YI_RADIXOR","YI","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","802","3578","0","3578","0","1","3578","802","6195","195","149","6392714","195","6392909","0.003050","149","6344","2.348676","0.969483568075","0.976513240858","0.999969497454","0.999946243726","0.988241369156","0.970881394183","0.972985707555","0.975099162627","0.947392567671","0.972992055990","0.972965163911","0.000053756274","0.972958803000","0.995691103897","0.996142223728","0.995916612726","0.995916612726"
"YI_RADIXOR","YI","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","802","3578","0","3489","89","3","3676","802","6344","0","0","6392909","0","6392909","0.000000","0","6344","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
"YI_RADIXOR","YI","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","802","3578","0","3489","89","3","3676","802","6344","389","0","6392520","389","6392909","0.006085","0","6344","0.000000","0.942224862617","1.000000000000","0.999939151332","0.999939211655","0.999969575666","0.953239572064","0.970253116158","0.987885016662","0.942224862617","0.970682678643","0.970653145819","0.000060788345","","","","",""
1 Stemmer Language Dictionary mode Output policy Applied dictionary rows Processed word forms Singleton dictionary rows Forms with one candidate Forms with multiple candidates Maximum candidates for one form Total candidate assignments Distinct output stems True-positive pairs False-positive pairs False-negative pairs True-negative pairs Over-stemming error pairs Over-stemming possible pairs Over-stemming percentage Under-stemming error pairs Under-stemming possible pairs Under-stemming percentage Pairwise precision Pairwise recall Pairwise specificity Pairwise accuracy Balanced accuracy Pairwise F0.5 Pairwise F1 Pairwise F2 Jaccard index Fowlkes-Mallows index Matthews correlation coefficient Pairwise error rate Adjusted Rand Index Homogeneity Completeness V-measure Normalized mutual information
2 CZECH_LUCENE_CZECH_STEM_FILTER CS_CZ ALL_WORDS PRIMARY_OUTPUT 5113 51676 2 51676 0 1 51676 9647 177249 14480 124586 1334862335 14480 1334876815 0.001085 124586 301835 41.276194 0.924476735392 0.587238060530 0.999989152557 0.999895844650 0.793613606543 0.829234311828 0.718241200736 0.633453389361 0.560355974266 0.736809286788 0.736765291417 0.000104155350 0.718191706079 0.993800637348 0.944976928457 0.968774025802 0.968774025802
3 CZECH_LUCENE_CZECH_STEM_FILTER CS_CZ LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 5038 50968 2 50968 0 1 50968 9558 174387 13950 124426 1298530265 13950 1298544215 0.001074 124426 298813 41.640089 0.925930645598 0.583599107134 0.999989257201 0.999893462107 0.791794182167 0.828708724235 0.715947860002 0.630197985095 0.557569149804 0.735100195918 0.735055396892 0.000106537893 0.715897321649 0.993897397445 0.944297456221 0.968462775946 0.968462775946
4 CZECH_RADIXOR CS_CZ ALL_WORDS PRIMARY_OUTPUT 5113 51676 2 51676 0 1 51676 5162 299762 3867 2073 1334872948 3867 1334876815 0.000290 2073 301835 0.686799 0.987264062392 0.993132009210 0.999997103103 0.999995551157 0.996564556157 0.988432097845 0.990189342389 0.991952846154 0.980569312599 0.990193689085 0.990191466141 0.000004448843 0.990187117482 0.998733220675 0.998685552738 0.998709386137 0.998709386137
5 CZECH_RADIXOR CS_CZ ALL_WORDS ANY_CANDIDATE 5113 51676 2 51080 596 4 52319 5166 301835 0 0 1334876815 0 1334876815 0.000000 0 301835 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
6 CZECH_RADIXOR CS_CZ ALL_WORDS ALL_CANDIDATES 5113 51676 2 51080 596 4 52319 5166 301835 5850 0 1334870965 5850 1334876815 0.000438 0 301835 0.000000 0.980987048442 1.000000000000 0.999995617573 0.999995618564 0.999997808787 0.984731579205 0.990402283764 0.996138677580 0.980987048442 0.990447902942 0.990445732657 0.000004381436
7 CZECH_RADIXOR CS_CZ LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 5038 50968 2 50968 0 1 50968 5037 297104 3863 1709 1298540352 3863 1298544215 0.000297 1709 298813 0.571930 0.987164705765 0.994280703985 0.999997025130 0.999995710028 0.997138864558 0.988579745136 0.990709926973 0.992849308824 0.981590876052 0.990716315904 0.990714173387 0.000004289972 0.990707781520 0.998726091764 0.999029907266 0.998877976413 0.998877976413
8 CZECH_RADIXOR CS_CZ LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 5038 50968 2 50428 540 4 51543 5040 298813 0 0 1298544215 0 1298544215 0.000000 0 298813 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
9 CZECH_RADIXOR CS_CZ LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 5038 50968 2 50428 540 4 51543 5040 298813 5782 0 1298538433 5782 1298544215 0.000445 0 298813 0.000000 0.981017416570 1.000000000000 0.999995547321 0.999995548346 0.999997773661 0.984756059381 0.990417760454 0.996144940117 0.981017416570 0.990463233325 0.990461028216 0.000004451654
10 DA_DK_RADIXOR DA_DK ALL_WORDS PRIMARY_OUTPUT 4179 28079 32 28079 0 1 28079 4184 89188 1165 707 394110021 1165 394111186 0.000296 707 89895 0.786473 0.987106128186 0.992135268925 0.999997043981 0.999995251155 0.996066156453 0.988107873355 0.989614309174 0.991125345329 0.979442126071 0.989617503860 0.989615130363 0.000004748845 0.989611934224 0.998465862775 0.998718664384 0.998592247580 0.998592247580
11 DA_DK_RADIXOR DA_DK ALL_WORDS ANY_CANDIDATE 4179 28079 32 27756 323 3 28405 4187 89895 0 0 394111186 0 394111186 0.000000 0 89895 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
12 DA_DK_RADIXOR DA_DK ALL_WORDS ALL_CANDIDATES 4179 28079 32 27756 323 3 28405 4187 89895 1849 0 394109337 1849 394111186 0.000469 0 89895 0.000000 0.979846093478 1.000000000000 0.999995308431 0.999995309500 0.999997654215 0.983811622975 0.989820468071 0.995903164910 0.979846093478 0.989871756076 0.989869434047 0.000004690500
13 DA_DK_RADIXOR DA_DK LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 4173 28033 32 28033 0 1 28033 4170 89077 1165 663 392819623 1165 392820788 0.000297 663 89740 0.738801 0.987090268389 0.992611990194 0.999997034271 0.999995347541 0.996304512232 0.988189692661 0.989843428787 0.991502709249 0.979891095099 0.989847279032 0.989844954043 0.000004652459 0.989841102043 0.998463063294 0.998811876590 0.998637439483 0.998637439483
14 DA_DK_RADIXOR DA_DK LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 4173 28033 32 27718 315 3 28351 4173 89740 0 0 392820788 0 392820788 0.000000 0 89740 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
15 DA_DK_RADIXOR DA_DK LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 4173 28033 32 27718 315 3 28351 4173 89740 1849 0 392818939 1849 392820788 0.000471 0 89740 0.000000 0.979811986156 1.000000000000 0.999995293019 0.999995294094 0.999997646509 0.983784115625 0.989803065147 0.995896117847 0.979811986156 0.989854527774 0.989852198158 0.000004705906
16 ENGLISH_LUCENE_KSTEM_FILTER US_UK ALL_WORDS PRIMARY_OUTPUT 396939 607439 250964 607439 0 1 607439 371125 237565 1368501 76305 184489083270 1368501 184490451771 0.000742 76305 313870 24.311020 0.147917333410 0.756889795138 0.999992582267 0.999992168681 0.878441188702 0.176283968232 0.247471790726 0.415099040868 0.141208449266 0.334599940499 0.334597833111 0.000007831319 0.247469648794 0.980686838187 0.992107972963 0.986364345289 0.986364345289
17 ENGLISH_LUCENE_KSTEM_FILTER US_UK LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 374384 583910 228735 583910 0 1 583910 347624 237551 1367069 74340 170473473135 1367069 170474840204 0.000802 74340 311891 23.835250 0.148041904002 0.761647498645 0.999991980817 0.999991544756 0.880819739731 0.176476898525 0.247899438093 0.416437018089 0.141486991947 0.335791223646 0.335788961354 0.000008455244 0.247897133948 0.979822223835 0.991990504792 0.985868818390 0.985868818390
18 ENGLISH_LUCENE_MINIMAL_FILTER US_UK ALL_WORDS PRIMARY_OUTPUT 396939 607439 250964 607439 0 1 607439 453328 137225 1122264 176645 184489329507 1122264 184490451771 0.000608 176645 313870 56.279670 0.108952916619 0.437203300730 0.999993916953 0.999992959490 0.718598608842 0.128203906480 0.174435713655 0.272816484020 0.095551668577 0.218253464509 0.218250987161 0.000007040510 0.174433464995 0.995202198233 0.981173943304 0.988138284715 0.988138284715
19 ENGLISH_LUCENE_MINIMAL_FILTER US_UK LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 374384 583910 228735 583910 0 1 583910 430129 136932 1120871 174959 170473719333 1120871 170474840204 0.000657 174959 311891 56.096200 0.108866014789 0.439037997249 0.999993425006 0.999992398716 0.719515711128 0.128139023335 0.174469673707 0.273277328232 0.095572048952 0.218623688336 0.218621020778 0.000007601284 0.174467253215 0.994993790771 0.980519680106 0.987703711280 0.987703711280
20 ENGLISH_LUCENE_PORTER_COPIED US_UK ALL_WORDS PRIMARY_OUTPUT 396939 607439 250964 607439 0 1 607439 319968 285390 1557406 28480 184488894365 1557406 184490451771 0.000844 28480 313870 9.073820 0.154867928951 0.909261796285 0.999991558338 0.999991403982 0.954626677312 0.185678591198 0.264658505304 0.460562583837 0.152510906996 0.375253902399 0.375251973425 0.000008596018 0.264656367392 0.969648379409 0.997199419831 0.983230936080 0.983230936080
21 ENGLISH_LUCENE_PORTER_COPIED US_UK LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 374384 583910 228735 583910 0 1 583910 298779 283761 1552702 28130 170473287502 1552702 170474840204 0.000911 28130 311891 9.019177 0.154514956196 0.909808234287 0.999990891899 0.999990726907 0.954899563093 0.185277176317 0.264165961476 0.460049474275 0.152183881415 0.374938634269 0.374936557303 0.000009273093 0.264163659878 0.968644600847 0.997108656044 0.982670549062 0.982670549062
22 ENGLISH_LUCENE_PORTER_FILTER US_UK ALL_WORDS PRIMARY_OUTPUT 396939 607439 250964 607439 0 1 607439 319968 285390 1557406 28480 184488894365 1557406 184490451771 0.000844 28480 313870 9.073820 0.154867928951 0.909261796285 0.999991558338 0.999991403982 0.954626677312 0.185678591198 0.264658505304 0.460562583837 0.152510906996 0.375253902399 0.375251973425 0.000008596018 0.264656367392 0.969648379409 0.997199419831 0.983230936080 0.983230936080
23 ENGLISH_LUCENE_PORTER_FILTER US_UK LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 374384 583910 228735 583910 0 1 583910 298779 283761 1552702 28130 170473287502 1552702 170474840204 0.000911 28130 311891 9.019177 0.154514956196 0.909808234287 0.999990891899 0.999990726907 0.954899563093 0.185277176317 0.264165961476 0.460049474275 0.152183881415 0.374938634269 0.374936557303 0.000009273093 0.264163659878 0.968644600847 0.997108656044 0.982670549062 0.982670549062
24 ENGLISH_LUCENE_POSSESSIVE_FILTER US_UK ALL_WORDS PRIMARY_OUTPUT 396939 607439 250964 607439 0 1 607439 591899 7 1115154 313863 184489336617 1115154 184490451771 0.000604 313863 313870 99.997770 0.000006277121 0.000022302227 0.999993955492 0.999992254263 0.500008128860 0.000007330589 0.000009796848 0.000014763939 0.000004898448 0.000011831896 0.000008625150 0.000007745737 0.000007141644 0.995789196698 0.958018540631 0.976538780935 0.976538780935
25 ENGLISH_LUCENE_POSSESSIVE_FILTER US_UK LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 374384 583910 228735 583910 0 1 583910 568400 5 1113773 311886 170473726431 1113773 170474840204 0.000653 311886 311891 99.998397 0.000004489225 0.000016031242 0.999993466643 0.999991637145 0.500004748942 0.000005244385 0.000007014251 0.000010587200 0.000003507138 0.000008483387 0.000005026087 0.000008362855 0.000004155674 0.995605378040 0.956423154691 0.975621022465 0.975621022465
26 ENGLISH_OPENNLP_PORTER US_UK ALL_WORDS PRIMARY_OUTPUT 396939 607439 250964 607439 0 1 607439 319968 285390 1557406 28480 184488894365 1557406 184490451771 0.000844 28480 313870 9.073820 0.154867928951 0.909261796285 0.999991558338 0.999991403982 0.954626677312 0.185678591198 0.264658505304 0.460562583837 0.152510906996 0.375253902399 0.375251973425 0.000008596018 0.264656367392 0.969648379409 0.997199419831 0.983230936080 0.983230936080
27 ENGLISH_OPENNLP_PORTER US_UK LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 374384 583910 228735 583910 0 1 583910 298779 283761 1552702 28130 170473287502 1552702 170474840204 0.000911 28130 311891 9.019177 0.154514956196 0.909808234287 0.999990891899 0.999990726907 0.954899563093 0.185277176317 0.264165961476 0.460049474275 0.152183881415 0.374938634269 0.374936557303 0.000009273093 0.264163659878 0.968644600847 0.997108656044 0.982670549062 0.982670549062
28 ENGLISH_PAICE_HUSK_LANCASTER US_UK ALL_WORDS PRIMARY_OUTPUT 396939 607439 250964 607439 0 1 607439 268169 283991 3062661 29879 184487389110 3062661 184490451771 0.001660 29879 313870 9.519546 0.084858240415 0.904804536910 0.999983399352 0.999983237427 0.952393968131 0.103642734217 0.155164208820 0.308542866654 0.084107327905 0.277092260667 0.277089454298 0.000016762573 0.155161580693 0.937768073854 0.996599815184 0.966289292109 0.966289292109
29 ENGLISH_PAICE_HUSK_LANCASTER US_UK LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 374384 583910 228735 583910 0 1 583910 249411 282398 3045870 29493 170471794334 3045870 170474840204 0.001787 29493 311891 9.456188 0.084848335531 0.905438117804 0.999982133023 0.999981960051 0.952710125414 0.103632575002 0.155156958803 0.308575577075 0.084103067491 0.277173081705 0.277170064389 0.000018039949 0.155154132315 0.936076835754 0.996486960554 0.965337716341 0.965337716341
30 ENGLISH_RADIXOR US_UK ALL_WORDS PRIMARY_OUTPUT 396939 607439 250964 607439 0 1 607439 390361 292001 1149886 21869 184489301885 1149886 184490451771 0.000623 21869 313870 6.967534 0.202513095686 0.930324656705 0.999993767233 0.999993648707 0.965159211969 0.240076409811 0.332621199859 0.541270431499 0.199487482886 0.434054059102 0.434052478080 0.000006351293 0.332619335001 0.994214506865 0.997769723414 0.995988942533 0.995988942533
31 ENGLISH_RADIXOR US_UK ALL_WORDS ANY_CANDIDATE 396939 607439 250964 578231 29208 1355 2838145 397392 313855 12 15 184490451759 12 184490451771 0.000000 15 313870 0.004779 0.999961767245 0.999952209513 0.999999999935 0.999999999854 0.999976104724 0.999959855684 0.999956988357 0.999954121045 0.999913980413 0.999956988368 0.999956988295 0.000000000146
32 ENGLISH_RADIXOR US_UK ALL_WORDS ALL_CANDIDATES 396939 607439 250964 578231 29208 1355 2838145 397392 313855 11482166 15 184478969605 11482166 184490451771 0.006224 15 313870 0.004779 0.026606853277 0.999952209513 0.999937762817 0.999937762842 0.999944986165 0.033038791524 0.051834488023 0.120237128281 0.026606819443 0.163112175274 0.163107098882 0.000062237158
33 ENGLISH_RADIXOR US_UK LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 374384 583910 228735 583910 0 1 583910 367590 290572 1148489 21319 170473691715 1148489 170474840204 0.000674 21319 311891 6.835401 0.201917778329 0.931645991709 0.999993263000 0.999993137956 0.965819627354 0.239424468968 0.331901731173 0.540775136091 0.198970131062 0.433723286019 0.433721583515 0.000006862044 0.331899721995 0.993959181482 0.997731171071 0.995841604460 0.995841604460
34 ENGLISH_RADIXOR US_UK LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 374384 583910 228735 555084 28826 1355 2812871 374506 311891 0 0 170474840204 0 170474840204 0.000000 0 311891 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
35 ENGLISH_RADIXOR US_UK LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 374384 583910 228735 555084 28826 1355 2812871 374506 311891 11470018 0 170463370186 11470018 170474840204 0.006728 0 311891 0.000000 0.026472025883 1.000000000000 0.999932717239 0.999932717362 0.999966358619 0.032872482055 0.051578660140 0.119686728696 0.026472025883 0.162702261457 0.162696787836 0.000067282638
36 ENGLISH_SNOWBALL_ORIGINAL_PORTER US_UK ALL_WORDS PRIMARY_OUTPUT 396939 607439 250964 607439 0 1 607439 321092 285304 1555293 28566 184488896478 1555293 184490451771 0.000843 28566 313870 9.101220 0.155006228957 0.908987797496 0.999991569791 0.999991414969 0.954489683644 0.185835337999 0.264848800190 0.460750814660 0.152637303435 0.375364850057 0.375362921954 0.000008585031 0.264846663203 0.969891477221 0.997192899073 0.983352728141 0.983352728141
37 ENGLISH_SNOWBALL_ORIGINAL_PORTER US_UK LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 374384 583910 228735 583910 0 1 583910 299877 283675 1550615 28216 170473289589 1550615 170474840204 0.000910 28216 311891 9.046750 0.154651118416 0.909532496930 0.999990904142 0.999990738644 0.954761700536 0.185431499934 0.264353286139 0.460234326480 0.152308234175 0.375046954242 0.375044878196 0.000009261356 0.264350985524 0.968893806180 0.997101844661 0.982795461434 0.982795461434
38 ENGLISH_SNOWBALL_PORTER2 US_UK ALL_WORDS PRIMARY_OUTPUT 396939 607439 250964 607439 0 1 607439 318385 285334 1566711 28536 184488885060 1566711 184490451771 0.000849 28536 313870 9.091662 0.154064291094 0.909083378469 0.999991507902 0.999991353242 0.954537443185 0.184752753479 0.263476636895 0.459101696688 0.151726514306 0.374242282819 0.374240346981 0.000008646758 0.263474493989 0.969037354042 0.997181597682 0.982908049045 0.982908049045
39 ENGLISH_SNOWBALL_PORTER2 US_UK LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 374384 583910 228735 583910 0 1 583910 297220 283730 1561891 28161 170473278313 1561891 170474840204 0.000916 28161 311891 9.029116 0.153731454074 0.909708840589 0.999990837997 0.999990672823 0.954849839293 0.184374949232 0.263015918336 0.458637294569 0.151421029768 0.373966392672 0.373964308569 0.000009327177 0.263013611479 0.968019617024 0.997095706553 0.982342555079 0.982342555079
40 FINNISH_LUCENE_FINNISH_LIGHT_STEM_FILTER FI_FI ALL_WORDS PRIMARY_OUTPUT 57027 1811717 292 1811717 0 1 1811717 439975 12355389 2223150 19168306 1641124591341 2223150 1641126814491 0.000135 19168306 31523695 60.806025 0.847505295284 0.391939745642 0.999998645351 0.999986965635 0.695969195497 0.687649407375 0.535999578676 0.439151826652 0.366119825424 0.576342788507 0.576337821084 0.000013034365 0.535993941880 0.988126027331 0.886473473160 0.934543630400 0.934543630400
41 FINNISH_LUCENE_FINNISH_LIGHT_STEM_FILTER FI_FI LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 54762 1757055 274 1757055 0 1 1757055 431848 11988389 1806392 18825444 1543587637760 1806392 1543589444152 0.000117 18825444 30813833 61.094133 0.869052506162 0.389058673746 0.999998829746 0.999986634125 0.694528751746 0.697056446146 0.537492108587 0.437372459518 0.367513988637 0.581474346349 0.581469391800 0.000013365875 0.537486398327 0.989268269625 0.885293761089 0.934397488899 0.934397488899
42 FINNISH_RADIXOR FI_FI ALL_WORDS PRIMARY_OUTPUT 57027 1811717 292 1811717 0 1 1811717 69091 30552427 731279 971268 1641126083212 731279 1641126814491 0.000045 971268 31523695 3.081073 0.976624284859 0.969189271753 0.999999554404 0.999998962594 0.984594413078 0.975128170336 0.972892573600 0.970667204156 0.947215985975 0.972899675927 0.972899157490 0.000001037406 0.972892054895 0.996084757586 0.993746341306 0.994914175412 0.994914175412
43 FINNISH_RADIXOR FI_FI ALL_WORDS ANY_CANDIDATE 57027 1811717 292 1754389 57328 6 1876272 69769 31523695 0 0 1641126814491 0 1641126814491 0.000000 0 31523695 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
44 FINNISH_RADIXOR FI_FI ALL_WORDS ALL_CANDIDATES 57027 1811717 292 1754389 57328 6 1876272 69769 31523695 1683575 0 1641125130916 1683575 1641126814491 0.000103 0 31523695 0.000000 0.949301011495 1.000000000000 0.999998974135 0.999998974154 0.999999487067 0.959025334376 0.973991195713 0.989431554710 0.949301011495 0.974320794962 0.974320295201 0.000001025846
45 FINNISH_RADIXOR FI_FI LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 54762 1757055 274 1757055 0 1 1757055 54633 30078528 730145 735305 1543588714007 730145 1543589444152 0.000047 735305 30813833 2.386282 0.976300667023 0.976137178390 0.999999526982 0.999999050641 0.988068352686 0.976267964916 0.976218915862 0.976169871736 0.953542638154 0.976218919284 0.976218444595 0.000000949359 0.976218441173 0.996000407428 0.996068984852 0.996034694959 0.996034694959
46 FINNISH_RADIXOR FI_FI LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 54762 1757055 274 1712724 44331 6 1805864 54984 30813833 0 0 1543589444152 0 1543589444152 0.000000 0 30813833 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
47 FINNISH_RADIXOR FI_FI LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 54762 1757055 274 1712724 44331 6 1805864 54984 30813833 1653320 0 1543587790832 1653320 1543589444152 0.000107 0 30813833 0.000000 0.949077148834 1.000000000000 0.999998928912 0.999998928933 0.999999464456 0.958842548108 0.973873352732 0.989382907677 0.949077148834 0.974205906795 0.974205385065 0.000001071067
48 FRENCH_LUCENE_FRENCH_LIGHT_STEM_FILTER FR_FR ALL_WORDS PRIMARY_OUTPUT 59240 425210 2301 425210 0 1 425210 245918 202782 276403 5251833 90395828427 276403 90396104830 0.000306 5251833 5454615 96.282377 0.423181026117 0.037176226003 0.999996942313 0.999938848002 0.518586584158 0.137547303040 0.068348107452 0.045471618191 0.035383242558 0.125428359900 0.125414592230 0.000061151998 0.068339028277 0.974109647704 0.812375827422 0.885921707253 0.885921707253
49 FRENCH_LUCENE_FRENCH_LIGHT_STEM_FILTER FR_FR LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 57698 421231 2133 421231 0 1 421231 245182 200690 262689 5239869 88711863817 262689 88712126506 0.000296 5239869 5440559 96.311225 0.433101197939 0.036887753630 0.999997038860 0.999937976681 0.518442396245 0.137570562409 0.067985131280 0.045148356975 0.035188720533 0.126396717862 0.126383026150 0.000062023319 0.067976159358 0.975085555241 0.811143708698 0.885591261484 0.885591261484
50 FRENCH_LUCENE_FRENCH_MINIMAL_STEM_FILTER FR_FR ALL_WORDS PRIMARY_OUTPUT 59240 425210 2301 425210 0 1 425210 269236 183612 160438 5271003 90395944392 160438 90396104830 0.000177 5271003 5454615 96.633823 0.533678244441 0.033661770812 0.999998225167 0.999939918724 0.516829997990 0.134399775137 0.063329059361 0.041424008382 0.032699958487 0.134031916915 0.134021061615 0.000060081276 0.063322352769 0.984019125555 0.810978546011 0.889158144694 0.889158144694
51 FRENCH_LUCENE_FRENCH_MINIMAL_STEM_FILTER FR_FR LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 57698 421231 2133 421231 0 1 421231 268411 181686 147476 5258873 88711979030 147476 88712126506 0.000166 5258873 5440559 96.660527 0.551965293685 0.033394730211 0.999998337589 0.999939061122 0.516696533900 0.134438681544 0.062979128454 0.041121435592 0.032513396928 0.135767198057 0.135756528540 0.000060938878 0.062972571968 0.985086367216 0.809773733549 0.888868235590 0.888868235590
52 FRENCH_RADIXOR FR_FR ALL_WORDS PRIMARY_OUTPUT 59240 425210 2301 425210 0 1 425210 60225 4985455 318767 469160 90395786063 318767 90396104830 0.000353 469160 5454615 8.601157 0.939903156391 0.913988429981 0.999996473664 0.999991284144 0.956992451823 0.934603310507 0.926764667966 0.919056419273 0.863524187383 0.926855226151 0.926850879168 0.000008715856 0.926760310630 0.988772235003 0.985214034569 0.986989927876 0.986989927876
53 FRENCH_RADIXOR FR_FR ALL_WORDS ANY_CANDIDATE 59240 425210 2301 382170 43040 56 477024 60383 5454383 12 232 90396104818 12 90396104830 0.000000 232 5454615 0.004253 0.999997799939 0.999957467209 0.999999999867 0.999999997301 0.999978733538 0.999989733133 0.999977633167 0.999965533495 0.999955267335 0.999977633371 0.999977632021 0.000000002699
54 FRENCH_RADIXOR FR_FR ALL_WORDS ALL_CANDIDATES 59240 425210 2301 382170 43040 56 477024 60383 5454383 1056255 232 90395048575 1056255 90396104830 0.001168 232 5454615 0.004253 0.837764747479 0.999957467209 0.999988315260 0.999988313399 0.999972891234 0.865852951156 0.911703747510 0.962682080453 0.837734895644 0.915275431226 0.915270082203 0.000011686601
55 FRENCH_RADIXOR FR_FR LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 57698 421231 2133 421231 0 1 421231 58069 4975123 315266 465436 88711811240 315266 88712126506 0.000355 465436 5440559 8.554930 0.940407784758 0.914450702584 0.999996446190 0.999991200142 0.957223574387 0.935099145312 0.927247620620 0.919526848134 0.864363145162 0.927338427699 0.927334038919 0.000008799858 0.927243221287 0.988915897225 0.985549842615 0.987230000708 0.987230000708
56 FRENCH_RADIXOR FR_FR LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 57698 421231 2133 380101 41130 56 468574 58208 5440559 0 0 88712126506 0 88712126506 0.000000 0 5440559 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
57 FRENCH_RADIXOR FR_FR LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 57698 421231 2133 380101 41130 56 468574 58208 5440559 938985 0 88711187521 938985 88712126506 0.001058 0 5440559 0.000000 0.852813147774 1.000000000000 0.999989415370 0.999989416019 0.999994707685 0.878679151458 0.920560336911 0.966633773699 0.852813147774 0.923478829088 0.923473941734 0.000010583981
58 GERMAN_CISTEM DE_DE ALL_WORDS PRIMARY_OUTPUT 54092 296974 1474 296974 0 1 296974 59097 1053889 477122 329983 44094768857 477122 44095245979 0.001082 329983 1383872 23.844908 0.688361481400 0.761550923785 0.999989179741 0.999981696901 0.880770051763 0.701851885397 0.723108954973 0.745693871888 0.566304351331 0.724031989665 0.724022910459 0.000018303099 0.723099826442 0.974048119240 0.975147027686 0.974597263694 0.974597263694
59 GERMAN_CISTEM DE_DE LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 16007 150098 228 150098 0 1 150098 23023 725447 156784 147964 11263599558 156784 11263756342 0.001392 147964 873411 16.940936 0.822286906717 0.830590638313 0.999986080665 0.999972946470 0.915288359489 0.823934343933 0.826417914358 0.828916502414 0.704184159310 0.826428343371 0.826414817348 0.000027053530 0.826404386881 0.985935685912 0.973569821618 0.979713735095 0.979713735095
60 GERMAN_LUCENE_GERMAN_LIGHT_STEM_FILTER DE_DE ALL_WORDS PRIMARY_OUTPUT 54092 296974 1474 296974 0 1 296974 98357 709263 205740 674609 44095040239 205740 44095245979 0.000467 674609 1383872 48.747933 0.775148278202 0.512520666651 0.999995334191 0.999980035912 0.756258000421 0.703092101246 0.617052253820 0.549774428024 0.446186239158 0.630301128270 0.630292039259 0.000019964088 0.617042686770 0.980753120457 0.936533167951 0.958133203614 0.958133203614
61 GERMAN_LUCENE_GERMAN_LIGHT_STEM_FILTER DE_DE LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 16007 150098 228 150098 0 1 150098 50335 471565 55477 401846 11263700865 55477 11263756342 0.000493 401846 873411 46.008809 0.894738939212 0.539911908597 0.999995074734 0.999959401861 0.769953491666 0.790797426464 0.673446377708 0.586423560557 0.507666155661 0.695039717114 0.695022690564 0.000040598139 0.673427319227 0.991320177896 0.915069562866 0.951669957566 0.951669957566
62 GERMAN_LUCENE_GERMAN_MINIMAL_STEM_FILTER DE_DE ALL_WORDS PRIMARY_OUTPUT 54092 296974 1474 296974 0 1 296974 140505 271626 110840 1112246 44095135139 110840 44095245979 0.000251 1112246 1383872 80.372029 0.710196461908 0.196279713731 0.999997486350 0.999972263504 0.598138600041 0.466112921692 0.307558349534 0.229493166050 0.181724639931 0.373359288402 0.373350267608 0.000027736496 0.307548938689 0.983615403456 0.896263607272 0.937910029995 0.937910029995
63 GERMAN_LUCENE_GERMAN_MINIMAL_STEM_FILTER DE_DE LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 16007 150098 228 150098 0 1 150098 80363 132221 21214 741190 11263735128 21214 11263756342 0.000188 741190 873411 84.861537 0.861739498811 0.151384628772 0.999998116614 0.999932318770 0.575691372693 0.444544636019 0.257528392768 0.181269722976 0.147794886125 0.361184321539 0.361168285320 0.000067681230 0.257511188319 0.992642524078 0.854402700840 0.918349417869 0.918349417869
64 GERMAN_LUCENE_GERMAN_STEM_FILTER DE_DE ALL_WORDS PRIMARY_OUTPUT 54092 296974 1474 296974 0 1 296974 81085 619354 331871 764518 44094914108 331871 44095245979 0.000753 764518 1383872 55.244849 0.651111987174 0.447551507654 0.999992473769 0.999975136671 0.723771990712 0.596821367368 0.530473894660 0.477402037056 0.360982967729 0.539820480819 0.539808754751 0.000024863329 0.530461889452 0.975549631706 0.942889706548 0.958941664321 0.958941664321
65 GERMAN_LUCENE_GERMAN_STEM_FILTER DE_DE LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 16007 150098 228 150098 0 1 150098 41574 378734 78723 494677 11263677619 78723 11263756342 0.000699 494677 873411 56.637368 0.827911694432 0.433626322545 0.999993010946 0.999949097306 0.716809666745 0.700518896035 0.569153364571 0.479276535831 0.397773842757 0.599169678345 0.599148958371 0.000050902694 0.569130398173 0.988583531594 0.918717729459 0.952371013508 0.952371013508
66 GERMAN_RADIXOR DE_DE ALL_WORDS PRIMARY_OUTPUT 54092 296974 1474 296974 0 1 296974 68104 1128969 98192 254903 44095147787 98192 44095245979 0.000223 254903 1383872 18.419550 0.919984419322 0.815804496370 0.999997773184 0.999991992699 0.907901134777 0.897072808397 0.864768082211 0.834709150216 0.761754553110 0.866329859738 0.866325955582 0.000008007301 0.864764092865 0.989946248415 0.975085217969 0.982459538105 0.982459538105
67 GERMAN_RADIXOR DE_DE ALL_WORDS ANY_CANDIDATE 54092 296974 1474 248400 48574 8 361016 70717 1272705 1375 111167 44095244604 1375 44095245979 0.000003 111167 1383872 8.033041 0.998920789903 0.919669593720 0.999999968818 0.999997447832 0.959834781269 0.981996366774 0.957658377578 0.934497606897 0.918756727140 0.958476435291 0.958475209548 0.000002552168
68 GERMAN_RADIXOR DE_DE ALL_WORDS ALL_CANDIDATES 54092 296974 1474 248400 48574 8 361016 70717 1272705 244817 111167 44095001162 244817 44095245979 0.000555 111167 1383872 8.033041 0.838673179038 0.919669593720 0.999994447996 0.999991927184 0.959832020858 0.853710645080 0.877305874349 0.902242446842 0.781429112618 0.878238135035 0.878234164088 0.000008072816
69 GERMAN_RADIXOR DE_DE LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 16007 150098 228 150098 0 1 150098 17264 814297 47898 59114 11263708444 47898 11263756342 0.000425 59114 873411 6.768177 0.944446441930 0.932318232768 0.999995747600 0.999990500176 0.966156990184 0.941995622128 0.938343149309 0.934718891125 0.883847872972 0.938362743125 0.938357995965 0.000009499824 0.938338399230 0.994062310308 0.990664418294 0.992360455671 0.992360455671
70 GERMAN_RADIXOR DE_DE LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 16007 150098 228 135120 14978 8 167157 18366 873411 0 0 11263756342 0 11263756342 0.000000 0 873411 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
71 GERMAN_RADIXOR DE_DE LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 16007 150098 228 135120 14978 8 167157 18366 873411 97544 0 11263658798 97544 11263756342 0.000866 0 873411 0.000000 0.899538083639 1.000000000000 0.999991340012 0.999991340683 0.999995670006 0.917982540684 0.947112449481 0.978151677228 0.899538083639 0.948439815507 0.948435708759 0.000008659317
72 HE_IL_RADIXOR HE_IL ALL_WORDS PRIMARY_OUTPUT 2358 58714 0 58714 0 1 58714 2358 688361 25234 19916 1722904030 25234 1722929264 0.001465 19916 708277 2.811894 0.964638205144 0.971881057835 0.999985354013 0.999973805398 0.985933205924 0.966078126522 0.968246086849 0.970423799230 0.938446730860 0.968252859146 0.968239762122 0.000026194602 0.968232984327 0.993166390361 0.993627509527 0.993396896433 0.993396896433
73 HE_IL_RADIXOR HE_IL ALL_WORDS ANY_CANDIDATE 2358 58714 0 56674 2040 40 62376 2358 708277 0 0 1722929264 0 1722929264 0.000000 0 708277 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
74 HE_IL_RADIXOR HE_IL ALL_WORDS ALL_CANDIDATES 2358 58714 0 56674 2040 40 62376 2358 708277 86628 0 1722842636 86628 1722929264 0.005028 0 708277 0.000000 0.891020939609 1.000000000000 0.999949720513 0.999949741174 0.999974860256 0.910874182109 0.942370251906 0.976122467036 0.891020939609 0.943939055029 0.943915324345 0.000050258826
75 HE_IL_RADIXOR HE_IL LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 2358 58714 0 58714 0 1 58714 2358 688361 25234 19916 1722904030 25234 1722929264 0.001465 19916 708277 2.811894 0.964638205144 0.971881057835 0.999985354013 0.999973805398 0.985933205924 0.966078126522 0.968246086849 0.970423799230 0.938446730860 0.968252859146 0.968239762122 0.000026194602 0.968232984327 0.993166390361 0.993627509527 0.993396896433 0.993396896433
76 HE_IL_RADIXOR HE_IL LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 2358 58714 0 56674 2040 40 62376 2358 708277 0 0 1722929264 0 1722929264 0.000000 0 708277 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
77 HE_IL_RADIXOR HE_IL LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 2358 58714 0 56674 2040 40 62376 2358 708277 86628 0 1722842636 86628 1722929264 0.005028 0 708277 0.000000 0.891020939609 1.000000000000 0.999949720513 0.999949741174 0.999974860256 0.910874182109 0.942370251906 0.976122467036 0.891020939609 0.943939055029 0.943915324345 0.000050258826
78 HUNGARIAN_LUCENE_HUNGARIAN_LIGHT_STEM_FILTER HU_HU ALL_WORDS PRIMARY_OUTPUT 19406 916344 1 916344 0 1 916344 94328 14036270 4132555 8125833 419816410338 4132555 419820542893 0.000984 8125833 22162103 36.665442 0.772546931351 0.633345580968 0.999990156377 0.999970802427 0.816667868673 0.740017627855 0.696054898613 0.657022705053 0.533806904809 0.699492090778 0.699477892426 0.000029197573 0.696040442258 0.982615378770 0.926771756762 0.953876941828 0.953876941828
79 HUNGARIAN_LUCENE_HUNGARIAN_LIGHT_STEM_FILTER HU_HU LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 18360 878513 1 878513 0 1 878513 91516 13492703 3639046 7918708 385867055871 3639046 385870694917 0.000943 7918708 21411411 36.983588 0.787584676848 0.630164121365 0.999990569261 0.999970049260 0.815077345313 0.750107959995 0.700134758022 0.656404225003 0.538621031944 0.704491026122 0.704476576404 0.000029950740 0.700119966552 0.983686881315 0.925487153321 0.953699929950 0.953699929950
80 HUNGARIAN_RADIXOR HU_HU ALL_WORDS PRIMARY_OUTPUT 19406 916344 1 916344 0 1 916344 20535 21962266 272900 199837 419820269993 272900 419820542893 0.000065 199837 22162103 0.901706 0.987726648859 0.990982940563 0.999999349960 0.999998874014 0.995491145262 0.988376194087 0.989352115329 0.990329965723 0.978928596533 0.989353455019 0.989352892139 0.000001125986 0.989351552308 0.998036093538 0.997808712909 0.997922390271 0.997922390271
81 HUNGARIAN_RADIXOR HU_HU ALL_WORDS ANY_CANDIDATE 19406 916344 1 904024 12320 5 929326 20567 22162103 0 0 419820542893 0 419820542893 0.000000 0 22162103 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
82 HUNGARIAN_RADIXOR HU_HU ALL_WORDS ALL_CANDIDATES 19406 916344 1 904024 12320 5 929326 20567 22162103 460158 0 419820082735 460158 419820542893 0.000110 0 22162103 0.000000 0.979659062372 1.000000000000 0.999998903917 0.999998903975 0.999999451959 0.983660778882 0.989725029923 0.995864516790 0.979659062372 0.989777279176 0.989776736737 0.000001096025
83 HUNGARIAN_RADIXOR HU_HU LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 18360 878513 1 878513 0 1 878513 18363 21247134 272775 164277 385870422142 272775 385870694917 0.000071 164277 21411411 0.767240 0.987324528185 0.992327595785 0.999999293092 0.999998867424 0.996163444439 0.988321101757 0.989819739994 0.991322930046 0.979844666523 0.989822900984 0.989822335019 0.000001132576 0.989819173678 0.997945135090 0.998273386381 0.998109233747 0.998109233747
84 HUNGARIAN_RADIXOR HU_HU LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 18360 878513 1 867360 11153 5 890245 18375 21411411 0 0 385870694917 0 385870694917 0.000000 0 21411411 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
85 HUNGARIAN_RADIXOR HU_HU LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 18360 878513 1 867360 11153 5 890245 18375 21411411 458462 0 385870236455 458462 385870694917 0.000119 0 21411411 0.000000 0.979036823854 1.000000000000 0.999998811877 0.999998811943 0.999999405938 0.983158850285 0.989407384494 0.995735852714 0.979036823854 0.989462896653 0.989462308851 0.000001188057
86 HUNSPELL_CZECH_LUCENE_FILTER CS_CZ ALL_WORDS PRIMARY_OUTPUT 5113 51676 2 51676 0 1 51676 10920 213552 11408 88283 1334865407 11408 1334876815 0.000855 88283 301835 29.248762 0.949288762447 0.707512382593 0.999991453893 0.999925335085 0.853751918243 0.888559718726 0.810759403563 0.745486280807 0.681745481942 0.819532521678 0.819499025505 0.000074664915 0.810722859062 0.995776551361 0.952852006662 0.973841506371 0.973841506371
87 HUNSPELL_CZECH_LUCENE_FILTER CS_CZ ALL_WORDS ANY_CANDIDATE 5113 51676 2 48359 3317 5 55596 11359 224312 10102 77523 1334866713 10102 1334876815 0.000757 77523 301835 25.683900 0.956905304291 0.743160998559 0.999992432261 0.999934372078 0.871576715410 0.904855299474 0.836596431881 0.777913569166 0.719093919606 0.843288029954 0.843258147533 0.000065627922
88 HUNSPELL_CZECH_LUCENE_FILTER CS_CZ ALL_WORDS ALL_CANDIDATES 5113 51676 2 48359 3317 5 55596 11359 224312 13917 77523 1334862898 13917 1334876815 0.001043 77523 301835 25.683900 0.941581419558 0.743160998559 0.999989574319 0.999931514783 0.871575286439 0.893850652440 0.830686733424 0.775860578084 0.710405634802 0.836508570179 0.836476906392 0.000068485217
89 HUNSPELL_CZECH_LUCENE_FILTER CS_CZ LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 5038 50968 2 50968 0 1 50968 10816 210827 11239 87986 1298532976 11239 1298544215 0.000866 87986 298813 29.445171 0.949388920411 0.705548286052 0.999991344923 0.999923605087 0.852769815487 0.888008949714 0.809504702628 0.743753342581 0.679973036781 0.818437368155 0.818403143837 0.000076394913 0.809467327086 0.995812473772 0.952393750423 0.973619286126 0.973619286126
90 HUNSPELL_CZECH_LUCENE_FILTER CS_CZ LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 5038 50968 2 47731 3237 5 54804 11240 221382 10028 77431 1298534187 10028 1298544215 0.000772 77431 298813 25.912862 0.956665658355 0.740871381098 0.999992277506 0.999932663918 0.870431829302 0.904003665310 0.835052421340 0.775874033233 0.716815448726 0.841882537861 0.841851913927 0.000067336082
91 HUNSPELL_CZECH_LUCENE_FILTER CS_CZ LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 5038 50968 2 47731 3237 5 54804 11240 221382 13601 77431 1298530614 13601 1298544215 0.001047 77431 298813 25.912862 0.942119217135 0.740871381098 0.999989525963 0.999929913009 0.870430453530 0.893573737936 0.829462940899 0.773935751817 0.708617411512 0.835457458856 0.835425115185 0.000070086991
92 HUNSPELL_DUTCH_LUCENE_FILTER NL_NL ALL_WORDS PRIMARY_OUTPUT 4992 26477 85 26477 0 1 26477 15909 18482 1333 46084 350436627 1333 350437960 0.000380 46084 64566 71.375027 0.932727731517 0.286249728960 0.999996196188 0.999864717095 0.643122962574 0.642512480358 0.438060700869 0.332315636923 0.280459491039 0.516713712165 0.516673857221 0.000135282905 0.438012080403 0.996931617211 0.889026094124 0.939891935467 0.939891935467
93 HUNSPELL_DUTCH_LUCENE_FILTER NL_NL ALL_WORDS ANY_CANDIDATE 4992 26477 85 25223 1254 3 27763 16027 21374 1164 43192 350436796 1164 350437960 0.000332 43192 64566 66.895889 0.948353891206 0.331041105226 0.999996678442 0.999873450270 0.665518891834 0.690740573172 0.490769654666 0.380588457347 0.325178761600 0.560307166017 0.560267948638 0.000126549730
94 HUNSPELL_DUTCH_LUCENE_FILTER NL_NL ALL_WORDS ALL_CANDIDATES 4992 26477 85 25223 1254 3 27763 16027 21374 1738 43192 350436222 1738 350437960 0.000496 43192 64566 66.895889 0.924800969193 0.331041105226 0.999995040492 0.999871812621 0.665518072859 0.680639942935 0.487556741714 0.379812066416 0.322363658301 0.553305643343 0.553264631551 0.000128187379
95 HUNSPELL_DUTCH_LUCENE_FILTER NL_NL LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 4796 25678 84 25678 0 1 25678 15258 18333 1310 44814 329602546 1310 329603856 0.000397 44814 63147 70.967742 0.933309575930 0.290322580645 0.999996025532 0.999860089122 0.645159303089 0.646808120294 0.442879574828 0.336717714000 0.284422172921 0.520538994337 0.520497519470 0.000139910878 0.442828931093 0.996884174988 0.889060613994 0.939890140871 0.939890140871
96 HUNSPELL_DUTCH_LUCENE_FILTER NL_NL LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 4796 25678 84 24492 1186 3 26896 15323 21212 1141 41935 329602715 1141 329603856 0.000346 41935 63147 66.408539 0.948955397486 0.335914611937 0.999996538269 0.999869334815 0.667955575103 0.695206444720 0.496187134503 0.385755489360 0.329952712792 0.564595416287 0.564554662442 0.000130665185
97 HUNSPELL_DUTCH_LUCENE_FILTER NL_NL LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 4796 25678 84 24492 1186 3 26896 15323 21212 1712 41935 329602144 1712 329603856 0.000519 41935 63147 66.408539 0.925318443553 0.335914611937 0.999994805886 0.999867602764 0.667954708912 0.684951854459 0.492895400309 0.384956009176 0.327047903915 0.557519493726 0.557476858392 0.000132397236
98 HUNSPELL_ENGLISH_LUCENE_FILTER US_UK ALL_WORDS PRIMARY_OUTPUT 396939 607439 250964 607439 0 1 607439 557518 46002 1981986 267868 184488469785 1981986 184490451771 0.001074 267868 313870 85.343614 0.022683566175 0.146563864020 0.999989256972 0.999987805059 0.573276560496 0.027298226808 0.039286754363 0.070050933952 0.020036970960 0.057659267324 0.057655308782 0.000012194941 0.039283923590 0.993096189204 0.963677172906 0.978165531652 0.978165531652
99 HUNSPELL_ENGLISH_LUCENE_FILTER US_UK ALL_WORDS ANY_CANDIDATE 396939 607439 250964 600602 6837 4 614296 557638 51229 1978852 262641 184488472919 1978852 184490451771 0.001073 262641 313870 83.678274 0.025234953679 0.163217255552 0.999989273960 0.999987850378 0.581603264756 0.030369825498 0.043711664621 0.077960810954 0.022344183028 0.064177721084 0.064173802046 0.000012149622
100 HUNSPELL_ENGLISH_LUCENE_FILTER US_UK ALL_WORDS ALL_CANDIDATES 396939 607439 250964 600602 6837 4 614296 557638 51229 2008917 262641 184488442854 2008917 184490451771 0.001089 262641 313870 83.678274 0.024866684206 0.163217255552 0.999989110997 0.999987687416 0.581603183275 0.029942881217 0.043158091605 0.077253888104 0.022054971033 0.063707707153 0.063703758400 0.000012312584
101 HUNSPELL_ENGLISH_LUCENE_FILTER US_UK LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 374384 583910 228735 583910 0 1 583910 535362 45926 1978041 265965 170472862163 1978041 170474840204 0.001160 265965 311891 85.274984 0.022691081426 0.147250161114 0.999988396874 0.999986836756 0.573619278994 0.027311677226 0.039322595808 0.070190378755 0.020055617372 0.057803679777 0.057799415161 0.000013163244 0.039319549964 0.993066314983 0.962316521867 0.977449637185 0.977449637185
102 HUNSPELL_ENGLISH_LUCENE_FILTER US_UK LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 374384 583910 228735 577124 6786 4 590716 535485 51150 1974950 260741 170472865254 1974950 170474840204 0.001158 260741 311891 83.600040 0.025245545630 0.163999602425 0.999988415006 0.999986885532 0.581994008716 0.030387494919 0.043755514884 0.078123472659 0.022367099418 0.064344847861 0.064340626006 0.000013114468
103 HUNSPELL_ENGLISH_LUCENE_FILTER US_UK LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 374384 583910 228735 577124 6786 4 590716 535485 51150 2004598 260741 170472835606 2004598 170474840204 0.001176 260741 311891 83.600040 0.024881454342 0.163999602425 0.999988241092 0.999986711618 0.581993921758 0.029965261387 0.043207600483 0.077422296168 0.022080830084 0.063879172034 0.063874918547 0.000013288382
104 HUNSPELL_FRENCH_LUCENE_FILTER FR_FR ALL_WORDS PRIMARY_OUTPUT 59240 425210 2301 425210 0 1 425210 154336 3422734 776728 2031881 90395328102 776728 90396104830 0.000859 2031881 5454615 37.250677 0.815041069547 0.627493232795 0.999991407506 0.999968931852 0.813742320150 0.769068574566 0.709075347131 0.657764674673 0.549277098051 0.715145268872 0.715130511338 0.000031068148 0.709060074832 0.978337247291 0.913705954219 0.944917713687 0.944917713687
105 HUNSPELL_FRENCH_LUCENE_FILTER FR_FR ALL_WORDS ANY_CANDIDATE 59240 425210 2301 411699 13511 4 439015 154718 3610612 745831 1844003 90395358999 745831 90396104830 0.000825 1844003 5454615 33.806291 0.828798173189 0.661937093635 0.999991749302 0.999971351888 0.830964421468 0.789018996925 0.736029080656 0.689708764155 0.582314885091 0.740683639600 0.740669908387 0.000028648112
106 HUNSPELL_FRENCH_LUCENE_FILTER FR_FR ALL_WORDS ALL_CANDIDATES 59240 425210 2301 411699 13511 4 439015 154718 3610612 1043199 1844003 90395061631 1043199 90396104830 0.001154 1844003 5454615 33.806291 0.775839843947 0.661937093635 0.999988459691 0.999968062476 0.830962776663 0.750027659074 0.714376699201 0.681961135862 0.555665643861 0.716629033342 0.716613365354 0.000031937524
107 HUNSPELL_FRENCH_LUCENE_FILTER FR_FR LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 57698 421231 2133 421231 0 1 421231 153822 3412548 763305 2028011 88711363201 763305 88712126506 0.000860 2028011 5440559 37.275784 0.817209801207 0.627242163903 0.999991395708 0.999968537054 0.813616779806 0.770536594362 0.709734150326 0.657825640123 0.550068151075 0.715952822518 0.715937898033 0.000031462946 0.709718690125 0.979328164393 0.913161860024 0.945088340370 0.945088340370
108 HUNSPELL_FRENCH_LUCENE_FILTER FR_FR LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 57698 421231 2133 407794 13437 4 434961 154205 3600083 733584 1840476 88711392922 733584 88712126506 0.000827 1840476 5440559 33.828803 0.830724418835 0.661711967465 0.999991730736 0.999970985904 0.830851849101 0.790350629656 0.736648201095 0.689779349655 0.583090317150 0.741417756470 0.741403865778 0.000029014096
109 HUNSPELL_FRENCH_LUCENE_FILTER FR_FR LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 57698 421231 2133 407794 13437 4 434961 154205 3600083 1027635 1840476 88711098871 1027635 88712126506 0.001158 1840476 5440559 33.828803 0.777939148410 0.661711967465 0.999988416071 0.999967671442 0.830850191768 0.751538185756 0.715133880405 0.682093458746 0.556582409247 0.717475884237 0.717460037184 0.000032328558
110 HUNSPELL_GERMAN_LUCENE_FILTER DE_DE ALL_WORDS PRIMARY_OUTPUT 54092 296974 1474 296974 0 1 296974 182774 391862 203883 992010 44095042096 203883 44095245979 0.000462 992010 1383872 71.683653 0.657768004767 0.283163471766 0.999995376304 0.999972880172 0.641579424035 0.520145203475 0.395896782054 0.319562150060 0.246802560848 0.431573715426 0.431562811676 0.000027119828 0.395885371175 0.980462638581 0.886872825924 0.931322397630 0.931322397630
111 HUNSPELL_GERMAN_LUCENE_FILTER DE_DE ALL_WORDS ANY_CANDIDATE 54092 296974 1474 289083 7891 3 305052 183111 408175 158403 975697 44095087576 158403 44095245979 0.000359 975697 1383872 70.504859 0.720421548313 0.294951411691 0.999996407708 0.999974281481 0.647473909700 0.559115650060 0.418544438463 0.334456395588 0.264657729653 0.460965674088 0.460955788036 0.000025718519
112 HUNSPELL_GERMAN_LUCENE_FILTER DE_DE ALL_WORDS ALL_CANDIDATES 54092 296974 1474 289083 7891 3 305052 183111 408175 242551 975697 44095003428 242551 44095245979 0.000550 975697 1383872 70.504859 0.627260936247 0.294951411691 0.999994499384 0.999972373218 0.647472955538 0.511911128190 0.401234052132 0.329906951166 0.250964847398 0.430129630047 0.430118032816 0.000027626782
113 HUNSPELL_GERMAN_LUCENE_FILTER DE_DE LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 16007 150098 228 150098 0 1 150098 86983 278093 84679 595318 11263671663 84679 11263756342 0.000752 595318 873411 68.160122 0.766577905682 0.318398783620 0.999992482170 0.999939634323 0.659195632895 0.598178360154 0.449922058465 0.360558871242 0.290257700216 0.494041974653 0.494019111671 0.000060365677 0.449897024669 0.988041339480 0.865580709092 0.922765807515 0.922765807515
114 HUNSPELL_GERMAN_LUCENE_FILTER DE_DE LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 16007 150098 228 145109 4989 3 155207 87393 288864 60996 584547 11263695346 60996 11263756342 0.000542 584547 873411 66.926911 0.825655976676 0.330730893016 0.999994584755 0.999942692923 0.665362738886 0.635466205220 0.472281285177 0.375782098835 0.309141519702 0.522560942370 0.522540242219 0.000057307077
115 HUNSPELL_GERMAN_LUCENE_FILTER DE_DE LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 16007 150098 228 145109 4989 3 155207 87393 288864 96545 584547 11263659797 96545 11263756342 0.000857 584547 873411 66.926911 0.749499881944 0.330730893016 0.999991428703 0.999939537116 0.665361160860 0.598050472724 0.458944090497 0.372338300095 0.297811447117 0.497878263505 0.497854575726 0.000060462884
116 HUNSPELL_POLISH_LUCENE_FILTER PL_PL ALL_WORDS PRIMARY_OUTPUT 9990 122341 1 122341 0 1 122341 18419 971262 52652 149705 7482425351 52652 7482478003 0.000704 149705 1120967 13.354987 0.948577712581 0.866450127435 0.999992963294 0.999972959935 0.933221545364 0.930929837176 0.905655838249 0.881717903868 0.827578626454 0.906584403102 0.906571161039 0.000027040065 0.905642343969 0.994545991966 0.970520439141 0.982386343372 0.982386343372
117 HUNSPELL_POLISH_LUCENE_FILTER PL_PL ALL_WORDS ANY_CANDIDATE 9990 122341 1 110894 11447 6 135231 19068 1040224 42213 80743 7482435790 42213 7482478003 0.000564 80743 1120967 7.202977 0.961001887408 0.927970225707 0.999994358420 0.999983569937 0.963982292063 0.954208759768 0.944197251162 0.934393641743 0.894293230626 0.944341642819 0.944333470354 0.000016430063
118 HUNSPELL_POLISH_LUCENE_FILTER PL_PL ALL_WORDS ALL_CANDIDATES 9990 122341 1 110894 11447 6 135231 19068 1040224 82745 80743 7482395258 82745 7482478003 0.001106 80743 1120967 7.202977 0.926315864463 0.927970225707 0.999988941498 0.999978153827 0.963979583602 0.926646264647 0.927142307089 0.927638880888 0.864180136112 0.927142676087 0.927131751477 0.000021846173
119 HUNSPELL_POLISH_LUCENE_FILTER PL_PL LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 9846 120925 1 120925 0 1 120925 18149 965984 51950 148667 7310200749 51950 7310252699 0.000711 148667 1114651 13.337538 0.948965257080 0.866624620621 0.999992893543 0.999972560946 0.933308757082 0.931268723294 0.905927782480 0.881929423296 0.828032892137 0.906860880124 0.906847444801 0.000027439054 0.905914089179 0.994583905165 0.970514203019 0.982401644006 0.982401644006
120 HUNSPELL_POLISH_LUCENE_FILTER PL_PL LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 9846 120925 1 109660 11265 6 133595 18789 1034283 41671 80368 7310211028 41671 7310252699 0.000570 80368 1114651 7.210149 0.961270649117 0.927898508143 0.999994299650 0.999983308321 0.963946403896 0.954405554191 0.944289819479 0.934386268967 0.894459328803 0.944437187555 0.944428885928 0.000016691679
121 HUNSPELL_POLISH_LUCENE_FILTER PL_PL LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 9846 120925 1 109660 11265 6 133595 18789 1034283 81865 80368 7310170834 81865 7310252699 0.001120 80368 1114651 7.210149 0.926653992123 0.927898508143 0.999988801345 0.999977810854 0.963943654744 0.926902628188 0.927275832560 0.927649337585 0.864412176686 0.927276041347 0.927264945147 0.000022189146
122 HUNSPELL_SPANISH_LUCENE_FILTER ES_ES ALL_WORDS PRIMARY_OUTPUT 65059 871332 3589 871332 0 1 871332 495840 9662476 536192 32310860 379566781918 536192 379567318110 0.000141 32310860 41973336 76.979490 0.947425291224 0.230205099733 0.999998587360 0.999913471422 0.615101843546 0.583708381625 0.370408466579 0.271277635967 0.227301418167 0.467014061518 0.466991649518 0.000086528578 0.370381248953 0.993314263125 0.790558492734 0.880413722434 0.880413722434
123 HUNSPELL_SPANISH_LUCENE_FILTER ES_ES ALL_WORDS ANY_CANDIDATE 65059 871332 3589 853455 17877 5 890999 496361 10079118 416345 31894218 379566901765 416345 379567318110 0.000110 31894218 41973336 75.986855 0.960330954432 0.240131449166 0.999998903106 0.999914884689 0.620065176136 0.600267728541 0.384194728757 0.282504215637 0.237772914592 0.480214185303 0.480192080762 0.000085115311
124 HUNSPELL_SPANISH_LUCENE_FILTER ES_ES ALL_WORDS ALL_CANDIDATES 65059 871332 3589 853455 17877 5 890999 496361 10079118 888077 31894218 379566430033 888077 379567318110 0.000234 31894218 41973336 75.986855 0.919024235459 0.240131449166 0.999997660291 0.999913642011 0.620064554728 0.587073016700 0.380771322449 0.281759130783 0.235155989841 0.469772946730 0.469749183448 0.000086357989
125 HUNSPELL_SPANISH_LUCENE_FILTER ES_ES LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 64918 869371 3525 869371 0 1 869371 495045 9628515 531181 32234855 377860138584 531181 377860669765 0.000141 32234855 41863370 77.000144 0.947716841134 0.229998564377 0.999998594241 0.999913295008 0.614998579309 0.583531128169 0.370163304101 0.271052948234 0.227116805648 0.466876335765 0.466853897372 0.000086704992 0.370136051001 0.993362468962 0.790499503024 0.880396073652 0.880396073652
126 HUNSPELL_SPANISH_LUCENE_FILTER ES_ES LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 64918 869371 3525 851564 17807 5 888962 495572 10041262 412198 31822108 377860257567 412198 377860669765 0.000109 31822108 41863370 76.014205 0.960568271175 0.239857947413 0.999998909127 0.999914702064 0.619928428270 0.599999808789 0.383863548307 0.282205460900 0.237519268813 0.479999931119 0.479977799265 0.000085297936
127 HUNSPELL_SPANISH_LUCENE_FILTER ES_ES LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 64918 869371 3525 851564 17807 5 888962 495572 10041262 878949 31822108 377859790816 878949 377860669765 0.000233 31822108 41863370 76.014205 0.919511720057 0.239857947413 0.999997673881 0.999913466955 0.619927810647 0.586904802235 0.380469146267 0.281467012980 0.234925531298 0.469629847641 0.469606065497 0.000086533045
128 HUNSPELL_UKRAINIAN_LUCENE_FILTER UK_UA ALL_WORDS PRIMARY_OUTPUT 1493 14245 4 14245 0 1 14245 3137 50416 794 14924 101386756 794 101387550 0.000783 14924 65340 22.840526 0.984495215778 0.771594735231 0.999992168664 0.999845070949 0.885793451947 0.933007624547 0.865139425139 0.806475349522 0.762331024889 0.871568313648 0.871498740996 0.000154929051 0.865063055969 0.998114340300 0.949803904722 0.973360047526 0.973360047526
129 HUNSPELL_UKRAINIAN_LUCENE_FILTER UK_UA ALL_WORDS ANY_CANDIDATE 1493 14245 4 12923 1322 6 15740 3311 55875 326 9465 101387224 326 101387550 0.000322 9465 65340 14.485767 0.994199391470 0.855142332415 0.999996784615 0.999903492153 0.927569558515 0.962883947281 0.919442821764 0.879752236578 0.850896963421 0.922053136488 0.922008115880 0.000096507847
130 HUNSPELL_UKRAINIAN_LUCENE_FILTER UK_UA ALL_WORDS ALL_CANDIDATES 1493 14245 4 12923 1322 6 15740 3311 55875 1271 9465 101386279 1271 101387550 0.001254 9465 65340 14.485767 0.977758723270 0.855142332415 0.999987463944 0.999894177485 0.927564898180 0.950500809733 0.912349166435 0.877142031861 0.838825419225 0.914397547654 0.914347195556 0.000105822515
131 HUNSPELL_UKRAINIAN_LUCENE_FILTER UK_UA LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 1491 14236 4 14236 0 1 14236 3134 50404 794 14920 101258612 794 101259406 0.000784 14920 65324 22.839998 0.984491581702 0.771600024493 0.999992158753 0.999844914465 0.885796091623 0.933006560145 0.865141346698 0.806479484406 0.762334008893 0.871569692311 0.871500048785 0.000155085535 0.865064900251 0.998112856419 0.949788155938 0.973351072054 0.973351072054
132 HUNSPELL_UKRAINIAN_LUCENE_FILTER UK_UA LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 1491 14236 4 12915 1321 6 15730 3308 55859 326 9465 101259080 326 101259406 0.000322 9465 65324 14.489315 0.994197739610 0.855106851999 0.999996780546 0.999903370085 0.927551816273 0.962873710629 0.919421606630 0.879721936116 0.850860624524 0.922033242016 0.921988165267 0.000096629915
133 HUNSPELL_UKRAINIAN_LUCENE_FILTER UK_UA LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 1491 14236 4 12915 1321 6 15730 3308 55859 1271 9465 101258135 1271 101259406 0.001255 9465 65324 14.489315 0.977752494311 0.855106851999 0.999987448080 0.999894043636 0.927547150039 0.950487333415 0.912326261290 0.877111165546 0.838786695698 0.914375665383 0.914325250217 0.000105956364
134 ITALIAN_LUCENE_ITALIAN_LIGHT_STEM_FILTER IT_IT ALL_WORDS PRIMARY_OUTPUT 10009 327551 0 327551 0 1 327551 244870 109684 10589 6034130 53638510622 10589 53638521211 0.000020 6034130 6143814 98.214725 0.911958627456 0.017852754006 0.999999802586 0.999887319289 0.508926278296 0.082781551919 0.035019947839 0.022207258650 0.017822037328 0.127596916262 0.127588341500 0.000112680711 0.035015703871 0.997481424185 0.737537113266 0.848036553212 0.848036553212
135 ITALIAN_LUCENE_ITALIAN_LIGHT_STEM_FILTER IT_IT LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 10007 327469 0 327469 0 1 327469 244808 109658 10588 6032516 53611656484 10588 53611667072 0.000020 6032516 6142174 98.214671 0.911947174958 0.017853287777 0.999999802506 0.999887292971 0.508926545141 0.082783771729 0.035020966336 0.022207918023 0.017822564890 0.127598022525 0.127589445488 0.000112707029 0.035016721203 0.997481197125 0.737534145120 0.848034509070 0.848034509070
136 ITALIAN_RADIXOR IT_IT ALL_WORDS PRIMARY_OUTPUT 10009 327551 0 327551 0 1 327551 10010 6100906 124172 42908 53638397039 124172 53638521211 0.000231 42908 6143814 0.698394 0.980052940702 0.993016064614 0.999997685022 0.999996885431 0.996506874818 0.982618418699 0.986491918597 0.990396078172 0.973343909830 0.986513210398 0.986511657877 0.000003114569 0.986490361200 0.995780270704 0.997112550353 0.996445965204 0.996445965204
137 ITALIAN_RADIXOR IT_IT ALL_WORDS ANY_CANDIDATE 10009 327551 0 321297 6254 4 334175 10012 6143734 0 80 53638521211 0 53638521211 0.000000 80 6143814 0.001302 1.000000000000 0.999986978772 1.000000000000 0.999999998509 0.999993489386 0.999997395727 0.999993489344 0.999989582991 0.999986978772 0.999993489365 0.999993488619 0.000000001491
138 ITALIAN_RADIXOR IT_IT ALL_WORDS ALL_CANDIDATES 10009 327551 0 321297 6254 4 334175 10012 6143734 170950 80 53638350261 170950 53638521211 0.000319 80 6143814 0.001302 0.972928178195 0.999986978772 0.999996812925 0.999996811799 0.999991895849 0.978222150749 0.986272020913 0.994455476443 0.972915852437 0.986364795335 0.986363222748 0.000003188201
139 ITALIAN_RADIXOR IT_IT LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 10007 327469 0 327469 0 1 327469 10007 6099346 124171 42828 53611542901 124171 53611667072 0.000232 42828 6142174 0.697278 0.980048098206 0.993027224563 0.999997683881 0.999996885382 0.996512454222 0.982616709845 0.986494972258 0.990403969991 0.973349855458 0.986516316590 0.986514764059 0.000003114618 0.986493414837 0.995779575755 0.997114909855 0.996446795437 0.996446795437
140 ITALIAN_RADIXOR IT_IT LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 10007 327469 0 321217 6252 4 334089 10007 6142174 0 0 53611667072 0 53611667072 0.000000 0 6142174 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
141 ITALIAN_RADIXOR IT_IT LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 10007 327469 0 321217 6252 4 334089 10007 6142174 170949 0 53611496123 170949 53611667072 0.000319 0 6142174 0.000000 0.972921642743 1.000000000000 0.999996811347 0.999996811712 0.999998405674 0.978219357390 0.986274996092 0.994464412864 0.972921642743 0.986367904356 0.986366331762 0.000003188288
142 NL_NL_RADIXOR NL_NL ALL_WORDS PRIMARY_OUTPUT 4992 26477 85 26477 0 1 26477 5015 63102 1214 1464 350436746 1214 350437960 0.000346 1464 64566 2.267447 0.981124448038 0.977325527367 0.999996535763 0.999992359542 0.988661031565 0.980362303079 0.979221303208 0.978082956166 0.959288537549 0.979223145453 0.979219325206 0.000007640458 0.979217482290 0.997464133435 0.997003025118 0.997233525974 0.997233525974
143 NL_NL_RADIXOR NL_NL ALL_WORDS ANY_CANDIDATE 4992 26477 85 25905 572 3 27061 5016 64566 0 0 350437960 0 350437960 0.000000 0 64566 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
144 NL_NL_RADIXOR NL_NL ALL_WORDS ALL_CANDIDATES 4992 26477 85 25905 572 3 27061 5016 64566 2651 0 350435309 2651 350437960 0.000756 0 64566 0.000000 0.960560572474 1.000000000000 0.999992435180 0.999992436574 0.999996217590 0.968197604323 0.979883596519 0.991855131329 0.960560572474 0.980081921308 0.980078214229 0.000007563426
145 NL_NL_RADIXOR NL_NL LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 4796 25678 84 25678 0 1 25678 4797 61763 1214 1384 329602642 1214 329603856 0.000368 1384 63147 2.191711 0.980723121139 0.978082885964 0.999996316791 0.999992119320 0.989039601378 0.980193934392 0.979401224192 0.978609795129 0.959633939808 0.979402113872 0.979398173122 0.000007880680 0.979397283106 0.997373193672 0.997139403762 0.997256285015 0.997256285015
146 NL_NL_RADIXOR NL_NL LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 4796 25678 84 25129 549 3 26239 4797 63147 0 0 329603856 0 329603856 0.000000 0 63147 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
147 NL_NL_RADIXOR NL_NL LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 4796 25678 84 25129 549 3 26239 4797 63147 2651 0 329601205 2651 329603856 0.000804 0 63147 0.000000 0.959710021581 1.000000000000 0.999991957012 0.999991958552 0.999995978506 0.967506182222 0.979440846873 0.991673628866 0.959710021581 0.979647906945 0.979643967288 0.000008041448
148 NN_NO_RADIXOR NN_NO ALL_WORDS PRIMARY_OUTPUT 4688 18250 23 18250 0 1 18250 4680 26716 6230 3936 166485243 6230 166491473 0.003742 3936 30652 12.840924 0.810902689249 0.871590760799 0.999962580666 0.999938951055 0.935776670732 0.822354650447 0.840152206044 0.858737158800 0.724364188493 0.840699287413 0.840668985911 0.000061048945 0.840121715471 0.983845117159 0.986801676045 0.985321178741 0.985321178741
149 NN_NO_RADIXOR NN_NO ALL_WORDS ANY_CANDIDATE 4688 18250 23 15846 2404 5 21513 4693 30652 0 0 166491473 0 166491473 0.000000 0 30652 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
150 NN_NO_RADIXOR NN_NO ALL_WORDS ALL_CANDIDATES 4688 18250 23 15846 2404 5 21513 4693 30652 13214 0 166478259 13214 166491473 0.007937 0 30652 0.000000 0.698764418912 1.000000000000 0.999920632572 0.999920647181 0.999960316286 0.743561877778 0.822673716418 0.920624241623 0.698764418912 0.835921299473 0.835888126353 0.000079352819
151 NN_NO_RADIXOR NN_NO LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 4681 18219 23 18219 0 1 18219 4668 26671 6230 3924 165920046 6230 165926276 0.003755 3924 30595 12.825625 0.810644053372 0.871743748979 0.999962453204 0.999938815429 0.935853101091 0.822169063928 0.840084414766 0.858797921188 0.724263408011 0.840638974932 0.840608609146 0.000061184571 0.840053856992 0.983814668550 0.986841730061 0.985325874420 0.985325874420
152 NN_NO_RADIXOR NN_NO LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 4681 18219 23 15820 2399 5 21477 4681 30595 0 0 165926276 0 165926276 0.000000 0 30595 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
153 NN_NO_RADIXOR NN_NO LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 4681 18219 23 15820 2399 5 21477 4681 30595 13214 0 165913062 13214 165926276 0.007964 0 30595 0.000000 0.698372480541 1.000000000000 0.999920362222 0.999920376903 0.999960181111 0.743206805583 0.822402021397 0.920488118949 0.698372480541 0.835686831618 0.835653554835 0.000079623097
154 NORWEGIAN_BOKMAL_LUCENE_NORWEGIAN_LIGHT_STEM_FILTER NB_NO ALL_WORDS PRIMARY_OUTPUT 17929 75310 252 75310 0 1 75310 25999 99529 25171 42651 2835593044 25171 2835618215 0.000888 42651 142180 29.997890 0.798147554130 0.700021100014 0.999991123276 0.999976083311 0.850006111645 0.776381478361 0.745870803357 0.717667503101 0.594732030284 0.747475838282 0.747464055956 0.000023916689 0.745858895755 0.987774378220 0.965622291196 0.976572729149 0.976572729149
155 NORWEGIAN_BOKMAL_LUCENE_NORWEGIAN_LIGHT_STEM_FILTER NB_NO LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 17914 75251 252 75251 0 1 75251 25985 99450 25118 42641 2831151666 25118 2831176784 0.000887 42641 142091 30.009642 0.798359129150 0.699903582915 0.999991128071 0.999976068044 0.849947355493 0.776512696705 0.745896444523 0.717602881668 0.594764635875 0.747512150366 0.747500361706 0.000023931956 0.745884529658 0.987789184092 0.965602788874 0.976569991255 0.976569991255
156 NORWEGIAN_BOKMAL_LUCENE_NORWEGIAN_MINIMAL_STEM_FILTER NB_NO ALL_WORDS PRIMARY_OUTPUT 17929 75310 252 75310 0 1 75310 27457 94526 14772 47654 2835603443 14772 2835618215 0.000521 47654 142180 33.516669 0.864846566268 0.664833309889 0.999994790554 0.999977986151 0.832414050221 0.815762584315 0.751763573752 0.697075888841 0.602260563739 0.758273568838 0.758263230800 0.000022013849 0.751752754540 0.992088894987 0.962515968891 0.977078714611 0.977078714611
157 NORWEGIAN_BOKMAL_LUCENE_NORWEGIAN_MINIMAL_STEM_FILTER NB_NO LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 17914 75251 252 75251 0 1 75251 27443 94447 14719 47644 2831162065 14719 2831176784 0.000520 47644 142091 33.530625 0.865168642251 0.664693752595 0.999994801102 0.999977973869 0.832344276848 0.815949754214 0.751795969864 0.696994967013 0.602302149098 0.758335144541 0.758324803793 0.000022026131 0.751785145440 0.992107474043 0.962494026490 0.977076419047 0.977076419047
158 NORWEGIAN_BOKMAL_RADIXOR NB_NO ALL_WORDS PRIMARY_OUTPUT 17929 75310 252 75310 0 1 75310 17886 135010 11482 7170 2835606733 11482 2835618215 0.000405 7170 142180 5.042903 0.921620293258 0.949570966381 0.999995950795 0.999993422575 0.974783458588 0.927078011613 0.935386875069 0.943846020481 0.878616704195 0.935491246621 0.935487968734 0.000006577425 0.935383586923 0.993354053349 0.994615320153 0.993984286646 0.993984286646
159 NORWEGIAN_BOKMAL_RADIXOR NB_NO ALL_WORDS ANY_CANDIDATE 17929 75310 252 71073 4237 9 79825 17962 142180 0 0 2835618215 0 2835618215 0.000000 0 142180 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
160 NORWEGIAN_BOKMAL_RADIXOR NB_NO ALL_WORDS ALL_CANDIDATES 17929 75310 252 71073 4237 9 79825 17962 142180 20161 0 2835598054 20161 2835618215 0.000711 0 142180 0.000000 0.875810793330 1.000000000000 0.999992890087 0.999992890443 0.999996445043 0.898118108406 0.933794385280 0.972422273928 0.875810793330 0.935847633608 0.935844306704 0.000007109557
161 NORWEGIAN_BOKMAL_RADIXOR NB_NO LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 17914 75251 252 75251 0 1 75251 17838 134987 11482 7104 2831165302 11482 2831176784 0.000406 7104 142091 4.999613 0.921607985307 0.950003870759 0.999995944443 0.999993435568 0.974999907601 0.927150543912 0.935590518436 0.944185565020 0.878976122105 0.935698217036 0.935694946007 0.000006564432 0.935587236809 0.993348241255 0.994693619298 0.994020475044 0.994020475044
162 NORWEGIAN_BOKMAL_RADIXOR NB_NO LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 17914 75251 252 71047 4204 9 79733 17914 142091 0 0 2831176784 0 2831176784 0.000000 0 142091 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
163 NORWEGIAN_BOKMAL_RADIXOR NB_NO LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 17914 75251 252 71047 4204 9 79733 17914 142091 20161 0 2831156623 20161 2831176784 0.000712 0 142091 0.000000 0.875742671893 1.000000000000 0.999992878933 0.999992879290 0.999996439466 0.898060798964 0.933755663840 0.972405477022 0.875742671893 0.935811237319 0.935807905326 0.000007120710
164 PERSIAN_LUCENE_PERSIAN_STEM_FILTER FA_IR ALL_WORDS PRIMARY_OUTPUT 69 3701 0 3701 0 1 3701 3190 430 179 98018 6748223 179 6748402 0.002652 98018 98448 99.563221 0.706075533662 0.004367788071 0.999973475202 0.985658076342 0.502170631636 0.021311605408 0.008681870034 0.005451304637 0.004359860890 0.055533668104 0.054800599292 0.014341923658 0.008506575635 0.985685778881 0.520346904570 0.681125382676 0.681125382676
165 PERSIAN_LUCENE_PERSIAN_STEM_FILTER FA_IR LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 69 3701 0 3701 0 1 3701 3190 430 179 98018 6748223 179 6748402 0.002652 98018 98448 99.563221 0.706075533662 0.004367788071 0.999973475202 0.985658076342 0.502170631636 0.021311605408 0.008681870034 0.005451304637 0.004359860890 0.055533668104 0.054800599292 0.014341923658 0.008506575635 0.985685778881 0.520346904570 0.681125382676 0.681125382676
166 PERSIAN_RADIXOR FA_IR ALL_WORDS PRIMARY_OUTPUT 69 3701 0 3701 0 1 3701 69 93636 8621 4812 6739781 8621 6748402 0.127749 4812 98448 4.887860 0.915692813206 0.951121404193 0.998722512381 0.998038075904 0.974921958287 0.922565796215 0.933070924989 0.943818050233 0.874538848780 0.933239001706 0.932248664283 0.001961924096 0.932075735269 0.980342969277 0.984586447427 0.982460126226 0.982460126226
167 PERSIAN_RADIXOR FA_IR ALL_WORDS ANY_CANDIDATE 69 3701 0 3387 314 2 4015 69 98448 0 0 6748402 0 6748402 0.000000 0 98448 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
168 PERSIAN_RADIXOR FA_IR ALL_WORDS ALL_CANDIDATES 69 3701 0 3387 314 2 4015 69 98448 13433 0 6734969 13433 6748402 0.199055 0 98448 0.000000 0.879934930864 1.000000000000 0.998009454683 0.998038075904 0.999004727341 0.901584696651 0.936133391021 0.973435401930 0.879934930864 0.938048469358 0.937114390300 0.001961924096
169 PERSIAN_RADIXOR FA_IR LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 69 3701 0 3701 0 1 3701 69 93636 8621 4812 6739781 8621 6748402 0.127749 4812 98448 4.887860 0.915692813206 0.951121404193 0.998722512381 0.998038075904 0.974921958287 0.922565796215 0.933070924989 0.943818050233 0.874538848780 0.933239001706 0.932248664283 0.001961924096 0.932075735269 0.980342969277 0.984586447427 0.982460126226 0.982460126226
170 PERSIAN_RADIXOR FA_IR LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 69 3701 0 3387 314 2 4015 69 98448 0 0 6748402 0 6748402 0.000000 0 98448 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
171 PERSIAN_RADIXOR FA_IR LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 69 3701 0 3387 314 2 4015 69 98448 13433 0 6734969 13433 6748402 0.199055 0 98448 0.000000 0.879934930864 1.000000000000 0.998009454683 0.998038075904 0.999004727341 0.901584696651 0.936133391021 0.973435401930 0.879934930864 0.938048469358 0.937114390300 0.001961924096
172 POLISH_LUCENE_MORFOLOGIK_FILTER PL_PL ALL_WORDS PRIMARY_OUTPUT 9990 122341 1 122341 0 1 122341 15519 1004747 99228 116220 7482378775 99228 7482478003 0.001326 116220 1120967 10.367834 0.910117529835 0.896321657997 0.999986738618 0.999971210643 0.948154198307 0.907324485129 0.903166914014 0.899047271013 0.823431500703 0.903193253581 0.903178865015 0.000028789357 0.903152518035 0.990022261217 0.977053921984 0.983495343428 0.983495343428
173 POLISH_LUCENE_MORFOLOGIK_FILTER PL_PL ALL_WORDS ANY_CANDIDATE 9990 122341 1 109468 12873 5 136636 16295 1093112 85532 27855 7482392471 85532 7482478003 0.001143 27855 1120967 2.484908 0.927431862377 0.975150918805 0.999988569028 0.999984848600 0.987569743916 0.936598359399 0.950692965028 0.965218263555 0.906019814355 0.950992130738 0.950984648194 0.000015151400
174 POLISH_LUCENE_MORFOLOGIK_FILTER PL_PL ALL_WORDS ALL_CANDIDATES 9990 122341 1 109468 12873 5 136636 16295 1093112 143096 27855 7482334907 143096 7482478003 0.001912 27855 1120967 2.484908 0.884246016852 0.975150918805 0.999980875854 0.999977156579 0.987565897330 0.901045352805 0.927476322293 0.955504786999 0.864760696263 0.928586730350 0.928575670129 0.000022843421
175 POLISH_LUCENE_MORFOLOGIK_FILTER PL_PL LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 9846 120925 1 120925 0 1 120925 15277 999138 99224 115513 7310153475 99224 7310252699 0.001357 115513 1114651 10.363154 0.909661841906 0.896368459724 0.999986426735 0.999970629707 0.948177443229 0.906971715650 0.902966227492 0.898995962905 0.823097930182 0.902990688822 0.902976009256 0.000029370293 0.902951540918 0.989889334726 0.977011745487 0.983408384376 0.983408384376
176 POLISH_LUCENE_MORFOLOGIK_FILTER PL_PL LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 9846 120925 1 108162 12763 5 135105 16044 1087157 85532 27494 7310167167 85532 7310252699 0.001170 27494 1114651 2.466602 0.927063356099 0.975333983462 0.999988299720 0.999984541059 0.987661141591 0.936331423447 0.950586270515 0.965281863330 0.905826028197 0.950892420848 0.950884788442 0.000015458941
177 POLISH_LUCENE_MORFOLOGIK_FILTER PL_PL LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 9846 120925 1 108162 12763 5 135105 16044 1087157 143085 27494 7310109614 143085 7310252699 0.001957 27494 1114651 2.466602 0.883693614752 0.975333983462 0.999980426805 0.999976669344 0.987657205134 0.900617649988 0.927255102898 0.955516285728 0.864376148890 0.928383764096 0.928372472930 0.000023330656
178 POLISH_LUCENE_STEMPEL_DIRECT PL_PL ALL_WORDS PRIMARY_OUTPUT 9990 122341 1 122341 0 1 122341 31432 797573 66669 323394 7482411334 66669 7482478003 0.000891 323394 1120967 28.849556 0.922858412343 0.711504442147 0.999991089984 0.999947877619 0.855747766065 0.871105640425 0.803515398127 0.745658746735 0.671563509358 0.810319603524 0.810295555502 0.000052122381 0.803489769425 0.991766794523 0.931068514076 0.960459620808 0.960459620808
179 POLISH_LUCENE_STEMPEL_DIRECT PL_PL LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 9846 120925 1 120925 0 1 120925 30830 794493 66274 320158 7310186425 66274 7310252699 0.000907 320158 1114651 28.722712 0.923005877316 0.712772876892 0.999990934103 0.999947146412 0.856381905497 0.871590591697 0.804379630033 0.746792242917 0.672771767894 0.811106376847 0.811081975971 0.000052853588 0.804353636298 0.991711086900 0.931317880193 0.960566151650 0.960566151650
180 POLISH_LUCENE_STEMPEL_FILTER PL_PL ALL_WORDS PRIMARY_OUTPUT 9990 122341 1 122341 0 1 122341 31432 797573 66669 323394 7482411334 66669 7482478003 0.000891 323394 1120967 28.849556 0.922858412343 0.711504442147 0.999991089984 0.999947877619 0.855747766065 0.871105640425 0.803515398127 0.745658746735 0.671563509358 0.810319603524 0.810295555502 0.000052122381 0.803489769425 0.991766794523 0.931068514076 0.960459620808 0.960459620808
181 POLISH_LUCENE_STEMPEL_FILTER PL_PL LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 9846 120925 1 120925 0 1 120925 30830 794493 66274 320158 7310186425 66274 7310252699 0.000907 320158 1114651 28.722712 0.923005877316 0.712772876892 0.999990934103 0.999947146412 0.856381905497 0.871590591697 0.804379630033 0.746792242917 0.672771767894 0.811106376847 0.811081975971 0.000052853588 0.804353636298 0.991711086900 0.931317880193 0.960566151650 0.960566151650
182 POLISH_RADIXOR PL_PL ALL_WORDS PRIMARY_OUTPUT 9990 122341 1 122341 0 1 122341 10074 1099420 13669 21547 7482464334 13669 7482478003 0.000183 21547 1120967 1.922180 0.987719760055 0.980778203105 0.999998173199 0.999995294243 0.990388188152 0.986323599045 0.984236742499 0.982158698021 0.968962733423 0.984242862020 0.984240510632 0.000004705757 0.984234389298 0.996967243455 0.996469409869 0.996718264498 0.996718264498
183 POLISH_RADIXOR PL_PL ALL_WORDS ANY_CANDIDATE 9990 122341 1 119475 2866 4 125778 10079 1120967 0 0 7482478003 0 7482478003 0.000000 0 1120967 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
184 POLISH_RADIXOR PL_PL ALL_WORDS ALL_CANDIDATES 9990 122341 1 119475 2866 4 125778 10079 1120967 38073 0 7482439930 38073 7482478003 0.000509 0 1120967 0.000000 0.967151263114 1.000000000000 0.999994911712 0.999994912475 0.999997455856 0.973547222425 0.983301367057 0.993252946885 0.967151263114 0.983438489746 0.983435987734 0.000005087525
185 POLISH_RADIXOR PL_PL LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 9846 120925 1 120925 0 1 120925 9844 1093651 13669 21000 7310239030 13669 7310252699 0.000187 21000 1114651 1.883998 0.987655781527 0.981160022285 0.999998130160 0.999995258206 0.990579076223 0.986349757961 0.984397186102 0.982452329568 0.969273787578 0.984402543989 0.984400174373 0.000004741794 0.984394814870 0.996926141446 0.996646530259 0.996786316244 0.996786316244
186 POLISH_RADIXOR PL_PL LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 9846 120925 1 118145 2780 4 124274 9847 1114651 0 0 7310252699 0 7310252699 0.000000 0 1114651 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
187 POLISH_RADIXOR PL_PL LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 9846 120925 1 118145 2780 4 124274 9847 1114651 38073 0 7310214626 38073 7310252699 0.000521 0 1114651 0.000000 0.966971278467 1.000000000000 0.999994791835 0.999994792629 0.999997395918 0.973401318686 0.983208335630 0.993214975136 0.966971278467 0.983346977657 0.983344416937 0.000005207371
188 PORTUGUESE_LUCENE_PORTUGUESE_LIGHT_STEM_FILTER PT_PT ALL_WORDS PRIMARY_OUTPUT 4001 211489 0 211489 0 1 211489 112814 149830 2511 5339230 22358201245 2511 22358203756 0.000011 5339230 5489060 97.270389 0.983517240927 0.027296112631 0.999999887692 0.999761142266 0.513648000162 0.122843213263 0.053118010934 0.033885033146 0.027283631587 0.163848092400 0.163827867245 0.000238857734 0.053105458848 0.999226179883 0.720580123442 0.837329788424 0.837329788424
189 PORTUGUESE_LUCENE_PORTUGUESE_LIGHT_STEM_FILTER PT_PT LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 4001 211489 0 211489 0 1 211489 112814 149830 2511 5339230 22358201245 2511 22358203756 0.000011 5339230 5489060 97.270389 0.983517240927 0.027296112631 0.999999887692 0.999761142266 0.513648000162 0.122843213263 0.053118010934 0.033885033146 0.027283631587 0.163848092400 0.163827867245 0.000238857734 0.053105458848 0.999226179883 0.720580123442 0.837329788424 0.837329788424
190 PORTUGUESE_LUCENE_PORTUGUESE_MINIMAL_STEM_FILTER PT_PT ALL_WORDS PRIMARY_OUTPUT 4001 211489 0 211489 0 1 211489 167745 43406 598 5445654 22358203158 598 22358203756 0.000003 5445654 5489060 99.209227 0.986410326334 0.007907729192 0.999999973254 0.999756469021 0.503953851223 0.038310165654 0.015689679353 0.009864890589 0.007906867787 0.088319113068 0.088308061848 0.000243530979 0.015685836581 0.999664059174 0.692382565086 0.818121623354 0.818121623354
191 PORTUGUESE_LUCENE_PORTUGUESE_MINIMAL_STEM_FILTER PT_PT LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 4001 211489 0 211489 0 1 211489 167745 43406 598 5445654 22358203158 598 22358203756 0.000003 5445654 5489060 99.209227 0.986410326334 0.007907729192 0.999999973254 0.999756469021 0.503953851223 0.038310165654 0.015689679353 0.009864890589 0.007906867787 0.088319113068 0.088308061848 0.000243530979 0.015685836581 0.999664059174 0.692382565086 0.818121623354 0.818121623354
192 PORTUGUESE_LUCENE_PORTUGUESE_STEM_FILTER PT_PT ALL_WORDS PRIMARY_OUTPUT 4001 211489 0 211489 0 1 211489 27586 3803488 99075 1685572 22358104681 99075 22358203756 0.000443 1685572 5489060 30.707844 0.974612837768 0.692921556696 0.999995568741 0.999920198913 0.846458562719 0.901329863268 0.809974591186 0.735433886866 0.680636384053 0.821784792219 0.821750382444 0.000079801087 0.809935821347 0.996728545383 0.918475110538 0.956003146785 0.956003146785
193 PORTUGUESE_LUCENE_PORTUGUESE_STEM_FILTER PT_PT LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 4001 211489 0 211489 0 1 211489 27586 3803488 99075 1685572 22358104681 99075 22358203756 0.000443 1685572 5489060 30.707844 0.974612837768 0.692921556696 0.999995568741 0.999920198913 0.846458562719 0.901329863268 0.809974591186 0.735433886866 0.680636384053 0.821784792219 0.821750382444 0.000079801087 0.809935821347 0.996728545383 0.918475110538 0.956003146785 0.956003146785
194 PORTUGUESE_RADIXOR PT_PT ALL_WORDS PRIMARY_OUTPUT 4001 211489 0 211489 0 1 211489 4001 5472616 20678 16444 22358183078 20678 22358203756 0.000092 16444 5489060 0.299578 0.996235774018 0.997004222945 0.999999075149 0.999998340077 0.998501649047 0.996389369023 0.996619850353 0.996850438335 0.993262474550 0.996619924417 0.996619094289 0.000001659923 0.996619020188 0.999299376330 0.999346803887 0.999323089546 0.999323089546
195 PORTUGUESE_RADIXOR PT_PT ALL_WORDS ANY_CANDIDATE 4001 211489 0 210699 790 3 212297 4001 5489060 0 0 22358203756 0 22358203756 0.000000 0 5489060 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
196 PORTUGUESE_RADIXOR PT_PT ALL_WORDS ALL_CANDIDATES 4001 211489 0 210699 790 3 212297 4001 5489060 38310 0 22358165446 38310 22358203756 0.000171 0 5489060 0.000000 0.993069036450 1.000000000000 0.999998286535 0.999998286956 0.999999143267 0.994447532369 0.996522466897 0.998606078314 0.993069036450 0.996528492543 0.996527638784 0.000001713044
197 PORTUGUESE_RADIXOR PT_PT LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 4001 211489 0 211489 0 1 211489 4001 5472616 20678 16444 22358183078 20678 22358203756 0.000092 16444 5489060 0.299578 0.996235774018 0.997004222945 0.999999075149 0.999998340077 0.998501649047 0.996389369023 0.996619850353 0.996850438335 0.993262474550 0.996619924417 0.996619094289 0.000001659923 0.996619020188 0.999299376330 0.999346803887 0.999323089546 0.999323089546
198 PORTUGUESE_RADIXOR PT_PT LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 4001 211489 0 210699 790 3 212297 4001 5489060 0 0 22358203756 0 22358203756 0.000000 0 5489060 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
199 PORTUGUESE_RADIXOR PT_PT LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 4001 211489 0 210699 790 3 212297 4001 5489060 38310 0 22358165446 38310 22358203756 0.000171 0 5489060 0.000000 0.993069036450 1.000000000000 0.999998286535 0.999998286956 0.999999143267 0.994447532369 0.996522466897 0.998606078314 0.993069036450 0.996528492543 0.996527638784 0.000001713044
200 RUSSIAN_LUCENE_RUSSIAN_LIGHT_STEM_FILTER RU_RU ALL_WORDS PRIMARY_OUTPUT 37410 768882 10 768882 0 1 768882 232250 3081067 321183 10008438 295575969833 321183 295576291016 0.000109 10008438 13089505 76.461547 0.905596884415 0.235384531348 0.999998913367 0.999965054154 0.617691722357 0.577011147253 0.373649378129 0.276277984307 0.229747124085 0.461696326851 0.461686629842 0.000034945846 0.373637933830 0.994310930069 0.870888421754 0.928516167212 0.928516167212
201 RUSSIAN_LUCENE_RUSSIAN_LIGHT_STEM_FILTER RU_RU LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 37297 768133 10 768133 0 1 768133 232143 3078888 318921 10008238 295000362731 318921 295000681652 0.000108 10008238 13087126 76.473918 0.906139220892 0.235260820443 0.999998918914 0.999964994315 0.617629869679 0.577038425373 0.373539598427 0.276151716079 0.229664120975 0.461713175622 0.461703471649 0.000035005685 0.373528142103 0.994349544377 0.870767393403 0.928464208705 0.928464208705
202 RUSSIAN_RADIXOR RU_RU ALL_WORDS PRIMARY_OUTPUT 37410 768882 10 768882 0 1 768882 37561 12823203 155850 266302 295576135166 155850 295576291016 0.000053 266302 13089505 2.034470 0.987992190185 0.979655304001 0.999999472725 0.999998571830 0.989827388363 0.986313480705 0.983806085477 0.981311406466 0.968128298562 0.983814916245 0.983814202914 0.000001428170 0.983805371373 0.997699288696 0.997273959852 0.997486578934 0.997486578934
203 RUSSIAN_RADIXOR RU_RU ALL_WORDS ANY_CANDIDATE 37410 768882 10 749720 19162 4 788492 37593 13089492 0 13 295576291016 0 295576291016 0.000000 13 13089505 0.000099 1.000000000000 0.999999006838 1.000000000000 0.999999999956 0.999999503419 0.999999801367 0.999999503419 0.999999205470 0.999999006838 0.999999503419 0.999999503397 0.000000000044
204 RUSSIAN_RADIXOR RU_RU ALL_WORDS ALL_CANDIDATES 37410 768882 10 749720 19162 4 788492 37593 13089492 434710 13 295575856306 434710 295576291016 0.000147 13 13089505 0.000099 0.967856883534 0.999999006838 0.999998529280 0.999998529301 0.999998768059 0.974118939969 0.983665447282 0.993400920813 0.967855953192 0.983796687479 0.983795964011 0.000001470699
205 RUSSIAN_RADIXOR RU_RU LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 37297 768133 10 768133 0 1 768133 37282 12821513 155850 265613 295000525802 155850 295000681652 0.000053 265613 13087126 2.029575 0.987990626447 0.979704252867 0.999999471696 0.999998571379 0.989851862281 0.986322156837 0.983829991833 0.981350389119 0.968174600634 0.983838715706 0.983838002141 0.000001428621 0.983829277503 0.997696524283 0.997321167437 0.997508810549 0.997508810549
206 RUSSIAN_RADIXOR RU_RU LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 37297 768133 10 749142 18991 4 787549 37306 13087126 0 0 295000681652 0 295000681652 0.000000 0 13087126 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
207 RUSSIAN_RADIXOR RU_RU LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 37297 768133 10 749142 18991 4 787549 37306 13087126 434710 0 295000246942 434710 295000681652 0.000147 0 13087126 0.000000 0.967851259252 1.000000000000 0.999998526410 0.999998526476 0.999999263205 0.974114570610 0.983663023007 0.993400519870 0.967851259252 0.983794317554 0.983793592699 0.000001473524
208 SNOWBALL_DANISH_DIRECT DA_DK ALL_WORDS PRIMARY_OUTPUT 4179 28079 32 28079 0 1 28079 5553 78732 6341 11163 394104845 6341 394111186 0.001609 11163 89895 12.417821 0.925464013259 0.875821792091 0.999983910632 0.999955596266 0.937902851361 0.915090414169 0.899958849618 0.885319563795 0.818113803566 0.900300811178 0.900278764621 0.000044403734 0.899936659693 0.994195946109 0.978579164615 0.986325742978 0.986325742978
209 SNOWBALL_DANISH_DIRECT DA_DK LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 4173 28033 32 28033 0 1 28033 5539 78627 6341 11113 392814447 6341 392820788 0.001614 11113 89740 12.383552 0.925371904717 0.876164475150 0.999983857779 0.999955577673 0.938074166465 0.915093153823 0.900096160451 0.885582797210 0.818340774971 0.900432112497 0.900410054086 0.000044422327 0.900073960926 0.994185354918 0.978644331817 0.986353630937 0.986353630937
210 SNOWBALL_DANISH_LUCENE_FILTER DA_DK ALL_WORDS PRIMARY_OUTPUT 4179 28079 32 28079 0 1 28079 5546 78744 6507 11151 394104679 6507 394111186 0.001651 11151 89895 12.404472 0.923672449590 0.875955281161 0.999983489431 0.999955205602 0.937969385296 0.913717599716 0.899181254496 0.885100184115 0.816829526358 0.899497504322 0.899475250557 0.000044794398 0.899158868074 0.994052746860 0.978603476314 0.986267614487 0.986267614487
211 SNOWBALL_DANISH_LUCENE_FILTER DA_DK LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 4173 28033 32 28033 0 1 28033 5539 78627 6341 11113 392814447 6341 392820788 0.001614 11113 89740 12.383552 0.925371904717 0.876164475150 0.999983857779 0.999955577673 0.938074166465 0.915093153823 0.900096160451 0.885582797210 0.818340774971 0.900432112497 0.900410054086 0.000044422327 0.900073960926 0.994185354918 0.978644331817 0.986353630937 0.986353630937
212 SNOWBALL_DUTCH_DIRECT NL_NL ALL_WORDS PRIMARY_OUTPUT 4992 26477 85 26477 0 1 26477 12051 29325 4382 35241 350433578 4382 350437960 0.001250 35241 64566 54.581359 0.869997329931 0.454186413902 0.999987495647 0.999886953739 0.727086954774 0.735353119953 0.596806854375 0.502190286022 0.425320531415 0.628602392126 0.628557411604 0.000113046261 0.596755898222 0.992814719235 0.917346281080 0.953589661124 0.953589661124
213 SNOWBALL_DUTCH_DIRECT NL_NL LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 4796 25678 84 25678 0 1 25678 11466 29111 4382 34036 329599474 4382 329603856 0.001329 34036 63147 53.899631 0.869166691547 0.461003689803 0.999986705253 0.999883464224 0.730495197528 0.738411822300 0.602462748344 0.508789468717 0.431088865524 0.633000040962 0.632953313739 0.000116535776 0.602409959779 0.992557044900 0.918058993802 0.953855618789 0.953855618789
214 SNOWBALL_DUTCH_LUCENE_FILTER NL_NL ALL_WORDS PRIMARY_OUTPUT 4992 26477 85 26477 0 1 26477 14573 15302 1588 49264 350436372 1588 350437960 0.000453 49264 64566 76.300220 0.905979869745 0.236997800700 0.999995468527 0.999854916880 0.618496634614 0.579068464950 0.375712040856 0.278062466837 0.231308764398 0.463373754768 0.463333378452 0.000145083120 0.375664346452 0.995827666179 0.888410302915 0.939057139355 0.939057139355
215 SNOWBALL_DUTCH_LUCENE_FILTER NL_NL LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 4796 25678 84 25678 0 1 25678 14116 14972 1544 48175 329602312 1544 329603856 0.000468 48175 63147 76.290243 0.906514894648 0.237097565997 0.999995315589 0.999849184178 0.618546440793 0.579362438183 0.375883408860 0.278182412747 0.231438685443 0.463608105042 0.463566154833 0.000150815822 0.375833834664 0.995816517119 0.887491664267 0.938538755173 0.938538755173
216 SNOWBALL_FINNISH_DIRECT FI_FI ALL_WORDS PRIMARY_OUTPUT 57027 1811717 292 1811717 0 1 1811717 381483 15114332 1544812 16409363 1641125269679 1544812 1641126814491 0.000094 16409363 31523695 52.054060 0.907269425128 0.479459403474 0.999999058688 0.999989060059 0.739729231081 0.769880311353 0.627374073993 0.529384116965 0.457061215373 0.659544431681 0.659540149918 0.000010939941 0.627369124557 0.991871857177 0.904138579582 0.945975396220 0.945975396220
217 SNOWBALL_FINNISH_DIRECT FI_FI LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 54762 1757055 274 1757055 0 1 1757055 372232 14692070 1513705 16121763 1543587930447 1513705 1543589444152 0.000098 16121763 30813833 52.319888 0.906594717007 0.476801117213 0.999999019360 0.999988575255 0.738400068286 0.768116957494 0.624933751043 0.526744348874 0.454475376380 0.657468914800 0.657464451566 0.000011424745 0.624928589970 0.991731678095 0.902933406396 0.945251664438 0.945251664438
218 SNOWBALL_FINNISH_LUCENE_FILTER FI_FI ALL_WORDS PRIMARY_OUTPUT 57027 1811717 292 1811717 0 1 1811717 377778 15153638 1922153 16370057 1641124892338 1922153 1641126814491 0.000117 16370057 31523695 51.929372 0.887434028678 0.480706275073 0.999998828760 0.999988854086 0.740352551917 0.758996033322 0.623613097472 0.529216231176 0.453079796332 0.653142485450 0.653138019077 0.000011145914 0.623608016975 0.990717710840 0.904385055188 0.945584912497 0.945584912497
219 SNOWBALL_FINNISH_LUCENE_FILTER FI_FI LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 54762 1757055 274 1757055 0 1 1757055 372232 14692070 1513705 16121763 1543587930447 1513705 1543589444152 0.000098 16121763 30813833 52.319888 0.906594717007 0.476801117213 0.999999019360 0.999988575255 0.738400068286 0.768116957494 0.624933751043 0.526744348874 0.454475376380 0.657468914800 0.657464451566 0.000011424745 0.624928589970 0.991731678095 0.902933406396 0.945251664438 0.945251664438
220 SNOWBALL_FRENCH_DIRECT FR_FR ALL_WORDS PRIMARY_OUTPUT 59240 425210 2301 425210 0 1 425210 85627 3766640 1654723 1687975 90394450107 1654723 90396104830 0.001831 1687975 5454615 30.945814 0.694777309691 0.690541862258 0.999981694753 0.999963023890 0.845261778506 0.693926068790 0.692653111288 0.691384815533 0.529815856272 0.692656348624 0.692637859933 0.000036976110 0.692634622294 0.959459328254 0.944947915186 0.952148333884 0.952148333884
221 SNOWBALL_FRENCH_DIRECT FR_FR LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 57698 421231 2133 421231 0 1 421231 84526 3758589 1646111 1681970 88710480395 1646111 88712126506 0.001856 1681970 5440559 30.915389 0.695429718578 0.690846106071 0.999981444352 0.999962486787 0.845413775212 0.694508136723 0.693130334647 0.691757988461 0.530374491828 0.693134123475 0.693115366288 0.000037513213 0.693111577099 0.959520798119 0.944537159644 0.951970023370 0.951970023370
222 SNOWBALL_FRENCH_LUCENE_FILTER FR_FR ALL_WORDS PRIMARY_OUTPUT 59240 425210 2301 425210 0 1 425210 85202 3763777 1661388 1690838 90394443442 1661388 90396104830 0.001838 1690838 5454615 30.998301 0.693762678186 0.690016985617 0.999981621022 0.999962918494 0.844999303320 0.693010289898 0.691884762376 0.690762884895 0.528917286853 0.691887297134 0.691868755638 0.000037081506 0.691866220643 0.958697792387 0.944714715363 0.951654891797 0.951654891797
223 SNOWBALL_FRENCH_LUCENE_FILTER FR_FR LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 57698 421231 2133 421231 0 1 421231 84810 3755856 1641925 1684703 88710484581 1641925 88712126506 0.001851 1684703 5440559 30.965623 0.695814817237 0.690343767984 0.999981491538 0.999962503165 0.845162629761 0.694713680979 0.693068495729 0.691431084156 0.530302080457 0.693073894149 0.693055145391 0.000037496835 0.693049746458 0.959566165512 0.944384738154 0.951914926182 0.951914926182
224 SNOWBALL_GERMAN_DIRECT DE_DE ALL_WORDS PRIMARY_OUTPUT 54092 296974 1474 296974 0 1 296974 81649 771138 190680 612734 44095055299 190680 44095245979 0.000432 612734 1383872 44.276783 0.801750435114 0.557232171762 0.999995675724 0.999981780603 0.778613923743 0.737064397386 0.657493530688 0.593429030432 0.489750735447 0.668401927114 0.668393541401 0.000018219397 0.657484715679 0.983724573695 0.949324273697 0.966218331938 0.966218331938
225 SNOWBALL_GERMAN_DIRECT DE_DE LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 16007 150098 228 150098 0 1 150098 37843 516936 87697 356475 11263668645 87697 11263756342 0.000779 356475 873411 40.814118 0.854958297017 0.591858815609 0.999992214231 0.999960569321 0.795925514920 0.785153327381 0.699486618802 0.630674793334 0.537854226580 0.711347035607 0.711329191110 0.000039430679 0.699467554207 0.988417636496 0.932451723900 0.959619376607 0.959619376607
226 SNOWBALL_GERMAN_LUCENE_FILTER DE_DE ALL_WORDS PRIMARY_OUTPUT 54092 296974 1474 296974 0 1 296974 86669 751056 295701 632816 44094950278 295701 44095245979 0.000671 632816 1383872 45.727929 0.717507501741 0.542720714054 0.999993294039 0.999978943584 0.771357004047 0.674088567377 0.617993120299 0.570516594262 0.447170798768 0.624024185176 0.624014089276 0.000021056416 0.617982794334 0.975844522648 0.942925157079 0.959102449371 0.959102449371
227 SNOWBALL_GERMAN_LUCENE_FILTER DE_DE LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 16007 150098 228 150098 0 1 150098 46077 481501 77653 391910 11263678689 77653 11263756342 0.000689 391910 873411 44.871200 0.861124126806 0.551287996144 0.999993105941 0.999958315274 0.775640551042 0.774110642769 0.672222202832 0.594035281304 0.506276128631 0.689004640259 0.688986412764 0.000041684726 0.672202362239 0.989021274644 0.919542244735 0.953017108147 0.953017108147
228 SNOWBALL_HUNGARIAN_DIRECT HU_HU ALL_WORDS PRIMARY_OUTPUT 19406 916344 1 916344 0 1 916344 116105 14287912 1506056 7874191 419819036837 1506056 419820542893 0.000359 7874191 22162103 35.529981 0.904643595580 0.644700189328 0.999996412620 0.999977657711 0.822348300974 0.837136808086 0.752865700984 0.684009307333 0.603676525918 0.763690969794 0.763680928295 0.000022342289 0.752854843818 0.991947513126 0.924304257644 0.956931989551 0.956931989551
229 SNOWBALL_HUNGARIAN_DIRECT HU_HU LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 18360 878513 1 878513 0 1 878513 111379 13776526 1496670 7634885 385869198247 1496670 385870694917 0.000388 7634885 21411411 35.658019 0.902006757459 0.643419810119 0.999996121317 0.999976336507 0.821707965718 0.834898516372 0.751079383241 0.682554714263 0.601382804609 0.761819543337 0.761808891723 0.000023663493 0.751067882221 0.991609896137 0.923288102987 0.956230170303 0.956230170303
230 SNOWBALL_HUNGARIAN_LUCENE_FILTER HU_HU ALL_WORDS PRIMARY_OUTPUT 19406 916344 1 916344 0 1 916344 114867 14299358 1792049 7862745 419818750844 1792049 419820542893 0.000427 7862745 22162103 35.478334 0.888633169244 0.645216656560 0.999995731393 0.999977003783 0.822606193976 0.826287586346 0.747610245439 0.682613266689 0.596946950992 0.757205997314 0.757195513225 0.000022996217 0.747599036407 0.990687085622 0.924490230693 0.956444632598 0.956444632598
231 SNOWBALL_HUNGARIAN_LUCENE_FILTER HU_HU LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 18360 878513 1 878513 0 1 878513 111379 13776526 1496670 7634885 385869198247 1496670 385870694917 0.000388 7634885 21411411 35.658019 0.902006757459 0.643419810119 0.999996121317 0.999976336507 0.821707965718 0.834898516372 0.751079383241 0.682554714263 0.601382804609 0.761819543337 0.761808891723 0.000023663493 0.751067882221 0.991609896137 0.923288102987 0.956230170303 0.956230170303
232 SNOWBALL_ITALIAN_DIRECT IT_IT ALL_WORDS PRIMARY_OUTPUT 10009 327551 0 327551 0 1 327551 46828 4499650 504775 1644164 53638016436 504775 53638521211 0.000941 1644164 6143814 26.761292 0.899134266174 0.732387080729 0.999990589319 0.999959941236 0.866188835024 0.859975076366 0.807239600802 0.760598128154 0.676782697802 0.811488952720 0.811469907061 0.000040058764 0.807219778599 0.987993752409 0.933407841859 0.959925420024 0.959925420024
233 SNOWBALL_ITALIAN_DIRECT IT_IT LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 10007 327469 0 327469 0 1 327469 46814 4498652 504774 1643522 53611162298 504774 53611667072 0.000942 1643522 6142174 26.757985 0.899114326863 0.732420149608 0.999990584624 0.999959933163 0.866205367116 0.859969602244 0.807251650876 0.760623806435 0.676799637969 0.811498274672 0.811479224597 0.000040066837 0.807231824542 0.987990915845 0.933413003102 0.959926810498 0.959926810498
234 SNOWBALL_ITALIAN_LUCENE_FILTER IT_IT ALL_WORDS PRIMARY_OUTPUT 10009 327551 0 327551 0 1 327551 46828 4499650 504775 1644164 53638016436 504775 53638521211 0.000941 1644164 6143814 26.761292 0.899134266174 0.732387080729 0.999990589319 0.999959941236 0.866188835024 0.859975076366 0.807239600802 0.760598128154 0.676782697802 0.811488952720 0.811469907061 0.000040058764 0.807219778599 0.987993752409 0.933407841859 0.959925420024 0.959925420024
235 SNOWBALL_ITALIAN_LUCENE_FILTER IT_IT LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 10007 327469 0 327469 0 1 327469 46814 4498652 504774 1643522 53611162298 504774 53611667072 0.000942 1643522 6142174 26.757985 0.899114326863 0.732420149608 0.999990584624 0.999959933163 0.866205367116 0.859969602244 0.807251650876 0.760623806435 0.676799637969 0.811498274672 0.811479224597 0.000040066837 0.807231824542 0.987990915845 0.933413003102 0.959926810498 0.959926810498
236 SNOWBALL_NORWEGIAN_BOKMAL_DIRECT NB_NO ALL_WORDS PRIMARY_OUTPUT 17929 75310 252 75310 0 1 75310 24394 106626 23997 35554 2835594218 23997 2835618215 0.000846 35554 142180 25.006330 0.816288096277 0.749936699958 0.999991537295 0.999978999989 0.874964118626 0.802094867845 0.781706946038 0.762329786671 0.641641141674 0.782409356499 0.782398932962 0.000021000011 0.781696464373 0.988328173631 0.971119668400 0.979648355684 0.979648355684
237 SNOWBALL_NORWEGIAN_BOKMAL_DIRECT NB_NO LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 17914 75251 252 75251 0 1 75251 24367 106567 23997 35524 2831152787 23997 2831176784 0.000848 35524 142091 25.000880 0.816205079501 0.749991202821 0.999991524019 0.999978977642 0.874991363420 0.802043209347 0.781698483431 0.762360357576 0.641629738452 0.782397999310 0.782387564360 0.000021022358 0.781687990535 0.988317966243 0.971127035574 0.979647089734 0.979647089734
238 SNOWBALL_NORWEGIAN_BOKMAL_LUCENE_FILTER NB_NO ALL_WORDS PRIMARY_OUTPUT 17929 75310 252 75310 0 1 75310 24396 106589 24046 35591 2835594169 24046 2835618215 0.000848 35591 142180 25.032353 0.815929880966 0.749676466451 0.999991520015 0.999978969662 0.874833993233 0.801758635215 0.781401315910 0.762052176648 0.641229410562 0.782101930719 0.782091491842 0.000021030338 0.781390819068 0.988295184140 0.971086244692 0.979615142710 0.979615142710
239 SNOWBALL_NORWEGIAN_BOKMAL_LUCENE_FILTER NB_NO LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 17914 75251 252 75251 0 1 75251 24381 106512 23993 35579 2831152791 23993 2831176784 0.000847 35579 142091 25.039587 0.816152637830 0.749604126933 0.999991525432 0.999978959629 0.874797826182 0.801914137847 0.781464144742 0.762031224736 0.641314033862 0.782170943927 0.782160500766 0.000021040371 0.781453643056 0.988310445474 0.971073741466 0.979616277822 0.979616277822
240 SNOWBALL_NORWEGIAN_NYNORSK_DIRECT NN_NO ALL_WORDS PRIMARY_OUTPUT 4688 18250 23 18250 0 1 18250 6138 22004 8274 8648 166483199 8274 166491473 0.004970 8648 30652 28.213493 0.726732280864 0.717865065901 0.999950303761 0.999898379870 0.858907684831 0.724941356316 0.722271459051 0.719621155632 0.565277706417 0.722285066089 0.722234252664 0.000101620130 0.722220641604 0.980542486408 0.964998187466 0.972708239744 0.972708239744
241 SNOWBALL_NORWEGIAN_NYNORSK_DIRECT NN_NO LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 4681 18219 23 18219 0 1 18219 6120 21971 8274 8624 165918002 8274 165926276 0.004987 8624 30595 28.187612 0.726434121342 0.718123876450 0.999950134480 0.999898178365 0.859037005465 0.724756721095 0.722255095332 0.719770679771 0.565257660346 0.722267047015 0.722216132089 0.000101821635 0.722204176866 0.980505813025 0.965064509051 0.972723884952 0.972723884952
242 SNOWBALL_NORWEGIAN_NYNORSK_LUCENE_FILTER NN_NO ALL_WORDS PRIMARY_OUTPUT 4688 18250 23 18250 0 1 18250 6144 21978 8295 8674 166483178 8295 166491473 0.004982 8674 30652 28.298317 0.725993459518 0.717016834138 0.999950177629 0.999898097625 0.858483505883 0.724180198229 0.721477226098 0.718794356395 0.564305338023 0.721491186328 0.721440231913 0.000101902375 0.721426267560 0.980461058483 0.964862123312 0.972599049418 0.972599049418
243 SNOWBALL_NORWEGIAN_NYNORSK_LUCENE_FILTER NN_NO LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 4681 18219 23 18219 0 1 18219 6130 21948 8274 8647 165918002 8274 165926276 0.004987 8647 30595 28.262788 0.726225928132 0.717372119627 0.999950134480 0.999898039774 0.858661127054 0.724437725685 0.721771872996 0.719125568472 0.564665929147 0.721785448310 0.721734464790 0.000101960226 0.721720885459 0.980505813025 0.964944952705 0.972663150405 0.972663150405
244 SNOWBALL_PORTUGUESE_DIRECT PT_PT ALL_WORDS PRIMARY_OUTPUT 4001 211489 0 211489 0 1 211489 11315 4817239 167230 671821 22358036526 167230 22358203756 0.000748 671821 5489060 12.239272 0.966449786326 0.877607277020 0.999992520419 0.999962481554 0.938799898719 0.947270839082 0.919888415834 0.894044585092 0.851660540743 0.920957852105 0.920939611009 0.000037518446 0.919869695779 0.996663145176 0.967923515462 0.982083116554 0.982083116554
245 SNOWBALL_PORTUGUESE_DIRECT PT_PT LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 4001 211489 0 211489 0 1 211489 11315 4817239 167230 671821 22358036526 167230 22358203756 0.000748 671821 5489060 12.239272 0.966449786326 0.877607277020 0.999992520419 0.999962481554 0.938799898719 0.947270839082 0.919888415834 0.894044585092 0.851660540743 0.920957852105 0.920939611009 0.000037518446 0.919869695779 0.996663145176 0.967923515462 0.982083116554 0.982083116554
246 SNOWBALL_PORTUGUESE_LUCENE_FILTER PT_PT ALL_WORDS PRIMARY_OUTPUT 4001 211489 0 211489 0 1 211489 11315 4817239 167230 671821 22358036526 167230 22358203756 0.000748 671821 5489060 12.239272 0.966449786326 0.877607277020 0.999992520419 0.999962481554 0.938799898719 0.947270839082 0.919888415834 0.894044585092 0.851660540743 0.920957852105 0.920939611009 0.000037518446 0.919869695779 0.996663145176 0.967923515462 0.982083116554 0.982083116554
247 SNOWBALL_PORTUGUESE_LUCENE_FILTER PT_PT LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 4001 211489 0 211489 0 1 211489 11315 4817239 167230 671821 22358036526 167230 22358203756 0.000748 671821 5489060 12.239272 0.966449786326 0.877607277020 0.999992520419 0.999962481554 0.938799898719 0.947270839082 0.919888415834 0.894044585092 0.851660540743 0.920957852105 0.920939611009 0.000037518446 0.919869695779 0.996663145176 0.967923515462 0.982083116554 0.982083116554
248 SNOWBALL_RUSSIAN_DIRECT RU_RU ALL_WORDS PRIMARY_OUTPUT 37410 768882 10 768882 0 1 768882 64358 8766656 3782908 4322849 295572508108 3782908 295576291016 0.001280 4322849 13089505 33.025305 0.698562595481 0.669746946122 0.999987201585 0.999972577645 0.834867073854 0.692602792505 0.683851352013 0.675318311031 0.519585195076 0.684003044583 0.683989349009 0.000027422355 0.683837646322 0.974179960240 0.953661001039 0.963811283954 0.963811283954
249 SNOWBALL_RUSSIAN_DIRECT RU_RU LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 37297 768133 10 768133 0 1 768133 64159 8764719 3782908 4322407 294996898744 3782908 295000681652 0.001282 4322407 13087126 33.027931 0.698516062041 0.669720685810 0.999987176613 0.999972525638 0.834853931211 0.692560581516 0.683815365804 0.675288254087 0.519543647630 0.683966853085 0.683953131513 0.000027474362 0.683801634112 0.974148936252 0.953633561396 0.963782086975 0.963782086975
250 SNOWBALL_RUSSIAN_LUCENE_FILTER RU_RU ALL_WORDS PRIMARY_OUTPUT 37410 768882 10 768882 0 1 768882 64266 8766889 3785790 4322616 295572505226 3785790 295576291016 0.001281 4322616 13089505 33.023525 0.698407806015 0.669764746642 0.999987191835 0.999972568683 0.834875969239 0.692484865100 0.683786451263 0.675303850911 0.519510266339 0.683936347366 0.683922647122 0.000027431317 0.683772741022 0.974131099393 0.953673855106 0.963793934411 0.963793934411
251 SNOWBALL_RUSSIAN_LUCENE_FILTER RU_RU LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 37297 768133 10 768133 0 1 768133 64159 8764719 3782908 4322407 294996898744 3782908 295000681652 0.001282 4322407 13087126 33.027931 0.698516062041 0.669720685810 0.999987176613 0.999972525638 0.834853931211 0.692560581516 0.683815365804 0.675288254087 0.519543647630 0.683966853085 0.683953131513 0.000027474362 0.683801634112 0.974148936252 0.953633561396 0.963782086975 0.963782086975
252 SNOWBALL_SPANISH_DIRECT ES_ES ALL_WORDS PRIMARY_OUTPUT 65059 871332 3589 871332 0 1 871332 195021 12811687 2228819 29161649 379565089291 2228819 379567318110 0.000587 29161649 41973336 69.476605 0.851812232913 0.305233946618 0.999994128001 0.999917308483 0.652614037309 0.627191552465 0.449423738186 0.350172671706 0.289843040458 0.509903921959 0.509876023351 0.000082691517 0.449391616998 0.981405614580 0.852462513401 0.912400934512 0.912400934512
253 SNOWBALL_SPANISH_DIRECT ES_ES LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 64918 869371 3525 869371 0 1 869371 194444 12787018 2201196 29076352 377858468569 2201196 377860669765 0.000583 29076352 41863370 69.455354 0.853138205793 0.305446455935 0.999994174583 0.999917233823 0.652720315259 0.627945981812 0.449838583213 0.350441220963 0.290188220622 0.510478247707 0.510450359519 0.000082766177 0.449806446076 0.981468761133 0.852555702466 0.912481600659 0.912481600659
254 SNOWBALL_SPANISH_LUCENE_FILTER ES_ES ALL_WORDS PRIMARY_OUTPUT 65059 871332 3589 871332 0 1 871332 194971 12811693 2230481 29161643 379565087629 2230481 379567318110 0.000588 29161643 41973336 69.476591 0.851718175843 0.305234089566 0.999994123622 0.999917304121 0.652614106594 0.627150877515 0.449410800675 0.350169642836 0.289832278511 0.509875888791 0.509847985527 0.000082695879 0.449378676109 0.981386049614 0.852460744495 0.912391466047 0.912391466047
255 SNOWBALL_SPANISH_LUCENE_FILTER ES_ES LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 64918 869371 3525 869371 0 1 869371 194444 12787018 2201196 29076352 377858468569 2201196 377860669765 0.000583 29076352 41863370 69.455354 0.853138205793 0.305446455935 0.999994174583 0.999917233823 0.652720315259 0.627945981812 0.449838583213 0.350441220963 0.290188220622 0.510478247707 0.510450359519 0.000082766177 0.449806446076 0.981468761133 0.852555702466 0.912481600659 0.912481600659
256 SNOWBALL_SWEDISH_DIRECT SV_SE ALL_WORDS PRIMARY_OUTPUT 12371 98108 68 98108 0 1 98108 25915 237017 67105 148325 4812088331 67105 4812155436 0.001394 148325 385342 38.491781 0.779348419384 0.615082186733 0.999986055105 0.999955235704 0.807534120919 0.739831942216 0.687539886056 0.642151948805 0.523855832838 0.692360693585 0.692339154006 0.000044764296 0.687517812951 0.984860422704 0.942685282143 0.963311451566 0.963311451566
257 SNOWBALL_SWEDISH_DIRECT SV_SE LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 12342 97881 68 97881 0 1 97881 25840 236588 67105 147975 4789844472 67105 4789911577 0.001401 147975 384563 38.478741 0.779036724587 0.615212591955 0.999985990347 0.999955100897 0.807599291151 0.739644914918 0.687500000000 0.642223301999 0.523809523810 0.692295603454 0.692273994517 0.000044899103 0.687477858823 0.984821307273 0.942694565973 0.963297587020 0.963297587020
258 SNOWBALL_SWEDISH_LUCENE_FILTER SV_SE ALL_WORDS PRIMARY_OUTPUT 12371 98108 68 98108 0 1 98108 26781 230676 64262 154666 4812091174 64262 4812155436 0.001335 154666 385342 40.137333 0.782116919488 0.598626674487 0.999986645901 0.999954508853 0.799306660194 0.736939762085 0.678179573117 0.628097931391 0.513064830384 0.684248529829 0.684226838572 0.000045491147 0.678157227687 0.985207247898 0.939659207875 0.961894327137 0.961894327137
259 SNOWBALL_SWEDISH_LUCENE_FILTER SV_SE LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 12342 97881 68 97881 0 1 97881 26706 230247 64262 154316 4789847315 64262 4789911577 0.001342 154316 384563 40.127625 0.781799537535 0.598723746174 0.999986583886 0.999954370671 0.799355165030 0.736743719918 0.678122496584 0.628142458291 0.512999498691 0.684165146635 0.684143384784 0.000045629329 0.678100081584 0.985169028543 0.939660518299 0.961876797342 0.961876797342
260 SNOWBALL_YIDDISH_DIRECT YI ALL_WORDS PRIMARY_OUTPUT 802 3578 0 3578 0 1 3578 1087 4962 1151 1382 6391758 1151 6392909 0.018004 1382 6344 21.784363 0.811712743334 0.782156368222 0.999819956768 0.999604172550 0.890988162495 0.805624107027 0.796660512162 0.787894185271 0.662041360907 0.796797522188 0.796599716782 0.000395827450 0.796462473806 0.982918530193 0.962014249878 0.972354049666 0.972354049666
261 SNOWBALL_YIDDISH_DIRECT YI LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 802 3578 0 3578 0 1 3578 1087 4962 1151 1382 6391758 1151 6392909 0.018004 1382 6344 21.784363 0.811712743334 0.782156368222 0.999819956768 0.999604172550 0.890988162495 0.805624107027 0.796660512162 0.787894185271 0.662041360907 0.796797522188 0.796599716782 0.000395827450 0.796462473806 0.982918530193 0.962014249878 0.972354049666 0.972354049666
262 SNOWBALL_YIDDISH_LUCENE_FILTER YI ALL_WORDS PRIMARY_OUTPUT 802 3578 0 3578 0 1 3578 1087 4962 1151 1382 6391758 1151 6392909 0.018004 1382 6344 21.784363 0.811712743334 0.782156368222 0.999819956768 0.999604172550 0.890988162495 0.805624107027 0.796660512162 0.787894185271 0.662041360907 0.796797522188 0.796599716782 0.000395827450 0.796462473806 0.982918530193 0.962014249878 0.972354049666 0.972354049666
263 SNOWBALL_YIDDISH_LUCENE_FILTER YI LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 802 3578 0 3578 0 1 3578 1087 4962 1151 1382 6391758 1151 6392909 0.018004 1382 6344 21.784363 0.811712743334 0.782156368222 0.999819956768 0.999604172550 0.890988162495 0.805624107027 0.796660512162 0.787894185271 0.662041360907 0.796797522188 0.796599716782 0.000395827450 0.796462473806 0.982918530193 0.962014249878 0.972354049666 0.972354049666
264 SPANISH_LUCENE_SPANISH_LIGHT_STEM_FILTER ES_ES ALL_WORDS PRIMARY_OUTPUT 65059 871332 3589 871332 0 1 871332 405552 1244317 147956 40729019 379567170154 147956 379567318110 0.000039 40729019 41973336 97.035458 0.893730611741 0.029645415842 0.999999610198 0.999892318297 0.514822513020 0.130863846499 0.057387272020 0.036752000024 0.029541282827 0.162772895888 0.162762055080 0.000107681703 0.057380579619 0.993823553768 0.756690454887 0.859195405761 0.859195405761
265 SPANISH_LUCENE_SPANISH_LIGHT_STEM_FILTER ES_ES LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 64918 869371 3525 869371 0 1 869371 404617 1241848 146613 40621522 377860523152 146613 377860669765 0.000039 40621522 41863370 97.033569 0.894406108634 0.029664310351 0.999999611992 0.999892119974 0.514831961171 0.130949068412 0.057424066047 0.036775459718 0.029560783207 0.162886280533 0.162875426941 0.000107880026 0.057417362063 0.993866319748 0.756725425267 0.859233931168 0.859233931168
266 SPANISH_LUCENE_SPANISH_MINIMAL_STEM_FILTER ES_ES ALL_WORDS PRIMARY_OUTPUT 65059 871332 3589 871332 0 1 871332 718633 148463 47859 41824873 379567270251 47859 379567318110 0.000013 41824873 41973336 99.646292 0.756221921130 0.003537078873 0.999999873912 0.999889695187 0.501768476392 0.017360591398 0.007041223811 0.004416184633 0.003533050405 0.051718628951 0.051713939443 0.000110304813 0.007040201537 0.995635307295 0.710609719902 0.829315972283 0.829315972283
267 SPANISH_LUCENE_SPANISH_MINIMAL_STEM_FILTER ES_ES LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 64918 869371 3525 869371 0 1 869371 717093 148226 47148 41715144 377860622617 47148 377860669765 0.000012 41715144 41863370 99.645929 0.758678227400 0.003540708739 0.999999875224 0.999889489251 0.501770291981 0.017379114288 0.007048522419 0.004420728101 0.003536725554 0.051829129163 0.051824445330 0.000110510749 0.007047500484 0.995675746140 0.710626433954 0.829341382913 0.829341382913
268 SPANISH_LUCENE_SPANISH_PLURAL_STEM_FILTER ES_ES ALL_WORDS PRIMARY_OUTPUT 65059 871332 3589 871332 0 1 871332 578805 325245 58578 41648091 379567259532 58578 379567318110 0.000015 41648091 41973336 99.225115 0.847382777999 0.007748847983 0.999999845672 0.999890132644 0.503874346827 0.037377069210 0.015357262275 0.009663967067 0.007738048760 0.081032341260 0.081026288458 0.000109867356 0.015355289170 0.995442321769 0.723731297627 0.838115191065 0.838115191065
269 SPANISH_LUCENE_SPANISH_PLURAL_STEM_FILTER ES_ES LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 64918 869371 3525 869371 0 1 869371 577533 324656 57716 41538714 377860612049 57716 377860669765 0.000015 41538714 41863370 99.224487 0.849057985417 0.007755132948 0.999999847256 0.999889928152 0.503877490102 0.037408921072 0.015369880354 0.009671831022 0.007744455857 0.081145286724 0.081139234944 0.000110071848 0.015367905834 0.995484117647 0.723752971494 0.838144538380 0.838144538380
270 SPANISH_RADIXOR ES_ES ALL_WORDS PRIMARY_OUTPUT 65059 871332 3589 871332 0 1 871332 64995 41074684 288483 898652 379567029627 288483 379567318110 0.000076 898652 41973336 2.141007 0.993025606574 0.978589931475 0.999999239969 0.999996872745 0.989294585722 0.990104500109 0.985754921826 0.981443392220 0.971909988067 0.985781345071 0.985779787115 0.000003127255 0.985753358111 0.995417814373 0.993266303762 0.994340895233 0.994340895233
271 SPANISH_RADIXOR ES_ES ALL_WORDS ANY_CANDIDATE 65059 871332 3589 828695 42637 21 916797 65118 41972710 2 626 379567318108 2 379567318110 0.000000 626 41973336 0.001491 0.999999952350 0.999985085770 0.999999999995 0.999999998346 0.999992542882 0.999996978999 0.999992519005 0.999988059050 0.999985038121 0.999992519032 0.999992518205 0.000000001654
272 SPANISH_RADIXOR ES_ES ALL_WORDS ALL_CANDIDATES 65059 871332 3589 828695 42637 21 916797 65118 41972710 1349800 626 379565968310 1349800 379567318110 0.000356 626 41973336 0.001491 0.968842987168 0.999985085770 0.999996443846 0.999996442590 0.999990764808 0.974915259157 0.984167740127 0.993597526064 0.968828987818 0.984290880594 0.984289129583 0.000003557410
273 SPANISH_RADIXOR ES_ES LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 64918 869371 3525 869371 0 1 869371 64814 40978337 276044 885033 377860393721 276044 377860669765 0.000073 885033 41863370 2.114099 0.993308734895 0.978859012067 0.999999269456 0.999996927576 0.989429140761 0.990384762162 0.986030938205 0.981715226337 0.972446769193 0.986057405488 0.986055874970 0.000003072424 0.986029401906 0.995463637710 0.993323040564 0.994392187139 0.994392187139
274 SPANISH_RADIXOR ES_ES LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 64918 869371 3525 826968 42403 21 914127 64933 41863370 0 0 377860669765 0 377860669765 0.000000 0 41863370 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
275 SPANISH_RADIXOR ES_ES LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 64918 869371 3525 826968 42403 21 914127 64933 41863370 1255381 0 377859414384 1255381 377860669765 0.000332 0 41863370 0.000000 0.970885497124 1.000000000000 0.999996677662 0.999996678030 0.999998338831 0.976571978660 0.985227704543 0.994038240493 0.970885497124 0.985335220686 0.985333583876 0.000003321970
276 SWEDISH_LUCENE_SWEDISH_LIGHT_STEM_FILTER SV_SE ALL_WORDS PRIMARY_OUTPUT 12371 98108 68 98108 0 1 98108 22392 218635 45941 166707 4812109495 45941 4812155436 0.000955 166707 385342 43.262089 0.826359911708 0.567379107390 0.999990453135 0.999955813777 0.783684780262 0.757232036109 0.672807954234 0.605320541501 0.506940918144 0.684733049508 0.684712936280 0.000044186223 0.672786622564 0.986795482859 0.942302523776 0.964035907715 0.964035907715
277 SWEDISH_LUCENE_SWEDISH_LIGHT_STEM_FILTER SV_SE LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 12342 97881 68 97881 0 1 97881 22338 218126 45941 166437 4789865636 45941 4789911577 0.000959 166437 384563 43.279515 0.826025213298 0.567204853301 0.999990408800 0.999955664954 0.783597631051 0.756945124029 0.672574503184 0.605125951621 0.506675896159 0.684489232882 0.684469049184 0.000044335046 0.672553099274 0.986761366955 0.942264620928 0.963999791833 0.963999791833
278 SWEDISH_LUCENE_SWEDISH_MINIMAL_STEM_FILTER SV_SE ALL_WORDS PRIMARY_OUTPUT 12371 98108 68 98108 0 1 98108 23360 228181 40227 157161 4812115209 40227 4812155436 0.000836 157161 385342 40.784809 0.850127417961 0.592151906618 0.999991640544 0.999958984659 0.796071773581 0.781991317186 0.698068068834 0.630412272016 0.536178622033 0.709510092538 0.709491456160 0.000041015341 0.698048215965 0.988492665376 0.944581755622 0.966038479572 0.966038479572
279 SWEDISH_LUCENE_SWEDISH_MINIMAL_STEM_FILTER SV_SE LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 12342 97881 68 97881 0 1 97881 23312 227624 40227 156939 4789871350 40227 4789911577 0.000840 156939 384563 40.809698 0.849815755775 0.591903017191 0.999991601724 0.999958840540 0.795947309457 0.781693541131 0.697790053555 0.630152322431 0.535850655618 0.709230928471 0.709212225121 0.000041159460 0.697770131116 0.988462934404 0.944527865197 0.965996098328 0.965996098328
280 SWEDISH_RADIXOR SV_SE ALL_WORDS PRIMARY_OUTPUT 12371 98108 68 98108 0 1 98108 12330 365796 24473 19546 4812130963 24473 4812155436 0.000509 19546 385342 5.072377 0.937291970410 0.949276227351 0.999994914337 0.999990853272 0.974635570844 0.939664553041 0.943246034417 0.946854921499 0.892588119029 0.943265066457 0.943260495884 0.000009146728 0.943241460869 0.992630770222 0.993394969179 0.993012722673 0.993012722673
281 SWEDISH_RADIXOR SV_SE ALL_WORDS ANY_CANDIDATE 12371 98108 68 92341 5767 5 104148 12371 385342 0 0 4812155436 0 4812155436 0.000000 0 385342 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
282 SWEDISH_RADIXOR SV_SE ALL_WORDS ALL_CANDIDATES 12371 98108 68 92341 5767 5 104148 12371 385342 47848 0 4812107588 47848 4812155436 0.000994 0 385342 0.000000 0.889545003347 1.000000000000 0.999990056847 0.999990057643 0.999995028423 0.909639856815 0.941544130223 0.975767741439 0.889545003347 0.943156934634 0.943152245645 0.000009942357
283 SWEDISH_RADIXOR SV_SE LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 12342 97881 68 97881 0 1 97881 12301 365017 24473 19546 4789887104 24473 4789911577 0.000511 19546 384563 5.082652 0.937166551131 0.949173477428 0.999994890720 0.999990810798 0.974584184074 0.939543572972 0.943131801052 0.946747541943 0.892383555482 0.943150907472 0.943146315681 0.000009189202 0.943127206266 0.992611730682 0.993377890892 0.992994663001 0.992994663001
284 SWEDISH_RADIXOR SV_SE LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 12342 97881 68 92114 5767 5 103921 12342 384563 0 0 4789911577 0 4789911577 0.000000 0 384563 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
285 SWEDISH_RADIXOR SV_SE LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 12342 97881 68 92114 5767 5 103921 12342 384563 47848 0 4789863729 47848 4789911577 0.000999 0 384563 0.000000 0.889346015712 1.000000000000 0.999990010672 0.999990011473 0.999995005336 0.909473386475 0.941432652692 0.975719846569 0.889346015712 0.943051438529 0.943046728292 0.000009988527
286 UKRAINIAN_LUCENE_MORFOLOGIK_FILTER UK_UA ALL_WORDS PRIMARY_OUTPUT 1493 14245 4 14245 0 1 14245 2358 56032 828 9308 101386722 828 101387550 0.000817 9308 65340 14.245485 0.985437917693 0.857545148454 0.999991833317 0.999900091560 0.928768490886 0.956895962839 0.917054009820 0.880397209478 0.846814169992 0.919270093835 0.919222898475 0.000099908440 0.917004266345 0.997989675681 0.970999849348 0.984309781661 0.984309781661
287 UKRAINIAN_LUCENE_MORFOLOGIK_FILTER UK_UA ALL_WORDS ANY_CANDIDATE 1493 14245 4 12038 2207 6 16937 2912 60394 122 4946 101387428 122 101387550 0.000120 4946 65340 7.569636 0.997984004230 0.924303642485 0.999998796696 0.999950045780 0.962151219591 0.982322936592 0.959731756929 0.938156308641 0.922581039382 0.960437530635 0.960413432420 0.000049954220
288 UKRAINIAN_LUCENE_MORFOLOGIK_FILTER UK_UA ALL_WORDS ALL_CANDIDATES 1493 14245 4 12038 2207 6 16937 2912 60394 1368 4946 101386182 1368 101387550 0.001349 4946 65340 7.569636 0.977850458211 0.924303642485 0.999986507219 0.999937764217 0.962145074852 0.966650447520 0.950323362339 0.934538657225 0.905348683816 0.950700131656 0.950669478973 0.000062235783
289 UKRAINIAN_LUCENE_MORFOLOGIK_FILTER UK_UA LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 1491 14236 4 14236 0 1 14236 2356 56016 828 9308 101258578 828 101259406 0.000818 9308 65324 14.248974 0.985433818873 0.857510256567 0.999991822982 0.999899965191 0.928751039775 0.956884181756 0.917032283413 0.880367133966 0.846777119361 0.919249480202 0.919202225823 0.000100034809 0.916982477117 0.997988093697 0.970977645923 0.984297603943 0.984297603943
290 UKRAINIAN_LUCENE_MORFOLOGIK_FILTER UK_UA LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 1491 14236 4 12029 2207 6 16928 2910 60378 122 4946 101259284 122 101259406 0.000120 4946 65324 7.571490 0.997983471074 0.924285101953 0.999998795174 0.999949982596 0.962141948563 0.982318335047 0.959721515768 0.938140934008 0.922562112276 0.960427641371 0.960403512884 0.000050017404
291 UKRAINIAN_LUCENE_MORFOLOGIK_FILTER UK_UA LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 1491 14236 4 12029 2207 6 16928 2910 60378 1368 4946 101258038 1368 101259406 0.001351 4946 65324 7.571490 0.977844718686 0.924285101953 0.999986490144 0.999937685499 0.962135796049 0.966641904786 0.950310852286 0.934522445998 0.905325976129 0.950687806541 0.950657115187 0.000062314501
292 UKRAINIAN_MORFOLOGIK_DIRECT UK_UA ALL_WORDS PRIMARY_OUTPUT 1493 14245 4 14245 0 1 14245 2365 56016 828 9324 101386722 828 101387550 0.000817 9324 65340 14.269972 0.985433818873 0.857300275482 0.999991833317 0.999899933851 0.928646054399 0.956831877998 0.916912197996 0.880190066750 0.846572361262 0.919136923635 0.919089660097 0.000100066149 0.916862376978 0.997989675681 0.970875945953 0.984246115917 0.984246115917
293 UKRAINIAN_MORFOLOGIK_DIRECT UK_UA ALL_WORDS ANY_CANDIDATE 1493 14245 4 12038 2207 6 16937 2919 60378 122 4962 101387428 122 101387550 0.000120 4962 65340 7.594123 0.997983471074 0.924058769513 0.999998796696 0.999949888071 0.962028783105 0.982267195939 0.959599491418 0.937954390108 0.922336622774 0.960310042786 0.960285871670 0.000050111929
294 UKRAINIAN_MORFOLOGIK_DIRECT UK_UA ALL_WORDS ALL_CANDIDATES 1493 14245 4 12038 2207 6 16937 2919 60378 1368 4962 101386182 1368 101387550 0.001349 4962 65340 7.594123 0.977844718686 0.924058769513 0.999986507219 0.999937606509 0.962022638366 0.966592384831 0.950191209102 0.934337338211 0.905108832524 0.950571400540 0.950540673331 0.000062393491
295 UKRAINIAN_MORFOLOGIK_DIRECT UK_UA LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 1491 14236 4 14236 0 1 14236 2356 56016 828 9308 101258578 828 101259406 0.000818 9308 65324 14.248974 0.985433818873 0.857510256567 0.999991822982 0.999899965191 0.928751039775 0.956884181756 0.917032283413 0.880367133966 0.846777119361 0.919249480202 0.919202225823 0.000100034809 0.916982477117 0.997988093697 0.970977645923 0.984297603943 0.984297603943
296 UKRAINIAN_MORFOLOGIK_DIRECT UK_UA LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 1491 14236 4 12029 2207 6 16928 2910 60378 122 4946 101259284 122 101259406 0.000120 4946 65324 7.571490 0.997983471074 0.924285101953 0.999998795174 0.999949982596 0.962141948563 0.982318335047 0.959721515768 0.938140934008 0.922562112276 0.960427641371 0.960403512884 0.000050017404
297 UKRAINIAN_MORFOLOGIK_DIRECT UK_UA LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 1491 14236 4 12029 2207 6 16928 2910 60378 1368 4946 101258038 1368 101259406 0.001351 4946 65324 7.571490 0.977844718686 0.924285101953 0.999986490144 0.999937685499 0.962135796049 0.966641904786 0.950310852286 0.934522445998 0.905325976129 0.950687806541 0.950657115187 0.000062314501
298 UKRAINIAN_RADIXOR UK_UA ALL_WORDS PRIMARY_OUTPUT 1493 14245 4 14245 0 1 14245 1493 64732 880 608 101386670 880 101387550 0.000868 608 65340 0.930517 0.986587819301 0.990694827058 0.999991320433 0.999985333094 0.995343073746 0.987406494442 0.988637057853 0.989870692292 0.977529447297 0.988639190514 0.988631855097 0.000014666906 0.988629719696 0.997993591453 0.998265712624 0.998129633491 0.998129633491
299 UKRAINIAN_RADIXOR UK_UA ALL_WORDS ANY_CANDIDATE 1493 14245 4 14055 190 2 14435 1493 65340 0 0 101387550 0 101387550 0.000000 0 65340 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
300 UKRAINIAN_RADIXOR UK_UA ALL_WORDS ALL_CANDIDATES 1493 14245 4 14055 190 2 14435 1493 65340 1490 0 101386060 1490 101387550 0.001470 0 65340 0.000000 0.977704623672 1.000000000000 0.999985303916 0.999985313380 0.999992651958 0.982083809295 0.988726639933 0.995459946982 0.977704623672 0.988789473888 0.988782208195 0.000014686620
301 UKRAINIAN_RADIXOR UK_UA LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 1491 14236 4 14236 0 1 14236 1491 64716 880 608 101258526 880 101259406 0.000869 608 65324 0.930745 0.986584547838 0.990692547915 0.999991309449 0.999985314543 0.995341928682 0.987403420118 0.988634280477 0.989868213355 0.977524016676 0.988636414174 0.988629069474 0.000014685457 0.988626933033 0.997992012551 0.998264347489 0.998128161444 0.998128161444
302 UKRAINIAN_RADIXOR UK_UA LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 1491 14236 4 14046 190 2 14426 1491 65324 0 0 101259406 0 101259406 0.000000 0 65324 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
303 UKRAINIAN_RADIXOR UK_UA LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 1491 14236 4 14046 190 2 14426 1491 65324 1490 0 101257916 1490 101259406 0.001471 0 65324 0.000000 0.977699284581 1.000000000000 0.999985285318 0.999985294804 0.999992642659 0.982079499669 0.988723909852 0.995458840023 0.977699284581 0.988786774073 0.988779499204 0.000014705196
304 YI_RADIXOR YI ALL_WORDS PRIMARY_OUTPUT 802 3578 0 3578 0 1 3578 802 6195 195 149 6392714 195 6392909 0.003050 149 6344 2.348676 0.969483568075 0.976513240858 0.999969497454 0.999946243726 0.988241369156 0.970881394183 0.972985707555 0.975099162627 0.947392567671 0.972992055990 0.972965163911 0.000053756274 0.972958803000 0.995691103897 0.996142223728 0.995916612726 0.995916612726
305 YI_RADIXOR YI ALL_WORDS ANY_CANDIDATE 802 3578 0 3489 89 3 3676 802 6344 0 0 6392909 0 6392909 0.000000 0 6344 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
306 YI_RADIXOR YI ALL_WORDS ALL_CANDIDATES 802 3578 0 3489 89 3 3676 802 6344 389 0 6392520 389 6392909 0.006085 0 6344 0.000000 0.942224862617 1.000000000000 0.999939151332 0.999939211655 0.999969575666 0.953239572064 0.970253116158 0.987885016662 0.942224862617 0.970682678643 0.970653145819 0.000060788345
307 YI_RADIXOR YI LOWERCASE_GROUPS_ONLY PRIMARY_OUTPUT 802 3578 0 3578 0 1 3578 802 6195 195 149 6392714 195 6392909 0.003050 149 6344 2.348676 0.969483568075 0.976513240858 0.999969497454 0.999946243726 0.988241369156 0.970881394183 0.972985707555 0.975099162627 0.947392567671 0.972992055990 0.972965163911 0.000053756274 0.972958803000 0.995691103897 0.996142223728 0.995916612726 0.995916612726
308 YI_RADIXOR YI LOWERCASE_GROUPS_ONLY ANY_CANDIDATE 802 3578 0 3489 89 3 3676 802 6344 0 0 6392909 0 6392909 0.000000 0 6344 0.000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 1.000000000000 0.000000000000
309 YI_RADIXOR YI LOWERCASE_GROUPS_ONLY ALL_CANDIDATES 802 3578 0 3489 89 3 3676 802 6344 389 0 6392520 389 6392909 0.006085 0 6344 0.000000 0.942224862617 1.000000000000 0.999939151332 0.999939211655 0.999969575666 0.953239572064 0.970253116158 0.987885016662 0.942224862617 0.970682678643 0.970653145819 0.000060788345

View File

@@ -0,0 +1 @@
5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28 stemming-quality.csv

View File

@@ -6,6 +6,8 @@ two layers:
- **benchmark reference pages**, which explain methodology, corpora, environment, candidate - **benchmark reference pages**, which explain methodology, corpora, environment, candidate
selection, and the English dictionary coverage experiment; selection, and the English dictionary coverage experiment;
- **language result pages**, which contain the actual same-language accuracy and throughput tables. - **language result pages**, which contain the actual same-language accuracy and throughput tables.
- **pairwise quality pages and generated sections**, which publish over-stemming, under-stemming,
candidate-policy, classification, and partition measurements from one checked result snapshot.
This structure keeps methodology separate from per-language result pages, while preserving all This structure keeps methodology separate from per-language result pages, while preserving all
measured data and the command-class analysis for each Radixor language resource. measured data and the command-class analysis for each Radixor language resource.
@@ -26,6 +28,9 @@ the preferred result measured by the accuracy pass.
| Page | Purpose | | Page | Purpose |
| --- | --- | | --- | --- |
| [Methodology](reference/methodology.md) | Workload design, normalization, speed metrics, quality metrics, and interpretation rules. | | [Methodology](reference/methodology.md) | Workload design, normalization, speed metrics, quality metrics, and interpretation rules. |
| [Linguistic quality methodology](reference/linguistic-quality.md) | Gold-standard groups, output policies, pairwise formulas, ranking rules, aggregation, and limitations. |
| [Tested stemmers](reference/tested-stemmers.md) | Versions, upstream attribution, evaluated coverage, adapters, preprocessing, and output capability. |
| [Reproducibility and raw data](reference/reproducibility.md) | Commands, versioned CSV snapshot, checksum, generated artifacts, and unavailable provenance. |
| [Corpora](reference/corpora.md) | Dictionary row counts, complete quality tokens, already-root tokens, changed speed tokens, and timing token counts. | | [Corpora](reference/corpora.md) | Dictionary row counts, complete quality tokens, already-root tokens, changed speed tokens, and timing token counts. |
| [Environment and reports](reference/environment.md) | Hardware, JVM, JMH settings, report files, and badge/report policy. | | [Environment and reports](reference/environment.md) | Hardware, JVM, JMH settings, report files, and badge/report policy. |
| [English dictionary coverage](reference/english-coverage.md) | Quality/speed operating curve for contracted Radixor tries built from 100% down to 10% of English dictionary rows. | | [English dictionary coverage](reference/english-coverage.md) | Quality/speed operating curve for contracted Radixor tries built from 100% down to 10% of English dictionary rows. |
@@ -53,3 +58,270 @@ keeps `92.868%` all-token exactness and `76.516%` changed-token exactness at `86
Those figures should not be reduced to a single speed badge. The professional interpretation is a Those figures should not be reduced to a single speed badge. The professional interpretation is a
quality/speed envelope: the amount and quality of dictionary knowledge affect stemming precision, quality/speed envelope: the amount and quality of dictionary knowledge affect stemming precision,
while contracted tries reduce lookup cost in uniform regions of the compiled graph. while contracted tries reduce lookup cost in uniform regions of the compiled graph.
## Quality versus performance
Each language page keeps exact-root accuracy, JMH latency, and pairwise linguistic-quality results in separate tables. No undocumented scalar combines them. The current repository checkout does not contain the dated machine-readable JMH CSV files named by the performance provenance page, so this revision preserves the existing performance tables but does not regenerate a cross-language Pareto frontier from rounded Markdown values. A defensible Pareto analysis requires the original unrounded JMH snapshot on the same hardware and JVM. Readers can still inspect the quality and speed dimensions side by side on every language page.
<!-- STEMMING-QUALITY-OVERVIEW:START -->
## Pairwise Quality Findings
The validated snapshot is a broad multilingual comparison covering the complete 20-language Radixor dictionary universe; 19 languages have existing benchmark pages. The direct ranking below uses only deterministic `PRIMARY_OUTPUT` rows over identical per-language inputs. Candidate-aware rows are intentionally excluded from this claim.
!!! success "Evidence-based primary-output result"
Radixor achieved the highest balanced accuracy among the evaluated deterministic stemmers for every documented language in both `ALL_WORDS` and `LOWERCASE_GROUPS_ONLY`: **38 wins in 38 language-mode comparisons, with no exact first-place ties**. This statement is limited to the evaluated implementations, versions, dictionaries, adapters, and balanced-accuracy metric; it is not a universal claim about every stemming use case.
### Per-language winner matrix
| Language | Dictionary mode | Winner | Balanced accuracy | Runner-up | Difference | Exact tie | Deterministic stemmers |
|---|---|---|---:|---|---:|---|---:|
|Czech (`CS_CZ`)|ALL_WORDS|Radixor|0.996565|HUNSPELL CZECH LUCENE FILTER|0.142812638|no|3|
|Czech (`CS_CZ`)|LOWERCASE_GROUPS_ONLY|Radixor|0.997139|HUNSPELL CZECH LUCENE FILTER|0.144369049|no|3|
|Danish (`DA_DK`)|ALL_WORDS|Radixor|0.996066|SNOWBALL DANISH LUCENE FILTER|0.058096771|no|3|
|Danish (`DA_DK`)|LOWERCASE_GROUPS_ONLY|Radixor|0.996305|SNOWBALL DANISH DIRECT|0.058230346|no|3|
|Dutch (`NL_NL`)|ALL_WORDS|Radixor|0.988661|SNOWBALL DUTCH DIRECT|0.261574077|no|4|
|Dutch (`NL_NL`)|LOWERCASE_GROUPS_ONLY|Radixor|0.989040|SNOWBALL DUTCH DIRECT|0.258544404|no|4|
|English (`US_UK`)|ALL_WORDS|Radixor|0.965159|ENGLISH LUCENE PORTER COPIED|0.010532535|no|11|
|English (`US_UK`)|LOWERCASE_GROUPS_ONLY|Radixor|0.965820|ENGLISH LUCENE PORTER COPIED|0.010920064|no|11|
|Finnish (`FI_FI`)|ALL_WORDS|Radixor|0.984594|SNOWBALL FINNISH LUCENE FILTER|0.244241861|no|4|
|Finnish (`FI_FI`)|LOWERCASE_GROUPS_ONLY|Radixor|0.988068|SNOWBALL FINNISH DIRECT|0.249668284|no|4|
|French (`FR_FR`)|ALL_WORDS|Radixor|0.956992|SNOWBALL FRENCH DIRECT|0.111730673|no|6|
|French (`FR_FR`)|LOWERCASE_GROUPS_ONLY|Radixor|0.957224|SNOWBALL FRENCH DIRECT|0.111809799|no|6|
|German (`DE_DE`)|ALL_WORDS|Radixor|0.907901|GERMAN CISTEM|0.027131083|no|8|
|German (`DE_DE`)|LOWERCASE_GROUPS_ONLY|Radixor|0.966157|GERMAN CISTEM|0.050868631|no|8|
|Hungarian (`HU_HU`)|ALL_WORDS|Radixor|0.995491|SNOWBALL HUNGARIAN LUCENE FILTER|0.172884951|no|4|
|Hungarian (`HU_HU`)|LOWERCASE_GROUPS_ONLY|Radixor|0.996163|SNOWBALL HUNGARIAN DIRECT|0.174455479|no|4|
|Italian (`IT_IT`)|ALL_WORDS|Radixor|0.996507|SNOWBALL ITALIAN DIRECT|0.130318040|no|4|
|Italian (`IT_IT`)|LOWERCASE_GROUPS_ONLY|Radixor|0.996512|SNOWBALL ITALIAN DIRECT|0.130307087|no|4|
|Norwegian Bokmal (`NB_NO`)|ALL_WORDS|Radixor|0.974783|SNOWBALL NORWEGIAN BOKMAL DIRECT|0.099819340|no|5|
|Norwegian Bokmal (`NB_NO`)|LOWERCASE_GROUPS_ONLY|Radixor|0.975000|SNOWBALL NORWEGIAN BOKMAL DIRECT|0.100008544|no|5|
|Norwegian Nynorsk (`NN_NO`)|ALL_WORDS|Radixor|0.935777|SNOWBALL NORWEGIAN NYNORSK DIRECT|0.076868986|no|3|
|Norwegian Nynorsk (`NN_NO`)|LOWERCASE_GROUPS_ONLY|Radixor|0.935853|SNOWBALL NORWEGIAN NYNORSK DIRECT|0.076816096|no|3|
|Persian (`FA_IR`)|ALL_WORDS|Radixor|0.974922|PERSIAN LUCENE PERSIAN STEM FILTER|0.472751327|no|2|
|Persian (`FA_IR`)|LOWERCASE_GROUPS_ONLY|Radixor|0.974922|PERSIAN LUCENE PERSIAN STEM FILTER|0.472751327|no|2|
|Polish (`PL_PL`)|ALL_WORDS|Radixor|0.990388|POLISH LUCENE MORFOLOGIK FILTER|0.042233990|no|5|
|Polish (`PL_PL`)|LOWERCASE_GROUPS_ONLY|Radixor|0.990579|POLISH LUCENE MORFOLOGIK FILTER|0.042401633|no|5|
|Portuguese (`PT_PT`)|ALL_WORDS|Radixor|0.998502|SNOWBALL PORTUGUESE DIRECT|0.059701750|no|6|
|Portuguese (`PT_PT`)|LOWERCASE_GROUPS_ONLY|Radixor|0.998502|SNOWBALL PORTUGUESE DIRECT|0.059701750|no|6|
|Russian (`RU_RU`)|ALL_WORDS|Radixor|0.989827|SNOWBALL RUSSIAN LUCENE FILTER|0.154951419|no|4|
|Russian (`RU_RU`)|LOWERCASE_GROUPS_ONLY|Radixor|0.989852|SNOWBALL RUSSIAN DIRECT|0.154997931|no|4|
|Spanish (`ES_ES`)|ALL_WORDS|Radixor|0.989295|SNOWBALL SPANISH LUCENE FILTER|0.336680479|no|7|
|Spanish (`ES_ES`)|LOWERCASE_GROUPS_ONLY|Radixor|0.989429|SNOWBALL SPANISH DIRECT|0.336708826|no|7|
|Swedish (`SV_SE`)|ALL_WORDS|Radixor|0.974636|SNOWBALL SWEDISH DIRECT|0.167101450|no|5|
|Swedish (`SV_SE`)|LOWERCASE_GROUPS_ONLY|Radixor|0.974584|SNOWBALL SWEDISH DIRECT|0.166984893|no|5|
|Ukrainian (`UK_UA`)|ALL_WORDS|Radixor|0.995343|UKRAINIAN LUCENE MORFOLOGIK FILTER|0.066574583|no|4|
|Ukrainian (`UK_UA`)|LOWERCASE_GROUPS_ONLY|Radixor|0.995342|UKRAINIAN LUCENE MORFOLOGIK FILTER|0.066590889|no|4|
|Yiddish (`YI`)|ALL_WORDS|Radixor|0.988241|SNOWBALL YIDDISH DIRECT|0.097253207|no|3|
|Yiddish (`YI`)|LOWERCASE_GROUPS_ONLY|Radixor|0.988241|SNOWBALL YIDDISH DIRECT|0.097253207|no|3|
### Secondary-metric trade-offs
Balanced-accuracy leadership does not imply leadership on every error trade-off. The table below lists all **15** deterministic primary-output language-mode-metric cases where a non-Radixor adapter has the best displayed value. Equal values are resolved by the authoritative row ordering and should be read as ties when the unrounded values are equal. Throughput leadership remains in the separate performance tables.
<details class="quality-details" markdown="1"><summary>Non-Radixor secondary-metric leaders</summary>
| Language | Dictionary mode | Metric | Leader | Value |
|---|---|---|---|---:|
|English|ALL_WORDS|Over-stemming percentage|ENGLISH LUCENE POSSESSIVE FILTER|0.000604|
|English|LOWERCASE_GROUPS_ONLY|Over-stemming percentage|ENGLISH LUCENE POSSESSIVE FILTER|0.000653|
|French|ALL_WORDS|Over-stemming percentage|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|0.000177|
|French|LOWERCASE_GROUPS_ONLY|Over-stemming percentage|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|0.000166|
|German|LOWERCASE_GROUPS_ONLY|Over-stemming percentage|GERMAN LUCENE GERMAN MINIMAL STEM FILTER|0.000188|
|Italian|ALL_WORDS|Over-stemming percentage|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|0.000020|
|Italian|LOWERCASE_GROUPS_ONLY|Over-stemming percentage|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|0.000020|
|Persian|ALL_WORDS|Over-stemming percentage|PERSIAN LUCENE PERSIAN STEM FILTER|0.002652|
|Persian|LOWERCASE_GROUPS_ONLY|Over-stemming percentage|PERSIAN LUCENE PERSIAN STEM FILTER|0.002652|
|Portuguese|ALL_WORDS|Over-stemming percentage|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|0.000003|
|Portuguese|LOWERCASE_GROUPS_ONLY|Over-stemming percentage|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|0.000003|
|Spanish|ALL_WORDS|Over-stemming percentage|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|0.000013|
|Spanish|LOWERCASE_GROUPS_ONLY|Over-stemming percentage|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|0.000012|
|Ukrainian|ALL_WORDS|Over-stemming percentage|HUNSPELL UKRAINIAN LUCENE FILTER|0.000783|
|Ukrainian|LOWERCASE_GROUPS_ONLY|Over-stemming percentage|HUNSPELL UKRAINIAN LUCENE FILTER|0.000784|
</details>
### Win, tie, and placement summary
Counts use `PRIMARY_OUTPUT` only and retain each adapter configuration as a separate stemmer except that language-specific Radixor identifiers are combined as Radixor. Coverage is displayed explicitly; unsupported languages are absent, not losses.
<details class="quality-details" markdown="1"><summary>ALL_WORDS placements</summary>
| Stemmer | Evaluated languages | Wins | Exact first-place ties | Top-three placements | Average rank | Median rank |
|---|---:|---:|---:|---:|---:|---:|
|Radixor|19|19|0|19|1.000|1.000|
|CZECH LUCENE CZECH STEM FILTER|1|0|0|1|3.000|3.000|
|ENGLISH LUCENE KSTEM FILTER|1|0|0|0|8.000|8.000|
|ENGLISH LUCENE MINIMAL FILTER|1|0|0|0|9.000|9.000|
|ENGLISH LUCENE PORTER COPIED|1|0|0|1|2.000|2.000|
|ENGLISH LUCENE PORTER FILTER|1|0|0|1|3.000|3.000|
|ENGLISH LUCENE POSSESSIVE FILTER|1|0|0|0|11.000|11.000|
|ENGLISH OPENNLP PORTER|1|0|0|0|4.000|4.000|
|ENGLISH PAICE HUSK LANCASTER|1|0|0|0|7.000|7.000|
|ENGLISH SNOWBALL ORIGINAL PORTER|1|0|0|0|6.000|6.000|
|ENGLISH SNOWBALL PORTER2|1|0|0|0|5.000|5.000|
|FINNISH LUCENE FINNISH LIGHT STEM FILTER|1|0|0|0|4.000|4.000|
|FRENCH LUCENE FRENCH LIGHT STEM FILTER|1|0|0|0|5.000|5.000|
|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|1|0|0|0|6.000|6.000|
|GERMAN CISTEM|1|0|0|1|2.000|2.000|
|GERMAN LUCENE GERMAN LIGHT STEM FILTER|1|0|0|0|5.000|5.000|
|GERMAN LUCENE GERMAN MINIMAL STEM FILTER|1|0|0|0|8.000|8.000|
|GERMAN LUCENE GERMAN STEM FILTER|1|0|0|0|6.000|6.000|
|HUNGARIAN LUCENE HUNGARIAN LIGHT STEM FILTER|1|0|0|0|4.000|4.000|
|HUNSPELL CZECH LUCENE FILTER|1|0|0|1|2.000|2.000|
|HUNSPELL DUTCH LUCENE FILTER|1|0|0|1|3.000|3.000|
|HUNSPELL ENGLISH LUCENE FILTER|1|0|0|0|10.000|10.000|
|HUNSPELL FRENCH LUCENE FILTER|1|0|0|0|4.000|4.000|
|HUNSPELL GERMAN LUCENE FILTER|1|0|0|0|7.000|7.000|
|HUNSPELL POLISH LUCENE FILTER|1|0|0|1|3.000|3.000|
|HUNSPELL SPANISH LUCENE FILTER|1|0|0|0|4.000|4.000|
|HUNSPELL UKRAINIAN LUCENE FILTER|1|0|0|0|4.000|4.000|
|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|1|0|0|0|4.000|4.000|
|NORWEGIAN BOKMAL LUCENE NORWEGIAN LIGHT STEM FILTER|1|0|0|0|4.000|4.000|
|NORWEGIAN BOKMAL LUCENE NORWEGIAN MINIMAL STEM FILTER|1|0|0|0|5.000|5.000|
|PERSIAN LUCENE PERSIAN STEM FILTER|1|0|0|1|2.000|2.000|
|POLISH LUCENE MORFOLOGIK FILTER|1|0|0|1|2.000|2.000|
|POLISH LUCENE STEMPEL DIRECT|1|0|0|0|4.000|4.000|
|POLISH LUCENE STEMPEL FILTER|1|0|0|0|5.000|5.000|
|PORTUGUESE LUCENE PORTUGUESE LIGHT STEM FILTER|1|0|0|0|5.000|5.000|
|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|1|0|0|0|6.000|6.000|
|PORTUGUESE LUCENE PORTUGUESE STEM FILTER|1|0|0|0|4.000|4.000|
|RUSSIAN LUCENE RUSSIAN LIGHT STEM FILTER|1|0|0|0|4.000|4.000|
|SNOWBALL DANISH DIRECT|1|0|0|1|3.000|3.000|
|SNOWBALL DANISH LUCENE FILTER|1|0|0|1|2.000|2.000|
|SNOWBALL DUTCH DIRECT|1|0|0|1|2.000|2.000|
|SNOWBALL DUTCH LUCENE FILTER|1|0|0|0|4.000|4.000|
|SNOWBALL FINNISH DIRECT|1|0|0|1|3.000|3.000|
|SNOWBALL FINNISH LUCENE FILTER|1|0|0|1|2.000|2.000|
|SNOWBALL FRENCH DIRECT|1|0|0|1|2.000|2.000|
|SNOWBALL FRENCH LUCENE FILTER|1|0|0|1|3.000|3.000|
|SNOWBALL GERMAN DIRECT|1|0|0|1|3.000|3.000|
|SNOWBALL GERMAN LUCENE FILTER|1|0|0|0|4.000|4.000|
|SNOWBALL HUNGARIAN DIRECT|1|0|0|1|3.000|3.000|
|SNOWBALL HUNGARIAN LUCENE FILTER|1|0|0|1|2.000|2.000|
|SNOWBALL ITALIAN DIRECT|1|0|0|1|2.000|2.000|
|SNOWBALL ITALIAN LUCENE FILTER|1|0|0|1|3.000|3.000|
|SNOWBALL NORWEGIAN BOKMAL DIRECT|1|0|0|1|2.000|2.000|
|SNOWBALL NORWEGIAN BOKMAL LUCENE FILTER|1|0|0|1|3.000|3.000|
|SNOWBALL NORWEGIAN NYNORSK DIRECT|1|0|0|1|2.000|2.000|
|SNOWBALL NORWEGIAN NYNORSK LUCENE FILTER|1|0|0|1|3.000|3.000|
|SNOWBALL PORTUGUESE DIRECT|1|0|0|1|2.000|2.000|
|SNOWBALL PORTUGUESE LUCENE FILTER|1|0|0|1|3.000|3.000|
|SNOWBALL RUSSIAN DIRECT|1|0|0|1|3.000|3.000|
|SNOWBALL RUSSIAN LUCENE FILTER|1|0|0|1|2.000|2.000|
|SNOWBALL SPANISH DIRECT|1|0|0|1|3.000|3.000|
|SNOWBALL SPANISH LUCENE FILTER|1|0|0|1|2.000|2.000|
|SNOWBALL SWEDISH DIRECT|1|0|0|1|2.000|2.000|
|SNOWBALL SWEDISH LUCENE FILTER|1|0|0|1|3.000|3.000|
|SNOWBALL YIDDISH DIRECT|1|0|0|1|2.000|2.000|
|SNOWBALL YIDDISH LUCENE FILTER|1|0|0|1|3.000|3.000|
|SPANISH LUCENE SPANISH LIGHT STEM FILTER|1|0|0|0|5.000|5.000|
|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|1|0|0|0|7.000|7.000|
|SPANISH LUCENE SPANISH PLURAL STEM FILTER|1|0|0|0|6.000|6.000|
|SWEDISH LUCENE SWEDISH LIGHT STEM FILTER|1|0|0|0|5.000|5.000|
|SWEDISH LUCENE SWEDISH MINIMAL STEM FILTER|1|0|0|0|4.000|4.000|
|UKRAINIAN LUCENE MORFOLOGIK FILTER|1|0|0|1|2.000|2.000|
|UKRAINIAN MORFOLOGIK DIRECT|1|0|0|1|3.000|3.000|
</details>
<details class="quality-details" markdown="1"><summary>LOWERCASE_GROUPS_ONLY placements</summary>
| Stemmer | Evaluated languages | Wins | Exact first-place ties | Top-three placements | Average rank | Median rank |
|---|---:|---:|---:|---:|---:|---:|
|Radixor|19|19|0|19|1.000|1.000|
|CZECH LUCENE CZECH STEM FILTER|1|0|0|1|3.000|3.000|
|ENGLISH LUCENE KSTEM FILTER|1|0|0|0|8.000|8.000|
|ENGLISH LUCENE MINIMAL FILTER|1|0|0|0|9.000|9.000|
|ENGLISH LUCENE PORTER COPIED|1|0|0|1|2.000|2.000|
|ENGLISH LUCENE PORTER FILTER|1|0|0|1|3.000|3.000|
|ENGLISH LUCENE POSSESSIVE FILTER|1|0|0|0|11.000|11.000|
|ENGLISH OPENNLP PORTER|1|0|0|0|4.000|4.000|
|ENGLISH PAICE HUSK LANCASTER|1|0|0|0|7.000|7.000|
|ENGLISH SNOWBALL ORIGINAL PORTER|1|0|0|0|6.000|6.000|
|ENGLISH SNOWBALL PORTER2|1|0|0|0|5.000|5.000|
|FINNISH LUCENE FINNISH LIGHT STEM FILTER|1|0|0|0|4.000|4.000|
|FRENCH LUCENE FRENCH LIGHT STEM FILTER|1|0|0|0|5.000|5.000|
|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|1|0|0|0|6.000|6.000|
|GERMAN CISTEM|1|0|0|1|2.000|2.000|
|GERMAN LUCENE GERMAN LIGHT STEM FILTER|1|0|0|0|5.000|5.000|
|GERMAN LUCENE GERMAN MINIMAL STEM FILTER|1|0|0|0|8.000|8.000|
|GERMAN LUCENE GERMAN STEM FILTER|1|0|0|0|6.000|6.000|
|HUNGARIAN LUCENE HUNGARIAN LIGHT STEM FILTER|1|0|0|0|4.000|4.000|
|HUNSPELL CZECH LUCENE FILTER|1|0|0|1|2.000|2.000|
|HUNSPELL DUTCH LUCENE FILTER|1|0|0|1|3.000|3.000|
|HUNSPELL ENGLISH LUCENE FILTER|1|0|0|0|10.000|10.000|
|HUNSPELL FRENCH LUCENE FILTER|1|0|0|0|4.000|4.000|
|HUNSPELL GERMAN LUCENE FILTER|1|0|0|0|7.000|7.000|
|HUNSPELL POLISH LUCENE FILTER|1|0|0|1|3.000|3.000|
|HUNSPELL SPANISH LUCENE FILTER|1|0|0|0|4.000|4.000|
|HUNSPELL UKRAINIAN LUCENE FILTER|1|0|0|0|4.000|4.000|
|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|1|0|0|0|4.000|4.000|
|NORWEGIAN BOKMAL LUCENE NORWEGIAN LIGHT STEM FILTER|1|0|0|0|4.000|4.000|
|NORWEGIAN BOKMAL LUCENE NORWEGIAN MINIMAL STEM FILTER|1|0|0|0|5.000|5.000|
|PERSIAN LUCENE PERSIAN STEM FILTER|1|0|0|1|2.000|2.000|
|POLISH LUCENE MORFOLOGIK FILTER|1|0|0|1|2.000|2.000|
|POLISH LUCENE STEMPEL DIRECT|1|0|0|0|4.000|4.000|
|POLISH LUCENE STEMPEL FILTER|1|0|0|0|5.000|5.000|
|PORTUGUESE LUCENE PORTUGUESE LIGHT STEM FILTER|1|0|0|0|5.000|5.000|
|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|1|0|0|0|6.000|6.000|
|PORTUGUESE LUCENE PORTUGUESE STEM FILTER|1|0|0|0|4.000|4.000|
|RUSSIAN LUCENE RUSSIAN LIGHT STEM FILTER|1|0|0|0|4.000|4.000|
|SNOWBALL DANISH DIRECT|1|0|0|1|2.000|2.000|
|SNOWBALL DANISH LUCENE FILTER|1|0|0|1|3.000|3.000|
|SNOWBALL DUTCH DIRECT|1|0|0|1|2.000|2.000|
|SNOWBALL DUTCH LUCENE FILTER|1|0|0|0|4.000|4.000|
|SNOWBALL FINNISH DIRECT|1|0|0|1|2.000|2.000|
|SNOWBALL FINNISH LUCENE FILTER|1|0|0|1|3.000|3.000|
|SNOWBALL FRENCH DIRECT|1|0|0|1|2.000|2.000|
|SNOWBALL FRENCH LUCENE FILTER|1|0|0|1|3.000|3.000|
|SNOWBALL GERMAN DIRECT|1|0|0|1|3.000|3.000|
|SNOWBALL GERMAN LUCENE FILTER|1|0|0|0|4.000|4.000|
|SNOWBALL HUNGARIAN DIRECT|1|0|0|1|2.000|2.000|
|SNOWBALL HUNGARIAN LUCENE FILTER|1|0|0|1|3.000|3.000|
|SNOWBALL ITALIAN DIRECT|1|0|0|1|2.000|2.000|
|SNOWBALL ITALIAN LUCENE FILTER|1|0|0|1|3.000|3.000|
|SNOWBALL NORWEGIAN BOKMAL DIRECT|1|0|0|1|2.000|2.000|
|SNOWBALL NORWEGIAN BOKMAL LUCENE FILTER|1|0|0|1|3.000|3.000|
|SNOWBALL NORWEGIAN NYNORSK DIRECT|1|0|0|1|2.000|2.000|
|SNOWBALL NORWEGIAN NYNORSK LUCENE FILTER|1|0|0|1|3.000|3.000|
|SNOWBALL PORTUGUESE DIRECT|1|0|0|1|2.000|2.000|
|SNOWBALL PORTUGUESE LUCENE FILTER|1|0|0|1|3.000|3.000|
|SNOWBALL RUSSIAN DIRECT|1|0|0|1|2.000|2.000|
|SNOWBALL RUSSIAN LUCENE FILTER|1|0|0|1|3.000|3.000|
|SNOWBALL SPANISH DIRECT|1|0|0|1|2.000|2.000|
|SNOWBALL SPANISH LUCENE FILTER|1|0|0|1|3.000|3.000|
|SNOWBALL SWEDISH DIRECT|1|0|0|1|2.000|2.000|
|SNOWBALL SWEDISH LUCENE FILTER|1|0|0|1|3.000|3.000|
|SNOWBALL YIDDISH DIRECT|1|0|0|1|2.000|2.000|
|SNOWBALL YIDDISH LUCENE FILTER|1|0|0|1|3.000|3.000|
|SPANISH LUCENE SPANISH LIGHT STEM FILTER|1|0|0|0|5.000|5.000|
|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|1|0|0|0|7.000|7.000|
|SPANISH LUCENE SPANISH PLURAL STEM FILTER|1|0|0|0|6.000|6.000|
|SWEDISH LUCENE SWEDISH LIGHT STEM FILTER|1|0|0|0|5.000|5.000|
|SWEDISH LUCENE SWEDISH MINIMAL STEM FILTER|1|0|0|0|4.000|4.000|
|UKRAINIAN LUCENE MORFOLOGIK FILTER|1|0|0|1|2.000|2.000|
|UKRAINIAN MORFOLOGIK DIRECT|1|0|0|1|3.000|3.000|
</details>
### Radixor full-coverage aggregates
These aggregates cover all 19 documented languages. Macro balanced accuracy gives each language equal weight. Micro metrics first sum raw pair counts across languages. Unsupported third-party languages are never inserted as zero results, so this full-coverage table is not presented as a cross-stemmer common-language ranking.
| Dictionary mode | Languages | Macro balanced accuracy | Micro balanced accuracy | Micro precision | Micro recall | Micro F1 |
|---|---:|---:|---:|---:|---:|---:|
|ALL_WORDS|19|0.978929|0.987664|0.975113|0.975328|0.975221|
|LOWERCASE_GROUPS_ONLY|19|0.982354|0.989366|0.975322|0.978734|0.977025|
### Reproducible data
- [Machine-readable quality snapshot](data/stemming-quality.csv)
- SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
- [Linguistic quality methodology](reference/linguistic-quality.md)
- [Tested stemmer inventory](reference/tested-stemmers.md)
- [Reproducibility and raw data](reference/reproducibility.md)
- Pearson and Spearman correlation files are generated under `build/reports/stemming-quality/`; they are separated by dictionary mode and output policy. Correlation does not establish metric equivalence.
<!-- STEMMING-QUALITY-OVERVIEW:END -->

View File

@@ -51,3 +51,368 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs. - Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available. - Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees. - Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
<!-- STEMMING-QUALITY:START -->
## Stemming Quality
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `CS_CZ` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
### Evaluation Scope and Key Findings
The dictionary resource is `src/main/resources/cs_cz/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.996565** among 3 deterministic stemmers. The runner-up is `HUNSPELL CZECH LUCENE FILTER` at 0.853752, a difference of 0.142813. This rank does not imply leadership in throughput or every secondary metric.
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.997139** among 3 deterministic stemmers. The runner-up is `HUNSPELL CZECH LUCENE FILTER` at 0.852770, a difference of 0.144369. This rank does not imply leadership in throughput or every secondary metric.
### `ALL_WORDS`
This mode contains **7 result rows**, **3 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.996565|3867 / 1334876815 (0.000290%)|2073 / 301835 (0.686799%)|0.988432|0.990189|0.990191|
|2|HUNSPELL CZECH LUCENE FILTER|PRIMARY_OUTPUT|0.853752|11408 / 1334876815 (0.000855%)|88283 / 301835 (29.248762%)|0.888560|0.810759|0.819499|
|3|CZECH LUCENE CZECH STEM FILTER|PRIMARY_OUTPUT|0.793614|14480 / 1334876815 (0.001085%)|124586 / 301835 (41.276194%)|0.829234|0.718241|0.736765|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.987264|0.993132|0.999997|0.996565|0.999996|0.000004|
|2|HUNSPELL CZECH LUCENE FILTER|PRIMARY_OUTPUT|0.949289|0.707512|0.999991|0.853752|0.999925|0.000075|
|3|CZECH LUCENE CZECH STEM FILTER|PRIMARY_OUTPUT|0.924477|0.587238|0.999989|0.793614|0.999896|0.000104|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.988432|0.990189|0.991953|0.980569|0.990194|0.990191|
|2|HUNSPELL CZECH LUCENE FILTER|PRIMARY_OUTPUT|0.888560|0.810759|0.745486|0.681745|0.819533|0.819499|
|3|CZECH LUCENE CZECH STEM FILTER|PRIMARY_OUTPUT|0.829234|0.718241|0.633453|0.560356|0.736809|0.736765|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.990187|0.998733|0.998686|0.998709|0.998709|
|2|HUNSPELL CZECH LUCENE FILTER|PRIMARY_OUTPUT|0.810723|0.995777|0.952852|0.973842|0.973842|
|3|CZECH LUCENE CZECH STEM FILTER|PRIMARY_OUTPUT|0.718192|0.993801|0.944977|0.968774|0.968774|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|299762|3867|2073|1334872948|3867 / 1334876815|2073 / 301835|
|2|HUNSPELL CZECH LUCENE FILTER|PRIMARY_OUTPUT|213552|11408|88283|1334865407|11408 / 1334876815|88283 / 301835|
|3|CZECH LUCENE CZECH STEM FILTER|PRIMARY_OUTPUT|177249|14480|124586|1334862335|14480 / 1334876815|124586 / 301835|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 1334876815 (0.000000%)|0 / 301835 (0.000000%)|1.000000|1.000000|1.000000|
|2|HUNSPELL CZECH LUCENE FILTER|ANY_CANDIDATE|0.871577|10102 / 1334876815 (0.000757%)|77523 / 301835 (25.683900%)|0.904855|0.836596|0.843258|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|2|HUNSPELL CZECH LUCENE FILTER|ANY_CANDIDATE|0.956905|0.743161|0.999992|0.871577|0.999934|0.000066|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|2|HUNSPELL CZECH LUCENE FILTER|ANY_CANDIDATE|0.904855|0.836596|0.777914|0.719094|0.843288|0.843258|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|2|HUNSPELL CZECH LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|301835|0|0|1334876815|0 / 1334876815|0 / 301835|
|2|HUNSPELL CZECH LUCENE FILTER|ANY_CANDIDATE|224312|10102|77523|1334866713|10102 / 1334876815|77523 / 301835|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999998|5850 / 1334876815 (0.000438%)|0 / 301835 (0.000000%)|0.984732|0.990402|0.990446|
|2|HUNSPELL CZECH LUCENE FILTER|ALL_CANDIDATES|0.871575|13917 / 1334876815 (0.001043%)|77523 / 301835 (25.683900%)|0.893851|0.830687|0.836477|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.980987|1.000000|0.999996|0.999998|0.999996|0.000004|
|2|HUNSPELL CZECH LUCENE FILTER|ALL_CANDIDATES|0.941581|0.743161|0.999990|0.871575|0.999932|0.000068|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.984732|0.990402|0.996139|0.980987|0.990448|0.990446|
|2|HUNSPELL CZECH LUCENE FILTER|ALL_CANDIDATES|0.893851|0.830687|0.775861|0.710406|0.836509|0.836477|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|2|HUNSPELL CZECH LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|301835|5850|0|1334870965|5850 / 1334876815|0 / 301835|
|2|HUNSPELL CZECH LUCENE FILTER|ALL_CANDIDATES|224312|13917|77523|1334862898|13917 / 1334876815|77523 / 301835|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|2073|3867|1983|596|1.153340%|4|52319|
|HUNSPELL CZECH LUCENE FILTER|10760|1306|2509|3317|6.418840%|5|55596|
### `LOWERCASE_GROUPS_ONLY`
This mode contains **7 result rows**, **3 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.997139|3863 / 1298544215 (0.000297%)|1709 / 298813 (0.571930%)|0.988580|0.990710|0.990714|
|2|HUNSPELL CZECH LUCENE FILTER|PRIMARY_OUTPUT|0.852770|11239 / 1298544215 (0.000866%)|87986 / 298813 (29.445171%)|0.888009|0.809505|0.818403|
|3|CZECH LUCENE CZECH STEM FILTER|PRIMARY_OUTPUT|0.791794|13950 / 1298544215 (0.001074%)|124426 / 298813 (41.640089%)|0.828709|0.715948|0.735055|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.987165|0.994281|0.999997|0.997139|0.999996|0.000004|
|2|HUNSPELL CZECH LUCENE FILTER|PRIMARY_OUTPUT|0.949389|0.705548|0.999991|0.852770|0.999924|0.000076|
|3|CZECH LUCENE CZECH STEM FILTER|PRIMARY_OUTPUT|0.925931|0.583599|0.999989|0.791794|0.999893|0.000107|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.988580|0.990710|0.992849|0.981591|0.990716|0.990714|
|2|HUNSPELL CZECH LUCENE FILTER|PRIMARY_OUTPUT|0.888009|0.809505|0.743753|0.679973|0.818437|0.818403|
|3|CZECH LUCENE CZECH STEM FILTER|PRIMARY_OUTPUT|0.828709|0.715948|0.630198|0.557569|0.735100|0.735055|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.990708|0.998726|0.999030|0.998878|0.998878|
|2|HUNSPELL CZECH LUCENE FILTER|PRIMARY_OUTPUT|0.809467|0.995812|0.952394|0.973619|0.973619|
|3|CZECH LUCENE CZECH STEM FILTER|PRIMARY_OUTPUT|0.715897|0.993897|0.944297|0.968463|0.968463|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|297104|3863|1709|1298540352|3863 / 1298544215|1709 / 298813|
|2|HUNSPELL CZECH LUCENE FILTER|PRIMARY_OUTPUT|210827|11239|87986|1298532976|11239 / 1298544215|87986 / 298813|
|3|CZECH LUCENE CZECH STEM FILTER|PRIMARY_OUTPUT|174387|13950|124426|1298530265|13950 / 1298544215|124426 / 298813|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 1298544215 (0.000000%)|0 / 298813 (0.000000%)|1.000000|1.000000|1.000000|
|2|HUNSPELL CZECH LUCENE FILTER|ANY_CANDIDATE|0.870432|10028 / 1298544215 (0.000772%)|77431 / 298813 (25.912862%)|0.904004|0.835052|0.841852|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|2|HUNSPELL CZECH LUCENE FILTER|ANY_CANDIDATE|0.956666|0.740871|0.999992|0.870432|0.999933|0.000067|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|2|HUNSPELL CZECH LUCENE FILTER|ANY_CANDIDATE|0.904004|0.835052|0.775874|0.716815|0.841883|0.841852|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|2|HUNSPELL CZECH LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|298813|0|0|1298544215|0 / 1298544215|0 / 298813|
|2|HUNSPELL CZECH LUCENE FILTER|ANY_CANDIDATE|221382|10028|77431|1298534187|10028 / 1298544215|77431 / 298813|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999998|5782 / 1298544215 (0.000445%)|0 / 298813 (0.000000%)|0.984756|0.990418|0.990461|
|2|HUNSPELL CZECH LUCENE FILTER|ALL_CANDIDATES|0.870430|13601 / 1298544215 (0.001047%)|77431 / 298813 (25.912862%)|0.893574|0.829463|0.835425|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.981017|1.000000|0.999996|0.999998|0.999996|0.000004|
|2|HUNSPELL CZECH LUCENE FILTER|ALL_CANDIDATES|0.942119|0.740871|0.999990|0.870430|0.999930|0.000070|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.984756|0.990418|0.996145|0.981017|0.990463|0.990461|
|2|HUNSPELL CZECH LUCENE FILTER|ALL_CANDIDATES|0.893574|0.829463|0.773936|0.708617|0.835457|0.835425|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|2|HUNSPELL CZECH LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|298813|5782|0|1298538433|5782 / 1298544215|0 / 298813|
|2|HUNSPELL CZECH LUCENE FILTER|ALL_CANDIDATES|221382|13601|77431|1298530614|13601 / 1298544215|77431 / 298813|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|1709|3863|1919|540|1.059488%|4|51543|
|HUNSPELL CZECH LUCENE FILTER|10555|1211|2362|3237|6.351044%|5|54804|
### Output Policies and Metric Definitions
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
- Jaccard index: `TP / (TP + FP + FN)`.
- FowlkesMallows index: `sqrt(precision * recall)`.
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
### Provenance
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
- Evaluation command: `./gradlew stemmingQuality`
- Dictionary language: `CS_CZ`
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
<!-- STEMMING-QUALITY:END -->

View File

@@ -51,3 +51,346 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs. - Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available. - Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees. - Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
<!-- STEMMING-QUALITY:START -->
## Stemming Quality
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `DA_DK` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
### Evaluation Scope and Key Findings
The dictionary resource is `src/main/resources/da_dk/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.996066** among 3 deterministic stemmers. The runner-up is `SNOWBALL DANISH LUCENE FILTER` at 0.937969, a difference of 0.058097. This rank does not imply leadership in throughput or every secondary metric.
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.996305** among 3 deterministic stemmers. The runner-up is `SNOWBALL DANISH DIRECT` at 0.938074, a difference of 0.058230. This rank does not imply leadership in throughput or every secondary metric.
### `ALL_WORDS`
This mode contains **5 result rows**, **3 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.996066|1165 / 394111186 (0.000296%)|707 / 89895 (0.786473%)|0.988108|0.989614|0.989615|
|2|SNOWBALL DANISH LUCENE FILTER|PRIMARY_OUTPUT|0.937969|6507 / 394111186 (0.001651%)|11151 / 89895 (12.404472%)|0.913718|0.899181|0.899475|
|3|SNOWBALL DANISH DIRECT|PRIMARY_OUTPUT|0.937903|6341 / 394111186 (0.001609%)|11163 / 89895 (12.417821%)|0.915090|0.899959|0.900279|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.987106|0.992135|0.999997|0.996066|0.999995|0.000005|
|2|SNOWBALL DANISH LUCENE FILTER|PRIMARY_OUTPUT|0.923672|0.875955|0.999983|0.937969|0.999955|0.000045|
|3|SNOWBALL DANISH DIRECT|PRIMARY_OUTPUT|0.925464|0.875822|0.999984|0.937903|0.999956|0.000044|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.988108|0.989614|0.991125|0.979442|0.989618|0.989615|
|2|SNOWBALL DANISH LUCENE FILTER|PRIMARY_OUTPUT|0.913718|0.899181|0.885100|0.816830|0.899498|0.899475|
|3|SNOWBALL DANISH DIRECT|PRIMARY_OUTPUT|0.915090|0.899959|0.885320|0.818114|0.900301|0.900279|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.989612|0.998466|0.998719|0.998592|0.998592|
|2|SNOWBALL DANISH LUCENE FILTER|PRIMARY_OUTPUT|0.899159|0.994053|0.978603|0.986268|0.986268|
|3|SNOWBALL DANISH DIRECT|PRIMARY_OUTPUT|0.899937|0.994196|0.978579|0.986326|0.986326|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|89188|1165|707|394110021|1165 / 394111186|707 / 89895|
|2|SNOWBALL DANISH LUCENE FILTER|PRIMARY_OUTPUT|78744|6507|11151|394104679|6507 / 394111186|11151 / 89895|
|3|SNOWBALL DANISH DIRECT|PRIMARY_OUTPUT|78732|6341|11163|394104845|6341 / 394111186|11163 / 89895|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 394111186 (0.000000%)|0 / 89895 (0.000000%)|1.000000|1.000000|1.000000|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|89895|0|0|394111186|0 / 394111186|0 / 89895|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999998|1849 / 394111186 (0.000469%)|0 / 89895 (0.000000%)|0.983812|0.989820|0.989869|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.979846|1.000000|0.999995|0.999998|0.999995|0.000005|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.983812|0.989820|0.995903|0.979846|0.989872|0.989869|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|89895|1849|0|394109337|1849 / 394111186|0 / 89895|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|707|1165|684|323|1.150326%|3|28405|
### `LOWERCASE_GROUPS_ONLY`
This mode contains **5 result rows**, **3 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.996305|1165 / 392820788 (0.000297%)|663 / 89740 (0.738801%)|0.988190|0.989843|0.989845|
|2|SNOWBALL DANISH DIRECT|PRIMARY_OUTPUT|0.938074|6341 / 392820788 (0.001614%)|11113 / 89740 (12.383552%)|0.915093|0.900096|0.900410|
|3|SNOWBALL DANISH LUCENE FILTER|PRIMARY_OUTPUT|0.938074|6341 / 392820788 (0.001614%)|11113 / 89740 (12.383552%)|0.915093|0.900096|0.900410|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.987090|0.992612|0.999997|0.996305|0.999995|0.000005|
|2|SNOWBALL DANISH DIRECT|PRIMARY_OUTPUT|0.925372|0.876164|0.999984|0.938074|0.999956|0.000044|
|3|SNOWBALL DANISH LUCENE FILTER|PRIMARY_OUTPUT|0.925372|0.876164|0.999984|0.938074|0.999956|0.000044|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.988190|0.989843|0.991503|0.979891|0.989847|0.989845|
|2|SNOWBALL DANISH DIRECT|PRIMARY_OUTPUT|0.915093|0.900096|0.885583|0.818341|0.900432|0.900410|
|3|SNOWBALL DANISH LUCENE FILTER|PRIMARY_OUTPUT|0.915093|0.900096|0.885583|0.818341|0.900432|0.900410|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.989841|0.998463|0.998812|0.998637|0.998637|
|2|SNOWBALL DANISH DIRECT|PRIMARY_OUTPUT|0.900074|0.994185|0.978644|0.986354|0.986354|
|3|SNOWBALL DANISH LUCENE FILTER|PRIMARY_OUTPUT|0.900074|0.994185|0.978644|0.986354|0.986354|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|89077|1165|663|392819623|1165 / 392820788|663 / 89740|
|2|SNOWBALL DANISH DIRECT|PRIMARY_OUTPUT|78627|6341|11113|392814447|6341 / 392820788|11113 / 89740|
|3|SNOWBALL DANISH LUCENE FILTER|PRIMARY_OUTPUT|78627|6341|11113|392814447|6341 / 392820788|11113 / 89740|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 392820788 (0.000000%)|0 / 89740 (0.000000%)|1.000000|1.000000|1.000000|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|89740|0|0|392820788|0 / 392820788|0 / 89740|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999998|1849 / 392820788 (0.000471%)|0 / 89740 (0.000000%)|0.983784|0.989803|0.989852|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.979812|1.000000|0.999995|0.999998|0.999995|0.000005|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.983784|0.989803|0.995896|0.979812|0.989855|0.989852|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|89740|1849|0|392818939|1849 / 392820788|0 / 89740|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|663|1165|684|315|1.123676%|3|28351|
### Output Policies and Metric Definitions
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
- Jaccard index: `TP / (TP + FP + FN)`.
- FowlkesMallows index: `sqrt(precision * recall)`.
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
### Provenance
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
- Evaluation command: `./gradlew stemmingQuality`
- Dictionary language: `DA_DK`
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
<!-- STEMMING-QUALITY:END -->

View File

@@ -53,3 +53,378 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs. - Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available. - Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees. - Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
<!-- STEMMING-QUALITY:START -->
## Stemming Quality
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `NL_NL` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
### Evaluation Scope and Key Findings
The dictionary resource is `src/main/resources/nl_nl/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.988661** among 4 deterministic stemmers. The runner-up is `SNOWBALL DUTCH DIRECT` at 0.727087, a difference of 0.261574. This rank does not imply leadership in throughput or every secondary metric.
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.989040** among 4 deterministic stemmers. The runner-up is `SNOWBALL DUTCH DIRECT` at 0.730495, a difference of 0.258544. This rank does not imply leadership in throughput or every secondary metric.
### `ALL_WORDS`
This mode contains **8 result rows**, **4 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.988661|1214 / 350437960 (0.000346%)|1464 / 64566 (2.267447%)|0.980362|0.979221|0.979219|
|2|SNOWBALL DUTCH DIRECT|PRIMARY_OUTPUT|0.727087|4382 / 350437960 (0.001250%)|35241 / 64566 (54.581359%)|0.735353|0.596807|0.628557|
|3|HUNSPELL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.643123|1333 / 350437960 (0.000380%)|46084 / 64566 (71.375027%)|0.642512|0.438061|0.516674|
|4|SNOWBALL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.618497|1588 / 350437960 (0.000453%)|49264 / 64566 (76.300220%)|0.579068|0.375712|0.463333|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.981124|0.977326|0.999997|0.988661|0.999992|0.000008|
|2|SNOWBALL DUTCH DIRECT|PRIMARY_OUTPUT|0.869997|0.454186|0.999987|0.727087|0.999887|0.000113|
|3|HUNSPELL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.932728|0.286250|0.999996|0.643123|0.999865|0.000135|
|4|SNOWBALL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.905980|0.236998|0.999995|0.618497|0.999855|0.000145|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.980362|0.979221|0.978083|0.959289|0.979223|0.979219|
|2|SNOWBALL DUTCH DIRECT|PRIMARY_OUTPUT|0.735353|0.596807|0.502190|0.425321|0.628602|0.628557|
|3|HUNSPELL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.642512|0.438061|0.332316|0.280459|0.516714|0.516674|
|4|SNOWBALL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.579068|0.375712|0.278062|0.231309|0.463374|0.463333|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.979217|0.997464|0.997003|0.997234|0.997234|
|2|SNOWBALL DUTCH DIRECT|PRIMARY_OUTPUT|0.596756|0.992815|0.917346|0.953590|0.953590|
|3|HUNSPELL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.438012|0.996932|0.889026|0.939892|0.939892|
|4|SNOWBALL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.375664|0.995828|0.888410|0.939057|0.939057|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|63102|1214|1464|350436746|1214 / 350437960|1464 / 64566|
|2|SNOWBALL DUTCH DIRECT|PRIMARY_OUTPUT|29325|4382|35241|350433578|4382 / 350437960|35241 / 64566|
|3|HUNSPELL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|18482|1333|46084|350436627|1333 / 350437960|46084 / 64566|
|4|SNOWBALL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|15302|1588|49264|350436372|1588 / 350437960|49264 / 64566|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 350437960 (0.000000%)|0 / 64566 (0.000000%)|1.000000|1.000000|1.000000|
|2|HUNSPELL DUTCH LUCENE FILTER|ANY_CANDIDATE|0.665519|1164 / 350437960 (0.000332%)|43192 / 64566 (66.895889%)|0.690741|0.490770|0.560268|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|2|HUNSPELL DUTCH LUCENE FILTER|ANY_CANDIDATE|0.948354|0.331041|0.999997|0.665519|0.999873|0.000127|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|2|HUNSPELL DUTCH LUCENE FILTER|ANY_CANDIDATE|0.690741|0.490770|0.380588|0.325179|0.560307|0.560268|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|2|HUNSPELL DUTCH LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|64566|0|0|350437960|0 / 350437960|0 / 64566|
|2|HUNSPELL DUTCH LUCENE FILTER|ANY_CANDIDATE|21374|1164|43192|350436796|1164 / 350437960|43192 / 64566|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999996|2651 / 350437960 (0.000756%)|0 / 64566 (0.000000%)|0.968198|0.979884|0.980078|
|2|HUNSPELL DUTCH LUCENE FILTER|ALL_CANDIDATES|0.665518|1738 / 350437960 (0.000496%)|43192 / 64566 (66.895889%)|0.680640|0.487557|0.553265|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.960561|1.000000|0.999992|0.999996|0.999992|0.000008|
|2|HUNSPELL DUTCH LUCENE FILTER|ALL_CANDIDATES|0.924801|0.331041|0.999995|0.665518|0.999872|0.000128|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.968198|0.979884|0.991855|0.960561|0.980082|0.980078|
|2|HUNSPELL DUTCH LUCENE FILTER|ALL_CANDIDATES|0.680640|0.487557|0.379812|0.322364|0.553306|0.553265|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|2|HUNSPELL DUTCH LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|64566|2651|0|350435309|2651 / 350437960|0 / 64566|
|2|HUNSPELL DUTCH LUCENE FILTER|ALL_CANDIDATES|21374|1738|43192|350436222|1738 / 350437960|43192 / 64566|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|HUNSPELL DUTCH LUCENE FILTER|2892|169|405|1254|4.736186%|3|27763|
|Radixor|1464|1214|1437|572|2.160366%|3|27061|
### `LOWERCASE_GROUPS_ONLY`
This mode contains **8 result rows**, **4 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.989040|1214 / 329603856 (0.000368%)|1384 / 63147 (2.191711%)|0.980194|0.979401|0.979398|
|2|SNOWBALL DUTCH DIRECT|PRIMARY_OUTPUT|0.730495|4382 / 329603856 (0.001329%)|34036 / 63147 (53.899631%)|0.738412|0.602463|0.632953|
|3|HUNSPELL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.645159|1310 / 329603856 (0.000397%)|44814 / 63147 (70.967742%)|0.646808|0.442880|0.520498|
|4|SNOWBALL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.618546|1544 / 329603856 (0.000468%)|48175 / 63147 (76.290243%)|0.579362|0.375883|0.463566|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.980723|0.978083|0.999996|0.989040|0.999992|0.000008|
|2|SNOWBALL DUTCH DIRECT|PRIMARY_OUTPUT|0.869167|0.461004|0.999987|0.730495|0.999883|0.000117|
|3|HUNSPELL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.933310|0.290323|0.999996|0.645159|0.999860|0.000140|
|4|SNOWBALL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.906515|0.237098|0.999995|0.618546|0.999849|0.000151|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.980194|0.979401|0.978610|0.959634|0.979402|0.979398|
|2|SNOWBALL DUTCH DIRECT|PRIMARY_OUTPUT|0.738412|0.602463|0.508789|0.431089|0.633000|0.632953|
|3|HUNSPELL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.646808|0.442880|0.336718|0.284422|0.520539|0.520498|
|4|SNOWBALL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.579362|0.375883|0.278182|0.231439|0.463608|0.463566|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.979397|0.997373|0.997139|0.997256|0.997256|
|2|SNOWBALL DUTCH DIRECT|PRIMARY_OUTPUT|0.602410|0.992557|0.918059|0.953856|0.953856|
|3|HUNSPELL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.442829|0.996884|0.889061|0.939890|0.939890|
|4|SNOWBALL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.375834|0.995817|0.887492|0.938539|0.938539|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|61763|1214|1384|329602642|1214 / 329603856|1384 / 63147|
|2|SNOWBALL DUTCH DIRECT|PRIMARY_OUTPUT|29111|4382|34036|329599474|4382 / 329603856|34036 / 63147|
|3|HUNSPELL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|18333|1310|44814|329602546|1310 / 329603856|44814 / 63147|
|4|SNOWBALL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|14972|1544|48175|329602312|1544 / 329603856|48175 / 63147|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 329603856 (0.000000%)|0 / 63147 (0.000000%)|1.000000|1.000000|1.000000|
|2|HUNSPELL DUTCH LUCENE FILTER|ANY_CANDIDATE|0.667956|1141 / 329603856 (0.000346%)|41935 / 63147 (66.408539%)|0.695206|0.496187|0.564555|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|2|HUNSPELL DUTCH LUCENE FILTER|ANY_CANDIDATE|0.948955|0.335915|0.999997|0.667956|0.999869|0.000131|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|2|HUNSPELL DUTCH LUCENE FILTER|ANY_CANDIDATE|0.695206|0.496187|0.385755|0.329953|0.564595|0.564555|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|2|HUNSPELL DUTCH LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|63147|0|0|329603856|0 / 329603856|0 / 63147|
|2|HUNSPELL DUTCH LUCENE FILTER|ANY_CANDIDATE|21212|1141|41935|329602715|1141 / 329603856|41935 / 63147|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999996|2651 / 329603856 (0.000804%)|0 / 63147 (0.000000%)|0.967506|0.979441|0.979644|
|2|HUNSPELL DUTCH LUCENE FILTER|ALL_CANDIDATES|0.667955|1712 / 329603856 (0.000519%)|41935 / 63147 (66.408539%)|0.684952|0.492895|0.557477|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.959710|1.000000|0.999992|0.999996|0.999992|0.000008|
|2|HUNSPELL DUTCH LUCENE FILTER|ALL_CANDIDATES|0.925318|0.335915|0.999995|0.667955|0.999868|0.000132|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.967506|0.979441|0.991674|0.959710|0.979648|0.979644|
|2|HUNSPELL DUTCH LUCENE FILTER|ALL_CANDIDATES|0.684952|0.492895|0.384956|0.327048|0.557519|0.557477|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|2|HUNSPELL DUTCH LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|63147|2651|0|329601205|2651 / 329603856|0 / 63147|
|2|HUNSPELL DUTCH LUCENE FILTER|ALL_CANDIDATES|21212|1712|41935|329602144|1712 / 329603856|41935 / 63147|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|HUNSPELL DUTCH LUCENE FILTER|2879|169|402|1186|4.618740%|3|26896|
|Radixor|1384|1214|1437|549|2.138017%|3|26239|
### Output Policies and Metric Definitions
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
- Jaccard index: `TP / (TP + FP + FN)`.
- FowlkesMallows index: `sqrt(precision * recall)`.
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
### Provenance
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
- Evaluation command: `./gradlew stemmingQuality`
- Dictionary language: `NL_NL`
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
<!-- STEMMING-QUALITY:END -->

View File

@@ -67,3 +67,448 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs. - Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available. - Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees. - Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
<!-- STEMMING-QUALITY:START -->
## Stemming Quality
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `US_UK` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
### Evaluation Scope and Key Findings
The dictionary resource is `src/main/resources/us_uk/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.965159** among 11 deterministic stemmers. The runner-up is `ENGLISH LUCENE PORTER COPIED` at 0.954627, a difference of 0.010533. This rank does not imply leadership in throughput or every secondary metric.
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.965820** among 11 deterministic stemmers. The runner-up is `ENGLISH LUCENE PORTER COPIED` at 0.954900, a difference of 0.010920. This rank does not imply leadership in throughput or every secondary metric.
### `ALL_WORDS`
This mode contains **15 result rows**, **11 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.965159|1149886 / 184490451771 (0.000623%)|21869 / 313870 (6.967534%)|0.240076|0.332621|0.434052|
|2|ENGLISH LUCENE PORTER COPIED|PRIMARY_OUTPUT|0.954627|1557406 / 184490451771 (0.000844%)|28480 / 313870 (9.073820%)|0.185679|0.264659|0.375252|
|3|ENGLISH LUCENE PORTER FILTER|PRIMARY_OUTPUT|0.954627|1557406 / 184490451771 (0.000844%)|28480 / 313870 (9.073820%)|0.185679|0.264659|0.375252|
|4|ENGLISH OPENNLP PORTER|PRIMARY_OUTPUT|0.954627|1557406 / 184490451771 (0.000844%)|28480 / 313870 (9.073820%)|0.185679|0.264659|0.375252|
|5|ENGLISH SNOWBALL PORTER2|PRIMARY_OUTPUT|0.954537|1566711 / 184490451771 (0.000849%)|28536 / 313870 (9.091662%)|0.184753|0.263477|0.374240|
|6|ENGLISH SNOWBALL ORIGINAL PORTER|PRIMARY_OUTPUT|0.954490|1555293 / 184490451771 (0.000843%)|28566 / 313870 (9.101220%)|0.185835|0.264849|0.375363|
|7|ENGLISH PAICE HUSK LANCASTER|PRIMARY_OUTPUT|0.952394|3062661 / 184490451771 (0.001660%)|29879 / 313870 (9.519546%)|0.103643|0.155164|0.277089|
|8|ENGLISH LUCENE KSTEM FILTER|PRIMARY_OUTPUT|0.878441|1368501 / 184490451771 (0.000742%)|76305 / 313870 (24.311020%)|0.176284|0.247472|0.334598|
|9|ENGLISH LUCENE MINIMAL FILTER|PRIMARY_OUTPUT|0.718599|1122264 / 184490451771 (0.000608%)|176645 / 313870 (56.279670%)|0.128204|0.174436|0.218251|
|10|HUNSPELL ENGLISH LUCENE FILTER|PRIMARY_OUTPUT|0.573277|1981986 / 184490451771 (0.001074%)|267868 / 313870 (85.343614%)|0.027298|0.039287|0.057655|
|11|ENGLISH LUCENE POSSESSIVE FILTER|PRIMARY_OUTPUT|0.500008|1115154 / 184490451771 (0.000604%)|313863 / 313870 (99.997770%)|0.000007|0.000010|0.000009|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.202513|0.930325|0.999994|0.965159|0.999994|0.000006|
|2|ENGLISH LUCENE PORTER COPIED|PRIMARY_OUTPUT|0.154868|0.909262|0.999992|0.954627|0.999991|0.000009|
|3|ENGLISH LUCENE PORTER FILTER|PRIMARY_OUTPUT|0.154868|0.909262|0.999992|0.954627|0.999991|0.000009|
|4|ENGLISH OPENNLP PORTER|PRIMARY_OUTPUT|0.154868|0.909262|0.999992|0.954627|0.999991|0.000009|
|5|ENGLISH SNOWBALL PORTER2|PRIMARY_OUTPUT|0.154064|0.909083|0.999992|0.954537|0.999991|0.000009|
|6|ENGLISH SNOWBALL ORIGINAL PORTER|PRIMARY_OUTPUT|0.155006|0.908988|0.999992|0.954490|0.999991|0.000009|
|7|ENGLISH PAICE HUSK LANCASTER|PRIMARY_OUTPUT|0.084858|0.904805|0.999983|0.952394|0.999983|0.000017|
|8|ENGLISH LUCENE KSTEM FILTER|PRIMARY_OUTPUT|0.147917|0.756890|0.999993|0.878441|0.999992|0.000008|
|9|ENGLISH LUCENE MINIMAL FILTER|PRIMARY_OUTPUT|0.108953|0.437203|0.999994|0.718599|0.999993|0.000007|
|10|HUNSPELL ENGLISH LUCENE FILTER|PRIMARY_OUTPUT|0.022684|0.146564|0.999989|0.573277|0.999988|0.000012|
|11|ENGLISH LUCENE POSSESSIVE FILTER|PRIMARY_OUTPUT|0.000006|0.000022|0.999994|0.500008|0.999992|0.000008|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.240076|0.332621|0.541270|0.199487|0.434054|0.434052|
|2|ENGLISH LUCENE PORTER COPIED|PRIMARY_OUTPUT|0.185679|0.264659|0.460563|0.152511|0.375254|0.375252|
|3|ENGLISH LUCENE PORTER FILTER|PRIMARY_OUTPUT|0.185679|0.264659|0.460563|0.152511|0.375254|0.375252|
|4|ENGLISH OPENNLP PORTER|PRIMARY_OUTPUT|0.185679|0.264659|0.460563|0.152511|0.375254|0.375252|
|5|ENGLISH SNOWBALL PORTER2|PRIMARY_OUTPUT|0.184753|0.263477|0.459102|0.151727|0.374242|0.374240|
|6|ENGLISH SNOWBALL ORIGINAL PORTER|PRIMARY_OUTPUT|0.185835|0.264849|0.460751|0.152637|0.375365|0.375363|
|7|ENGLISH PAICE HUSK LANCASTER|PRIMARY_OUTPUT|0.103643|0.155164|0.308543|0.084107|0.277092|0.277089|
|8|ENGLISH LUCENE KSTEM FILTER|PRIMARY_OUTPUT|0.176284|0.247472|0.415099|0.141208|0.334600|0.334598|
|9|ENGLISH LUCENE MINIMAL FILTER|PRIMARY_OUTPUT|0.128204|0.174436|0.272816|0.095552|0.218253|0.218251|
|10|HUNSPELL ENGLISH LUCENE FILTER|PRIMARY_OUTPUT|0.027298|0.039287|0.070051|0.020037|0.057659|0.057655|
|11|ENGLISH LUCENE POSSESSIVE FILTER|PRIMARY_OUTPUT|0.000007|0.000010|0.000015|0.000005|0.000012|0.000009|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.332619|0.994215|0.997770|0.995989|0.995989|
|2|ENGLISH LUCENE PORTER COPIED|PRIMARY_OUTPUT|0.264656|0.969648|0.997199|0.983231|0.983231|
|3|ENGLISH LUCENE PORTER FILTER|PRIMARY_OUTPUT|0.264656|0.969648|0.997199|0.983231|0.983231|
|4|ENGLISH OPENNLP PORTER|PRIMARY_OUTPUT|0.264656|0.969648|0.997199|0.983231|0.983231|
|5|ENGLISH SNOWBALL PORTER2|PRIMARY_OUTPUT|0.263474|0.969037|0.997182|0.982908|0.982908|
|6|ENGLISH SNOWBALL ORIGINAL PORTER|PRIMARY_OUTPUT|0.264847|0.969891|0.997193|0.983353|0.983353|
|7|ENGLISH PAICE HUSK LANCASTER|PRIMARY_OUTPUT|0.155162|0.937768|0.996600|0.966289|0.966289|
|8|ENGLISH LUCENE KSTEM FILTER|PRIMARY_OUTPUT|0.247470|0.980687|0.992108|0.986364|0.986364|
|9|ENGLISH LUCENE MINIMAL FILTER|PRIMARY_OUTPUT|0.174433|0.995202|0.981174|0.988138|0.988138|
|10|HUNSPELL ENGLISH LUCENE FILTER|PRIMARY_OUTPUT|0.039284|0.993096|0.963677|0.978166|0.978166|
|11|ENGLISH LUCENE POSSESSIVE FILTER|PRIMARY_OUTPUT|0.000007|0.995789|0.958019|0.976539|0.976539|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|292001|1149886|21869|184489301885|1149886 / 184490451771|21869 / 313870|
|2|ENGLISH LUCENE PORTER COPIED|PRIMARY_OUTPUT|285390|1557406|28480|184488894365|1557406 / 184490451771|28480 / 313870|
|3|ENGLISH LUCENE PORTER FILTER|PRIMARY_OUTPUT|285390|1557406|28480|184488894365|1557406 / 184490451771|28480 / 313870|
|4|ENGLISH OPENNLP PORTER|PRIMARY_OUTPUT|285390|1557406|28480|184488894365|1557406 / 184490451771|28480 / 313870|
|5|ENGLISH SNOWBALL PORTER2|PRIMARY_OUTPUT|285334|1566711|28536|184488885060|1566711 / 184490451771|28536 / 313870|
|6|ENGLISH SNOWBALL ORIGINAL PORTER|PRIMARY_OUTPUT|285304|1555293|28566|184488896478|1555293 / 184490451771|28566 / 313870|
|7|ENGLISH PAICE HUSK LANCASTER|PRIMARY_OUTPUT|283991|3062661|29879|184487389110|3062661 / 184490451771|29879 / 313870|
|8|ENGLISH LUCENE KSTEM FILTER|PRIMARY_OUTPUT|237565|1368501|76305|184489083270|1368501 / 184490451771|76305 / 313870|
|9|ENGLISH LUCENE MINIMAL FILTER|PRIMARY_OUTPUT|137225|1122264|176645|184489329507|1122264 / 184490451771|176645 / 313870|
|10|HUNSPELL ENGLISH LUCENE FILTER|PRIMARY_OUTPUT|46002|1981986|267868|184488469785|1981986 / 184490451771|267868 / 313870|
|11|ENGLISH LUCENE POSSESSIVE FILTER|PRIMARY_OUTPUT|7|1115154|313863|184489336617|1115154 / 184490451771|313863 / 313870|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|0.999976|12 / 184490451771 (0.000000%)|15 / 313870 (0.004779%)|0.999960|0.999957|0.999957|
|2|HUNSPELL ENGLISH LUCENE FILTER|ANY_CANDIDATE|0.581603|1978852 / 184490451771 (0.001073%)|262641 / 313870 (83.678274%)|0.030370|0.043712|0.064174|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|0.999962|0.999952|1.000000|0.999976|1.000000|0.000000|
|2|HUNSPELL ENGLISH LUCENE FILTER|ANY_CANDIDATE|0.025235|0.163217|0.999989|0.581603|0.999988|0.000012|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|0.999960|0.999957|0.999954|0.999914|0.999957|0.999957|
|2|HUNSPELL ENGLISH LUCENE FILTER|ANY_CANDIDATE|0.030370|0.043712|0.077961|0.022344|0.064178|0.064174|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|2|HUNSPELL ENGLISH LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|313855|12|15|184490451759|12 / 184490451771|15 / 313870|
|2|HUNSPELL ENGLISH LUCENE FILTER|ANY_CANDIDATE|51229|1978852|262641|184488472919|1978852 / 184490451771|262641 / 313870|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999945|11482166 / 184490451771 (0.006224%)|15 / 313870 (0.004779%)|0.033039|0.051834|0.163107|
|2|HUNSPELL ENGLISH LUCENE FILTER|ALL_CANDIDATES|0.581603|2008917 / 184490451771 (0.001089%)|262641 / 313870 (83.678274%)|0.029943|0.043158|0.063704|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.026607|0.999952|0.999938|0.999945|0.999938|0.000062|
|2|HUNSPELL ENGLISH LUCENE FILTER|ALL_CANDIDATES|0.024867|0.163217|0.999989|0.581603|0.999988|0.000012|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.033039|0.051834|0.120237|0.026607|0.163112|0.163107|
|2|HUNSPELL ENGLISH LUCENE FILTER|ALL_CANDIDATES|0.029943|0.043158|0.077254|0.022055|0.063708|0.063704|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|2|HUNSPELL ENGLISH LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|313855|11482166|15|184478969605|11482166 / 184490451771|15 / 313870|
|2|HUNSPELL ENGLISH LUCENE FILTER|ALL_CANDIDATES|51229|2008917|262641|184488442854|2008917 / 184490451771|262641 / 313870|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|21854|1149874|10332280|29208|4.808384%|1355|2838145|
|HUNSPELL ENGLISH LUCENE FILTER|5227|3134|26931|6837|1.125545%|4|614296|
### `LOWERCASE_GROUPS_ONLY`
This mode contains **15 result rows**, **11 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.965820|1148489 / 170474840204 (0.000674%)|21319 / 311891 (6.835401%)|0.239424|0.331902|0.433722|
|2|ENGLISH LUCENE PORTER COPIED|PRIMARY_OUTPUT|0.954900|1552702 / 170474840204 (0.000911%)|28130 / 311891 (9.019177%)|0.185277|0.264166|0.374937|
|3|ENGLISH LUCENE PORTER FILTER|PRIMARY_OUTPUT|0.954900|1552702 / 170474840204 (0.000911%)|28130 / 311891 (9.019177%)|0.185277|0.264166|0.374937|
|4|ENGLISH OPENNLP PORTER|PRIMARY_OUTPUT|0.954900|1552702 / 170474840204 (0.000911%)|28130 / 311891 (9.019177%)|0.185277|0.264166|0.374937|
|5|ENGLISH SNOWBALL PORTER2|PRIMARY_OUTPUT|0.954850|1561891 / 170474840204 (0.000916%)|28161 / 311891 (9.029116%)|0.184375|0.263016|0.373964|
|6|ENGLISH SNOWBALL ORIGINAL PORTER|PRIMARY_OUTPUT|0.954762|1550615 / 170474840204 (0.000910%)|28216 / 311891 (9.046750%)|0.185431|0.264353|0.375045|
|7|ENGLISH PAICE HUSK LANCASTER|PRIMARY_OUTPUT|0.952710|3045870 / 170474840204 (0.001787%)|29493 / 311891 (9.456188%)|0.103633|0.155157|0.277170|
|8|ENGLISH LUCENE KSTEM FILTER|PRIMARY_OUTPUT|0.880820|1367069 / 170474840204 (0.000802%)|74340 / 311891 (23.835250%)|0.176477|0.247899|0.335789|
|9|ENGLISH LUCENE MINIMAL FILTER|PRIMARY_OUTPUT|0.719516|1120871 / 170474840204 (0.000657%)|174959 / 311891 (56.096200%)|0.128139|0.174470|0.218621|
|10|HUNSPELL ENGLISH LUCENE FILTER|PRIMARY_OUTPUT|0.573619|1978041 / 170474840204 (0.001160%)|265965 / 311891 (85.274984%)|0.027312|0.039323|0.057799|
|11|ENGLISH LUCENE POSSESSIVE FILTER|PRIMARY_OUTPUT|0.500005|1113773 / 170474840204 (0.000653%)|311886 / 311891 (99.998397%)|0.000005|0.000007|0.000005|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.201918|0.931646|0.999993|0.965820|0.999993|0.000007|
|2|ENGLISH LUCENE PORTER COPIED|PRIMARY_OUTPUT|0.154515|0.909808|0.999991|0.954900|0.999991|0.000009|
|3|ENGLISH LUCENE PORTER FILTER|PRIMARY_OUTPUT|0.154515|0.909808|0.999991|0.954900|0.999991|0.000009|
|4|ENGLISH OPENNLP PORTER|PRIMARY_OUTPUT|0.154515|0.909808|0.999991|0.954900|0.999991|0.000009|
|5|ENGLISH SNOWBALL PORTER2|PRIMARY_OUTPUT|0.153731|0.909709|0.999991|0.954850|0.999991|0.000009|
|6|ENGLISH SNOWBALL ORIGINAL PORTER|PRIMARY_OUTPUT|0.154651|0.909532|0.999991|0.954762|0.999991|0.000009|
|7|ENGLISH PAICE HUSK LANCASTER|PRIMARY_OUTPUT|0.084848|0.905438|0.999982|0.952710|0.999982|0.000018|
|8|ENGLISH LUCENE KSTEM FILTER|PRIMARY_OUTPUT|0.148042|0.761647|0.999992|0.880820|0.999992|0.000008|
|9|ENGLISH LUCENE MINIMAL FILTER|PRIMARY_OUTPUT|0.108866|0.439038|0.999993|0.719516|0.999992|0.000008|
|10|HUNSPELL ENGLISH LUCENE FILTER|PRIMARY_OUTPUT|0.022691|0.147250|0.999988|0.573619|0.999987|0.000013|
|11|ENGLISH LUCENE POSSESSIVE FILTER|PRIMARY_OUTPUT|0.000004|0.000016|0.999993|0.500005|0.999992|0.000008|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.239424|0.331902|0.540775|0.198970|0.433723|0.433722|
|2|ENGLISH LUCENE PORTER COPIED|PRIMARY_OUTPUT|0.185277|0.264166|0.460049|0.152184|0.374939|0.374937|
|3|ENGLISH LUCENE PORTER FILTER|PRIMARY_OUTPUT|0.185277|0.264166|0.460049|0.152184|0.374939|0.374937|
|4|ENGLISH OPENNLP PORTER|PRIMARY_OUTPUT|0.185277|0.264166|0.460049|0.152184|0.374939|0.374937|
|5|ENGLISH SNOWBALL PORTER2|PRIMARY_OUTPUT|0.184375|0.263016|0.458637|0.151421|0.373966|0.373964|
|6|ENGLISH SNOWBALL ORIGINAL PORTER|PRIMARY_OUTPUT|0.185431|0.264353|0.460234|0.152308|0.375047|0.375045|
|7|ENGLISH PAICE HUSK LANCASTER|PRIMARY_OUTPUT|0.103633|0.155157|0.308576|0.084103|0.277173|0.277170|
|8|ENGLISH LUCENE KSTEM FILTER|PRIMARY_OUTPUT|0.176477|0.247899|0.416437|0.141487|0.335791|0.335789|
|9|ENGLISH LUCENE MINIMAL FILTER|PRIMARY_OUTPUT|0.128139|0.174470|0.273277|0.095572|0.218624|0.218621|
|10|HUNSPELL ENGLISH LUCENE FILTER|PRIMARY_OUTPUT|0.027312|0.039323|0.070190|0.020056|0.057804|0.057799|
|11|ENGLISH LUCENE POSSESSIVE FILTER|PRIMARY_OUTPUT|0.000005|0.000007|0.000011|0.000004|0.000008|0.000005|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.331900|0.993959|0.997731|0.995842|0.995842|
|2|ENGLISH LUCENE PORTER COPIED|PRIMARY_OUTPUT|0.264164|0.968645|0.997109|0.982671|0.982671|
|3|ENGLISH LUCENE PORTER FILTER|PRIMARY_OUTPUT|0.264164|0.968645|0.997109|0.982671|0.982671|
|4|ENGLISH OPENNLP PORTER|PRIMARY_OUTPUT|0.264164|0.968645|0.997109|0.982671|0.982671|
|5|ENGLISH SNOWBALL PORTER2|PRIMARY_OUTPUT|0.263014|0.968020|0.997096|0.982343|0.982343|
|6|ENGLISH SNOWBALL ORIGINAL PORTER|PRIMARY_OUTPUT|0.264351|0.968894|0.997102|0.982795|0.982795|
|7|ENGLISH PAICE HUSK LANCASTER|PRIMARY_OUTPUT|0.155154|0.936077|0.996487|0.965338|0.965338|
|8|ENGLISH LUCENE KSTEM FILTER|PRIMARY_OUTPUT|0.247897|0.979822|0.991991|0.985869|0.985869|
|9|ENGLISH LUCENE MINIMAL FILTER|PRIMARY_OUTPUT|0.174467|0.994994|0.980520|0.987704|0.987704|
|10|HUNSPELL ENGLISH LUCENE FILTER|PRIMARY_OUTPUT|0.039320|0.993066|0.962317|0.977450|0.977450|
|11|ENGLISH LUCENE POSSESSIVE FILTER|PRIMARY_OUTPUT|0.000004|0.995605|0.956423|0.975621|0.975621|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|290572|1148489|21319|170473691715|1148489 / 170474840204|21319 / 311891|
|2|ENGLISH LUCENE PORTER COPIED|PRIMARY_OUTPUT|283761|1552702|28130|170473287502|1552702 / 170474840204|28130 / 311891|
|3|ENGLISH LUCENE PORTER FILTER|PRIMARY_OUTPUT|283761|1552702|28130|170473287502|1552702 / 170474840204|28130 / 311891|
|4|ENGLISH OPENNLP PORTER|PRIMARY_OUTPUT|283761|1552702|28130|170473287502|1552702 / 170474840204|28130 / 311891|
|5|ENGLISH SNOWBALL PORTER2|PRIMARY_OUTPUT|283730|1561891|28161|170473278313|1561891 / 170474840204|28161 / 311891|
|6|ENGLISH SNOWBALL ORIGINAL PORTER|PRIMARY_OUTPUT|283675|1550615|28216|170473289589|1550615 / 170474840204|28216 / 311891|
|7|ENGLISH PAICE HUSK LANCASTER|PRIMARY_OUTPUT|282398|3045870|29493|170471794334|3045870 / 170474840204|29493 / 311891|
|8|ENGLISH LUCENE KSTEM FILTER|PRIMARY_OUTPUT|237551|1367069|74340|170473473135|1367069 / 170474840204|74340 / 311891|
|9|ENGLISH LUCENE MINIMAL FILTER|PRIMARY_OUTPUT|136932|1120871|174959|170473719333|1120871 / 170474840204|174959 / 311891|
|10|HUNSPELL ENGLISH LUCENE FILTER|PRIMARY_OUTPUT|45926|1978041|265965|170472862163|1978041 / 170474840204|265965 / 311891|
|11|ENGLISH LUCENE POSSESSIVE FILTER|PRIMARY_OUTPUT|5|1113773|311886|170473726431|1113773 / 170474840204|311886 / 311891|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 170474840204 (0.000000%)|0 / 311891 (0.000000%)|1.000000|1.000000|1.000000|
|2|HUNSPELL ENGLISH LUCENE FILTER|ANY_CANDIDATE|0.581994|1974950 / 170474840204 (0.001158%)|260741 / 311891 (83.600040%)|0.030387|0.043756|0.064341|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|2|HUNSPELL ENGLISH LUCENE FILTER|ANY_CANDIDATE|0.025246|0.164000|0.999988|0.581994|0.999987|0.000013|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|2|HUNSPELL ENGLISH LUCENE FILTER|ANY_CANDIDATE|0.030387|0.043756|0.078123|0.022367|0.064345|0.064341|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|2|HUNSPELL ENGLISH LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|311891|0|0|170474840204|0 / 170474840204|0 / 311891|
|2|HUNSPELL ENGLISH LUCENE FILTER|ANY_CANDIDATE|51150|1974950|260741|170472865254|1974950 / 170474840204|260741 / 311891|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999966|11470018 / 170474840204 (0.006728%)|0 / 311891 (0.000000%)|0.032872|0.051579|0.162697|
|2|HUNSPELL ENGLISH LUCENE FILTER|ALL_CANDIDATES|0.581994|2004598 / 170474840204 (0.001176%)|260741 / 311891 (83.600040%)|0.029965|0.043208|0.063875|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.026472|1.000000|0.999933|0.999966|0.999933|0.000067|
|2|HUNSPELL ENGLISH LUCENE FILTER|ALL_CANDIDATES|0.024881|0.164000|0.999988|0.581994|0.999987|0.000013|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.032872|0.051579|0.119687|0.026472|0.162702|0.162697|
|2|HUNSPELL ENGLISH LUCENE FILTER|ALL_CANDIDATES|0.029965|0.043208|0.077422|0.022081|0.063879|0.063875|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|2|HUNSPELL ENGLISH LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|311891|11470018|0|170463370186|11470018 / 170474840204|0 / 311891|
|2|HUNSPELL ENGLISH LUCENE FILTER|ALL_CANDIDATES|51150|2004598|260741|170472835606|2004598 / 170474840204|260741 / 311891|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|21319|1148489|10321529|28826|4.936720%|1355|2812871|
|HUNSPELL ENGLISH LUCENE FILTER|5224|3091|26557|6786|1.162165%|4|590716|
### Output Policies and Metric Definitions
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
- Jaccard index: `TP / (TP + FP + FN)`.
- FowlkesMallows index: `sqrt(precision * recall)`.
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
### Provenance
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
- Evaluation command: `./gradlew stemmingQuality`
- Dictionary language: `US_UK`
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
<!-- STEMMING-QUALITY:END -->

View File

@@ -53,3 +53,356 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs. - Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available. - Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees. - Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
<!-- STEMMING-QUALITY:START -->
## Stemming Quality
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `FI_FI` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
### Evaluation Scope and Key Findings
The dictionary resource is `src/main/resources/fi_fi/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.984594** among 4 deterministic stemmers. The runner-up is `SNOWBALL FINNISH LUCENE FILTER` at 0.740353, a difference of 0.244242. This rank does not imply leadership in throughput or every secondary metric.
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.988068** among 4 deterministic stemmers. The runner-up is `SNOWBALL FINNISH DIRECT` at 0.738400, a difference of 0.249668. This rank does not imply leadership in throughput or every secondary metric.
### `ALL_WORDS`
This mode contains **6 result rows**, **4 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.984594|731279 / 1641126814491 (0.000045%)|971268 / 31523695 (3.081073%)|0.975128|0.972893|0.972899|
|2|SNOWBALL FINNISH LUCENE FILTER|PRIMARY_OUTPUT|0.740353|1922153 / 1641126814491 (0.000117%)|16370057 / 31523695 (51.929372%)|0.758996|0.623613|0.653138|
|3|SNOWBALL FINNISH DIRECT|PRIMARY_OUTPUT|0.739729|1544812 / 1641126814491 (0.000094%)|16409363 / 31523695 (52.054060%)|0.769880|0.627374|0.659540|
|4|FINNISH LUCENE FINNISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.695969|2223150 / 1641126814491 (0.000135%)|19168306 / 31523695 (60.806025%)|0.687649|0.536000|0.576338|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.976624|0.969189|1.000000|0.984594|0.999999|0.000001|
|2|SNOWBALL FINNISH LUCENE FILTER|PRIMARY_OUTPUT|0.887434|0.480706|0.999999|0.740353|0.999989|0.000011|
|3|SNOWBALL FINNISH DIRECT|PRIMARY_OUTPUT|0.907269|0.479459|0.999999|0.739729|0.999989|0.000011|
|4|FINNISH LUCENE FINNISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.847505|0.391940|0.999999|0.695969|0.999987|0.000013|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.975128|0.972893|0.970667|0.947216|0.972900|0.972899|
|2|SNOWBALL FINNISH LUCENE FILTER|PRIMARY_OUTPUT|0.758996|0.623613|0.529216|0.453080|0.653142|0.653138|
|3|SNOWBALL FINNISH DIRECT|PRIMARY_OUTPUT|0.769880|0.627374|0.529384|0.457061|0.659544|0.659540|
|4|FINNISH LUCENE FINNISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.687649|0.536000|0.439152|0.366120|0.576343|0.576338|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.972892|0.996085|0.993746|0.994914|0.994914|
|2|SNOWBALL FINNISH LUCENE FILTER|PRIMARY_OUTPUT|0.623608|0.990718|0.904385|0.945585|0.945585|
|3|SNOWBALL FINNISH DIRECT|PRIMARY_OUTPUT|0.627369|0.991872|0.904139|0.945975|0.945975|
|4|FINNISH LUCENE FINNISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.535994|0.988126|0.886473|0.934544|0.934544|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|30552427|731279|971268|1641126083212|731279 / 1641126814491|971268 / 31523695|
|2|SNOWBALL FINNISH LUCENE FILTER|PRIMARY_OUTPUT|15153638|1922153|16370057|1641124892338|1922153 / 1641126814491|16370057 / 31523695|
|3|SNOWBALL FINNISH DIRECT|PRIMARY_OUTPUT|15114332|1544812|16409363|1641125269679|1544812 / 1641126814491|16409363 / 31523695|
|4|FINNISH LUCENE FINNISH LIGHT STEM FILTER|PRIMARY_OUTPUT|12355389|2223150|19168306|1641124591341|2223150 / 1641126814491|19168306 / 31523695|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 1641126814491 (0.000000%)|0 / 31523695 (0.000000%)|1.000000|1.000000|1.000000|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|31523695|0|0|1641126814491|0 / 1641126814491|0 / 31523695|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999999|1683575 / 1641126814491 (0.000103%)|0 / 31523695 (0.000000%)|0.959025|0.973991|0.974320|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.949301|1.000000|0.999999|0.999999|0.999999|0.000001|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.959025|0.973991|0.989432|0.949301|0.974321|0.974320|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|31523695|1683575|0|1641125130916|1683575 / 1641126814491|0 / 31523695|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|971268|731279|952296|57328|3.164291%|6|1876272|
### `LOWERCASE_GROUPS_ONLY`
This mode contains **6 result rows**, **4 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.988068|730145 / 1543589444152 (0.000047%)|735305 / 30813833 (2.386282%)|0.976268|0.976219|0.976218|
|2|SNOWBALL FINNISH DIRECT|PRIMARY_OUTPUT|0.738400|1513705 / 1543589444152 (0.000098%)|16121763 / 30813833 (52.319888%)|0.768117|0.624934|0.657464|
|3|SNOWBALL FINNISH LUCENE FILTER|PRIMARY_OUTPUT|0.738400|1513705 / 1543589444152 (0.000098%)|16121763 / 30813833 (52.319888%)|0.768117|0.624934|0.657464|
|4|FINNISH LUCENE FINNISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.694529|1806392 / 1543589444152 (0.000117%)|18825444 / 30813833 (61.094133%)|0.697056|0.537492|0.581469|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.976301|0.976137|1.000000|0.988068|0.999999|0.000001|
|2|SNOWBALL FINNISH DIRECT|PRIMARY_OUTPUT|0.906595|0.476801|0.999999|0.738400|0.999989|0.000011|
|3|SNOWBALL FINNISH LUCENE FILTER|PRIMARY_OUTPUT|0.906595|0.476801|0.999999|0.738400|0.999989|0.000011|
|4|FINNISH LUCENE FINNISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.869053|0.389059|0.999999|0.694529|0.999987|0.000013|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.976268|0.976219|0.976170|0.953543|0.976219|0.976218|
|2|SNOWBALL FINNISH DIRECT|PRIMARY_OUTPUT|0.768117|0.624934|0.526744|0.454475|0.657469|0.657464|
|3|SNOWBALL FINNISH LUCENE FILTER|PRIMARY_OUTPUT|0.768117|0.624934|0.526744|0.454475|0.657469|0.657464|
|4|FINNISH LUCENE FINNISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.697056|0.537492|0.437372|0.367514|0.581474|0.581469|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.976218|0.996000|0.996069|0.996035|0.996035|
|2|SNOWBALL FINNISH DIRECT|PRIMARY_OUTPUT|0.624929|0.991732|0.902933|0.945252|0.945252|
|3|SNOWBALL FINNISH LUCENE FILTER|PRIMARY_OUTPUT|0.624929|0.991732|0.902933|0.945252|0.945252|
|4|FINNISH LUCENE FINNISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.537486|0.989268|0.885294|0.934397|0.934397|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|30078528|730145|735305|1543588714007|730145 / 1543589444152|735305 / 30813833|
|2|SNOWBALL FINNISH DIRECT|PRIMARY_OUTPUT|14692070|1513705|16121763|1543587930447|1513705 / 1543589444152|16121763 / 30813833|
|3|SNOWBALL FINNISH LUCENE FILTER|PRIMARY_OUTPUT|14692070|1513705|16121763|1543587930447|1513705 / 1543589444152|16121763 / 30813833|
|4|FINNISH LUCENE FINNISH LIGHT STEM FILTER|PRIMARY_OUTPUT|11988389|1806392|18825444|1543587637760|1806392 / 1543589444152|18825444 / 30813833|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 1543589444152 (0.000000%)|0 / 30813833 (0.000000%)|1.000000|1.000000|1.000000|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|30813833|0|0|1543589444152|0 / 1543589444152|0 / 30813833|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999999|1653320 / 1543589444152 (0.000107%)|0 / 30813833 (0.000000%)|0.958843|0.973873|0.974205|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.949077|1.000000|0.999999|0.999999|0.999999|0.000001|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.958843|0.973873|0.989383|0.949077|0.974206|0.974205|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|30813833|1653320|0|1543587790832|1653320 / 1543589444152|0 / 30813833|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|735305|730145|923175|44331|2.523029%|6|1805864|
### Output Policies and Metric Definitions
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
- Jaccard index: `TP / (TP + FP + FN)`.
- FowlkesMallows index: `sqrt(precision * recall)`.
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
### Provenance
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
- Evaluation command: `./gradlew stemmingQuality`
- Dictionary language: `FI_FI`
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
<!-- STEMMING-QUALITY:END -->

View File

@@ -57,3 +57,398 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs. - Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available. - Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees. - Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
<!-- STEMMING-QUALITY:START -->
## Stemming Quality
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `FR_FR` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
### Evaluation Scope and Key Findings
The dictionary resource is `src/main/resources/fr_fr/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.956992** among 6 deterministic stemmers. The runner-up is `SNOWBALL FRENCH DIRECT` at 0.845262, a difference of 0.111731. This rank does not imply leadership in throughput or every secondary metric.
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.957224** among 6 deterministic stemmers. The runner-up is `SNOWBALL FRENCH DIRECT` at 0.845414, a difference of 0.111810. This rank does not imply leadership in throughput or every secondary metric.
### `ALL_WORDS`
This mode contains **10 result rows**, **6 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.956992|318767 / 90396104830 (0.000353%)|469160 / 5454615 (8.601157%)|0.934603|0.926765|0.926851|
|2|SNOWBALL FRENCH DIRECT|PRIMARY_OUTPUT|0.845262|1654723 / 90396104830 (0.001831%)|1687975 / 5454615 (30.945814%)|0.693926|0.692653|0.692638|
|3|SNOWBALL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.844999|1661388 / 90396104830 (0.001838%)|1690838 / 5454615 (30.998301%)|0.693010|0.691885|0.691869|
|4|HUNSPELL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.813742|776728 / 90396104830 (0.000859%)|2031881 / 5454615 (37.250677%)|0.769069|0.709075|0.715131|
|5|FRENCH LUCENE FRENCH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.518587|276403 / 90396104830 (0.000306%)|5251833 / 5454615 (96.282377%)|0.137547|0.068348|0.125415|
|6|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.516830|160438 / 90396104830 (0.000177%)|5271003 / 5454615 (96.633823%)|0.134400|0.063329|0.134021|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.939903|0.913988|0.999996|0.956992|0.999991|0.000009|
|2|SNOWBALL FRENCH DIRECT|PRIMARY_OUTPUT|0.694777|0.690542|0.999982|0.845262|0.999963|0.000037|
|3|SNOWBALL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.693763|0.690017|0.999982|0.844999|0.999963|0.000037|
|4|HUNSPELL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.815041|0.627493|0.999991|0.813742|0.999969|0.000031|
|5|FRENCH LUCENE FRENCH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.423181|0.037176|0.999997|0.518587|0.999939|0.000061|
|6|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.533678|0.033662|0.999998|0.516830|0.999940|0.000060|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.934603|0.926765|0.919056|0.863524|0.926855|0.926851|
|2|SNOWBALL FRENCH DIRECT|PRIMARY_OUTPUT|0.693926|0.692653|0.691385|0.529816|0.692656|0.692638|
|3|SNOWBALL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.693010|0.691885|0.690763|0.528917|0.691887|0.691869|
|4|HUNSPELL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.769069|0.709075|0.657765|0.549277|0.715145|0.715131|
|5|FRENCH LUCENE FRENCH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.137547|0.068348|0.045472|0.035383|0.125428|0.125415|
|6|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.134400|0.063329|0.041424|0.032700|0.134032|0.134021|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.926760|0.988772|0.985214|0.986990|0.986990|
|2|SNOWBALL FRENCH DIRECT|PRIMARY_OUTPUT|0.692635|0.959459|0.944948|0.952148|0.952148|
|3|SNOWBALL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.691866|0.958698|0.944715|0.951655|0.951655|
|4|HUNSPELL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.709060|0.978337|0.913706|0.944918|0.944918|
|5|FRENCH LUCENE FRENCH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.068339|0.974110|0.812376|0.885922|0.885922|
|6|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.063322|0.984019|0.810979|0.889158|0.889158|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|4985455|318767|469160|90395786063|318767 / 90396104830|469160 / 5454615|
|2|SNOWBALL FRENCH DIRECT|PRIMARY_OUTPUT|3766640|1654723|1687975|90394450107|1654723 / 90396104830|1687975 / 5454615|
|3|SNOWBALL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|3763777|1661388|1690838|90394443442|1661388 / 90396104830|1690838 / 5454615|
|4|HUNSPELL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|3422734|776728|2031881|90395328102|776728 / 90396104830|2031881 / 5454615|
|5|FRENCH LUCENE FRENCH LIGHT STEM FILTER|PRIMARY_OUTPUT|202782|276403|5251833|90395828427|276403 / 90396104830|5251833 / 5454615|
|6|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|PRIMARY_OUTPUT|183612|160438|5271003|90395944392|160438 / 90396104830|5271003 / 5454615|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|0.999979|12 / 90396104830 (0.000000%)|232 / 5454615 (0.004253%)|0.999990|0.999978|0.999978|
|2|HUNSPELL FRENCH LUCENE FILTER|ANY_CANDIDATE|0.830964|745831 / 90396104830 (0.000825%)|1844003 / 5454615 (33.806291%)|0.789019|0.736029|0.740670|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|0.999998|0.999957|1.000000|0.999979|1.000000|0.000000|
|2|HUNSPELL FRENCH LUCENE FILTER|ANY_CANDIDATE|0.828798|0.661937|0.999992|0.830964|0.999971|0.000029|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|0.999990|0.999978|0.999966|0.999955|0.999978|0.999978|
|2|HUNSPELL FRENCH LUCENE FILTER|ANY_CANDIDATE|0.789019|0.736029|0.689709|0.582315|0.740684|0.740670|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|2|HUNSPELL FRENCH LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|5454383|12|232|90396104818|12 / 90396104830|232 / 5454615|
|2|HUNSPELL FRENCH LUCENE FILTER|ANY_CANDIDATE|3610612|745831|1844003|90395358999|745831 / 90396104830|1844003 / 5454615|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999973|1056255 / 90396104830 (0.001168%)|232 / 5454615 (0.004253%)|0.865853|0.911704|0.915270|
|2|HUNSPELL FRENCH LUCENE FILTER|ALL_CANDIDATES|0.830963|1043199 / 90396104830 (0.001154%)|1844003 / 5454615 (33.806291%)|0.750028|0.714377|0.716613|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.837765|0.999957|0.999988|0.999973|0.999988|0.000012|
|2|HUNSPELL FRENCH LUCENE FILTER|ALL_CANDIDATES|0.775840|0.661937|0.999988|0.830963|0.999968|0.000032|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.865853|0.911704|0.962682|0.837735|0.915275|0.915270|
|2|HUNSPELL FRENCH LUCENE FILTER|ALL_CANDIDATES|0.750028|0.714377|0.681961|0.555666|0.716629|0.716613|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|2|HUNSPELL FRENCH LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|5454383|1056255|232|90395048575|1056255 / 90396104830|232 / 5454615|
|2|HUNSPELL FRENCH LUCENE FILTER|ALL_CANDIDATES|3610612|1043199|1844003|90395061631|1043199 / 90396104830|1844003 / 5454615|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|468928|318755|737488|43040|10.122057%|56|477024|
|HUNSPELL FRENCH LUCENE FILTER|187878|30897|266471|13511|3.177489%|4|439015|
### `LOWERCASE_GROUPS_ONLY`
This mode contains **10 result rows**, **6 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.957224|315266 / 88712126506 (0.000355%)|465436 / 5440559 (8.554930%)|0.935099|0.927248|0.927334|
|2|SNOWBALL FRENCH DIRECT|PRIMARY_OUTPUT|0.845414|1646111 / 88712126506 (0.001856%)|1681970 / 5440559 (30.915389%)|0.694508|0.693130|0.693115|
|3|SNOWBALL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.845163|1641925 / 88712126506 (0.001851%)|1684703 / 5440559 (30.965623%)|0.694714|0.693068|0.693055|
|4|HUNSPELL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.813617|763305 / 88712126506 (0.000860%)|2028011 / 5440559 (37.275784%)|0.770537|0.709734|0.715938|
|5|FRENCH LUCENE FRENCH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.518442|262689 / 88712126506 (0.000296%)|5239869 / 5440559 (96.311225%)|0.137571|0.067985|0.126383|
|6|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.516697|147476 / 88712126506 (0.000166%)|5258873 / 5440559 (96.660527%)|0.134439|0.062979|0.135757|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.940408|0.914451|0.999996|0.957224|0.999991|0.000009|
|2|SNOWBALL FRENCH DIRECT|PRIMARY_OUTPUT|0.695430|0.690846|0.999981|0.845414|0.999962|0.000038|
|3|SNOWBALL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.695815|0.690344|0.999981|0.845163|0.999963|0.000037|
|4|HUNSPELL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.817210|0.627242|0.999991|0.813617|0.999969|0.000031|
|5|FRENCH LUCENE FRENCH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.433101|0.036888|0.999997|0.518442|0.999938|0.000062|
|6|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.551965|0.033395|0.999998|0.516697|0.999939|0.000061|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.935099|0.927248|0.919527|0.864363|0.927338|0.927334|
|2|SNOWBALL FRENCH DIRECT|PRIMARY_OUTPUT|0.694508|0.693130|0.691758|0.530374|0.693134|0.693115|
|3|SNOWBALL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.694714|0.693068|0.691431|0.530302|0.693074|0.693055|
|4|HUNSPELL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.770537|0.709734|0.657826|0.550068|0.715953|0.715938|
|5|FRENCH LUCENE FRENCH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.137571|0.067985|0.045148|0.035189|0.126397|0.126383|
|6|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.134439|0.062979|0.041121|0.032513|0.135767|0.135757|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.927243|0.988916|0.985550|0.987230|0.987230|
|2|SNOWBALL FRENCH DIRECT|PRIMARY_OUTPUT|0.693112|0.959521|0.944537|0.951970|0.951970|
|3|SNOWBALL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.693050|0.959566|0.944385|0.951915|0.951915|
|4|HUNSPELL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.709719|0.979328|0.913162|0.945088|0.945088|
|5|FRENCH LUCENE FRENCH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.067976|0.975086|0.811144|0.885591|0.885591|
|6|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.062973|0.985086|0.809774|0.888868|0.888868|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|4975123|315266|465436|88711811240|315266 / 88712126506|465436 / 5440559|
|2|SNOWBALL FRENCH DIRECT|PRIMARY_OUTPUT|3758589|1646111|1681970|88710480395|1646111 / 88712126506|1681970 / 5440559|
|3|SNOWBALL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|3755856|1641925|1684703|88710484581|1641925 / 88712126506|1684703 / 5440559|
|4|HUNSPELL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|3412548|763305|2028011|88711363201|763305 / 88712126506|2028011 / 5440559|
|5|FRENCH LUCENE FRENCH LIGHT STEM FILTER|PRIMARY_OUTPUT|200690|262689|5239869|88711863817|262689 / 88712126506|5239869 / 5440559|
|6|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|PRIMARY_OUTPUT|181686|147476|5258873|88711979030|147476 / 88712126506|5258873 / 5440559|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 88712126506 (0.000000%)|0 / 5440559 (0.000000%)|1.000000|1.000000|1.000000|
|2|HUNSPELL FRENCH LUCENE FILTER|ANY_CANDIDATE|0.830852|733584 / 88712126506 (0.000827%)|1840476 / 5440559 (33.828803%)|0.790351|0.736648|0.741404|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|2|HUNSPELL FRENCH LUCENE FILTER|ANY_CANDIDATE|0.830724|0.661712|0.999992|0.830852|0.999971|0.000029|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|2|HUNSPELL FRENCH LUCENE FILTER|ANY_CANDIDATE|0.790351|0.736648|0.689779|0.583090|0.741418|0.741404|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|2|HUNSPELL FRENCH LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|5440559|0|0|88712126506|0 / 88712126506|0 / 5440559|
|2|HUNSPELL FRENCH LUCENE FILTER|ANY_CANDIDATE|3600083|733584|1840476|88711392922|733584 / 88712126506|1840476 / 5440559|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999995|938985 / 88712126506 (0.001058%)|0 / 5440559 (0.000000%)|0.878679|0.920560|0.923474|
|2|HUNSPELL FRENCH LUCENE FILTER|ALL_CANDIDATES|0.830850|1027635 / 88712126506 (0.001158%)|1840476 / 5440559 (33.828803%)|0.751538|0.715134|0.717460|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.852813|1.000000|0.999989|0.999995|0.999989|0.000011|
|2|HUNSPELL FRENCH LUCENE FILTER|ALL_CANDIDATES|0.777939|0.661712|0.999988|0.830850|0.999968|0.000032|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.878679|0.920560|0.966634|0.852813|0.923479|0.923474|
|2|HUNSPELL FRENCH LUCENE FILTER|ALL_CANDIDATES|0.751538|0.715134|0.682093|0.556582|0.717476|0.717460|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|2|HUNSPELL FRENCH LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|5440559|938985|0|88711187521|938985 / 88712126506|0 / 5440559|
|2|HUNSPELL FRENCH LUCENE FILTER|ALL_CANDIDATES|3600083|1027635|1840476|88711098871|1027635 / 88712126506|1840476 / 5440559|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|465436|315266|623719|41130|9.764239%|56|468574|
|HUNSPELL FRENCH LUCENE FILTER|187535|29721|264330|13437|3.189936%|4|434961|
### Output Policies and Metric Definitions
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
- Jaccard index: `TP / (TP + FP + FN)`.
- FowlkesMallows index: `sqrt(precision * recall)`.
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
### Provenance
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
- Evaluation command: `./gradlew stemmingQuality`
- Dictionary language: `FR_FR`
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
<!-- STEMMING-QUALITY:END -->

View File

@@ -61,3 +61,418 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs. - Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available. - Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees. - Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
<!-- STEMMING-QUALITY:START -->
## Stemming Quality
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `DE_DE` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
### Evaluation Scope and Key Findings
The dictionary resource is `src/main/resources/de_de/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.907901** among 8 deterministic stemmers. The runner-up is `GERMAN CISTEM` at 0.880770, a difference of 0.027131. This rank does not imply leadership in throughput or every secondary metric.
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.966157** among 8 deterministic stemmers. The runner-up is `GERMAN CISTEM` at 0.915288, a difference of 0.050869. This rank does not imply leadership in throughput or every secondary metric.
### `ALL_WORDS`
This mode contains **12 result rows**, **8 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.907901|98192 / 44095245979 (0.000223%)|254903 / 1383872 (18.419550%)|0.897073|0.864768|0.866326|
|2|GERMAN CISTEM|PRIMARY_OUTPUT|0.880770|477122 / 44095245979 (0.001082%)|329983 / 1383872 (23.844908%)|0.701852|0.723109|0.724023|
|3|SNOWBALL GERMAN DIRECT|PRIMARY_OUTPUT|0.778614|190680 / 44095245979 (0.000432%)|612734 / 1383872 (44.276783%)|0.737064|0.657494|0.668394|
|4|SNOWBALL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.771357|295701 / 44095245979 (0.000671%)|632816 / 1383872 (45.727929%)|0.674089|0.617993|0.624014|
|5|GERMAN LUCENE GERMAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.756258|205740 / 44095245979 (0.000467%)|674609 / 1383872 (48.747933%)|0.703092|0.617052|0.630292|
|6|GERMAN LUCENE GERMAN STEM FILTER|PRIMARY_OUTPUT|0.723772|331871 / 44095245979 (0.000753%)|764518 / 1383872 (55.244849%)|0.596821|0.530474|0.539809|
|7|HUNSPELL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.641579|203883 / 44095245979 (0.000462%)|992010 / 1383872 (71.683653%)|0.520145|0.395897|0.431563|
|8|GERMAN LUCENE GERMAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.598139|110840 / 44095245979 (0.000251%)|1112246 / 1383872 (80.372029%)|0.466113|0.307558|0.373350|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.919984|0.815804|0.999998|0.907901|0.999992|0.000008|
|2|GERMAN CISTEM|PRIMARY_OUTPUT|0.688361|0.761551|0.999989|0.880770|0.999982|0.000018|
|3|SNOWBALL GERMAN DIRECT|PRIMARY_OUTPUT|0.801750|0.557232|0.999996|0.778614|0.999982|0.000018|
|4|SNOWBALL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.717508|0.542721|0.999993|0.771357|0.999979|0.000021|
|5|GERMAN LUCENE GERMAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.775148|0.512521|0.999995|0.756258|0.999980|0.000020|
|6|GERMAN LUCENE GERMAN STEM FILTER|PRIMARY_OUTPUT|0.651112|0.447552|0.999992|0.723772|0.999975|0.000025|
|7|HUNSPELL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.657768|0.283163|0.999995|0.641579|0.999973|0.000027|
|8|GERMAN LUCENE GERMAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.710196|0.196280|0.999997|0.598139|0.999972|0.000028|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.897073|0.864768|0.834709|0.761755|0.866330|0.866326|
|2|GERMAN CISTEM|PRIMARY_OUTPUT|0.701852|0.723109|0.745694|0.566304|0.724032|0.724023|
|3|SNOWBALL GERMAN DIRECT|PRIMARY_OUTPUT|0.737064|0.657494|0.593429|0.489751|0.668402|0.668394|
|4|SNOWBALL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.674089|0.617993|0.570517|0.447171|0.624024|0.624014|
|5|GERMAN LUCENE GERMAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.703092|0.617052|0.549774|0.446186|0.630301|0.630292|
|6|GERMAN LUCENE GERMAN STEM FILTER|PRIMARY_OUTPUT|0.596821|0.530474|0.477402|0.360983|0.539820|0.539809|
|7|HUNSPELL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.520145|0.395897|0.319562|0.246803|0.431574|0.431563|
|8|GERMAN LUCENE GERMAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.466113|0.307558|0.229493|0.181725|0.373359|0.373350|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.864764|0.989946|0.975085|0.982460|0.982460|
|2|GERMAN CISTEM|PRIMARY_OUTPUT|0.723100|0.974048|0.975147|0.974597|0.974597|
|3|SNOWBALL GERMAN DIRECT|PRIMARY_OUTPUT|0.657485|0.983725|0.949324|0.966218|0.966218|
|4|SNOWBALL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.617983|0.975845|0.942925|0.959102|0.959102|
|5|GERMAN LUCENE GERMAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.617043|0.980753|0.936533|0.958133|0.958133|
|6|GERMAN LUCENE GERMAN STEM FILTER|PRIMARY_OUTPUT|0.530462|0.975550|0.942890|0.958942|0.958942|
|7|HUNSPELL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.395885|0.980463|0.886873|0.931322|0.931322|
|8|GERMAN LUCENE GERMAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.307549|0.983615|0.896264|0.937910|0.937910|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|1128969|98192|254903|44095147787|98192 / 44095245979|254903 / 1383872|
|2|GERMAN CISTEM|PRIMARY_OUTPUT|1053889|477122|329983|44094768857|477122 / 44095245979|329983 / 1383872|
|3|SNOWBALL GERMAN DIRECT|PRIMARY_OUTPUT|771138|190680|612734|44095055299|190680 / 44095245979|612734 / 1383872|
|4|SNOWBALL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|751056|295701|632816|44094950278|295701 / 44095245979|632816 / 1383872|
|5|GERMAN LUCENE GERMAN LIGHT STEM FILTER|PRIMARY_OUTPUT|709263|205740|674609|44095040239|205740 / 44095245979|674609 / 1383872|
|6|GERMAN LUCENE GERMAN STEM FILTER|PRIMARY_OUTPUT|619354|331871|764518|44094914108|331871 / 44095245979|764518 / 1383872|
|7|HUNSPELL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|391862|203883|992010|44095042096|203883 / 44095245979|992010 / 1383872|
|8|GERMAN LUCENE GERMAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|271626|110840|1112246|44095135139|110840 / 44095245979|1112246 / 1383872|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|0.959835|1375 / 44095245979 (0.000003%)|111167 / 1383872 (8.033041%)|0.981996|0.957658|0.958475|
|2|HUNSPELL GERMAN LUCENE FILTER|ANY_CANDIDATE|0.647474|158403 / 44095245979 (0.000359%)|975697 / 1383872 (70.504859%)|0.559116|0.418544|0.460956|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|0.998921|0.919670|1.000000|0.959835|0.999997|0.000003|
|2|HUNSPELL GERMAN LUCENE FILTER|ANY_CANDIDATE|0.720422|0.294951|0.999996|0.647474|0.999974|0.000026|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|0.981996|0.957658|0.934498|0.918757|0.958476|0.958475|
|2|HUNSPELL GERMAN LUCENE FILTER|ANY_CANDIDATE|0.559116|0.418544|0.334456|0.264658|0.460966|0.460956|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|2|HUNSPELL GERMAN LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1272705|1375|111167|44095244604|1375 / 44095245979|111167 / 1383872|
|2|HUNSPELL GERMAN LUCENE FILTER|ANY_CANDIDATE|408175|158403|975697|44095087576|158403 / 44095245979|975697 / 1383872|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.959832|244817 / 44095245979 (0.000555%)|111167 / 1383872 (8.033041%)|0.853711|0.877306|0.878234|
|2|HUNSPELL GERMAN LUCENE FILTER|ALL_CANDIDATES|0.647473|242551 / 44095245979 (0.000550%)|975697 / 1383872 (70.504859%)|0.511911|0.401234|0.430118|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.838673|0.919670|0.999994|0.959832|0.999992|0.000008|
|2|HUNSPELL GERMAN LUCENE FILTER|ALL_CANDIDATES|0.627261|0.294951|0.999994|0.647473|0.999972|0.000028|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.853711|0.877306|0.902242|0.781429|0.878238|0.878234|
|2|HUNSPELL GERMAN LUCENE FILTER|ALL_CANDIDATES|0.511911|0.401234|0.329907|0.250965|0.430130|0.430118|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|2|HUNSPELL GERMAN LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|1272705|244817|111167|44095001162|244817 / 44095245979|111167 / 1383872|
|2|HUNSPELL GERMAN LUCENE FILTER|ALL_CANDIDATES|408175|242551|975697|44095003428|242551 / 44095245979|975697 / 1383872|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|143736|96817|146625|48574|16.356314%|8|361016|
|HUNSPELL GERMAN LUCENE FILTER|16313|45480|38668|7891|2.657135%|3|305052|
### `LOWERCASE_GROUPS_ONLY`
This mode contains **12 result rows**, **8 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.966157|47898 / 11263756342 (0.000425%)|59114 / 873411 (6.768177%)|0.941996|0.938343|0.938358|
|2|GERMAN CISTEM|PRIMARY_OUTPUT|0.915288|156784 / 11263756342 (0.001392%)|147964 / 873411 (16.940936%)|0.823934|0.826418|0.826415|
|3|SNOWBALL GERMAN DIRECT|PRIMARY_OUTPUT|0.795926|87697 / 11263756342 (0.000779%)|356475 / 873411 (40.814118%)|0.785153|0.699487|0.711329|
|4|SNOWBALL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.775641|77653 / 11263756342 (0.000689%)|391910 / 873411 (44.871200%)|0.774111|0.672222|0.688986|
|5|GERMAN LUCENE GERMAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.769953|55477 / 11263756342 (0.000493%)|401846 / 873411 (46.008809%)|0.790797|0.673446|0.695023|
|6|GERMAN LUCENE GERMAN STEM FILTER|PRIMARY_OUTPUT|0.716810|78723 / 11263756342 (0.000699%)|494677 / 873411 (56.637368%)|0.700519|0.569153|0.599149|
|7|HUNSPELL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.659196|84679 / 11263756342 (0.000752%)|595318 / 873411 (68.160122%)|0.598178|0.449922|0.494019|
|8|GERMAN LUCENE GERMAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.575691|21214 / 11263756342 (0.000188%)|741190 / 873411 (84.861537%)|0.444545|0.257528|0.361168|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.944446|0.932318|0.999996|0.966157|0.999991|0.000009|
|2|GERMAN CISTEM|PRIMARY_OUTPUT|0.822287|0.830591|0.999986|0.915288|0.999973|0.000027|
|3|SNOWBALL GERMAN DIRECT|PRIMARY_OUTPUT|0.854958|0.591859|0.999992|0.795926|0.999961|0.000039|
|4|SNOWBALL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.861124|0.551288|0.999993|0.775641|0.999958|0.000042|
|5|GERMAN LUCENE GERMAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.894739|0.539912|0.999995|0.769953|0.999959|0.000041|
|6|GERMAN LUCENE GERMAN STEM FILTER|PRIMARY_OUTPUT|0.827912|0.433626|0.999993|0.716810|0.999949|0.000051|
|7|HUNSPELL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.766578|0.318399|0.999992|0.659196|0.999940|0.000060|
|8|GERMAN LUCENE GERMAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.861739|0.151385|0.999998|0.575691|0.999932|0.000068|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.941996|0.938343|0.934719|0.883848|0.938363|0.938358|
|2|GERMAN CISTEM|PRIMARY_OUTPUT|0.823934|0.826418|0.828917|0.704184|0.826428|0.826415|
|3|SNOWBALL GERMAN DIRECT|PRIMARY_OUTPUT|0.785153|0.699487|0.630675|0.537854|0.711347|0.711329|
|4|SNOWBALL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.774111|0.672222|0.594035|0.506276|0.689005|0.688986|
|5|GERMAN LUCENE GERMAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.790797|0.673446|0.586424|0.507666|0.695040|0.695023|
|6|GERMAN LUCENE GERMAN STEM FILTER|PRIMARY_OUTPUT|0.700519|0.569153|0.479277|0.397774|0.599170|0.599149|
|7|HUNSPELL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.598178|0.449922|0.360559|0.290258|0.494042|0.494019|
|8|GERMAN LUCENE GERMAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.444545|0.257528|0.181270|0.147795|0.361184|0.361168|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.938338|0.994062|0.990664|0.992360|0.992360|
|2|GERMAN CISTEM|PRIMARY_OUTPUT|0.826404|0.985936|0.973570|0.979714|0.979714|
|3|SNOWBALL GERMAN DIRECT|PRIMARY_OUTPUT|0.699468|0.988418|0.932452|0.959619|0.959619|
|4|SNOWBALL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.672202|0.989021|0.919542|0.953017|0.953017|
|5|GERMAN LUCENE GERMAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.673427|0.991320|0.915070|0.951670|0.951670|
|6|GERMAN LUCENE GERMAN STEM FILTER|PRIMARY_OUTPUT|0.569130|0.988584|0.918718|0.952371|0.952371|
|7|HUNSPELL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.449897|0.988041|0.865581|0.922766|0.922766|
|8|GERMAN LUCENE GERMAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.257511|0.992643|0.854403|0.918349|0.918349|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|814297|47898|59114|11263708444|47898 / 11263756342|59114 / 873411|
|2|GERMAN CISTEM|PRIMARY_OUTPUT|725447|156784|147964|11263599558|156784 / 11263756342|147964 / 873411|
|3|SNOWBALL GERMAN DIRECT|PRIMARY_OUTPUT|516936|87697|356475|11263668645|87697 / 11263756342|356475 / 873411|
|4|SNOWBALL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|481501|77653|391910|11263678689|77653 / 11263756342|391910 / 873411|
|5|GERMAN LUCENE GERMAN LIGHT STEM FILTER|PRIMARY_OUTPUT|471565|55477|401846|11263700865|55477 / 11263756342|401846 / 873411|
|6|GERMAN LUCENE GERMAN STEM FILTER|PRIMARY_OUTPUT|378734|78723|494677|11263677619|78723 / 11263756342|494677 / 873411|
|7|HUNSPELL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|278093|84679|595318|11263671663|84679 / 11263756342|595318 / 873411|
|8|GERMAN LUCENE GERMAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|132221|21214|741190|11263735128|21214 / 11263756342|741190 / 873411|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 11263756342 (0.000000%)|0 / 873411 (0.000000%)|1.000000|1.000000|1.000000|
|2|HUNSPELL GERMAN LUCENE FILTER|ANY_CANDIDATE|0.665363|60996 / 11263756342 (0.000542%)|584547 / 873411 (66.926911%)|0.635466|0.472281|0.522540|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|2|HUNSPELL GERMAN LUCENE FILTER|ANY_CANDIDATE|0.825656|0.330731|0.999995|0.665363|0.999943|0.000057|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|2|HUNSPELL GERMAN LUCENE FILTER|ANY_CANDIDATE|0.635466|0.472281|0.375782|0.309142|0.522561|0.522540|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|2|HUNSPELL GERMAN LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|873411|0|0|11263756342|0 / 11263756342|0 / 873411|
|2|HUNSPELL GERMAN LUCENE FILTER|ANY_CANDIDATE|288864|60996|584547|11263695346|60996 / 11263756342|584547 / 873411|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999996|97544 / 11263756342 (0.000866%)|0 / 873411 (0.000000%)|0.917983|0.947112|0.948436|
|2|HUNSPELL GERMAN LUCENE FILTER|ALL_CANDIDATES|0.665361|96545 / 11263756342 (0.000857%)|584547 / 873411 (66.926911%)|0.598050|0.458944|0.497855|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.899538|1.000000|0.999991|0.999996|0.999991|0.000009|
|2|HUNSPELL GERMAN LUCENE FILTER|ALL_CANDIDATES|0.749500|0.330731|0.999991|0.665361|0.999940|0.000060|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.917983|0.947112|0.978152|0.899538|0.948440|0.948436|
|2|HUNSPELL GERMAN LUCENE FILTER|ALL_CANDIDATES|0.598050|0.458944|0.372338|0.297811|0.497878|0.497855|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|2|HUNSPELL GERMAN LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|873411|97544|0|11263658798|97544 / 11263756342|0 / 873411|
|2|HUNSPELL GERMAN LUCENE FILTER|ALL_CANDIDATES|288864|96545|584547|11263659797|96545 / 11263756342|584547 / 873411|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|59114|47898|49646|14978|9.978814%|8|167157|
|HUNSPELL GERMAN LUCENE FILTER|10771|23683|11866|4989|3.323828%|3|155207|
### Output Policies and Metric Definitions
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
- Jaccard index: `TP / (TP + FP + FN)`.
- FowlkesMallows index: `sqrt(precision * recall)`.
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
### Provenance
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
- Evaluation command: `./gradlew stemmingQuality`
- Dictionary language: `DE_DE`
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
<!-- STEMMING-QUALITY:END -->

View File

@@ -53,3 +53,356 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs. - Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available. - Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees. - Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
<!-- STEMMING-QUALITY:START -->
## Stemming Quality
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `HU_HU` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
### Evaluation Scope and Key Findings
The dictionary resource is `src/main/resources/hu_hu/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.995491** among 4 deterministic stemmers. The runner-up is `SNOWBALL HUNGARIAN LUCENE FILTER` at 0.822606, a difference of 0.172885. This rank does not imply leadership in throughput or every secondary metric.
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.996163** among 4 deterministic stemmers. The runner-up is `SNOWBALL HUNGARIAN DIRECT` at 0.821708, a difference of 0.174455. This rank does not imply leadership in throughput or every secondary metric.
### `ALL_WORDS`
This mode contains **6 result rows**, **4 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.995491|272900 / 419820542893 (0.000065%)|199837 / 22162103 (0.901706%)|0.988376|0.989352|0.989353|
|2|SNOWBALL HUNGARIAN LUCENE FILTER|PRIMARY_OUTPUT|0.822606|1792049 / 419820542893 (0.000427%)|7862745 / 22162103 (35.478334%)|0.826288|0.747610|0.757196|
|3|SNOWBALL HUNGARIAN DIRECT|PRIMARY_OUTPUT|0.822348|1506056 / 419820542893 (0.000359%)|7874191 / 22162103 (35.529981%)|0.837137|0.752866|0.763681|
|4|HUNGARIAN LUCENE HUNGARIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.816668|4132555 / 419820542893 (0.000984%)|8125833 / 22162103 (36.665442%)|0.740018|0.696055|0.699478|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.987727|0.990983|0.999999|0.995491|0.999999|0.000001|
|2|SNOWBALL HUNGARIAN LUCENE FILTER|PRIMARY_OUTPUT|0.888633|0.645217|0.999996|0.822606|0.999977|0.000023|
|3|SNOWBALL HUNGARIAN DIRECT|PRIMARY_OUTPUT|0.904644|0.644700|0.999996|0.822348|0.999978|0.000022|
|4|HUNGARIAN LUCENE HUNGARIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.772547|0.633346|0.999990|0.816668|0.999971|0.000029|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.988376|0.989352|0.990330|0.978929|0.989353|0.989353|
|2|SNOWBALL HUNGARIAN LUCENE FILTER|PRIMARY_OUTPUT|0.826288|0.747610|0.682613|0.596947|0.757206|0.757196|
|3|SNOWBALL HUNGARIAN DIRECT|PRIMARY_OUTPUT|0.837137|0.752866|0.684009|0.603677|0.763691|0.763681|
|4|HUNGARIAN LUCENE HUNGARIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.740018|0.696055|0.657023|0.533807|0.699492|0.699478|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.989352|0.998036|0.997809|0.997922|0.997922|
|2|SNOWBALL HUNGARIAN LUCENE FILTER|PRIMARY_OUTPUT|0.747599|0.990687|0.924490|0.956445|0.956445|
|3|SNOWBALL HUNGARIAN DIRECT|PRIMARY_OUTPUT|0.752855|0.991948|0.924304|0.956932|0.956932|
|4|HUNGARIAN LUCENE HUNGARIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.696040|0.982615|0.926772|0.953877|0.953877|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|21962266|272900|199837|419820269993|272900 / 419820542893|199837 / 22162103|
|2|SNOWBALL HUNGARIAN LUCENE FILTER|PRIMARY_OUTPUT|14299358|1792049|7862745|419818750844|1792049 / 419820542893|7862745 / 22162103|
|3|SNOWBALL HUNGARIAN DIRECT|PRIMARY_OUTPUT|14287912|1506056|7874191|419819036837|1506056 / 419820542893|7874191 / 22162103|
|4|HUNGARIAN LUCENE HUNGARIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|14036270|4132555|8125833|419816410338|4132555 / 419820542893|8125833 / 22162103|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 419820542893 (0.000000%)|0 / 22162103 (0.000000%)|1.000000|1.000000|1.000000|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|22162103|0|0|419820542893|0 / 419820542893|0 / 22162103|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999999|460158 / 419820542893 (0.000110%)|0 / 22162103 (0.000000%)|0.983661|0.989725|0.989777|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.979659|1.000000|0.999999|0.999999|0.999999|0.000001|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.983661|0.989725|0.995865|0.979659|0.989777|0.989777|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|22162103|460158|0|419820082735|460158 / 419820542893|0 / 22162103|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|199837|272900|187258|12320|1.344473%|5|929326|
### `LOWERCASE_GROUPS_ONLY`
This mode contains **6 result rows**, **4 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.996163|272775 / 385870694917 (0.000071%)|164277 / 21411411 (0.767240%)|0.988321|0.989820|0.989822|
|2|SNOWBALL HUNGARIAN DIRECT|PRIMARY_OUTPUT|0.821708|1496670 / 385870694917 (0.000388%)|7634885 / 21411411 (35.658019%)|0.834899|0.751079|0.761809|
|3|SNOWBALL HUNGARIAN LUCENE FILTER|PRIMARY_OUTPUT|0.821708|1496670 / 385870694917 (0.000388%)|7634885 / 21411411 (35.658019%)|0.834899|0.751079|0.761809|
|4|HUNGARIAN LUCENE HUNGARIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.815077|3639046 / 385870694917 (0.000943%)|7918708 / 21411411 (36.983588%)|0.750108|0.700135|0.704477|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.987325|0.992328|0.999999|0.996163|0.999999|0.000001|
|2|SNOWBALL HUNGARIAN DIRECT|PRIMARY_OUTPUT|0.902007|0.643420|0.999996|0.821708|0.999976|0.000024|
|3|SNOWBALL HUNGARIAN LUCENE FILTER|PRIMARY_OUTPUT|0.902007|0.643420|0.999996|0.821708|0.999976|0.000024|
|4|HUNGARIAN LUCENE HUNGARIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.787585|0.630164|0.999991|0.815077|0.999970|0.000030|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.988321|0.989820|0.991323|0.979845|0.989823|0.989822|
|2|SNOWBALL HUNGARIAN DIRECT|PRIMARY_OUTPUT|0.834899|0.751079|0.682555|0.601383|0.761820|0.761809|
|3|SNOWBALL HUNGARIAN LUCENE FILTER|PRIMARY_OUTPUT|0.834899|0.751079|0.682555|0.601383|0.761820|0.761809|
|4|HUNGARIAN LUCENE HUNGARIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.750108|0.700135|0.656404|0.538621|0.704491|0.704477|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.989819|0.997945|0.998273|0.998109|0.998109|
|2|SNOWBALL HUNGARIAN DIRECT|PRIMARY_OUTPUT|0.751068|0.991610|0.923288|0.956230|0.956230|
|3|SNOWBALL HUNGARIAN LUCENE FILTER|PRIMARY_OUTPUT|0.751068|0.991610|0.923288|0.956230|0.956230|
|4|HUNGARIAN LUCENE HUNGARIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.700120|0.983687|0.925487|0.953700|0.953700|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|21247134|272775|164277|385870422142|272775 / 385870694917|164277 / 21411411|
|2|SNOWBALL HUNGARIAN DIRECT|PRIMARY_OUTPUT|13776526|1496670|7634885|385869198247|1496670 / 385870694917|7634885 / 21411411|
|3|SNOWBALL HUNGARIAN LUCENE FILTER|PRIMARY_OUTPUT|13776526|1496670|7634885|385869198247|1496670 / 385870694917|7634885 / 21411411|
|4|HUNGARIAN LUCENE HUNGARIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|13492703|3639046|7918708|385867055871|3639046 / 385870694917|7918708 / 21411411|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 385870694917 (0.000000%)|0 / 21411411 (0.000000%)|1.000000|1.000000|1.000000|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|21411411|0|0|385870694917|0 / 385870694917|0 / 21411411|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999999|458462 / 385870694917 (0.000119%)|0 / 21411411 (0.000000%)|0.983159|0.989407|0.989462|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.979037|1.000000|0.999999|0.999999|0.999999|0.000001|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.983159|0.989407|0.995736|0.979037|0.989463|0.989462|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|21411411|458462|0|385870236455|458462 / 385870694917|0 / 21411411|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|164277|272775|185687|11153|1.269532%|5|890245|
### Output Policies and Metric Definitions
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
- Jaccard index: `TP / (TP + FP + FN)`.
- FowlkesMallows index: `sqrt(precision * recall)`.
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
### Provenance
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
- Evaluation command: `./gradlew stemmingQuality`
- Dictionary language: `HU_HU`
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
<!-- STEMMING-QUALITY:END -->

View File

@@ -1,12 +1,12 @@
# Language Benchmark Pages # Language Benchmark Pages
This section splits Radixor stemmer benchmark results by language. Each language page lists accuracy first and speed second. This section splits Radixor stemmer benchmark results by language. Each language page preserves the existing exact-root accuracy and runtime-performance results and adds pairwise stemming-quality tables for both dictionary-processing modes.
## Reference Pages ## Reference Pages
| Page | Purpose | | Page | Purpose |
| --- | --- | | --- | --- |
| [Methodology](../reference/methodology.md) | Workload design, normalization, speed metrics, and quality metrics. | | [Methodology](../reference/methodology.md) | Workload design, normalization, speed metrics, and exact-root quality metrics. Pairwise quality definitions are also reproduced on every language page. |
| [Corpora](../reference/corpora.md) | Dictionary sizes and changed-token timing workloads. | | [Corpora](../reference/corpora.md) | Dictionary sizes and changed-token timing workloads. |
| [Environment and reports](../reference/environment.md) | Hardware, JVM, JMH settings, report files, and badge policy. | | [Environment and reports](../reference/environment.md) | Hardware, JVM, JMH settings, report files, and badge policy. |
| [English dictionary coverage](../reference/english-coverage.md) | Quality/speed operating curve for contracted Radixor tries built from 100% down to 10% of English dictionary rows. | | [English dictionary coverage](../reference/english-coverage.md) | Quality/speed operating curve for contracted Radixor tries built from 100% down to 10% of English dictionary rows. |

View File

@@ -52,3 +52,356 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs. - Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available. - Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees. - Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
<!-- STEMMING-QUALITY:START -->
## Stemming Quality
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `IT_IT` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
### Evaluation Scope and Key Findings
The dictionary resource is `src/main/resources/it_it/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.996507** among 4 deterministic stemmers. The runner-up is `SNOWBALL ITALIAN DIRECT` at 0.866189, a difference of 0.130318. This rank does not imply leadership in throughput or every secondary metric.
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.996512** among 4 deterministic stemmers. The runner-up is `SNOWBALL ITALIAN DIRECT` at 0.866205, a difference of 0.130307. This rank does not imply leadership in throughput or every secondary metric.
### `ALL_WORDS`
This mode contains **6 result rows**, **4 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.996507|124172 / 53638521211 (0.000231%)|42908 / 6143814 (0.698394%)|0.982618|0.986492|0.986512|
|2|SNOWBALL ITALIAN DIRECT|PRIMARY_OUTPUT|0.866189|504775 / 53638521211 (0.000941%)|1644164 / 6143814 (26.761292%)|0.859975|0.807240|0.811470|
|3|SNOWBALL ITALIAN LUCENE FILTER|PRIMARY_OUTPUT|0.866189|504775 / 53638521211 (0.000941%)|1644164 / 6143814 (26.761292%)|0.859975|0.807240|0.811470|
|4|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.508926|10589 / 53638521211 (0.000020%)|6034130 / 6143814 (98.214725%)|0.082782|0.035020|0.127588|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.980053|0.993016|0.999998|0.996507|0.999997|0.000003|
|2|SNOWBALL ITALIAN DIRECT|PRIMARY_OUTPUT|0.899134|0.732387|0.999991|0.866189|0.999960|0.000040|
|3|SNOWBALL ITALIAN LUCENE FILTER|PRIMARY_OUTPUT|0.899134|0.732387|0.999991|0.866189|0.999960|0.000040|
|4|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.911959|0.017853|1.000000|0.508926|0.999887|0.000113|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.982618|0.986492|0.990396|0.973344|0.986513|0.986512|
|2|SNOWBALL ITALIAN DIRECT|PRIMARY_OUTPUT|0.859975|0.807240|0.760598|0.676783|0.811489|0.811470|
|3|SNOWBALL ITALIAN LUCENE FILTER|PRIMARY_OUTPUT|0.859975|0.807240|0.760598|0.676783|0.811489|0.811470|
|4|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.082782|0.035020|0.022207|0.017822|0.127597|0.127588|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.986490|0.995780|0.997113|0.996446|0.996446|
|2|SNOWBALL ITALIAN DIRECT|PRIMARY_OUTPUT|0.807220|0.987994|0.933408|0.959925|0.959925|
|3|SNOWBALL ITALIAN LUCENE FILTER|PRIMARY_OUTPUT|0.807220|0.987994|0.933408|0.959925|0.959925|
|4|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.035016|0.997481|0.737537|0.848037|0.848037|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|6100906|124172|42908|53638397039|124172 / 53638521211|42908 / 6143814|
|2|SNOWBALL ITALIAN DIRECT|PRIMARY_OUTPUT|4499650|504775|1644164|53638016436|504775 / 53638521211|1644164 / 6143814|
|3|SNOWBALL ITALIAN LUCENE FILTER|PRIMARY_OUTPUT|4499650|504775|1644164|53638016436|504775 / 53638521211|1644164 / 6143814|
|4|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|109684|10589|6034130|53638510622|10589 / 53638521211|6034130 / 6143814|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|0.999993|0 / 53638521211 (0.000000%)|80 / 6143814 (0.001302%)|0.999997|0.999993|0.999993|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0.999987|1.000000|0.999993|1.000000|0.000000|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|0.999997|0.999993|0.999990|0.999987|0.999993|0.999993|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|6143734|0|80|53638521211|0 / 53638521211|80 / 6143814|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999992|170950 / 53638521211 (0.000319%)|80 / 6143814 (0.001302%)|0.978222|0.986272|0.986363|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.972928|0.999987|0.999997|0.999992|0.999997|0.000003|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.978222|0.986272|0.994455|0.972916|0.986365|0.986363|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|6143734|170950|80|53638350261|170950 / 53638521211|80 / 6143814|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|42828|124172|46778|6254|1.909321%|4|334175|
### `LOWERCASE_GROUPS_ONLY`
This mode contains **6 result rows**, **4 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.996512|124171 / 53611667072 (0.000232%)|42828 / 6142174 (0.697278%)|0.982617|0.986495|0.986515|
|2|SNOWBALL ITALIAN DIRECT|PRIMARY_OUTPUT|0.866205|504774 / 53611667072 (0.000942%)|1643522 / 6142174 (26.757985%)|0.859970|0.807252|0.811479|
|3|SNOWBALL ITALIAN LUCENE FILTER|PRIMARY_OUTPUT|0.866205|504774 / 53611667072 (0.000942%)|1643522 / 6142174 (26.757985%)|0.859970|0.807252|0.811479|
|4|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.508927|10588 / 53611667072 (0.000020%)|6032516 / 6142174 (98.214671%)|0.082784|0.035021|0.127589|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.980048|0.993027|0.999998|0.996512|0.999997|0.000003|
|2|SNOWBALL ITALIAN DIRECT|PRIMARY_OUTPUT|0.899114|0.732420|0.999991|0.866205|0.999960|0.000040|
|3|SNOWBALL ITALIAN LUCENE FILTER|PRIMARY_OUTPUT|0.899114|0.732420|0.999991|0.866205|0.999960|0.000040|
|4|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.911947|0.017853|1.000000|0.508927|0.999887|0.000113|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.982617|0.986495|0.990404|0.973350|0.986516|0.986515|
|2|SNOWBALL ITALIAN DIRECT|PRIMARY_OUTPUT|0.859970|0.807252|0.760624|0.676800|0.811498|0.811479|
|3|SNOWBALL ITALIAN LUCENE FILTER|PRIMARY_OUTPUT|0.859970|0.807252|0.760624|0.676800|0.811498|0.811479|
|4|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.082784|0.035021|0.022208|0.017823|0.127598|0.127589|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.986493|0.995780|0.997115|0.996447|0.996447|
|2|SNOWBALL ITALIAN DIRECT|PRIMARY_OUTPUT|0.807232|0.987991|0.933413|0.959927|0.959927|
|3|SNOWBALL ITALIAN LUCENE FILTER|PRIMARY_OUTPUT|0.807232|0.987991|0.933413|0.959927|0.959927|
|4|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.035017|0.997481|0.737534|0.848035|0.848035|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|6099346|124171|42828|53611542901|124171 / 53611667072|42828 / 6142174|
|2|SNOWBALL ITALIAN DIRECT|PRIMARY_OUTPUT|4498652|504774|1643522|53611162298|504774 / 53611667072|1643522 / 6142174|
|3|SNOWBALL ITALIAN LUCENE FILTER|PRIMARY_OUTPUT|4498652|504774|1643522|53611162298|504774 / 53611667072|1643522 / 6142174|
|4|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|109658|10588|6032516|53611656484|10588 / 53611667072|6032516 / 6142174|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 53611667072 (0.000000%)|0 / 6142174 (0.000000%)|1.000000|1.000000|1.000000|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|6142174|0|0|53611667072|0 / 53611667072|0 / 6142174|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999998|170949 / 53611667072 (0.000319%)|0 / 6142174 (0.000000%)|0.978219|0.986275|0.986366|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.972922|1.000000|0.999997|0.999998|0.999997|0.000003|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.978219|0.986275|0.994464|0.972922|0.986368|0.986366|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|6142174|170949|0|53611496123|170949 / 53611667072|0 / 6142174|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|42828|124171|46778|6252|1.909188%|4|334089|
### Output Policies and Metric Definitions
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
- Jaccard index: `TP / (TP + FP + FN)`.
- FowlkesMallows index: `sqrt(precision * recall)`.
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
### Provenance
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
- Evaluation command: `./gradlew stemmingQuality`
- Dictionary language: `IT_IT`
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
<!-- STEMMING-QUALITY:END -->

View File

@@ -55,3 +55,366 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs. - Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available. - Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees. - Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
<!-- STEMMING-QUALITY:START -->
## Stemming Quality
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `NB_NO` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
### Evaluation Scope and Key Findings
The dictionary resource is `src/main/resources/nb_no/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.974783** among 5 deterministic stemmers. The runner-up is `SNOWBALL NORWEGIAN BOKMAL DIRECT` at 0.874964, a difference of 0.099819. This rank does not imply leadership in throughput or every secondary metric.
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.975000** among 5 deterministic stemmers. The runner-up is `SNOWBALL NORWEGIAN BOKMAL DIRECT` at 0.874991, a difference of 0.100009. This rank does not imply leadership in throughput or every secondary metric.
### `ALL_WORDS`
This mode contains **7 result rows**, **5 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.974783|11482 / 2835618215 (0.000405%)|7170 / 142180 (5.042903%)|0.927078|0.935387|0.935488|
|2|SNOWBALL NORWEGIAN BOKMAL DIRECT|PRIMARY_OUTPUT|0.874964|23997 / 2835618215 (0.000846%)|35554 / 142180 (25.006330%)|0.802095|0.781707|0.782399|
|3|SNOWBALL NORWEGIAN BOKMAL LUCENE FILTER|PRIMARY_OUTPUT|0.874834|24046 / 2835618215 (0.000848%)|35591 / 142180 (25.032353%)|0.801759|0.781401|0.782091|
|4|NORWEGIAN BOKMAL LUCENE NORWEGIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.850006|25171 / 2835618215 (0.000888%)|42651 / 142180 (29.997890%)|0.776381|0.745871|0.747464|
|5|NORWEGIAN BOKMAL LUCENE NORWEGIAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.832414|14772 / 2835618215 (0.000521%)|47654 / 142180 (33.516669%)|0.815763|0.751764|0.758263|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.921620|0.949571|0.999996|0.974783|0.999993|0.000007|
|2|SNOWBALL NORWEGIAN BOKMAL DIRECT|PRIMARY_OUTPUT|0.816288|0.749937|0.999992|0.874964|0.999979|0.000021|
|3|SNOWBALL NORWEGIAN BOKMAL LUCENE FILTER|PRIMARY_OUTPUT|0.815930|0.749676|0.999992|0.874834|0.999979|0.000021|
|4|NORWEGIAN BOKMAL LUCENE NORWEGIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.798148|0.700021|0.999991|0.850006|0.999976|0.000024|
|5|NORWEGIAN BOKMAL LUCENE NORWEGIAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.864847|0.664833|0.999995|0.832414|0.999978|0.000022|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.927078|0.935387|0.943846|0.878617|0.935491|0.935488|
|2|SNOWBALL NORWEGIAN BOKMAL DIRECT|PRIMARY_OUTPUT|0.802095|0.781707|0.762330|0.641641|0.782409|0.782399|
|3|SNOWBALL NORWEGIAN BOKMAL LUCENE FILTER|PRIMARY_OUTPUT|0.801759|0.781401|0.762052|0.641229|0.782102|0.782091|
|4|NORWEGIAN BOKMAL LUCENE NORWEGIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.776381|0.745871|0.717668|0.594732|0.747476|0.747464|
|5|NORWEGIAN BOKMAL LUCENE NORWEGIAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.815763|0.751764|0.697076|0.602261|0.758274|0.758263|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.935384|0.993354|0.994615|0.993984|0.993984|
|2|SNOWBALL NORWEGIAN BOKMAL DIRECT|PRIMARY_OUTPUT|0.781696|0.988328|0.971120|0.979648|0.979648|
|3|SNOWBALL NORWEGIAN BOKMAL LUCENE FILTER|PRIMARY_OUTPUT|0.781391|0.988295|0.971086|0.979615|0.979615|
|4|NORWEGIAN BOKMAL LUCENE NORWEGIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.745859|0.987774|0.965622|0.976573|0.976573|
|5|NORWEGIAN BOKMAL LUCENE NORWEGIAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.751753|0.992089|0.962516|0.977079|0.977079|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|135010|11482|7170|2835606733|11482 / 2835618215|7170 / 142180|
|2|SNOWBALL NORWEGIAN BOKMAL DIRECT|PRIMARY_OUTPUT|106626|23997|35554|2835594218|23997 / 2835618215|35554 / 142180|
|3|SNOWBALL NORWEGIAN BOKMAL LUCENE FILTER|PRIMARY_OUTPUT|106589|24046|35591|2835594169|24046 / 2835618215|35591 / 142180|
|4|NORWEGIAN BOKMAL LUCENE NORWEGIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|99529|25171|42651|2835593044|25171 / 2835618215|42651 / 142180|
|5|NORWEGIAN BOKMAL LUCENE NORWEGIAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|94526|14772|47654|2835603443|14772 / 2835618215|47654 / 142180|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 2835618215 (0.000000%)|0 / 142180 (0.000000%)|1.000000|1.000000|1.000000|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|142180|0|0|2835618215|0 / 2835618215|0 / 142180|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999996|20161 / 2835618215 (0.000711%)|0 / 142180 (0.000000%)|0.898118|0.933794|0.935844|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.875811|1.000000|0.999993|0.999996|0.999993|0.000007|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.898118|0.933794|0.972422|0.875811|0.935848|0.935844|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|142180|20161|0|2835598054|20161 / 2835618215|0 / 142180|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|7170|11482|8679|4237|5.626079%|9|79825|
### `LOWERCASE_GROUPS_ONLY`
This mode contains **7 result rows**, **5 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.975000|11482 / 2831176784 (0.000406%)|7104 / 142091 (4.999613%)|0.927151|0.935591|0.935695|
|2|SNOWBALL NORWEGIAN BOKMAL DIRECT|PRIMARY_OUTPUT|0.874991|23997 / 2831176784 (0.000848%)|35524 / 142091 (25.000880%)|0.802043|0.781698|0.782388|
|3|SNOWBALL NORWEGIAN BOKMAL LUCENE FILTER|PRIMARY_OUTPUT|0.874798|23993 / 2831176784 (0.000847%)|35579 / 142091 (25.039587%)|0.801914|0.781464|0.782161|
|4|NORWEGIAN BOKMAL LUCENE NORWEGIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.849947|25118 / 2831176784 (0.000887%)|42641 / 142091 (30.009642%)|0.776513|0.745896|0.747500|
|5|NORWEGIAN BOKMAL LUCENE NORWEGIAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.832344|14719 / 2831176784 (0.000520%)|47644 / 142091 (33.530625%)|0.815950|0.751796|0.758325|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.921608|0.950004|0.999996|0.975000|0.999993|0.000007|
|2|SNOWBALL NORWEGIAN BOKMAL DIRECT|PRIMARY_OUTPUT|0.816205|0.749991|0.999992|0.874991|0.999979|0.000021|
|3|SNOWBALL NORWEGIAN BOKMAL LUCENE FILTER|PRIMARY_OUTPUT|0.816153|0.749604|0.999992|0.874798|0.999979|0.000021|
|4|NORWEGIAN BOKMAL LUCENE NORWEGIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.798359|0.699904|0.999991|0.849947|0.999976|0.000024|
|5|NORWEGIAN BOKMAL LUCENE NORWEGIAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.865169|0.664694|0.999995|0.832344|0.999978|0.000022|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.927151|0.935591|0.944186|0.878976|0.935698|0.935695|
|2|SNOWBALL NORWEGIAN BOKMAL DIRECT|PRIMARY_OUTPUT|0.802043|0.781698|0.762360|0.641630|0.782398|0.782388|
|3|SNOWBALL NORWEGIAN BOKMAL LUCENE FILTER|PRIMARY_OUTPUT|0.801914|0.781464|0.762031|0.641314|0.782171|0.782161|
|4|NORWEGIAN BOKMAL LUCENE NORWEGIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.776513|0.745896|0.717603|0.594765|0.747512|0.747500|
|5|NORWEGIAN BOKMAL LUCENE NORWEGIAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.815950|0.751796|0.696995|0.602302|0.758335|0.758325|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.935587|0.993348|0.994694|0.994020|0.994020|
|2|SNOWBALL NORWEGIAN BOKMAL DIRECT|PRIMARY_OUTPUT|0.781688|0.988318|0.971127|0.979647|0.979647|
|3|SNOWBALL NORWEGIAN BOKMAL LUCENE FILTER|PRIMARY_OUTPUT|0.781454|0.988310|0.971074|0.979616|0.979616|
|4|NORWEGIAN BOKMAL LUCENE NORWEGIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.745885|0.987789|0.965603|0.976570|0.976570|
|5|NORWEGIAN BOKMAL LUCENE NORWEGIAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.751785|0.992107|0.962494|0.977076|0.977076|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|134987|11482|7104|2831165302|11482 / 2831176784|7104 / 142091|
|2|SNOWBALL NORWEGIAN BOKMAL DIRECT|PRIMARY_OUTPUT|106567|23997|35524|2831152787|23997 / 2831176784|35524 / 142091|
|3|SNOWBALL NORWEGIAN BOKMAL LUCENE FILTER|PRIMARY_OUTPUT|106512|23993|35579|2831152791|23993 / 2831176784|35579 / 142091|
|4|NORWEGIAN BOKMAL LUCENE NORWEGIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|99450|25118|42641|2831151666|25118 / 2831176784|42641 / 142091|
|5|NORWEGIAN BOKMAL LUCENE NORWEGIAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|94447|14719|47644|2831162065|14719 / 2831176784|47644 / 142091|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 2831176784 (0.000000%)|0 / 142091 (0.000000%)|1.000000|1.000000|1.000000|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|142091|0|0|2831176784|0 / 2831176784|0 / 142091|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999996|20161 / 2831176784 (0.000712%)|0 / 142091 (0.000000%)|0.898061|0.933756|0.935808|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.875743|1.000000|0.999993|0.999996|0.999993|0.000007|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.898061|0.933756|0.972405|0.875743|0.935811|0.935808|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|142091|20161|0|2831156623|20161 / 2831176784|0 / 142091|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|7104|11482|8679|4204|5.586637%|9|79733|
### Output Policies and Metric Definitions
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
- Jaccard index: `TP / (TP + FP + FN)`.
- FowlkesMallows index: `sqrt(precision * recall)`.
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
### Provenance
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
- Evaluation command: `./gradlew stemmingQuality`
- Dictionary language: `NB_NO`
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
<!-- STEMMING-QUALITY:END -->

View File

@@ -51,3 +51,346 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs. - Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available. - Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees. - Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
<!-- STEMMING-QUALITY:START -->
## Stemming Quality
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `NN_NO` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
### Evaluation Scope and Key Findings
The dictionary resource is `src/main/resources/nn_no/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.935777** among 3 deterministic stemmers. The runner-up is `SNOWBALL NORWEGIAN NYNORSK DIRECT` at 0.858908, a difference of 0.076869. This rank does not imply leadership in throughput or every secondary metric.
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.935853** among 3 deterministic stemmers. The runner-up is `SNOWBALL NORWEGIAN NYNORSK DIRECT` at 0.859037, a difference of 0.076816. This rank does not imply leadership in throughput or every secondary metric.
### `ALL_WORDS`
This mode contains **5 result rows**, **3 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.935777|6230 / 166491473 (0.003742%)|3936 / 30652 (12.840924%)|0.822355|0.840152|0.840669|
|2|SNOWBALL NORWEGIAN NYNORSK DIRECT|PRIMARY_OUTPUT|0.858908|8274 / 166491473 (0.004970%)|8648 / 30652 (28.213493%)|0.724941|0.722271|0.722234|
|3|SNOWBALL NORWEGIAN NYNORSK LUCENE FILTER|PRIMARY_OUTPUT|0.858484|8295 / 166491473 (0.004982%)|8674 / 30652 (28.298317%)|0.724180|0.721477|0.721440|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.810903|0.871591|0.999963|0.935777|0.999939|0.000061|
|2|SNOWBALL NORWEGIAN NYNORSK DIRECT|PRIMARY_OUTPUT|0.726732|0.717865|0.999950|0.858908|0.999898|0.000102|
|3|SNOWBALL NORWEGIAN NYNORSK LUCENE FILTER|PRIMARY_OUTPUT|0.725993|0.717017|0.999950|0.858484|0.999898|0.000102|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.822355|0.840152|0.858737|0.724364|0.840699|0.840669|
|2|SNOWBALL NORWEGIAN NYNORSK DIRECT|PRIMARY_OUTPUT|0.724941|0.722271|0.719621|0.565278|0.722285|0.722234|
|3|SNOWBALL NORWEGIAN NYNORSK LUCENE FILTER|PRIMARY_OUTPUT|0.724180|0.721477|0.718794|0.564305|0.721491|0.721440|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.840122|0.983845|0.986802|0.985321|0.985321|
|2|SNOWBALL NORWEGIAN NYNORSK DIRECT|PRIMARY_OUTPUT|0.722221|0.980542|0.964998|0.972708|0.972708|
|3|SNOWBALL NORWEGIAN NYNORSK LUCENE FILTER|PRIMARY_OUTPUT|0.721426|0.980461|0.964862|0.972599|0.972599|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|26716|6230|3936|166485243|6230 / 166491473|3936 / 30652|
|2|SNOWBALL NORWEGIAN NYNORSK DIRECT|PRIMARY_OUTPUT|22004|8274|8648|166483199|8274 / 166491473|8648 / 30652|
|3|SNOWBALL NORWEGIAN NYNORSK LUCENE FILTER|PRIMARY_OUTPUT|21978|8295|8674|166483178|8295 / 166491473|8674 / 30652|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 166491473 (0.000000%)|0 / 30652 (0.000000%)|1.000000|1.000000|1.000000|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|30652|0|0|166491473|0 / 166491473|0 / 30652|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999960|13214 / 166491473 (0.007937%)|0 / 30652 (0.000000%)|0.743562|0.822674|0.835888|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.698764|1.000000|0.999921|0.999960|0.999921|0.000079|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.743562|0.822674|0.920624|0.698764|0.835921|0.835888|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|30652|13214|0|166478259|13214 / 166491473|0 / 30652|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|3936|6230|6984|2404|13.172603%|5|21513|
### `LOWERCASE_GROUPS_ONLY`
This mode contains **5 result rows**, **3 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.935853|6230 / 165926276 (0.003755%)|3924 / 30595 (12.825625%)|0.822169|0.840084|0.840609|
|2|SNOWBALL NORWEGIAN NYNORSK DIRECT|PRIMARY_OUTPUT|0.859037|8274 / 165926276 (0.004987%)|8624 / 30595 (28.187612%)|0.724757|0.722255|0.722216|
|3|SNOWBALL NORWEGIAN NYNORSK LUCENE FILTER|PRIMARY_OUTPUT|0.858661|8274 / 165926276 (0.004987%)|8647 / 30595 (28.262788%)|0.724438|0.721772|0.721734|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.810644|0.871744|0.999962|0.935853|0.999939|0.000061|
|2|SNOWBALL NORWEGIAN NYNORSK DIRECT|PRIMARY_OUTPUT|0.726434|0.718124|0.999950|0.859037|0.999898|0.000102|
|3|SNOWBALL NORWEGIAN NYNORSK LUCENE FILTER|PRIMARY_OUTPUT|0.726226|0.717372|0.999950|0.858661|0.999898|0.000102|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.822169|0.840084|0.858798|0.724263|0.840639|0.840609|
|2|SNOWBALL NORWEGIAN NYNORSK DIRECT|PRIMARY_OUTPUT|0.724757|0.722255|0.719771|0.565258|0.722267|0.722216|
|3|SNOWBALL NORWEGIAN NYNORSK LUCENE FILTER|PRIMARY_OUTPUT|0.724438|0.721772|0.719126|0.564666|0.721785|0.721734|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.840054|0.983815|0.986842|0.985326|0.985326|
|2|SNOWBALL NORWEGIAN NYNORSK DIRECT|PRIMARY_OUTPUT|0.722204|0.980506|0.965065|0.972724|0.972724|
|3|SNOWBALL NORWEGIAN NYNORSK LUCENE FILTER|PRIMARY_OUTPUT|0.721721|0.980506|0.964945|0.972663|0.972663|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|26671|6230|3924|165920046|6230 / 165926276|3924 / 30595|
|2|SNOWBALL NORWEGIAN NYNORSK DIRECT|PRIMARY_OUTPUT|21971|8274|8624|165918002|8274 / 165926276|8624 / 30595|
|3|SNOWBALL NORWEGIAN NYNORSK LUCENE FILTER|PRIMARY_OUTPUT|21948|8274|8647|165918002|8274 / 165926276|8647 / 30595|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 165926276 (0.000000%)|0 / 30595 (0.000000%)|1.000000|1.000000|1.000000|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|30595|0|0|165926276|0 / 165926276|0 / 30595|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999960|13214 / 165926276 (0.007964%)|0 / 30595 (0.000000%)|0.743207|0.822402|0.835654|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.698372|1.000000|0.999920|0.999960|0.999920|0.000080|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.743207|0.822402|0.920488|0.698372|0.835687|0.835654|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|30595|13214|0|165913062|13214 / 165926276|0 / 30595|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|3924|6230|6984|2399|13.167572%|5|21477|
### Output Policies and Metric Definitions
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
- Jaccard index: `TP / (TP + FP + FN)`.
- FowlkesMallows index: `sqrt(precision * recall)`.
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
### Provenance
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
- Evaluation command: `./gradlew stemmingQuality`
- Dictionary language: `NN_NO`
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
<!-- STEMMING-QUALITY:END -->

View File

@@ -47,3 +47,336 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs. - Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available. - Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees. - Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
<!-- STEMMING-QUALITY:START -->
## Stemming Quality
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `FA_IR` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
### Evaluation Scope and Key Findings
The dictionary resource is `src/main/resources/fa_ir/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.974922** among 2 deterministic stemmers. The runner-up is `PERSIAN LUCENE PERSIAN STEM FILTER` at 0.502171, a difference of 0.472751. This rank does not imply leadership in throughput or every secondary metric.
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.974922** among 2 deterministic stemmers. The runner-up is `PERSIAN LUCENE PERSIAN STEM FILTER` at 0.502171, a difference of 0.472751. This rank does not imply leadership in throughput or every secondary metric.
### `ALL_WORDS`
This mode contains **4 result rows**, **2 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.974922|8621 / 6748402 (0.127749%)|4812 / 98448 (4.887860%)|0.922566|0.933071|0.932249|
|2|PERSIAN LUCENE PERSIAN STEM FILTER|PRIMARY_OUTPUT|0.502171|179 / 6748402 (0.002652%)|98018 / 98448 (99.563221%)|0.021312|0.008682|0.054801|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.915693|0.951121|0.998723|0.974922|0.998038|0.001962|
|2|PERSIAN LUCENE PERSIAN STEM FILTER|PRIMARY_OUTPUT|0.706076|0.004368|0.999973|0.502171|0.985658|0.014342|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.922566|0.933071|0.943818|0.874539|0.933239|0.932249|
|2|PERSIAN LUCENE PERSIAN STEM FILTER|PRIMARY_OUTPUT|0.021312|0.008682|0.005451|0.004360|0.055534|0.054801|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.932076|0.980343|0.984586|0.982460|0.982460|
|2|PERSIAN LUCENE PERSIAN STEM FILTER|PRIMARY_OUTPUT|0.008507|0.985686|0.520347|0.681125|0.681125|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|93636|8621|4812|6739781|8621 / 6748402|4812 / 98448|
|2|PERSIAN LUCENE PERSIAN STEM FILTER|PRIMARY_OUTPUT|430|179|98018|6748223|179 / 6748402|98018 / 98448|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 6748402 (0.000000%)|0 / 98448 (0.000000%)|1.000000|1.000000|1.000000|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|98448|0|0|6748402|0 / 6748402|0 / 98448|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999005|13433 / 6748402 (0.199055%)|0 / 98448 (0.000000%)|0.901585|0.936133|0.937114|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.879935|1.000000|0.998009|0.999005|0.998038|0.001962|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.901585|0.936133|0.973435|0.879935|0.938048|0.937114|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|98448|13433|0|6734969|13433 / 6748402|0 / 98448|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|4812|8621|4812|314|8.484193%|2|4015|
### `LOWERCASE_GROUPS_ONLY`
This mode contains **4 result rows**, **2 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.974922|8621 / 6748402 (0.127749%)|4812 / 98448 (4.887860%)|0.922566|0.933071|0.932249|
|2|PERSIAN LUCENE PERSIAN STEM FILTER|PRIMARY_OUTPUT|0.502171|179 / 6748402 (0.002652%)|98018 / 98448 (99.563221%)|0.021312|0.008682|0.054801|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.915693|0.951121|0.998723|0.974922|0.998038|0.001962|
|2|PERSIAN LUCENE PERSIAN STEM FILTER|PRIMARY_OUTPUT|0.706076|0.004368|0.999973|0.502171|0.985658|0.014342|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.922566|0.933071|0.943818|0.874539|0.933239|0.932249|
|2|PERSIAN LUCENE PERSIAN STEM FILTER|PRIMARY_OUTPUT|0.021312|0.008682|0.005451|0.004360|0.055534|0.054801|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.932076|0.980343|0.984586|0.982460|0.982460|
|2|PERSIAN LUCENE PERSIAN STEM FILTER|PRIMARY_OUTPUT|0.008507|0.985686|0.520347|0.681125|0.681125|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|93636|8621|4812|6739781|8621 / 6748402|4812 / 98448|
|2|PERSIAN LUCENE PERSIAN STEM FILTER|PRIMARY_OUTPUT|430|179|98018|6748223|179 / 6748402|98018 / 98448|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 6748402 (0.000000%)|0 / 98448 (0.000000%)|1.000000|1.000000|1.000000|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|98448|0|0|6748402|0 / 6748402|0 / 98448|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999005|13433 / 6748402 (0.199055%)|0 / 98448 (0.000000%)|0.901585|0.936133|0.937114|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.879935|1.000000|0.998009|0.999005|0.998038|0.001962|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.901585|0.936133|0.973435|0.879935|0.938048|0.937114|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|98448|13433|0|6734969|13433 / 6748402|0 / 98448|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|4812|8621|4812|314|8.484193%|2|4015|
### Output Policies and Metric Definitions
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
- Jaccard index: `TP / (TP + FP + FN)`.
- FowlkesMallows index: `sqrt(precision * recall)`.
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
### Provenance
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
- Evaluation command: `./gradlew stemmingQuality`
- Dictionary language: `FA_IR`
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
<!-- STEMMING-QUALITY:END -->

View File

@@ -55,3 +55,410 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs. - Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available. - Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees. - Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
<!-- STEMMING-QUALITY:START -->
## Stemming Quality
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `PL_PL` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
### Evaluation Scope and Key Findings
The dictionary resource is `src/main/resources/pl_pl/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.990388** among 5 deterministic stemmers. The runner-up is `POLISH LUCENE MORFOLOGIK FILTER` at 0.948154, a difference of 0.042234. This rank does not imply leadership in throughput or every secondary metric.
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.990579** among 5 deterministic stemmers. The runner-up is `POLISH LUCENE MORFOLOGIK FILTER` at 0.948177, a difference of 0.042402. This rank does not imply leadership in throughput or every secondary metric.
### `ALL_WORDS`
This mode contains **11 result rows**, **5 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.990388|13669 / 7482478003 (0.000183%)|21547 / 1120967 (1.922180%)|0.986324|0.984237|0.984241|
|2|POLISH LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.948154|99228 / 7482478003 (0.001326%)|116220 / 1120967 (10.367834%)|0.907324|0.903167|0.903179|
|3|HUNSPELL POLISH LUCENE FILTER|PRIMARY_OUTPUT|0.933222|52652 / 7482478003 (0.000704%)|149705 / 1120967 (13.354987%)|0.930930|0.905656|0.906571|
|4|POLISH LUCENE STEMPEL DIRECT|PRIMARY_OUTPUT|0.855748|66669 / 7482478003 (0.000891%)|323394 / 1120967 (28.849556%)|0.871106|0.803515|0.810296|
|5|POLISH LUCENE STEMPEL FILTER|PRIMARY_OUTPUT|0.855748|66669 / 7482478003 (0.000891%)|323394 / 1120967 (28.849556%)|0.871106|0.803515|0.810296|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.987720|0.980778|0.999998|0.990388|0.999995|0.000005|
|2|POLISH LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.910118|0.896322|0.999987|0.948154|0.999971|0.000029|
|3|HUNSPELL POLISH LUCENE FILTER|PRIMARY_OUTPUT|0.948578|0.866450|0.999993|0.933222|0.999973|0.000027|
|4|POLISH LUCENE STEMPEL DIRECT|PRIMARY_OUTPUT|0.922858|0.711504|0.999991|0.855748|0.999948|0.000052|
|5|POLISH LUCENE STEMPEL FILTER|PRIMARY_OUTPUT|0.922858|0.711504|0.999991|0.855748|0.999948|0.000052|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.986324|0.984237|0.982159|0.968963|0.984243|0.984241|
|2|POLISH LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.907324|0.903167|0.899047|0.823432|0.903193|0.903179|
|3|HUNSPELL POLISH LUCENE FILTER|PRIMARY_OUTPUT|0.930930|0.905656|0.881718|0.827579|0.906584|0.906571|
|4|POLISH LUCENE STEMPEL DIRECT|PRIMARY_OUTPUT|0.871106|0.803515|0.745659|0.671564|0.810320|0.810296|
|5|POLISH LUCENE STEMPEL FILTER|PRIMARY_OUTPUT|0.871106|0.803515|0.745659|0.671564|0.810320|0.810296|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.984234|0.996967|0.996469|0.996718|0.996718|
|2|POLISH LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.903153|0.990022|0.977054|0.983495|0.983495|
|3|HUNSPELL POLISH LUCENE FILTER|PRIMARY_OUTPUT|0.905642|0.994546|0.970520|0.982386|0.982386|
|4|POLISH LUCENE STEMPEL DIRECT|PRIMARY_OUTPUT|0.803490|0.991767|0.931069|0.960460|0.960460|
|5|POLISH LUCENE STEMPEL FILTER|PRIMARY_OUTPUT|0.803490|0.991767|0.931069|0.960460|0.960460|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|1099420|13669|21547|7482464334|13669 / 7482478003|21547 / 1120967|
|2|POLISH LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|1004747|99228|116220|7482378775|99228 / 7482478003|116220 / 1120967|
|3|HUNSPELL POLISH LUCENE FILTER|PRIMARY_OUTPUT|971262|52652|149705|7482425351|52652 / 7482478003|149705 / 1120967|
|4|POLISH LUCENE STEMPEL DIRECT|PRIMARY_OUTPUT|797573|66669|323394|7482411334|66669 / 7482478003|323394 / 1120967|
|5|POLISH LUCENE STEMPEL FILTER|PRIMARY_OUTPUT|797573|66669|323394|7482411334|66669 / 7482478003|323394 / 1120967|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 7482478003 (0.000000%)|0 / 1120967 (0.000000%)|1.000000|1.000000|1.000000|
|2|POLISH LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|0.987570|85532 / 7482478003 (0.001143%)|27855 / 1120967 (2.484908%)|0.936598|0.950693|0.950985|
|3|HUNSPELL POLISH LUCENE FILTER|ANY_CANDIDATE|0.963982|42213 / 7482478003 (0.000564%)|80743 / 1120967 (7.202977%)|0.954209|0.944197|0.944333|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|2|POLISH LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|0.927432|0.975151|0.999989|0.987570|0.999985|0.000015|
|3|HUNSPELL POLISH LUCENE FILTER|ANY_CANDIDATE|0.961002|0.927970|0.999994|0.963982|0.999984|0.000016|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|2|POLISH LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|0.936598|0.950693|0.965218|0.906020|0.950992|0.950985|
|3|HUNSPELL POLISH LUCENE FILTER|ANY_CANDIDATE|0.954209|0.944197|0.934394|0.894293|0.944342|0.944333|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|2|POLISH LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|3|HUNSPELL POLISH LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1120967|0|0|7482478003|0 / 7482478003|0 / 1120967|
|2|POLISH LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|1093112|85532|27855|7482392471|85532 / 7482478003|27855 / 1120967|
|3|HUNSPELL POLISH LUCENE FILTER|ANY_CANDIDATE|1040224|42213|80743|7482435790|42213 / 7482478003|80743 / 1120967|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999997|38073 / 7482478003 (0.000509%)|0 / 1120967 (0.000000%)|0.973547|0.983301|0.983436|
|2|POLISH LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|0.987566|143096 / 7482478003 (0.001912%)|27855 / 1120967 (2.484908%)|0.901045|0.927476|0.928576|
|3|HUNSPELL POLISH LUCENE FILTER|ALL_CANDIDATES|0.963980|82745 / 7482478003 (0.001106%)|80743 / 1120967 (7.202977%)|0.926646|0.927142|0.927132|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.967151|1.000000|0.999995|0.999997|0.999995|0.000005|
|2|POLISH LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|0.884246|0.975151|0.999981|0.987566|0.999977|0.000023|
|3|HUNSPELL POLISH LUCENE FILTER|ALL_CANDIDATES|0.926316|0.927970|0.999989|0.963980|0.999978|0.000022|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.973547|0.983301|0.993253|0.967151|0.983438|0.983436|
|2|POLISH LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|0.901045|0.927476|0.955505|0.864761|0.928587|0.928576|
|3|HUNSPELL POLISH LUCENE FILTER|ALL_CANDIDATES|0.926646|0.927142|0.927639|0.864180|0.927143|0.927132|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|2|POLISH LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|3|HUNSPELL POLISH LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|1120967|38073|0|7482439930|38073 / 7482478003|0 / 1120967|
|2|POLISH LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|1093112|143096|27855|7482334907|143096 / 7482478003|27855 / 1120967|
|3|HUNSPELL POLISH LUCENE FILTER|ALL_CANDIDATES|1040224|82745|80743|7482395258|82745 / 7482478003|80743 / 1120967|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|HUNSPELL POLISH LUCENE FILTER|68962|10439|30093|11447|9.356634%|6|135231|
|POLISH LUCENE MORFOLOGIK FILTER|88365|13696|43868|12873|10.522229%|5|136636|
|Radixor|21547|13669|24404|2866|2.342632%|4|125778|
### `LOWERCASE_GROUPS_ONLY`
This mode contains **11 result rows**, **5 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.990579|13669 / 7310252699 (0.000187%)|21000 / 1114651 (1.883998%)|0.986350|0.984397|0.984400|
|2|POLISH LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.948177|99224 / 7310252699 (0.001357%)|115513 / 1114651 (10.363154%)|0.906972|0.902966|0.902976|
|3|HUNSPELL POLISH LUCENE FILTER|PRIMARY_OUTPUT|0.933309|51950 / 7310252699 (0.000711%)|148667 / 1114651 (13.337538%)|0.931269|0.905928|0.906847|
|4|POLISH LUCENE STEMPEL DIRECT|PRIMARY_OUTPUT|0.856382|66274 / 7310252699 (0.000907%)|320158 / 1114651 (28.722712%)|0.871591|0.804380|0.811082|
|5|POLISH LUCENE STEMPEL FILTER|PRIMARY_OUTPUT|0.856382|66274 / 7310252699 (0.000907%)|320158 / 1114651 (28.722712%)|0.871591|0.804380|0.811082|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.987656|0.981160|0.999998|0.990579|0.999995|0.000005|
|2|POLISH LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.909662|0.896368|0.999986|0.948177|0.999971|0.000029|
|3|HUNSPELL POLISH LUCENE FILTER|PRIMARY_OUTPUT|0.948965|0.866625|0.999993|0.933309|0.999973|0.000027|
|4|POLISH LUCENE STEMPEL DIRECT|PRIMARY_OUTPUT|0.923006|0.712773|0.999991|0.856382|0.999947|0.000053|
|5|POLISH LUCENE STEMPEL FILTER|PRIMARY_OUTPUT|0.923006|0.712773|0.999991|0.856382|0.999947|0.000053|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.986350|0.984397|0.982452|0.969274|0.984403|0.984400|
|2|POLISH LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.906972|0.902966|0.898996|0.823098|0.902991|0.902976|
|3|HUNSPELL POLISH LUCENE FILTER|PRIMARY_OUTPUT|0.931269|0.905928|0.881929|0.828033|0.906861|0.906847|
|4|POLISH LUCENE STEMPEL DIRECT|PRIMARY_OUTPUT|0.871591|0.804380|0.746792|0.672772|0.811106|0.811082|
|5|POLISH LUCENE STEMPEL FILTER|PRIMARY_OUTPUT|0.871591|0.804380|0.746792|0.672772|0.811106|0.811082|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.984395|0.996926|0.996647|0.996786|0.996786|
|2|POLISH LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.902952|0.989889|0.977012|0.983408|0.983408|
|3|HUNSPELL POLISH LUCENE FILTER|PRIMARY_OUTPUT|0.905914|0.994584|0.970514|0.982402|0.982402|
|4|POLISH LUCENE STEMPEL DIRECT|PRIMARY_OUTPUT|0.804354|0.991711|0.931318|0.960566|0.960566|
|5|POLISH LUCENE STEMPEL FILTER|PRIMARY_OUTPUT|0.804354|0.991711|0.931318|0.960566|0.960566|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|1093651|13669|21000|7310239030|13669 / 7310252699|21000 / 1114651|
|2|POLISH LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|999138|99224|115513|7310153475|99224 / 7310252699|115513 / 1114651|
|3|HUNSPELL POLISH LUCENE FILTER|PRIMARY_OUTPUT|965984|51950|148667|7310200749|51950 / 7310252699|148667 / 1114651|
|4|POLISH LUCENE STEMPEL DIRECT|PRIMARY_OUTPUT|794493|66274|320158|7310186425|66274 / 7310252699|320158 / 1114651|
|5|POLISH LUCENE STEMPEL FILTER|PRIMARY_OUTPUT|794493|66274|320158|7310186425|66274 / 7310252699|320158 / 1114651|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 7310252699 (0.000000%)|0 / 1114651 (0.000000%)|1.000000|1.000000|1.000000|
|2|POLISH LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|0.987661|85532 / 7310252699 (0.001170%)|27494 / 1114651 (2.466602%)|0.936331|0.950586|0.950885|
|3|HUNSPELL POLISH LUCENE FILTER|ANY_CANDIDATE|0.963946|41671 / 7310252699 (0.000570%)|80368 / 1114651 (7.210149%)|0.954406|0.944290|0.944429|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|2|POLISH LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|0.927063|0.975334|0.999988|0.987661|0.999985|0.000015|
|3|HUNSPELL POLISH LUCENE FILTER|ANY_CANDIDATE|0.961271|0.927899|0.999994|0.963946|0.999983|0.000017|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|2|POLISH LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|0.936331|0.950586|0.965282|0.905826|0.950892|0.950885|
|3|HUNSPELL POLISH LUCENE FILTER|ANY_CANDIDATE|0.954406|0.944290|0.934386|0.894459|0.944437|0.944429|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|2|POLISH LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|3|HUNSPELL POLISH LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1114651|0|0|7310252699|0 / 7310252699|0 / 1114651|
|2|POLISH LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|1087157|85532|27494|7310167167|85532 / 7310252699|27494 / 1114651|
|3|HUNSPELL POLISH LUCENE FILTER|ANY_CANDIDATE|1034283|41671|80368|7310211028|41671 / 7310252699|80368 / 1114651|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999997|38073 / 7310252699 (0.000521%)|0 / 1114651 (0.000000%)|0.973401|0.983208|0.983344|
|2|POLISH LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|0.987657|143085 / 7310252699 (0.001957%)|27494 / 1114651 (2.466602%)|0.900618|0.927255|0.928372|
|3|HUNSPELL POLISH LUCENE FILTER|ALL_CANDIDATES|0.963944|81865 / 7310252699 (0.001120%)|80368 / 1114651 (7.210149%)|0.926903|0.927276|0.927265|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.966971|1.000000|0.999995|0.999997|0.999995|0.000005|
|2|POLISH LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|0.883694|0.975334|0.999980|0.987657|0.999977|0.000023|
|3|HUNSPELL POLISH LUCENE FILTER|ALL_CANDIDATES|0.926654|0.927899|0.999989|0.963944|0.999978|0.000022|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.973401|0.983208|0.993215|0.966971|0.983347|0.983344|
|2|POLISH LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|0.900618|0.927255|0.955516|0.864376|0.928384|0.928372|
|3|HUNSPELL POLISH LUCENE FILTER|ALL_CANDIDATES|0.926903|0.927276|0.927649|0.864412|0.927276|0.927265|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|2|POLISH LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|3|HUNSPELL POLISH LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|1114651|38073|0|7310214626|38073 / 7310252699|0 / 1114651|
|2|POLISH LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|1087157|143085|27494|7310109614|143085 / 7310252699|27494 / 1114651|
|3|HUNSPELL POLISH LUCENE FILTER|ALL_CANDIDATES|1034283|81865|80368|7310170834|81865 / 7310252699|80368 / 1114651|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|HUNSPELL POLISH LUCENE FILTER|68299|10279|29915|11265|9.315692%|6|133595|
|POLISH LUCENE MORFOLOGIK FILTER|88019|13692|43861|12763|10.554476%|5|135105|
|Radixor|21000|13669|24404|2780|2.298946%|4|124274|
### Output Policies and Metric Definitions
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
- Jaccard index: `TP / (TP + FP + FN)`.
- FowlkesMallows index: `sqrt(precision * recall)`.
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
### Provenance
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
- Evaluation command: `./gradlew stemmingQuality`
- Dictionary language: `PL_PL`
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
<!-- STEMMING-QUALITY:END -->

View File

@@ -57,3 +57,376 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs. - Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available. - Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees. - Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
<!-- STEMMING-QUALITY:START -->
## Stemming Quality
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `PT_PT` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
### Evaluation Scope and Key Findings
The dictionary resource is `src/main/resources/pt_pt/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.998502** among 6 deterministic stemmers. The runner-up is `SNOWBALL PORTUGUESE DIRECT` at 0.938800, a difference of 0.059702. This rank does not imply leadership in throughput or every secondary metric.
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.998502** among 6 deterministic stemmers. The runner-up is `SNOWBALL PORTUGUESE DIRECT` at 0.938800, a difference of 0.059702. This rank does not imply leadership in throughput or every secondary metric.
### `ALL_WORDS`
This mode contains **8 result rows**, **6 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.998502|20678 / 22358203756 (0.000092%)|16444 / 5489060 (0.299578%)|0.996389|0.996620|0.996619|
|2|SNOWBALL PORTUGUESE DIRECT|PRIMARY_OUTPUT|0.938800|167230 / 22358203756 (0.000748%)|671821 / 5489060 (12.239272%)|0.947271|0.919888|0.920940|
|3|SNOWBALL PORTUGUESE LUCENE FILTER|PRIMARY_OUTPUT|0.938800|167230 / 22358203756 (0.000748%)|671821 / 5489060 (12.239272%)|0.947271|0.919888|0.920940|
|4|PORTUGUESE LUCENE PORTUGUESE STEM FILTER|PRIMARY_OUTPUT|0.846459|99075 / 22358203756 (0.000443%)|1685572 / 5489060 (30.707844%)|0.901330|0.809975|0.821750|
|5|PORTUGUESE LUCENE PORTUGUESE LIGHT STEM FILTER|PRIMARY_OUTPUT|0.513648|2511 / 22358203756 (0.000011%)|5339230 / 5489060 (97.270389%)|0.122843|0.053118|0.163828|
|6|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.503954|598 / 22358203756 (0.000003%)|5445654 / 5489060 (99.209227%)|0.038310|0.015690|0.088308|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.996236|0.997004|0.999999|0.998502|0.999998|0.000002|
|2|SNOWBALL PORTUGUESE DIRECT|PRIMARY_OUTPUT|0.966450|0.877607|0.999993|0.938800|0.999962|0.000038|
|3|SNOWBALL PORTUGUESE LUCENE FILTER|PRIMARY_OUTPUT|0.966450|0.877607|0.999993|0.938800|0.999962|0.000038|
|4|PORTUGUESE LUCENE PORTUGUESE STEM FILTER|PRIMARY_OUTPUT|0.974613|0.692922|0.999996|0.846459|0.999920|0.000080|
|5|PORTUGUESE LUCENE PORTUGUESE LIGHT STEM FILTER|PRIMARY_OUTPUT|0.983517|0.027296|1.000000|0.513648|0.999761|0.000239|
|6|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.986410|0.007908|1.000000|0.503954|0.999756|0.000244|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.996389|0.996620|0.996850|0.993262|0.996620|0.996619|
|2|SNOWBALL PORTUGUESE DIRECT|PRIMARY_OUTPUT|0.947271|0.919888|0.894045|0.851661|0.920958|0.920940|
|3|SNOWBALL PORTUGUESE LUCENE FILTER|PRIMARY_OUTPUT|0.947271|0.919888|0.894045|0.851661|0.920958|0.920940|
|4|PORTUGUESE LUCENE PORTUGUESE STEM FILTER|PRIMARY_OUTPUT|0.901330|0.809975|0.735434|0.680636|0.821785|0.821750|
|5|PORTUGUESE LUCENE PORTUGUESE LIGHT STEM FILTER|PRIMARY_OUTPUT|0.122843|0.053118|0.033885|0.027284|0.163848|0.163828|
|6|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.038310|0.015690|0.009865|0.007907|0.088319|0.088308|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.996619|0.999299|0.999347|0.999323|0.999323|
|2|SNOWBALL PORTUGUESE DIRECT|PRIMARY_OUTPUT|0.919870|0.996663|0.967924|0.982083|0.982083|
|3|SNOWBALL PORTUGUESE LUCENE FILTER|PRIMARY_OUTPUT|0.919870|0.996663|0.967924|0.982083|0.982083|
|4|PORTUGUESE LUCENE PORTUGUESE STEM FILTER|PRIMARY_OUTPUT|0.809936|0.996729|0.918475|0.956003|0.956003|
|5|PORTUGUESE LUCENE PORTUGUESE LIGHT STEM FILTER|PRIMARY_OUTPUT|0.053105|0.999226|0.720580|0.837330|0.837330|
|6|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.015686|0.999664|0.692383|0.818122|0.818122|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|5472616|20678|16444|22358183078|20678 / 22358203756|16444 / 5489060|
|2|SNOWBALL PORTUGUESE DIRECT|PRIMARY_OUTPUT|4817239|167230|671821|22358036526|167230 / 22358203756|671821 / 5489060|
|3|SNOWBALL PORTUGUESE LUCENE FILTER|PRIMARY_OUTPUT|4817239|167230|671821|22358036526|167230 / 22358203756|671821 / 5489060|
|4|PORTUGUESE LUCENE PORTUGUESE STEM FILTER|PRIMARY_OUTPUT|3803488|99075|1685572|22358104681|99075 / 22358203756|1685572 / 5489060|
|5|PORTUGUESE LUCENE PORTUGUESE LIGHT STEM FILTER|PRIMARY_OUTPUT|149830|2511|5339230|22358201245|2511 / 22358203756|5339230 / 5489060|
|6|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|PRIMARY_OUTPUT|43406|598|5445654|22358203158|598 / 22358203756|5445654 / 5489060|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 22358203756 (0.000000%)|0 / 5489060 (0.000000%)|1.000000|1.000000|1.000000|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|5489060|0|0|22358203756|0 / 22358203756|0 / 5489060|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999999|38310 / 22358203756 (0.000171%)|0 / 5489060 (0.000000%)|0.994448|0.996522|0.996528|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.993069|1.000000|0.999998|0.999999|0.999998|0.000002|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.994448|0.996522|0.998606|0.993069|0.996528|0.996528|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|5489060|38310|0|22358165446|38310 / 22358203756|0 / 5489060|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|16444|20678|17632|790|0.373542%|3|212297|
### `LOWERCASE_GROUPS_ONLY`
This mode contains **8 result rows**, **6 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.998502|20678 / 22358203756 (0.000092%)|16444 / 5489060 (0.299578%)|0.996389|0.996620|0.996619|
|2|SNOWBALL PORTUGUESE DIRECT|PRIMARY_OUTPUT|0.938800|167230 / 22358203756 (0.000748%)|671821 / 5489060 (12.239272%)|0.947271|0.919888|0.920940|
|3|SNOWBALL PORTUGUESE LUCENE FILTER|PRIMARY_OUTPUT|0.938800|167230 / 22358203756 (0.000748%)|671821 / 5489060 (12.239272%)|0.947271|0.919888|0.920940|
|4|PORTUGUESE LUCENE PORTUGUESE STEM FILTER|PRIMARY_OUTPUT|0.846459|99075 / 22358203756 (0.000443%)|1685572 / 5489060 (30.707844%)|0.901330|0.809975|0.821750|
|5|PORTUGUESE LUCENE PORTUGUESE LIGHT STEM FILTER|PRIMARY_OUTPUT|0.513648|2511 / 22358203756 (0.000011%)|5339230 / 5489060 (97.270389%)|0.122843|0.053118|0.163828|
|6|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.503954|598 / 22358203756 (0.000003%)|5445654 / 5489060 (99.209227%)|0.038310|0.015690|0.088308|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.996236|0.997004|0.999999|0.998502|0.999998|0.000002|
|2|SNOWBALL PORTUGUESE DIRECT|PRIMARY_OUTPUT|0.966450|0.877607|0.999993|0.938800|0.999962|0.000038|
|3|SNOWBALL PORTUGUESE LUCENE FILTER|PRIMARY_OUTPUT|0.966450|0.877607|0.999993|0.938800|0.999962|0.000038|
|4|PORTUGUESE LUCENE PORTUGUESE STEM FILTER|PRIMARY_OUTPUT|0.974613|0.692922|0.999996|0.846459|0.999920|0.000080|
|5|PORTUGUESE LUCENE PORTUGUESE LIGHT STEM FILTER|PRIMARY_OUTPUT|0.983517|0.027296|1.000000|0.513648|0.999761|0.000239|
|6|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.986410|0.007908|1.000000|0.503954|0.999756|0.000244|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.996389|0.996620|0.996850|0.993262|0.996620|0.996619|
|2|SNOWBALL PORTUGUESE DIRECT|PRIMARY_OUTPUT|0.947271|0.919888|0.894045|0.851661|0.920958|0.920940|
|3|SNOWBALL PORTUGUESE LUCENE FILTER|PRIMARY_OUTPUT|0.947271|0.919888|0.894045|0.851661|0.920958|0.920940|
|4|PORTUGUESE LUCENE PORTUGUESE STEM FILTER|PRIMARY_OUTPUT|0.901330|0.809975|0.735434|0.680636|0.821785|0.821750|
|5|PORTUGUESE LUCENE PORTUGUESE LIGHT STEM FILTER|PRIMARY_OUTPUT|0.122843|0.053118|0.033885|0.027284|0.163848|0.163828|
|6|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.038310|0.015690|0.009865|0.007907|0.088319|0.088308|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.996619|0.999299|0.999347|0.999323|0.999323|
|2|SNOWBALL PORTUGUESE DIRECT|PRIMARY_OUTPUT|0.919870|0.996663|0.967924|0.982083|0.982083|
|3|SNOWBALL PORTUGUESE LUCENE FILTER|PRIMARY_OUTPUT|0.919870|0.996663|0.967924|0.982083|0.982083|
|4|PORTUGUESE LUCENE PORTUGUESE STEM FILTER|PRIMARY_OUTPUT|0.809936|0.996729|0.918475|0.956003|0.956003|
|5|PORTUGUESE LUCENE PORTUGUESE LIGHT STEM FILTER|PRIMARY_OUTPUT|0.053105|0.999226|0.720580|0.837330|0.837330|
|6|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.015686|0.999664|0.692383|0.818122|0.818122|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|5472616|20678|16444|22358183078|20678 / 22358203756|16444 / 5489060|
|2|SNOWBALL PORTUGUESE DIRECT|PRIMARY_OUTPUT|4817239|167230|671821|22358036526|167230 / 22358203756|671821 / 5489060|
|3|SNOWBALL PORTUGUESE LUCENE FILTER|PRIMARY_OUTPUT|4817239|167230|671821|22358036526|167230 / 22358203756|671821 / 5489060|
|4|PORTUGUESE LUCENE PORTUGUESE STEM FILTER|PRIMARY_OUTPUT|3803488|99075|1685572|22358104681|99075 / 22358203756|1685572 / 5489060|
|5|PORTUGUESE LUCENE PORTUGUESE LIGHT STEM FILTER|PRIMARY_OUTPUT|149830|2511|5339230|22358201245|2511 / 22358203756|5339230 / 5489060|
|6|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|PRIMARY_OUTPUT|43406|598|5445654|22358203158|598 / 22358203756|5445654 / 5489060|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 22358203756 (0.000000%)|0 / 5489060 (0.000000%)|1.000000|1.000000|1.000000|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|5489060|0|0|22358203756|0 / 22358203756|0 / 5489060|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999999|38310 / 22358203756 (0.000171%)|0 / 5489060 (0.000000%)|0.994448|0.996522|0.996528|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.993069|1.000000|0.999998|0.999999|0.999998|0.000002|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.994448|0.996522|0.998606|0.993069|0.996528|0.996528|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|5489060|38310|0|22358165446|38310 / 22358203756|0 / 5489060|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|16444|20678|17632|790|0.373542%|3|212297|
### Output Policies and Metric Definitions
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
- Jaccard index: `TP / (TP + FP + FN)`.
- FowlkesMallows index: `sqrt(precision * recall)`.
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
### Provenance
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
- Evaluation command: `./gradlew stemmingQuality`
- Dictionary language: `PT_PT`
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
<!-- STEMMING-QUALITY:END -->

View File

@@ -53,3 +53,356 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs. - Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available. - Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees. - Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
<!-- STEMMING-QUALITY:START -->
## Stemming Quality
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `RU_RU` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
### Evaluation Scope and Key Findings
The dictionary resource is `src/main/resources/ru_ru/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.989827** among 4 deterministic stemmers. The runner-up is `SNOWBALL RUSSIAN LUCENE FILTER` at 0.834876, a difference of 0.154951. This rank does not imply leadership in throughput or every secondary metric.
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.989852** among 4 deterministic stemmers. The runner-up is `SNOWBALL RUSSIAN DIRECT` at 0.834854, a difference of 0.154998. This rank does not imply leadership in throughput or every secondary metric.
### `ALL_WORDS`
This mode contains **6 result rows**, **4 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.989827|155850 / 295576291016 (0.000053%)|266302 / 13089505 (2.034470%)|0.986313|0.983806|0.983814|
|2|SNOWBALL RUSSIAN LUCENE FILTER|PRIMARY_OUTPUT|0.834876|3785790 / 295576291016 (0.001281%)|4322616 / 13089505 (33.023525%)|0.692485|0.683786|0.683923|
|3|SNOWBALL RUSSIAN DIRECT|PRIMARY_OUTPUT|0.834867|3782908 / 295576291016 (0.001280%)|4322849 / 13089505 (33.025305%)|0.692603|0.683851|0.683989|
|4|RUSSIAN LUCENE RUSSIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.617692|321183 / 295576291016 (0.000109%)|10008438 / 13089505 (76.461547%)|0.577011|0.373649|0.461687|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.987992|0.979655|0.999999|0.989827|0.999999|0.000001|
|2|SNOWBALL RUSSIAN LUCENE FILTER|PRIMARY_OUTPUT|0.698408|0.669765|0.999987|0.834876|0.999973|0.000027|
|3|SNOWBALL RUSSIAN DIRECT|PRIMARY_OUTPUT|0.698563|0.669747|0.999987|0.834867|0.999973|0.000027|
|4|RUSSIAN LUCENE RUSSIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.905597|0.235385|0.999999|0.617692|0.999965|0.000035|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.986313|0.983806|0.981311|0.968128|0.983815|0.983814|
|2|SNOWBALL RUSSIAN LUCENE FILTER|PRIMARY_OUTPUT|0.692485|0.683786|0.675304|0.519510|0.683936|0.683923|
|3|SNOWBALL RUSSIAN DIRECT|PRIMARY_OUTPUT|0.692603|0.683851|0.675318|0.519585|0.684003|0.683989|
|4|RUSSIAN LUCENE RUSSIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.577011|0.373649|0.276278|0.229747|0.461696|0.461687|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.983805|0.997699|0.997274|0.997487|0.997487|
|2|SNOWBALL RUSSIAN LUCENE FILTER|PRIMARY_OUTPUT|0.683773|0.974131|0.953674|0.963794|0.963794|
|3|SNOWBALL RUSSIAN DIRECT|PRIMARY_OUTPUT|0.683838|0.974180|0.953661|0.963811|0.963811|
|4|RUSSIAN LUCENE RUSSIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.373638|0.994311|0.870888|0.928516|0.928516|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|12823203|155850|266302|295576135166|155850 / 295576291016|266302 / 13089505|
|2|SNOWBALL RUSSIAN LUCENE FILTER|PRIMARY_OUTPUT|8766889|3785790|4322616|295572505226|3785790 / 295576291016|4322616 / 13089505|
|3|SNOWBALL RUSSIAN DIRECT|PRIMARY_OUTPUT|8766656|3782908|4322849|295572508108|3782908 / 295576291016|4322849 / 13089505|
|4|RUSSIAN LUCENE RUSSIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|3081067|321183|10008438|295575969833|321183 / 295576291016|10008438 / 13089505|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 295576291016 (0.000000%)|13 / 13089505 (0.000099%)|1.000000|1.000000|1.000000|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0.999999|1.000000|1.000000|1.000000|0.000000|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|0.999999|0.999999|1.000000|1.000000|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|13089492|0|13|295576291016|0 / 295576291016|13 / 13089505|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999999|434710 / 295576291016 (0.000147%)|13 / 13089505 (0.000099%)|0.974119|0.983665|0.983796|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.967857|0.999999|0.999999|0.999999|0.999999|0.000001|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.974119|0.983665|0.993401|0.967856|0.983797|0.983796|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|13089492|434710|13|295575856306|434710 / 295576291016|13 / 13089505|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|266289|155850|278860|19162|2.492190%|4|788492|
### `LOWERCASE_GROUPS_ONLY`
This mode contains **6 result rows**, **4 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.989852|155850 / 295000681652 (0.000053%)|265613 / 13087126 (2.029575%)|0.986322|0.983830|0.983838|
|2|SNOWBALL RUSSIAN DIRECT|PRIMARY_OUTPUT|0.834854|3782908 / 295000681652 (0.001282%)|4322407 / 13087126 (33.027931%)|0.692561|0.683815|0.683953|
|3|SNOWBALL RUSSIAN LUCENE FILTER|PRIMARY_OUTPUT|0.834854|3782908 / 295000681652 (0.001282%)|4322407 / 13087126 (33.027931%)|0.692561|0.683815|0.683953|
|4|RUSSIAN LUCENE RUSSIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.617630|318921 / 295000681652 (0.000108%)|10008238 / 13087126 (76.473918%)|0.577038|0.373540|0.461703|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.987991|0.979704|0.999999|0.989852|0.999999|0.000001|
|2|SNOWBALL RUSSIAN DIRECT|PRIMARY_OUTPUT|0.698516|0.669721|0.999987|0.834854|0.999973|0.000027|
|3|SNOWBALL RUSSIAN LUCENE FILTER|PRIMARY_OUTPUT|0.698516|0.669721|0.999987|0.834854|0.999973|0.000027|
|4|RUSSIAN LUCENE RUSSIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.906139|0.235261|0.999999|0.617630|0.999965|0.000035|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.986322|0.983830|0.981350|0.968175|0.983839|0.983838|
|2|SNOWBALL RUSSIAN DIRECT|PRIMARY_OUTPUT|0.692561|0.683815|0.675288|0.519544|0.683967|0.683953|
|3|SNOWBALL RUSSIAN LUCENE FILTER|PRIMARY_OUTPUT|0.692561|0.683815|0.675288|0.519544|0.683967|0.683953|
|4|RUSSIAN LUCENE RUSSIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.577038|0.373540|0.276152|0.229664|0.461713|0.461703|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.983829|0.997697|0.997321|0.997509|0.997509|
|2|SNOWBALL RUSSIAN DIRECT|PRIMARY_OUTPUT|0.683802|0.974149|0.953634|0.963782|0.963782|
|3|SNOWBALL RUSSIAN LUCENE FILTER|PRIMARY_OUTPUT|0.683802|0.974149|0.953634|0.963782|0.963782|
|4|RUSSIAN LUCENE RUSSIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.373528|0.994350|0.870767|0.928464|0.928464|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|12821513|155850|265613|295000525802|155850 / 295000681652|265613 / 13087126|
|2|SNOWBALL RUSSIAN DIRECT|PRIMARY_OUTPUT|8764719|3782908|4322407|294996898744|3782908 / 295000681652|4322407 / 13087126|
|3|SNOWBALL RUSSIAN LUCENE FILTER|PRIMARY_OUTPUT|8764719|3782908|4322407|294996898744|3782908 / 295000681652|4322407 / 13087126|
|4|RUSSIAN LUCENE RUSSIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|3078888|318921|10008238|295000362731|318921 / 295000681652|10008238 / 13087126|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 295000681652 (0.000000%)|0 / 13087126 (0.000000%)|1.000000|1.000000|1.000000|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|13087126|0|0|295000681652|0 / 295000681652|0 / 13087126|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999999|434710 / 295000681652 (0.000147%)|0 / 13087126 (0.000000%)|0.974115|0.983663|0.983794|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.967851|1.000000|0.999999|0.999999|0.999999|0.000001|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.974115|0.983663|0.993401|0.967851|0.983794|0.983794|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|13087126|434710|0|295000246942|434710 / 295000681652|0 / 13087126|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|265613|155850|278860|18991|2.472358%|4|787549|
### Output Policies and Metric Definitions
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
- Jaccard index: `TP / (TP + FP + FN)`.
- FowlkesMallows index: `sqrt(precision * recall)`.
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
### Provenance
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
- Evaluation command: `./gradlew stemmingQuality`
- Dictionary language: `RU_RU`
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
<!-- STEMMING-QUALITY:END -->

View File

@@ -59,3 +59,408 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs. - Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available. - Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees. - Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
<!-- STEMMING-QUALITY:START -->
## Stemming Quality
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `ES_ES` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
### Evaluation Scope and Key Findings
The dictionary resource is `src/main/resources/es_es/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.989295** among 7 deterministic stemmers. The runner-up is `SNOWBALL SPANISH LUCENE FILTER` at 0.652614, a difference of 0.336680. This rank does not imply leadership in throughput or every secondary metric.
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.989429** among 7 deterministic stemmers. The runner-up is `SNOWBALL SPANISH DIRECT` at 0.652720, a difference of 0.336709. This rank does not imply leadership in throughput or every secondary metric.
### `ALL_WORDS`
This mode contains **11 result rows**, **7 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.989295|288483 / 379567318110 (0.000076%)|898652 / 41973336 (2.141007%)|0.990105|0.985755|0.985780|
|2|SNOWBALL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.652614|2230481 / 379567318110 (0.000588%)|29161643 / 41973336 (69.476591%)|0.627151|0.449411|0.509848|
|3|SNOWBALL SPANISH DIRECT|PRIMARY_OUTPUT|0.652614|2228819 / 379567318110 (0.000587%)|29161649 / 41973336 (69.476605%)|0.627192|0.449424|0.509876|
|4|HUNSPELL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.615102|536192 / 379567318110 (0.000141%)|32310860 / 41973336 (76.979490%)|0.583708|0.370408|0.466992|
|5|SPANISH LUCENE SPANISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.514823|147956 / 379567318110 (0.000039%)|40729019 / 41973336 (97.035458%)|0.130864|0.057387|0.162762|
|6|SPANISH LUCENE SPANISH PLURAL STEM FILTER|PRIMARY_OUTPUT|0.503874|58578 / 379567318110 (0.000015%)|41648091 / 41973336 (99.225115%)|0.037377|0.015357|0.081026|
|7|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.501768|47859 / 379567318110 (0.000013%)|41824873 / 41973336 (99.646292%)|0.017361|0.007041|0.051714|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.993026|0.978590|0.999999|0.989295|0.999997|0.000003|
|2|SNOWBALL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.851718|0.305234|0.999994|0.652614|0.999917|0.000083|
|3|SNOWBALL SPANISH DIRECT|PRIMARY_OUTPUT|0.851812|0.305234|0.999994|0.652614|0.999917|0.000083|
|4|HUNSPELL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.947425|0.230205|0.999999|0.615102|0.999913|0.000087|
|5|SPANISH LUCENE SPANISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.893731|0.029645|1.000000|0.514823|0.999892|0.000108|
|6|SPANISH LUCENE SPANISH PLURAL STEM FILTER|PRIMARY_OUTPUT|0.847383|0.007749|1.000000|0.503874|0.999890|0.000110|
|7|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.756222|0.003537|1.000000|0.501768|0.999890|0.000110|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.990105|0.985755|0.981443|0.971910|0.985781|0.985780|
|2|SNOWBALL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.627151|0.449411|0.350170|0.289832|0.509876|0.509848|
|3|SNOWBALL SPANISH DIRECT|PRIMARY_OUTPUT|0.627192|0.449424|0.350173|0.289843|0.509904|0.509876|
|4|HUNSPELL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.583708|0.370408|0.271278|0.227301|0.467014|0.466992|
|5|SPANISH LUCENE SPANISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.130864|0.057387|0.036752|0.029541|0.162773|0.162762|
|6|SPANISH LUCENE SPANISH PLURAL STEM FILTER|PRIMARY_OUTPUT|0.037377|0.015357|0.009664|0.007738|0.081032|0.081026|
|7|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.017361|0.007041|0.004416|0.003533|0.051719|0.051714|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.985753|0.995418|0.993266|0.994341|0.994341|
|2|SNOWBALL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.449379|0.981386|0.852461|0.912391|0.912391|
|3|SNOWBALL SPANISH DIRECT|PRIMARY_OUTPUT|0.449392|0.981406|0.852463|0.912401|0.912401|
|4|HUNSPELL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.370381|0.993314|0.790558|0.880414|0.880414|
|5|SPANISH LUCENE SPANISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.057381|0.993824|0.756690|0.859195|0.859195|
|6|SPANISH LUCENE SPANISH PLURAL STEM FILTER|PRIMARY_OUTPUT|0.015355|0.995442|0.723731|0.838115|0.838115|
|7|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.007040|0.995635|0.710610|0.829316|0.829316|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|41074684|288483|898652|379567029627|288483 / 379567318110|898652 / 41973336|
|2|SNOWBALL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|12811693|2230481|29161643|379565087629|2230481 / 379567318110|29161643 / 41973336|
|3|SNOWBALL SPANISH DIRECT|PRIMARY_OUTPUT|12811687|2228819|29161649|379565089291|2228819 / 379567318110|29161649 / 41973336|
|4|HUNSPELL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|9662476|536192|32310860|379566781918|536192 / 379567318110|32310860 / 41973336|
|5|SPANISH LUCENE SPANISH LIGHT STEM FILTER|PRIMARY_OUTPUT|1244317|147956|40729019|379567170154|147956 / 379567318110|40729019 / 41973336|
|6|SPANISH LUCENE SPANISH PLURAL STEM FILTER|PRIMARY_OUTPUT|325245|58578|41648091|379567259532|58578 / 379567318110|41648091 / 41973336|
|7|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|148463|47859|41824873|379567270251|47859 / 379567318110|41824873 / 41973336|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|0.999993|2 / 379567318110 (0.000000%)|626 / 41973336 (0.001491%)|0.999997|0.999993|0.999993|
|2|HUNSPELL SPANISH LUCENE FILTER|ANY_CANDIDATE|0.620065|416345 / 379567318110 (0.000110%)|31894218 / 41973336 (75.986855%)|0.600268|0.384195|0.480192|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0.999985|1.000000|0.999993|1.000000|0.000000|
|2|HUNSPELL SPANISH LUCENE FILTER|ANY_CANDIDATE|0.960331|0.240131|0.999999|0.620065|0.999915|0.000085|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|0.999997|0.999993|0.999988|0.999985|0.999993|0.999993|
|2|HUNSPELL SPANISH LUCENE FILTER|ANY_CANDIDATE|0.600268|0.384195|0.282504|0.237773|0.480214|0.480192|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|2|HUNSPELL SPANISH LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|41972710|2|626|379567318108|2 / 379567318110|626 / 41973336|
|2|HUNSPELL SPANISH LUCENE FILTER|ANY_CANDIDATE|10079118|416345|31894218|379566901765|416345 / 379567318110|31894218 / 41973336|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999991|1349800 / 379567318110 (0.000356%)|626 / 41973336 (0.001491%)|0.974915|0.984168|0.984289|
|2|HUNSPELL SPANISH LUCENE FILTER|ALL_CANDIDATES|0.620065|888077 / 379567318110 (0.000234%)|31894218 / 41973336 (75.986855%)|0.587073|0.380771|0.469749|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.968843|0.999985|0.999996|0.999991|0.999996|0.000004|
|2|HUNSPELL SPANISH LUCENE FILTER|ALL_CANDIDATES|0.919024|0.240131|0.999998|0.620065|0.999914|0.000086|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.974915|0.984168|0.993598|0.968829|0.984291|0.984289|
|2|HUNSPELL SPANISH LUCENE FILTER|ALL_CANDIDATES|0.587073|0.380771|0.281759|0.235156|0.469773|0.469749|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|2|HUNSPELL SPANISH LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|41972710|1349800|626|379565968310|1349800 / 379567318110|626 / 41973336|
|2|HUNSPELL SPANISH LUCENE FILTER|ALL_CANDIDATES|10079118|888077|31894218|379566430033|888077 / 379567318110|31894218 / 41973336|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|HUNSPELL SPANISH LUCENE FILTER|416642|119847|351885|17877|2.051686%|5|890999|
|Radixor|898026|288481|1061317|42637|4.893313%|21|916797|
### `LOWERCASE_GROUPS_ONLY`
This mode contains **11 result rows**, **7 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.989429|276044 / 377860669765 (0.000073%)|885033 / 41863370 (2.114099%)|0.990385|0.986031|0.986056|
|2|SNOWBALL SPANISH DIRECT|PRIMARY_OUTPUT|0.652720|2201196 / 377860669765 (0.000583%)|29076352 / 41863370 (69.455354%)|0.627946|0.449839|0.510450|
|3|SNOWBALL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.652720|2201196 / 377860669765 (0.000583%)|29076352 / 41863370 (69.455354%)|0.627946|0.449839|0.510450|
|4|HUNSPELL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.614999|531181 / 377860669765 (0.000141%)|32234855 / 41863370 (77.000144%)|0.583531|0.370163|0.466854|
|5|SPANISH LUCENE SPANISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.514832|146613 / 377860669765 (0.000039%)|40621522 / 41863370 (97.033569%)|0.130949|0.057424|0.162875|
|6|SPANISH LUCENE SPANISH PLURAL STEM FILTER|PRIMARY_OUTPUT|0.503877|57716 / 377860669765 (0.000015%)|41538714 / 41863370 (99.224487%)|0.037409|0.015370|0.081139|
|7|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.501770|47148 / 377860669765 (0.000012%)|41715144 / 41863370 (99.645929%)|0.017379|0.007049|0.051824|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.993309|0.978859|0.999999|0.989429|0.999997|0.000003|
|2|SNOWBALL SPANISH DIRECT|PRIMARY_OUTPUT|0.853138|0.305446|0.999994|0.652720|0.999917|0.000083|
|3|SNOWBALL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.853138|0.305446|0.999994|0.652720|0.999917|0.000083|
|4|HUNSPELL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.947717|0.229999|0.999999|0.614999|0.999913|0.000087|
|5|SPANISH LUCENE SPANISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.894406|0.029664|1.000000|0.514832|0.999892|0.000108|
|6|SPANISH LUCENE SPANISH PLURAL STEM FILTER|PRIMARY_OUTPUT|0.849058|0.007755|1.000000|0.503877|0.999890|0.000110|
|7|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.758678|0.003541|1.000000|0.501770|0.999889|0.000111|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.990385|0.986031|0.981715|0.972447|0.986057|0.986056|
|2|SNOWBALL SPANISH DIRECT|PRIMARY_OUTPUT|0.627946|0.449839|0.350441|0.290188|0.510478|0.510450|
|3|SNOWBALL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.627946|0.449839|0.350441|0.290188|0.510478|0.510450|
|4|HUNSPELL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.583531|0.370163|0.271053|0.227117|0.466876|0.466854|
|5|SPANISH LUCENE SPANISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.130949|0.057424|0.036775|0.029561|0.162886|0.162875|
|6|SPANISH LUCENE SPANISH PLURAL STEM FILTER|PRIMARY_OUTPUT|0.037409|0.015370|0.009672|0.007744|0.081145|0.081139|
|7|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.017379|0.007049|0.004421|0.003537|0.051829|0.051824|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.986029|0.995464|0.993323|0.994392|0.994392|
|2|SNOWBALL SPANISH DIRECT|PRIMARY_OUTPUT|0.449806|0.981469|0.852556|0.912482|0.912482|
|3|SNOWBALL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.449806|0.981469|0.852556|0.912482|0.912482|
|4|HUNSPELL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.370136|0.993362|0.790500|0.880396|0.880396|
|5|SPANISH LUCENE SPANISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.057417|0.993866|0.756725|0.859234|0.859234|
|6|SPANISH LUCENE SPANISH PLURAL STEM FILTER|PRIMARY_OUTPUT|0.015368|0.995484|0.723753|0.838145|0.838145|
|7|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.007048|0.995676|0.710626|0.829341|0.829341|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|40978337|276044|885033|377860393721|276044 / 377860669765|885033 / 41863370|
|2|SNOWBALL SPANISH DIRECT|PRIMARY_OUTPUT|12787018|2201196|29076352|377858468569|2201196 / 377860669765|29076352 / 41863370|
|3|SNOWBALL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|12787018|2201196|29076352|377858468569|2201196 / 377860669765|29076352 / 41863370|
|4|HUNSPELL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|9628515|531181|32234855|377860138584|531181 / 377860669765|32234855 / 41863370|
|5|SPANISH LUCENE SPANISH LIGHT STEM FILTER|PRIMARY_OUTPUT|1241848|146613|40621522|377860523152|146613 / 377860669765|40621522 / 41863370|
|6|SPANISH LUCENE SPANISH PLURAL STEM FILTER|PRIMARY_OUTPUT|324656|57716|41538714|377860612049|57716 / 377860669765|41538714 / 41863370|
|7|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|148226|47148|41715144|377860622617|47148 / 377860669765|41715144 / 41863370|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 377860669765 (0.000000%)|0 / 41863370 (0.000000%)|1.000000|1.000000|1.000000|
|2|HUNSPELL SPANISH LUCENE FILTER|ANY_CANDIDATE|0.619928|412198 / 377860669765 (0.000109%)|31822108 / 41863370 (76.014205%)|0.600000|0.383864|0.479978|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|2|HUNSPELL SPANISH LUCENE FILTER|ANY_CANDIDATE|0.960568|0.239858|0.999999|0.619928|0.999915|0.000085|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|2|HUNSPELL SPANISH LUCENE FILTER|ANY_CANDIDATE|0.600000|0.383864|0.282205|0.237519|0.480000|0.479978|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|2|HUNSPELL SPANISH LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|41863370|0|0|377860669765|0 / 377860669765|0 / 41863370|
|2|HUNSPELL SPANISH LUCENE FILTER|ANY_CANDIDATE|10041262|412198|31822108|377860257567|412198 / 377860669765|31822108 / 41863370|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999998|1255381 / 377860669765 (0.000332%)|0 / 41863370 (0.000000%)|0.976572|0.985228|0.985334|
|2|HUNSPELL SPANISH LUCENE FILTER|ALL_CANDIDATES|0.619928|878949 / 377860669765 (0.000233%)|31822108 / 41863370 (76.014205%)|0.586905|0.380469|0.469606|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.970885|1.000000|0.999997|0.999998|0.999997|0.000003|
|2|HUNSPELL SPANISH LUCENE FILTER|ALL_CANDIDATES|0.919512|0.239858|0.999998|0.619928|0.999913|0.000087|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.976572|0.985228|0.994038|0.970885|0.985335|0.985334|
|2|HUNSPELL SPANISH LUCENE FILTER|ALL_CANDIDATES|0.586905|0.380469|0.281467|0.234926|0.469630|0.469606|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|2|HUNSPELL SPANISH LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|41863370|1255381|0|377859414384|1255381 / 377860669765|0 / 41863370|
|2|HUNSPELL SPANISH LUCENE FILTER|ALL_CANDIDATES|10041262|878949|31822108|377859790816|878949 / 377860669765|31822108 / 41863370|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|HUNSPELL SPANISH LUCENE FILTER|412747|118983|347768|17807|2.048262%|5|888962|
|Radixor|885033|276044|979337|42403|4.877434%|21|914127|
### Output Policies and Metric Definitions
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
- Jaccard index: `TP / (TP + FP + FN)`.
- FowlkesMallows index: `sqrt(precision * recall)`.
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
### Provenance
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
- Evaluation command: `./gradlew stemmingQuality`
- Dictionary language: `ES_ES`
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
<!-- STEMMING-QUALITY:END -->

View File

@@ -55,3 +55,366 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs. - Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available. - Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees. - Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
<!-- STEMMING-QUALITY:START -->
## Stemming Quality
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `SV_SE` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
### Evaluation Scope and Key Findings
The dictionary resource is `src/main/resources/sv_se/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.974636** among 5 deterministic stemmers. The runner-up is `SNOWBALL SWEDISH DIRECT` at 0.807534, a difference of 0.167101. This rank does not imply leadership in throughput or every secondary metric.
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.974584** among 5 deterministic stemmers. The runner-up is `SNOWBALL SWEDISH DIRECT` at 0.807599, a difference of 0.166985. This rank does not imply leadership in throughput or every secondary metric.
### `ALL_WORDS`
This mode contains **7 result rows**, **5 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.974636|24473 / 4812155436 (0.000509%)|19546 / 385342 (5.072377%)|0.939665|0.943246|0.943260|
|2|SNOWBALL SWEDISH DIRECT|PRIMARY_OUTPUT|0.807534|67105 / 4812155436 (0.001394%)|148325 / 385342 (38.491781%)|0.739832|0.687540|0.692339|
|3|SNOWBALL SWEDISH LUCENE FILTER|PRIMARY_OUTPUT|0.799307|64262 / 4812155436 (0.001335%)|154666 / 385342 (40.137333%)|0.736940|0.678180|0.684227|
|4|SWEDISH LUCENE SWEDISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.796072|40227 / 4812155436 (0.000836%)|157161 / 385342 (40.784809%)|0.781991|0.698068|0.709491|
|5|SWEDISH LUCENE SWEDISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.783685|45941 / 4812155436 (0.000955%)|166707 / 385342 (43.262089%)|0.757232|0.672808|0.684713|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.937292|0.949276|0.999995|0.974636|0.999991|0.000009|
|2|SNOWBALL SWEDISH DIRECT|PRIMARY_OUTPUT|0.779348|0.615082|0.999986|0.807534|0.999955|0.000045|
|3|SNOWBALL SWEDISH LUCENE FILTER|PRIMARY_OUTPUT|0.782117|0.598627|0.999987|0.799307|0.999955|0.000045|
|4|SWEDISH LUCENE SWEDISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.850127|0.592152|0.999992|0.796072|0.999959|0.000041|
|5|SWEDISH LUCENE SWEDISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.826360|0.567379|0.999990|0.783685|0.999956|0.000044|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.939665|0.943246|0.946855|0.892588|0.943265|0.943260|
|2|SNOWBALL SWEDISH DIRECT|PRIMARY_OUTPUT|0.739832|0.687540|0.642152|0.523856|0.692361|0.692339|
|3|SNOWBALL SWEDISH LUCENE FILTER|PRIMARY_OUTPUT|0.736940|0.678180|0.628098|0.513065|0.684249|0.684227|
|4|SWEDISH LUCENE SWEDISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.781991|0.698068|0.630412|0.536179|0.709510|0.709491|
|5|SWEDISH LUCENE SWEDISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.757232|0.672808|0.605321|0.506941|0.684733|0.684713|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.943241|0.992631|0.993395|0.993013|0.993013|
|2|SNOWBALL SWEDISH DIRECT|PRIMARY_OUTPUT|0.687518|0.984860|0.942685|0.963311|0.963311|
|3|SNOWBALL SWEDISH LUCENE FILTER|PRIMARY_OUTPUT|0.678157|0.985207|0.939659|0.961894|0.961894|
|4|SWEDISH LUCENE SWEDISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.698048|0.988493|0.944582|0.966038|0.966038|
|5|SWEDISH LUCENE SWEDISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.672787|0.986795|0.942303|0.964036|0.964036|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|365796|24473|19546|4812130963|24473 / 4812155436|19546 / 385342|
|2|SNOWBALL SWEDISH DIRECT|PRIMARY_OUTPUT|237017|67105|148325|4812088331|67105 / 4812155436|148325 / 385342|
|3|SNOWBALL SWEDISH LUCENE FILTER|PRIMARY_OUTPUT|230676|64262|154666|4812091174|64262 / 4812155436|154666 / 385342|
|4|SWEDISH LUCENE SWEDISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|228181|40227|157161|4812115209|40227 / 4812155436|157161 / 385342|
|5|SWEDISH LUCENE SWEDISH LIGHT STEM FILTER|PRIMARY_OUTPUT|218635|45941|166707|4812109495|45941 / 4812155436|166707 / 385342|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 4812155436 (0.000000%)|0 / 385342 (0.000000%)|1.000000|1.000000|1.000000|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|385342|0|0|4812155436|0 / 4812155436|0 / 385342|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999995|47848 / 4812155436 (0.000994%)|0 / 385342 (0.000000%)|0.909640|0.941544|0.943152|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.889545|1.000000|0.999990|0.999995|0.999990|0.000010|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.909640|0.941544|0.975768|0.889545|0.943157|0.943152|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|385342|47848|0|4812107588|47848 / 4812155436|0 / 385342|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|19546|24473|23375|5767|5.878216%|5|104148|
### `LOWERCASE_GROUPS_ONLY`
This mode contains **7 result rows**, **5 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.974584|24473 / 4789911577 (0.000511%)|19546 / 384563 (5.082652%)|0.939544|0.943132|0.943146|
|2|SNOWBALL SWEDISH DIRECT|PRIMARY_OUTPUT|0.807599|67105 / 4789911577 (0.001401%)|147975 / 384563 (38.478741%)|0.739645|0.687500|0.692274|
|3|SNOWBALL SWEDISH LUCENE FILTER|PRIMARY_OUTPUT|0.799355|64262 / 4789911577 (0.001342%)|154316 / 384563 (40.127625%)|0.736744|0.678122|0.684143|
|4|SWEDISH LUCENE SWEDISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.795947|40227 / 4789911577 (0.000840%)|156939 / 384563 (40.809698%)|0.781694|0.697790|0.709212|
|5|SWEDISH LUCENE SWEDISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.783598|45941 / 4789911577 (0.000959%)|166437 / 384563 (43.279515%)|0.756945|0.672575|0.684469|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.937167|0.949173|0.999995|0.974584|0.999991|0.000009|
|2|SNOWBALL SWEDISH DIRECT|PRIMARY_OUTPUT|0.779037|0.615213|0.999986|0.807599|0.999955|0.000045|
|3|SNOWBALL SWEDISH LUCENE FILTER|PRIMARY_OUTPUT|0.781800|0.598724|0.999987|0.799355|0.999954|0.000046|
|4|SWEDISH LUCENE SWEDISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.849816|0.591903|0.999992|0.795947|0.999959|0.000041|
|5|SWEDISH LUCENE SWEDISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.826025|0.567205|0.999990|0.783598|0.999956|0.000044|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.939544|0.943132|0.946748|0.892384|0.943151|0.943146|
|2|SNOWBALL SWEDISH DIRECT|PRIMARY_OUTPUT|0.739645|0.687500|0.642223|0.523810|0.692296|0.692274|
|3|SNOWBALL SWEDISH LUCENE FILTER|PRIMARY_OUTPUT|0.736744|0.678122|0.628142|0.512999|0.684165|0.684143|
|4|SWEDISH LUCENE SWEDISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.781694|0.697790|0.630152|0.535851|0.709231|0.709212|
|5|SWEDISH LUCENE SWEDISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.756945|0.672575|0.605126|0.506676|0.684489|0.684469|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.943127|0.992612|0.993378|0.992995|0.992995|
|2|SNOWBALL SWEDISH DIRECT|PRIMARY_OUTPUT|0.687478|0.984821|0.942695|0.963298|0.963298|
|3|SNOWBALL SWEDISH LUCENE FILTER|PRIMARY_OUTPUT|0.678100|0.985169|0.939661|0.961877|0.961877|
|4|SWEDISH LUCENE SWEDISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.697770|0.988463|0.944528|0.965996|0.965996|
|5|SWEDISH LUCENE SWEDISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.672553|0.986761|0.942265|0.964000|0.964000|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|365017|24473|19546|4789887104|24473 / 4789911577|19546 / 384563|
|2|SNOWBALL SWEDISH DIRECT|PRIMARY_OUTPUT|236588|67105|147975|4789844472|67105 / 4789911577|147975 / 384563|
|3|SNOWBALL SWEDISH LUCENE FILTER|PRIMARY_OUTPUT|230247|64262|154316|4789847315|64262 / 4789911577|154316 / 384563|
|4|SWEDISH LUCENE SWEDISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|227624|40227|156939|4789871350|40227 / 4789911577|156939 / 384563|
|5|SWEDISH LUCENE SWEDISH LIGHT STEM FILTER|PRIMARY_OUTPUT|218126|45941|166437|4789865636|45941 / 4789911577|166437 / 384563|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 4789911577 (0.000000%)|0 / 384563 (0.000000%)|1.000000|1.000000|1.000000|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|384563|0|0|4789911577|0 / 4789911577|0 / 384563|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999995|47848 / 4789911577 (0.000999%)|0 / 384563 (0.000000%)|0.909473|0.941433|0.943047|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.889346|1.000000|0.999990|0.999995|0.999990|0.000010|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.909473|0.941433|0.975720|0.889346|0.943051|0.943047|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|384563|47848|0|4789863729|47848 / 4789911577|0 / 384563|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|19546|24473|23375|5767|5.891848%|5|103921|
### Output Policies and Metric Definitions
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
- Jaccard index: `TP / (TP + FP + FN)`.
- FowlkesMallows index: `sqrt(precision * recall)`.
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
### Provenance
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
- Evaluation command: `./gradlew stemmingQuality`
- Dictionary language: `SV_SE`
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
<!-- STEMMING-QUALITY:END -->

View File

@@ -53,3 +53,422 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs. - Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available. - Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees. - Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
<!-- STEMMING-QUALITY:START -->
## Stemming Quality
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `UK_UA` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
### Evaluation Scope and Key Findings
The dictionary resource is `src/main/resources/uk_ua/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.995343** among 4 deterministic stemmers. The runner-up is `UKRAINIAN LUCENE MORFOLOGIK FILTER` at 0.928768, a difference of 0.066575. This rank does not imply leadership in throughput or every secondary metric.
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.995342** among 4 deterministic stemmers. The runner-up is `UKRAINIAN LUCENE MORFOLOGIK FILTER` at 0.928751, a difference of 0.066591. This rank does not imply leadership in throughput or every secondary metric.
### `ALL_WORDS`
This mode contains **12 result rows**, **4 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.995343|880 / 101387550 (0.000868%)|608 / 65340 (0.930517%)|0.987406|0.988637|0.988632|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.928768|828 / 101387550 (0.000817%)|9308 / 65340 (14.245485%)|0.956896|0.917054|0.919223|
|3|UKRAINIAN MORFOLOGIK DIRECT|PRIMARY_OUTPUT|0.928646|828 / 101387550 (0.000817%)|9324 / 65340 (14.269972%)|0.956832|0.916912|0.919090|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|PRIMARY_OUTPUT|0.885793|794 / 101387550 (0.000783%)|14924 / 65340 (22.840526%)|0.933008|0.865139|0.871499|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.986588|0.990695|0.999991|0.995343|0.999985|0.000015|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.985438|0.857545|0.999992|0.928768|0.999900|0.000100|
|3|UKRAINIAN MORFOLOGIK DIRECT|PRIMARY_OUTPUT|0.985434|0.857300|0.999992|0.928646|0.999900|0.000100|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|PRIMARY_OUTPUT|0.984495|0.771595|0.999992|0.885793|0.999845|0.000155|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.987406|0.988637|0.989871|0.977529|0.988639|0.988632|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.956896|0.917054|0.880397|0.846814|0.919270|0.919223|
|3|UKRAINIAN MORFOLOGIK DIRECT|PRIMARY_OUTPUT|0.956832|0.916912|0.880190|0.846572|0.919137|0.919090|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|PRIMARY_OUTPUT|0.933008|0.865139|0.806475|0.762331|0.871568|0.871499|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.988630|0.997994|0.998266|0.998130|0.998130|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.917004|0.997990|0.971000|0.984310|0.984310|
|3|UKRAINIAN MORFOLOGIK DIRECT|PRIMARY_OUTPUT|0.916862|0.997990|0.970876|0.984246|0.984246|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|PRIMARY_OUTPUT|0.865063|0.998114|0.949804|0.973360|0.973360|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|64732|880|608|101386670|880 / 101387550|608 / 65340|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|56032|828|9308|101386722|828 / 101387550|9308 / 65340|
|3|UKRAINIAN MORFOLOGIK DIRECT|PRIMARY_OUTPUT|56016|828|9324|101386722|828 / 101387550|9324 / 65340|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|PRIMARY_OUTPUT|50416|794|14924|101386756|794 / 101387550|14924 / 65340|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 101387550 (0.000000%)|0 / 65340 (0.000000%)|1.000000|1.000000|1.000000|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|0.962151|122 / 101387550 (0.000120%)|4946 / 65340 (7.569636%)|0.982323|0.959732|0.960413|
|3|UKRAINIAN MORFOLOGIK DIRECT|ANY_CANDIDATE|0.962029|122 / 101387550 (0.000120%)|4962 / 65340 (7.594123%)|0.982267|0.959599|0.960286|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ANY_CANDIDATE|0.927570|326 / 101387550 (0.000322%)|9465 / 65340 (14.485767%)|0.962884|0.919443|0.922008|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|0.997984|0.924304|0.999999|0.962151|0.999950|0.000050|
|3|UKRAINIAN MORFOLOGIK DIRECT|ANY_CANDIDATE|0.997983|0.924059|0.999999|0.962029|0.999950|0.000050|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ANY_CANDIDATE|0.994199|0.855142|0.999997|0.927570|0.999903|0.000097|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|0.982323|0.959732|0.938156|0.922581|0.960438|0.960413|
|3|UKRAINIAN MORFOLOGIK DIRECT|ANY_CANDIDATE|0.982267|0.959599|0.937954|0.922337|0.960310|0.960286|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ANY_CANDIDATE|0.962884|0.919443|0.879752|0.850897|0.922053|0.922008|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|3|UKRAINIAN MORFOLOGIK DIRECT|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|65340|0|0|101387550|0 / 101387550|0 / 65340|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|60394|122|4946|101387428|122 / 101387550|4946 / 65340|
|3|UKRAINIAN MORFOLOGIK DIRECT|ANY_CANDIDATE|60378|122|4962|101387428|122 / 101387550|4962 / 65340|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ANY_CANDIDATE|55875|326|9465|101387224|326 / 101387550|9465 / 65340|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999993|1490 / 101387550 (0.001470%)|0 / 65340 (0.000000%)|0.982084|0.988727|0.988782|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|0.962145|1368 / 101387550 (0.001349%)|4946 / 65340 (7.569636%)|0.966650|0.950323|0.950669|
|3|UKRAINIAN MORFOLOGIK DIRECT|ALL_CANDIDATES|0.962023|1368 / 101387550 (0.001349%)|4962 / 65340 (7.594123%)|0.966592|0.950191|0.950541|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ALL_CANDIDATES|0.927565|1271 / 101387550 (0.001254%)|9465 / 65340 (14.485767%)|0.950501|0.912349|0.914347|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.977705|1.000000|0.999985|0.999993|0.999985|0.000015|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|0.977850|0.924304|0.999987|0.962145|0.999938|0.000062|
|3|UKRAINIAN MORFOLOGIK DIRECT|ALL_CANDIDATES|0.977845|0.924059|0.999987|0.962023|0.999938|0.000062|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ALL_CANDIDATES|0.977759|0.855142|0.999987|0.927565|0.999894|0.000106|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.982084|0.988727|0.995460|0.977705|0.988789|0.988782|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|0.966650|0.950323|0.934539|0.905349|0.950700|0.950669|
|3|UKRAINIAN MORFOLOGIK DIRECT|ALL_CANDIDATES|0.966592|0.950191|0.934337|0.905109|0.950571|0.950541|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ALL_CANDIDATES|0.950501|0.912349|0.877142|0.838825|0.914398|0.914347|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|3|UKRAINIAN MORFOLOGIK DIRECT|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|65340|1490|0|101386060|1490 / 101387550|0 / 65340|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|60394|1368|4946|101386182|1368 / 101387550|4946 / 65340|
|3|UKRAINIAN MORFOLOGIK DIRECT|ALL_CANDIDATES|60378|1368|4962|101386182|1368 / 101387550|4962 / 65340|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ALL_CANDIDATES|55875|1271|9465|101386279|1271 / 101387550|9465 / 65340|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|HUNSPELL UKRAINIAN LUCENE FILTER|5459|468|477|1322|9.280449%|6|15740|
|UKRAINIAN LUCENE MORFOLOGIK FILTER|4362|706|540|2207|15.493155%|6|16937|
|UKRAINIAN MORFOLOGIK DIRECT|4362|706|540|2207|15.493155%|6|16937|
|Radixor|608|880|610|190|1.333801%|2|14435|
### `LOWERCASE_GROUPS_ONLY`
This mode contains **12 result rows**, **4 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.995342|880 / 101259406 (0.000869%)|608 / 65324 (0.930745%)|0.987403|0.988634|0.988629|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.928751|828 / 101259406 (0.000818%)|9308 / 65324 (14.248974%)|0.956884|0.917032|0.919202|
|3|UKRAINIAN MORFOLOGIK DIRECT|PRIMARY_OUTPUT|0.928751|828 / 101259406 (0.000818%)|9308 / 65324 (14.248974%)|0.956884|0.917032|0.919202|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|PRIMARY_OUTPUT|0.885796|794 / 101259406 (0.000784%)|14920 / 65324 (22.839998%)|0.933007|0.865141|0.871500|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.986585|0.990693|0.999991|0.995342|0.999985|0.000015|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.985434|0.857510|0.999992|0.928751|0.999900|0.000100|
|3|UKRAINIAN MORFOLOGIK DIRECT|PRIMARY_OUTPUT|0.985434|0.857510|0.999992|0.928751|0.999900|0.000100|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|PRIMARY_OUTPUT|0.984492|0.771600|0.999992|0.885796|0.999845|0.000155|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.987403|0.988634|0.989868|0.977524|0.988636|0.988629|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.956884|0.917032|0.880367|0.846777|0.919249|0.919202|
|3|UKRAINIAN MORFOLOGIK DIRECT|PRIMARY_OUTPUT|0.956884|0.917032|0.880367|0.846777|0.919249|0.919202|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|PRIMARY_OUTPUT|0.933007|0.865141|0.806479|0.762334|0.871570|0.871500|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.988627|0.997992|0.998264|0.998128|0.998128|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.916982|0.997988|0.970978|0.984298|0.984298|
|3|UKRAINIAN MORFOLOGIK DIRECT|PRIMARY_OUTPUT|0.916982|0.997988|0.970978|0.984298|0.984298|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|PRIMARY_OUTPUT|0.865065|0.998113|0.949788|0.973351|0.973351|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|64716|880|608|101258526|880 / 101259406|608 / 65324|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|56016|828|9308|101258578|828 / 101259406|9308 / 65324|
|3|UKRAINIAN MORFOLOGIK DIRECT|PRIMARY_OUTPUT|56016|828|9308|101258578|828 / 101259406|9308 / 65324|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|PRIMARY_OUTPUT|50404|794|14920|101258612|794 / 101259406|14920 / 65324|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 101259406 (0.000000%)|0 / 65324 (0.000000%)|1.000000|1.000000|1.000000|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|0.962142|122 / 101259406 (0.000120%)|4946 / 65324 (7.571490%)|0.982318|0.959722|0.960404|
|3|UKRAINIAN MORFOLOGIK DIRECT|ANY_CANDIDATE|0.962142|122 / 101259406 (0.000120%)|4946 / 65324 (7.571490%)|0.982318|0.959722|0.960404|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ANY_CANDIDATE|0.927552|326 / 101259406 (0.000322%)|9465 / 65324 (14.489315%)|0.962874|0.919422|0.921988|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|0.997983|0.924285|0.999999|0.962142|0.999950|0.000050|
|3|UKRAINIAN MORFOLOGIK DIRECT|ANY_CANDIDATE|0.997983|0.924285|0.999999|0.962142|0.999950|0.000050|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ANY_CANDIDATE|0.994198|0.855107|0.999997|0.927552|0.999903|0.000097|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|0.982318|0.959722|0.938141|0.922562|0.960428|0.960404|
|3|UKRAINIAN MORFOLOGIK DIRECT|ANY_CANDIDATE|0.982318|0.959722|0.938141|0.922562|0.960428|0.960404|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ANY_CANDIDATE|0.962874|0.919422|0.879722|0.850861|0.922033|0.921988|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|3|UKRAINIAN MORFOLOGIK DIRECT|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|65324|0|0|101259406|0 / 101259406|0 / 65324|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|60378|122|4946|101259284|122 / 101259406|4946 / 65324|
|3|UKRAINIAN MORFOLOGIK DIRECT|ANY_CANDIDATE|60378|122|4946|101259284|122 / 101259406|4946 / 65324|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ANY_CANDIDATE|55859|326|9465|101259080|326 / 101259406|9465 / 65324|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999993|1490 / 101259406 (0.001471%)|0 / 65324 (0.000000%)|0.982079|0.988724|0.988779|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|0.962136|1368 / 101259406 (0.001351%)|4946 / 65324 (7.571490%)|0.966642|0.950311|0.950657|
|3|UKRAINIAN MORFOLOGIK DIRECT|ALL_CANDIDATES|0.962136|1368 / 101259406 (0.001351%)|4946 / 65324 (7.571490%)|0.966642|0.950311|0.950657|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ALL_CANDIDATES|0.927547|1271 / 101259406 (0.001255%)|9465 / 65324 (14.489315%)|0.950487|0.912326|0.914325|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.977699|1.000000|0.999985|0.999993|0.999985|0.000015|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|0.977845|0.924285|0.999986|0.962136|0.999938|0.000062|
|3|UKRAINIAN MORFOLOGIK DIRECT|ALL_CANDIDATES|0.977845|0.924285|0.999986|0.962136|0.999938|0.000062|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ALL_CANDIDATES|0.977752|0.855107|0.999987|0.927547|0.999894|0.000106|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.982079|0.988724|0.995459|0.977699|0.988787|0.988779|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|0.966642|0.950311|0.934522|0.905326|0.950688|0.950657|
|3|UKRAINIAN MORFOLOGIK DIRECT|ALL_CANDIDATES|0.966642|0.950311|0.934522|0.905326|0.950688|0.950657|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ALL_CANDIDATES|0.950487|0.912326|0.877111|0.838787|0.914376|0.914325|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|3|UKRAINIAN MORFOLOGIK DIRECT|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|65324|1490|0|101257916|1490 / 101259406|0 / 65324|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|60378|1368|4946|101258038|1368 / 101259406|4946 / 65324|
|3|UKRAINIAN MORFOLOGIK DIRECT|ALL_CANDIDATES|60378|1368|4946|101258038|1368 / 101259406|4946 / 65324|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ALL_CANDIDATES|55859|1271|9465|101258135|1271 / 101259406|9465 / 65324|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|HUNSPELL UKRAINIAN LUCENE FILTER|5455|468|477|1321|9.279292%|6|15730|
|UKRAINIAN LUCENE MORFOLOGIK FILTER|4362|706|540|2207|15.502950%|6|16928|
|UKRAINIAN MORFOLOGIK DIRECT|4362|706|540|2207|15.502950%|6|16928|
|Radixor|608|880|610|190|1.334645%|2|14426|
### Output Policies and Metric Definitions
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
- Jaccard index: `TP / (TP + FP + FN)`.
- FowlkesMallows index: `sqrt(precision * recall)`.
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
### Provenance
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
- Evaluation command: `./gradlew stemmingQuality`
- Dictionary language: `UK_UA`
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
<!-- STEMMING-QUALITY:END -->

View File

@@ -50,3 +50,346 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs. - Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available. - Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees. - Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
<!-- STEMMING-QUALITY:START -->
## Stemming Quality
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `YI` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
### Evaluation Scope and Key Findings
The dictionary resource is `src/main/resources/yi/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.988241** among 3 deterministic stemmers. The runner-up is `SNOWBALL YIDDISH DIRECT` at 0.890988, a difference of 0.097253. This rank does not imply leadership in throughput or every secondary metric.
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.988241** among 3 deterministic stemmers. The runner-up is `SNOWBALL YIDDISH DIRECT` at 0.890988, a difference of 0.097253. This rank does not imply leadership in throughput or every secondary metric.
### `ALL_WORDS`
This mode contains **5 result rows**, **3 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.988241|195 / 6392909 (0.003050%)|149 / 6344 (2.348676%)|0.970881|0.972986|0.972965|
|2|SNOWBALL YIDDISH DIRECT|PRIMARY_OUTPUT|0.890988|1151 / 6392909 (0.018004%)|1382 / 6344 (21.784363%)|0.805624|0.796661|0.796600|
|3|SNOWBALL YIDDISH LUCENE FILTER|PRIMARY_OUTPUT|0.890988|1151 / 6392909 (0.018004%)|1382 / 6344 (21.784363%)|0.805624|0.796661|0.796600|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.969484|0.976513|0.999969|0.988241|0.999946|0.000054|
|2|SNOWBALL YIDDISH DIRECT|PRIMARY_OUTPUT|0.811713|0.782156|0.999820|0.890988|0.999604|0.000396|
|3|SNOWBALL YIDDISH LUCENE FILTER|PRIMARY_OUTPUT|0.811713|0.782156|0.999820|0.890988|0.999604|0.000396|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.970881|0.972986|0.975099|0.947393|0.972992|0.972965|
|2|SNOWBALL YIDDISH DIRECT|PRIMARY_OUTPUT|0.805624|0.796661|0.787894|0.662041|0.796798|0.796600|
|3|SNOWBALL YIDDISH LUCENE FILTER|PRIMARY_OUTPUT|0.805624|0.796661|0.787894|0.662041|0.796798|0.796600|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.972959|0.995691|0.996142|0.995917|0.995917|
|2|SNOWBALL YIDDISH DIRECT|PRIMARY_OUTPUT|0.796462|0.982919|0.962014|0.972354|0.972354|
|3|SNOWBALL YIDDISH LUCENE FILTER|PRIMARY_OUTPUT|0.796462|0.982919|0.962014|0.972354|0.972354|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|6195|195|149|6392714|195 / 6392909|149 / 6344|
|2|SNOWBALL YIDDISH DIRECT|PRIMARY_OUTPUT|4962|1151|1382|6391758|1151 / 6392909|1382 / 6344|
|3|SNOWBALL YIDDISH LUCENE FILTER|PRIMARY_OUTPUT|4962|1151|1382|6391758|1151 / 6392909|1382 / 6344|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 6392909 (0.000000%)|0 / 6344 (0.000000%)|1.000000|1.000000|1.000000|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|6344|0|0|6392909|0 / 6392909|0 / 6344|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999970|389 / 6392909 (0.006085%)|0 / 6344 (0.000000%)|0.953240|0.970253|0.970653|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.942225|1.000000|0.999939|0.999970|0.999939|0.000061|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.953240|0.970253|0.987885|0.942225|0.970683|0.970653|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|6344|389|0|6392520|389 / 6392909|0 / 6344|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|149|195|194|89|2.487423%|3|3676|
### `LOWERCASE_GROUPS_ONLY`
This mode contains **5 result rows**, **3 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
#### `PRIMARY_OUTPUT` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.988241|195 / 6392909 (0.003050%)|149 / 6344 (2.348676%)|0.970881|0.972986|0.972965|
|2|SNOWBALL YIDDISH DIRECT|PRIMARY_OUTPUT|0.890988|1151 / 6392909 (0.018004%)|1382 / 6344 (21.784363%)|0.805624|0.796661|0.796600|
|3|SNOWBALL YIDDISH LUCENE FILTER|PRIMARY_OUTPUT|0.890988|1151 / 6392909 (0.018004%)|1382 / 6344 (21.784363%)|0.805624|0.796661|0.796600|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.969484|0.976513|0.999969|0.988241|0.999946|0.000054|
|2|SNOWBALL YIDDISH DIRECT|PRIMARY_OUTPUT|0.811713|0.782156|0.999820|0.890988|0.999604|0.000396|
|3|SNOWBALL YIDDISH LUCENE FILTER|PRIMARY_OUTPUT|0.811713|0.782156|0.999820|0.890988|0.999604|0.000396|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.970881|0.972986|0.975099|0.947393|0.972992|0.972965|
|2|SNOWBALL YIDDISH DIRECT|PRIMARY_OUTPUT|0.805624|0.796661|0.787894|0.662041|0.796798|0.796600|
|3|SNOWBALL YIDDISH LUCENE FILTER|PRIMARY_OUTPUT|0.805624|0.796661|0.787894|0.662041|0.796798|0.796600|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|0.972959|0.995691|0.996142|0.995917|0.995917|
|2|SNOWBALL YIDDISH DIRECT|PRIMARY_OUTPUT|0.796462|0.982919|0.962014|0.972354|0.972354|
|3|SNOWBALL YIDDISH LUCENE FILTER|PRIMARY_OUTPUT|0.796462|0.982919|0.962014|0.972354|0.972354|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|PRIMARY_OUTPUT|6195|195|149|6392714|195 / 6392909|149 / 6344|
|2|SNOWBALL YIDDISH DIRECT|PRIMARY_OUTPUT|4962|1151|1382|6391758|1151 / 6392909|1382 / 6344|
|3|SNOWBALL YIDDISH LUCENE FILTER|PRIMARY_OUTPUT|4962|1151|1382|6391758|1151 / 6392909|1382 / 6344|
</details>
#### `ANY_CANDIDATE` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 6392909 (0.000000%)|0 / 6344 (0.000000%)|1.000000|1.000000|1.000000|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ANY_CANDIDATE|6344|0|0|6392909|0 / 6392909|0 / 6344|
</details>
#### `ALL_CANDIDATES` ranking
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.999970|389 / 6392909 (0.006085%)|0 / 6344 (0.000000%)|0.953240|0.970253|0.970653|
</div>
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.942225|1.000000|0.999939|0.999970|0.999939|0.000061|
</details>
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|0.953240|0.970253|0.987885|0.942225|0.970683|0.970653|
</details>
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|---:|---|---|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
</details>
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|---:|---|---|---:|---:|---:|---:|---:|---:|
|1|Radixor|ALL_CANDIDATES|6344|389|0|6392520|389 / 6392909|0 / 6344|
</details>
#### Multi-output analysis
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|---|---:|---:|---:|---:|---:|---:|---:|
|Radixor|149|195|194|89|2.487423%|3|3676|
### Output Policies and Metric Definitions
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
- Jaccard index: `TP / (TP + FP + FN)`.
- FowlkesMallows index: `sqrt(precision * recall)`.
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
### Provenance
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
- Evaluation command: `./gradlew stemmingQuality`
- Dictionary language: `YI`
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
<!-- STEMMING-QUALITY:END -->

View File

@@ -0,0 +1,82 @@
# Linguistic Quality Methodology
This evaluation measures agreement between the relation predicted by a stemmer and the gold-standard relation represented by Radixor dictionary groups. It does not require a generated stem to equal one predetermined lemma string. Runtime performance and linguistic quality are separate measurements.
## Scope and fair-comparison rules
The authoritative Radixor language universe is the reconciled set of `stemmer.gz` resources under `src/main/resources` and `StemmerPatchTrieLoader.Language`. Radixor is evaluated for every reconciled language. A third-party adapter is evaluated only for languages supported by its tested implementation and having a compatible Radixor dictionary; unsupported combinations are absent rather than assigned zero quality.
Within one language and dictionary mode, every adapter receives the same original included forms. Exact duplicates are removed only within one dictionary row. Identical surface forms in different rows remain distinct entries. Candidate strings use exact `String.equals`, with no evaluation-only lowercasing, normalization, accent removal, or gold-label-aware selection. Adapter preprocessing and lifecycle match the JMH comparison path.
## Gold-standard pairs
Every usable dictionary row is a gold-standard equivalence group. An unordered pair from the same row is positive; a pair from different rows is negative. For group size `n`, `C2(n) = n * (n - 1) / 2`.
- `TP = underPossiblePairs - underErrorPairs`: same-group pairs correctly related.
- `FN = underErrorPairs`: same-group pairs incorrectly separated.
- `FP = overErrorPairs`: different-group pairs incorrectly related.
- `TN = overPossiblePairs - overErrorPairs`: different-group pairs correctly separated.
Under-stemming is the false-negative relation among same-group pairs. Over-stemming is the false-positive relation among different-group pairs. Their percentages use different denominators and must not be added or averaged without an explicitly defined composite.
## Dictionary-processing modes
- `ALL_WORDS` includes every valid group and preserves every original form.
- `LOWERCASE_GROUPS_ONLY` excludes an entire group if any Unicode code point is uppercase or titlecase. Retained forms are not converted to lowercase. Digits, punctuation, combining marks, and characters without case distinctions do not exclude a group by themselves.
## Output policies
`PRIMARY_OUTPUT` uses the adapter's deterministic primary stem. It defines a strict predicted partition and is the principal direct comparison between implementations.
`ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound. Same-group pairs succeed when candidate sets intersect. Different-group pairs avoid an error whenever a non-colliding candidate selection exists. Selection may differ between pairs, so this policy is not deterministic runtime behaviour and may not correspond to one globally realizable assignment.
`ALL_CANDIDATES` treats every returned candidate as active. Two forms are related when their candidate sets intersect. Alternatives can recover same-group relationships while introducing cross-group collisions. This overlapping relation need not be transitive or form a partition.
Candidate-aware policies are reported as capability analyses. They are not mixed into the principal `PRIMARY_OUTPUT` ranking.
## Relation metrics
Undefined denominators produce `n/a`, never zero, `NaN`, or infinity. Metrics are calculated from unrounded raw counts and displayed with six decimals.
| Metric | Formula | Range and interpretation | Sensitivity and applicability |
| --- | --- | --- | --- |
| Under-stemming rate | `FN / (TP + FN)` | `[0, 1]`; lower is better. False-negative rate over same-group pairs. | Sensitive to splitting large gold groups. All policies. |
| Over-stemming rate | `FP / (TN + FP)` | `[0, 1]`; lower is better. False-positive rate over different-group pairs. | The denominator is usually very large. All policies. |
| Precision | `TP / (TP + FP)` | `[0, 1]`; higher is better. Fraction of predicted relations that are gold-positive. | Penalizes over-stemming. All policies, with oracle-assisted interpretation for `ANY_CANDIDATE`. |
| Recall | `TP / (TP + FN)` | `[0, 1]`; higher is better. Fraction of gold-positive pairs recovered. | Equivalent to one minus the under-stemming rate. All policies. |
| Specificity | `TN / (TN + FP)` | `[0, 1]`; higher is better. Fraction of negative pairs separated. | Sensitive to cross-group collisions. All policies. |
| Balanced accuracy | `(recall + specificity) / 2` | `[0, 1]`; higher is better. Equal weight for positive and negative classes. | Primary navigation metric; less dominated by TN than ordinary accuracy, but not uniquely authoritative. |
| Pairwise accuracy | `(TP + TN) / (TP + TN + FP + FN)` | `[0, 1]`; higher is better. | Can be dominated by the very large TN class and is not the default ranking metric. |
| Pairwise error rate | `(FP + FN) / (TP + TN + FP + FN)` | `[0, 1]`; lower is better. | Also sensitive to the number of negative pairs. |
| F0.5 | `1.25 TP / (1.25 TP + 0.25 FN + FP)` | `[0, 1]`; higher is better. | Gives greater weight to precision and over-stemming avoidance. |
| F1 | `2 TP / (2 TP + FN + FP)` | `[0, 1]`; higher is better. | Equal precision/recall emphasis. |
| F2 | `5 TP / (5 TP + 4 FN + FP)` | `[0, 1]`; higher is better. | Gives greater weight to recall and under-stemming avoidance. |
| Jaccard | `TP / (TP + FP + FN)` | `[0, 1]`; higher is better. | Excludes TN. All policies. |
| FowlkesMallows | `sqrt(precision * recall)` | `[0, 1]`; higher is better. | Geometric balance of precision and recall. All policies. |
| MCC | `(TP TN - FP FN) / sqrt((TP+FP)(TP+FN)(TN+FP)(TN+FN))` | `[-1, 1]`; higher is better. Uses all four counts. | Informative under imbalance; undefined for a zero product denominator. All policies with policy-specific interpretation. |
The general F-beta formula is `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`.
## Partition-only metrics
These metrics apply only to `PRIMARY_OUTPUT`. Candidate relations are not forced into artificial partitions.
- Adjusted Rand Index is the Rand agreement corrected for agreement expected from the gold/predicted contingency-table marginals. Its usual range is `[-1, 1]`, with `1` indicating identical partitions.
- Homogeneity is `1 - H(gold | predicted) / H(gold)`, in `[0, 1]`; each predicted cluster ideally contains one gold group.
- Completeness is `1 - H(predicted | gold) / H(predicted)`, in `[0, 1]`; each gold group ideally maps to one predicted cluster.
- V-measure is the harmonic mean of homogeneity and completeness, in `[0, 1]`.
- Normalized mutual information uses arithmetic-mean entropy normalization: `MI / ((H(gold) + H(predicted)) / 2)`, in `[0, 1]` under this implementation.
Entropy zero cases follow the evaluator's explicit perfect/undefined conventions. Language tables render inapplicable candidate-policy values as `n/a`.
## Aggregation and ranking
Macro metrics average defined per-language values, giving each language equal weight. Micro metrics sum TP, FP, FN, and TN before calculating a metric. Cross-stemmer aggregate comparisons require the exact common supported-language intersection; unsupported languages are not zero-filled.
Language tables sort by unrounded balanced accuracy, then MCC, F1, over-stemming rate, over-stemming error count, under-stemming rate, stemmer name, and stable policy order. Display rounding never controls rank.
Multiple metrics and Pearson/Spearman correlation datasets are published because metric suitability and correlation remain analytical questions. Strong correlation does not establish equivalence.
## Limitations
Dictionary groups encode the available annotation, not every linguistic distinction. Homographs may occur in different groups, singleton rows contribute no under-stemming pair, and group size affects pair counts. `ANY_CANDIDATE` is optimistic; `ALL_CANDIDATES` measures an overlapping graph; neither is a deterministic global assignment. Results characterize the tested versions, adapters, dictionaries, and preprocessing, not every deployment or domain.

View File

@@ -21,11 +21,9 @@ timePerChangedTokenNs = JMH score ns/op / changedTimingTokenCount
This is necessary because Radixor dictionaries have different token counts by language. This is necessary because Radixor dictionaries have different token counts by language.
## Quality And Search Interpretation ## Exact-root quality and interpretation
Radixor speed must be interpreted together with exact-root quality. A slower Radixor row must not be read as a simple performance weakness when Radixor is also the row with accuracy close to 100% and competing stemmers are much lower. Runtime and exact-root agreement must be interpreted separately. Light, minimal, possessive, and aggressive rule-based implementations deliberately address different scopes and may achieve lower latency by performing fewer transformations. A throughput advantage does not establish higher linguistic quality, and higher dictionary agreement does not establish lower operational cost.
Many fast light, minimal, possessive, or aggressive rule-based stemmers are fast because they do much less linguistic work. The measured Radixor cost buys dictionary-trained precision, and that precision is what improves search quality when queries and indexed text are reduced to the same intended roots.
The [English dictionary coverage benchmark](english-coverage.md) shows this operating curve explicitly: contracted tries reduce lookup cost in uniform regions, while reduced dictionary coverage still lowers changed-form precision. The [English dictionary coverage benchmark](english-coverage.md) shows this operating curve explicitly: contracted tries reduce lookup cost in uniform regions, while reduced dictionary coverage still lowers changed-form precision.
@@ -57,3 +55,5 @@ rootPreservedPercent = rootPreservedMatches / rootEvaluatedTokens * 100
Morfologik can emit multiple terms for one input token. The quality benchmark uses the first emitted term for exact-root accounting when no ranking weight is exposed. Throughput benchmarks for Morfologik TokenFilter paths consume all emitted terms. Morfologik can emit multiple terms for one input token. The quality benchmark uses the first emitted term for exact-root accounting when no ranking weight is exposed. Throughput benchmarks for Morfologik TokenFilter paths consume all emitted terms.
Quality reports use JMH auxiliary counter rows. Exact-root accounting is deterministic for a fixed corpus and stemmer, so repeated measurement samples duplicate the same counters; documentation uses the counter ratios and does not interpret quality benchmark timing scores. Quality reports use JMH auxiliary counter rows. Exact-root accounting is deterministic for a fixed corpus and stemmer, so repeated measurement samples duplicate the same counters; documentation uses the counter ratios and does not interpret quality benchmark timing scores.
Pairwise over-stemming, under-stemming, candidate-aware policies, balanced accuracy, and partition comparison are a separate analytical evaluation. See [Linguistic Quality Methodology](linguistic-quality.md); exact-root accuracy must not be interpreted as the complement of pairwise under-stemming.

View File

@@ -0,0 +1,61 @@
# Reproducibility and Raw Data
## Published quality snapshot
- Machine-readable CSV: [stemming-quality.csv](../data/stemming-quality.csv)
- SHA-256 record: [stemming-quality.sha256](../data/stemming-quality.sha256)
- SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
- Complete scenarios: 308
- Authoritative language universe: 20 languages
- Language-page scenarios: 302 across 19 existing benchmark pages
The six remaining scenarios are the three Radixor policies in two modes for `HE_IL`. Hebrew is present in the complete result snapshot but has no existing language benchmark page.
The CSV contains raw TP, FP, FN, and TN counts; raw over/under numerators and denominators; candidate statistics; relation metrics; and partition-only metrics. Documentation is regenerated from this file rather than manually transcribed.
## Commands
```bash
./gradlew stemmingQuality
./gradlew publishStemmingQualityDocumentation
./gradlew verifyStemmingQualityDocumentation
./gradlew test
mkdocs build --strict
```
`stemmingQuality` performs the expensive complete evaluation and is intentionally not attached to `test` or `check`. It prepares JMH third-party dependencies automatically and writes:
- `build/reports/stemming-quality/stemming-quality.csv`
- `build/reports/stemming-quality/stemming-quality.md`
- `build/reports/stemming-quality/metric-correlations-pearson.csv`
- `build/reports/stemming-quality/metric-correlations-spearman.csv`
Audit mode is enabled with `-PstemmingQualityAudit=true`. Language, stemmer, dictionary-mode, output-policy, and ranking filters are documented on the central [stemming-quality page](../../stemming-quality.md). Filtered reports use separate filenames and cannot be accepted as publication sources.
`publishStemmingQualityDocumentation` validates the complete build CSV, copies a versioned documentation snapshot, and replaces only marked generated sections. `verifyStemmingQualityDocumentation` re-renders from the checked-in snapshot and fails on changed values, ordering, missing pages, duplicate keys, arithmetic inconsistencies, policy violations, or stale sections.
## Performance benchmark reproduction
The JMH comparison command family is:
```bash
./gradlew jmh -Pjmh.includes='.*StemmerComparisonBenchmark.*' --no-daemon
```
The exact JMH configuration, hardware, operating system, and JDK captured for the published performance tables are listed in [Environment and reports](environment.md). Quality and performance reports are separate datasets and are not combined into an undocumented scalar.
## Recorded and unavailable provenance
The performance documentation records its 2026-07-06 environment, JDK 25.0.3, operating system, and hardware. The quality CSV records the evaluated identifiers and counts but does not embed the Radixor Git revision, generation date, JDK, operating system, dictionary content hash, or immutable upstream revisions for every downloaded source. These fields are explicitly unavailable for this snapshot and are not reconstructed from filesystem timestamps.
Dependency versions that are reproducible from repository configuration include Apache Lucene 10.5.0, Morfologik 2.1.9, the Ukrainian dictionary artifact 4.9.1, and JMH 1.37. Other upstream branches or downloaded dictionary revisions should be pinned and embedded in a future result schema.
## Correlation and audit data
Pearson and Spearman files are generated from unrounded metric values in cohorts separated by dictionary mode and output policy. A missing coefficient means too few observations, undefined input, or zero variance. Correlation is descriptive and does not demonstrate that two metrics are scientifically interchangeable.
Audit reports preserve original multilingual forms and identify high-contributing dictionary groups. They are build artifacts rather than checked-in publication data because of their size. No documentation value is manually altered after generation.
## JMH badge compatibility
The quality documentation generator does not invoke JMH, change JMH result formats, or modify badge tooling. Existing JMH result paths and historical badge-compatible inputs remain independent. The repository currently publishes coverage and mutation badge metadata and retains JMH TXT/CSV artifacts as documented in [Environment and reports](environment.md).

View File

@@ -0,0 +1,29 @@
# Tested Stemmer Inventory
The JMH adapter registry is authoritative for evaluated implementations and language mappings. Names below describe the implementation actually invoked, not an abstract algorithm in every possible implementation. Unsupported language combinations are omitted rather than scored as failures.
| Family or implementation | Upstream / attribution | Tested version or revision | Evaluated scope | Output capability and adapter behaviour | Interpretation notes |
| --- | --- | --- | --- | --- | --- |
| Radixor | Egothor / Radixor project | Current repository revision; exact revision was not embedded in the quality CSV | All 20 reconciled Radixor dictionary languages; 19 have benchmark pages | Deterministic preferred patch via `get`; ranked distinct alternatives via `getAll`; primary is always included | Dictionary-derived compiled patch trie. Quality depends on dictionary coverage and annotation. |
| Apache Lucene language stem filters | Apache Lucene project | 10.5.0 | Adapter-declared language-specific subsets | TokenFilter lifecycle and language normalization match JMH; normally single-output | Light, minimal, possessive, and language stem filters deliberately implement different scopes. Narrow scope is not a defect. |
| Apache Lucene SnowballFilter | Apache Lucene project using Snowball algorithms | Lucene 10.5.0 | Snowball-supported subset of Radixor languages | Single primary token emitted through the Lucene TokenFilter path | Includes TokenStream overhead and required normalization. |
| Official Snowball Java | Snowball project | Repository preparation downloads the configured upstream Java distribution; an immutable revision was not recorded in the quality CSV | Same-language adapter subset | Direct generated Java API; single output | Rule-based suffix algorithms provide broad baselines rather than dictionary-root guarantees. |
| Lucene Stempel | Apache Lucene / Polish stemming tables | Lucene 10.5.0 | Polish | Direct and TokenFilter paths where registered; single primary output | Table-driven Polish implementation. |
| Morfologik | Morfologik project; Lucene integration by Apache Lucene | Morfologik 2.1.9, Lucene integration 10.5.0; Ukrainian dictionary artifact 4.9.1 | Registered Polish and Ukrainian paths | Deterministic first lemma for primary comparison; all distinct lemma strings for candidate policies | Several analyses may share a lemma and are deduplicated by exact string equality. |
| Hunspell via Lucene | Hunspell dictionaries from the `wooorm/dictionaries` repository; adapter by Apache Lucene | Lucene 10.5.0; dictionary repository revision was not recorded | Configured German, English, Spanish, French, Dutch, Polish, and Ukrainian dictionaries | First emitted stem is primary; all distinct stems at the token position are candidates | Dictionary content and affix rules differ by language. |
| CISTEM | Leonie Weissweiler, CISTEM project | Upstream `master` source path used by preparation; immutable commit not recorded | German | Single output | German stemming algorithm; benchmark-only implementation and gold-standard preparation remain under JMH infrastructure. |
| OpenNLP Porter | Apache OpenNLP project | Version resolved by `gradle/opennlp-benchmarks.gradle` and `gradle.lockfile` | English | Direct single output | Porter-family English baseline. |
| Lucene Porter source copy | Apache Lucene project | 10.5.0 source artifact | English | Package-isolated benchmark-only generated source; single output | Generated into the JMH build tree, never production code. |
| Paice/Husk Lancaster | Upstream Java implementation from `Hopper262/paice-husk-stemmer` | Configured upstream branch/revision in `gradle/paicehusk-benchmarks.gradle`; immutable commit not recorded | English | Direct single output | Aggressive rule-based English baseline; benchmark-only generated source. |
## Preprocessing and lifecycle
The quality evaluator calls the same adapter matrix used by JMH. Each language mapping is explicit. Retained dictionary forms are not evaluation-lowercased or normalized. Where an implementation requires preprocessing, such as Lucene German or Persian normalization, that operation is part of its documented adapter path. Stateful TokenFilters are reset through the same sequential lifecycle used by the benchmark and are not invoked concurrently.
Candidate sets are non-null, non-empty, contain the deterministic primary output, contain no null strings, and are deduplicated using exact Java string equality. Gold-standard group identity never selects, removes, or ranks a candidate.
## Coverage fairness
Radixor coverage is derived independently from its resources and language enumeration. Third-party coverage is the intersection of that universe with actual adapter support. Absence therefore means “not supported or not configured for this language,” not “zero quality.” Consult each language page for the exact evaluated rows.
Project authors and organizations are named only where repository configuration or source notices establish attribution. No broader authorship or license claim is inferred when metadata was not captured.

82
docs/stemming-quality.md Normal file
View File

@@ -0,0 +1,82 @@
# Stemming quality evaluation
The explicit `stemmingQuality` analysis measures agreement between stemmer outputs and gold-standard equivalence classes represented by bundled multilingual dictionary rows. Dictionary text remains unchanged; reports and diagnostics use English.
JMH adapters, registries, third-party versions, language mappings, and preparation remain in `src/jmh`. The evaluator, reports, audits, and tests reside in the standard `src/test` source set. The former `src/stemmingQualityTest` source set was removed, and neither analytical nor JMH classes enter the production JAR.
## Language and adapter coverage
The authoritative Radixor universe is the validated one-to-one reconciliation of `src/main/resources/*/stemmer.gz` and every `StemmerPatchTrieLoader.Language` value. All 20 current values have exactly one resource; no sentinel or alias is excluded. Radixor is evaluated for all 20 languages, independently of third-party support. Third-party combinations come only from explicit JMH adapter metadata. Unsupported combinations are documented and never fabricated as zero-valued rows.
The expected matrix is constructed before evaluation from stemmer, language, dictionary mode, and supported output policy. Generation fails on missing, duplicate, unexpected, or stale keys.
## Dictionary groups and modes
Every usable parsed row is one gold-standard group. Exact duplicate strings are removed only within that row; identical forms in different rows remain distinct. `ALL_WORDS` preserves every valid form. `LOWERCASE_GROUPS_ONLY` excludes a complete group containing an uppercase or titlecase Unicode code point. Retained words are not lowercased or normalized by the evaluator.
## Output policies
`PRIMARY_OUTPUT` uses the deterministic JMH output and defines a strict partition.
For multi-output adapters, `C(w)` is the immutable, sorted, exactly deduplicated candidate set. It is non-null, non-empty, contains no null, and contains the primary output. Radixor obtains alternatives through `getAll`. The repository's Morphologik lookups can return distinct lemma strings and are multi-output. Configured Hunspell filters can emit several stems at one token position. Other adapters emit only primary rows.
`ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound. A same-group pair succeeds when its sets intersect. A cross-group pair is an error only when both sets are the same singleton; otherwise unequal candidates can be selected for that pair. Choices may vary between pairs and need not form one realizable global assignment.
`ALL_CANDIDATES` activates every candidate. Two forms are related when their sets intersect, for both same-group and cross-group pairs. This relation can overlap and need not be transitive. A pair sharing several candidates is counted once.
The evaluator verifies:
```text
ANY under <= PRIMARY under
ALL under <= PRIMARY under
ANY under = ALL under
ANY over <= PRIMARY over
ALL over >= PRIMARY over
```
## Pair definitions and efficient counting
For `C2(n) = n(n-1)/2`:
```text
underPossible = sum_g C2(n_g)
overPossible = C2(N) - sum_g C2(n_g)
```
Under-stemming counts unrelated same-group pairs. Over-stemming counts related cross-group pairs. Primary output uses global and per-group stem frequencies. Candidate sets are canonical signatures counted globally and per group. An inverted candidate-to-signature index discovers intersections, and signature pairs shared through several candidates are deduplicated. `ANY_CANDIDATE` over-stemming uses only equal singleton signatures. All pair arithmetic uses checked `long` operations; complete production word pairs are never enumerated.
## Confusion and aggregate metrics
```text
TP = underPossible - underError
FN = underError
FP = overError
TN = overPossible - overError
```
Under-stemming is `FN/(TP+FN)` and over-stemming is `FP/(TN+FP)`; their denominators differ. The CSV also publishes precision, recall, specificity, accuracy, balanced accuracy, F0.5, F1, F2, Jaccard, Fowlkes-Mallows, Matthews correlation coefficient, and pairwise error rate. F0.5 emphasizes precision and over-stemming, F1 balances precision and recall, and F2 emphasizes recall and under-stemming. Accuracy and error rate can be dominated by the large cross-group true-negative population. Metrics use raw counts, not rounded rates. Zero denominators produce `n/a` in Markdown and empty CSV fields.
Only `PRIMARY_OUTPUT` receives partition metrics: Adjusted Rand Index, homogeneity, completeness, V-measure, and normalized mutual information with arithmetic-mean entropy normalization. Candidate policies remain inapplicable rather than being forced into artificial partitions.
Micro summaries sum confusion counts before calculation. Macro summaries average defined language values and retain coverage counts. Common-language comparisons use the exact language intersection and never score unsupported languages as zero. Rankings are separated by policy and metric; the default F0.5 choice is navigation, not a universal scientific preference.
Pearson and average-tie-rank Spearman reports use unrounded values and separate dictionary-mode and output-policy cohorts. Fewer than three observations, undefined inputs, and zero variance produce documented missing values. The reports provide reproducible data and make no automatic scientific conclusion.
## Exact accuracy and pairwise under-stemming
Exact textual accuracy and pairwise grouping use different denominators. One erroneous form in a 12-form group creates 11 erroneous pairs: with 88 singleton groups, exact accuracy can be 99% while pairwise under-stemming is `11/C2(12) = 16.666667%`. Singleton groups affect word accuracy but add no within-group pairs.
## Running the analysis
```bash
./gradlew stemmingQuality
./gradlew stemmingQuality -PstemmingQualityStemmer=Radixor -PstemmingQualityLanguage=DE_DE -PstemmingQualityMode=ALL_WORDS -PstemmingQualityAudit=true
```
Optional properties are `stemmingQualityLanguage`, `stemmingQualityStemmer`, `stemmingQualityMode`, `stemmingQualityOutputPolicy`, `stemmingQualityRankMetric`, `stemmingQualityAudit`, and `stemmingQualityAuditLimit`. Policies are `PRIMARY_OUTPUT`, `ANY_CANDIDATE`, and `ALL_CANDIDATES`. Filtered reports carry `-filtered` and cannot overwrite complete output.
Generated files under `build/reports/stemming-quality/` include `stemming-quality.md`, `stemming-quality.csv`, `metric-correlations-pearson.csv`, `metric-correlations-spearman.csv`, and optional audit Markdown.
## Limitations
These measurements evaluate agreement with the available dictionary grouping. They do not capture every semantic, morphological, downstream, or dataset-specific property. `ANY_CANDIDATE` is optimistic and may not be globally realizable. `ALL_CANDIDATES` measures an overlap graph rather than a partition. Language coverage must remain visible in cross-stemmer comparisons. No single published metric establishes universal superiority; multiple metrics and their correlations are provided for transparent scientific assessment.

View File

@@ -13,7 +13,7 @@ net.jqwik:jqwik-time:1.9.3=jmhRuntimeClasspath,testCompileClasspath,testRuntimeC
net.jqwik:jqwik-web:1.9.3=jmhRuntimeClasspath,testCompileClasspath,testRuntimeClasspath net.jqwik:jqwik-web:1.9.3=jmhRuntimeClasspath,testCompileClasspath,testRuntimeClasspath
net.jqwik:jqwik:1.9.3=jmhRuntimeClasspath,testCompileClasspath,testRuntimeClasspath net.jqwik:jqwik:1.9.3=jmhRuntimeClasspath,testCompileClasspath,testRuntimeClasspath
net.sf.jopt-simple:jopt-simple:4.9=pitest net.sf.jopt-simple:jopt-simple:4.9=pitest
net.sf.jopt-simple:jopt-simple:5.0.4=jmh,jmhCompileClasspath,jmhRuntimeClasspath net.sf.jopt-simple:jopt-simple:5.0.4=jmh,jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
net.sf.saxon:Saxon-HE:12.9=pmd net.sf.saxon:Saxon-HE:12.9=pmd
net.sourceforge.pmd:pmd-ant:7.20.0=pmd net.sourceforge.pmd:pmd-ant:7.20.0=pmd
net.sourceforge.pmd:pmd-core:7.20.0=pmd net.sourceforge.pmd:pmd-core:7.20.0=pmd
@@ -22,17 +22,17 @@ org.antlr:antlr4-runtime:4.9.3=pmd
org.antlr:stringtemplate:3.2.1=pitest org.antlr:stringtemplate:3.2.1=pitest
org.apache.commons:commons-lang3:3.18.0=pitest org.apache.commons:commons-lang3:3.18.0=pitest
org.apache.commons:commons-lang3:3.20.0=pmd org.apache.commons:commons-lang3:3.20.0=pmd
org.apache.commons:commons-math3:3.6.1=jmh,jmhCompileClasspath,jmhRuntimeClasspath org.apache.commons:commons-math3:3.6.1=jmh,jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
org.apache.commons:commons-text:1.14.0=pitest org.apache.commons:commons-text:1.14.0=pitest
org.apache.lucene:lucene-analysis-common:10.5.0=jmhCompileClasspath,jmhRuntimeClasspath org.apache.lucene:lucene-analysis-common:10.5.0=jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
org.apache.lucene:lucene-analysis-morfologik:10.5.0=jmhCompileClasspath,jmhRuntimeClasspath org.apache.lucene:lucene-analysis-morfologik:10.5.0=jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
org.apache.lucene:lucene-analysis-stempel:10.5.0=jmhCompileClasspath,jmhRuntimeClasspath org.apache.lucene:lucene-analysis-stempel:10.5.0=jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
org.apache.lucene:lucene-core:10.5.0=jmhCompileClasspath,jmhRuntimeClasspath org.apache.lucene:lucene-core:10.5.0=jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
org.apache.opennlp:opennlp-tools:2.5.4=jmhCompileClasspath,jmhRuntimeClasspath org.apache.opennlp:opennlp-tools:2.5.4=jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
org.apiguardian:apiguardian-api:1.1.2=jmhRuntimeClasspath,testCompileClasspath,testRuntimeClasspath org.apiguardian:apiguardian-api:1.1.2=jmhRuntimeClasspath,testCompileClasspath,testRuntimeClasspath
org.carrot2:morfologik-fsa:2.1.9=jmhCompileClasspath,jmhRuntimeClasspath org.carrot2:morfologik-fsa:2.1.9=jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
org.carrot2:morfologik-polish:2.1.9=jmhRuntimeClasspath org.carrot2:morfologik-polish:2.1.9=jmhRuntimeClasspath,stemmingQualityJmhRuntime
org.carrot2:morfologik-stemming:2.1.9=jmhCompileClasspath,jmhRuntimeClasspath org.carrot2:morfologik-stemming:2.1.9=jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
org.checkerframework:checker-qual:3.52.1=pmd org.checkerframework:checker-qual:3.52.1=pmd
org.jacoco:org.jacoco.agent:0.8.14=jacocoAgent,jacocoAnt org.jacoco:org.jacoco.agent:0.8.14=jacocoAgent,jacocoAnt
org.jacoco:org.jacoco.ant:0.8.14=jacocoAnt org.jacoco:org.jacoco.ant:0.8.14=jacocoAnt
@@ -49,10 +49,10 @@ org.junit:junit-bom:5.14.3=jmhRuntimeClasspath,testCompileClasspath,testRuntimeC
org.mockito:mockito-core:5.23.0=jmhRuntimeClasspath,mockitoAgent,testCompileClasspath,testRuntimeClasspath org.mockito:mockito-core:5.23.0=jmhRuntimeClasspath,mockitoAgent,testCompileClasspath,testRuntimeClasspath
org.mockito:mockito-junit-jupiter:5.23.0=jmhRuntimeClasspath,testCompileClasspath,testRuntimeClasspath org.mockito:mockito-junit-jupiter:5.23.0=jmhRuntimeClasspath,testCompileClasspath,testRuntimeClasspath
org.objenesis:objenesis:3.3=jmhRuntimeClasspath,testRuntimeClasspath org.objenesis:objenesis:3.3=jmhRuntimeClasspath,testRuntimeClasspath
org.openjdk.jmh:jmh-core:1.37=jmh,jmhCompileClasspath,jmhRuntimeClasspath org.openjdk.jmh:jmh-core:1.37=jmh,jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
org.openjdk.jmh:jmh-generator-asm:1.37=jmh,jmhCompileClasspath,jmhRuntimeClasspath org.openjdk.jmh:jmh-generator-asm:1.37=jmh,jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
org.openjdk.jmh:jmh-generator-bytecode:1.37=jmh,jmhCompileClasspath,jmhRuntimeClasspath org.openjdk.jmh:jmh-generator-bytecode:1.37=jmh,jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
org.openjdk.jmh:jmh-generator-reflection:1.37=jmh,jmhCompileClasspath,jmhRuntimeClasspath org.openjdk.jmh:jmh-generator-reflection:1.37=jmh,jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
org.opentest4j:opentest4j:1.3.0=jmhRuntimeClasspath,testCompileClasspath,testRuntimeClasspath org.opentest4j:opentest4j:1.3.0=jmhRuntimeClasspath,testCompileClasspath,testRuntimeClasspath
org.ow2.asm:asm-analysis:9.9.1=pitest org.ow2.asm:asm-analysis:9.9.1=pitest
org.ow2.asm:asm-commons:9.9=jacocoAnt org.ow2.asm:asm-commons:9.9=jacocoAnt
@@ -60,7 +60,7 @@ org.ow2.asm:asm-commons:9.9.1=pitest
org.ow2.asm:asm-tree:9.9=jacocoAnt org.ow2.asm:asm-tree:9.9=jacocoAnt
org.ow2.asm:asm-tree:9.9.1=pitest org.ow2.asm:asm-tree:9.9.1=pitest
org.ow2.asm:asm-util:9.9.1=pitest org.ow2.asm:asm-util:9.9.1=pitest
org.ow2.asm:asm:9.0=jmh,jmhCompileClasspath,jmhRuntimeClasspath org.ow2.asm:asm:9.0=jmh,jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
org.ow2.asm:asm:9.9=jacocoAnt org.ow2.asm:asm:9.9=jacocoAnt
org.ow2.asm:asm:9.9.1=pitest,pmd org.ow2.asm:asm:9.9.1=pitest,pmd
org.pcollections:pcollections:4.0.2=pmd org.pcollections:pcollections:4.0.2=pmd
@@ -70,7 +70,7 @@ org.pitest:pitest-html-report:1.22.1=pitest
org.pitest:pitest-junit5-plugin:1.2.3=pitest org.pitest:pitest-junit5-plugin:1.2.3=pitest
org.pitest:pitest:1.22.1=pitest org.pitest:pitest:1.22.1=pitest
org.slf4j:jul-to-slf4j:1.7.36=pmd org.slf4j:jul-to-slf4j:1.7.36=pmd
org.slf4j:slf4j-api:2.0.17=jmhCompileClasspath,jmhRuntimeClasspath org.slf4j:slf4j-api:2.0.17=jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
org.xmlresolver:xmlresolver:5.3.3=pmd org.xmlresolver:xmlresolver:5.3.3=pmd
ua.net.nlp:morfologik-ukrainian-search:4.9.1=jmhRuntimeClasspath ua.net.nlp:morfologik-ukrainian-search:4.9.1=jmhRuntimeClasspath,stemmingQualityJmhRuntime
empty=annotationProcessor,compileClasspath,cyclonedxBom,jmhAnnotationProcessor,mainPmdAuxClasspath,runtimeClasspath,testAnnotationProcessor empty=annotationProcessor,compileClasspath,cyclonedxBom,jmhAnnotationProcessor,mainPmdAuxClasspath,runtimeClasspath,testAnnotationProcessor

View File

@@ -68,6 +68,9 @@ nav:
- Benchmark Results: benchmarks/index.md - Benchmark Results: benchmarks/index.md
- Reference: - Reference:
- Methodology: benchmarks/reference/methodology.md - Methodology: benchmarks/reference/methodology.md
- Linguistic Quality Methodology: benchmarks/reference/linguistic-quality.md
- Tested Stemmers: benchmarks/reference/tested-stemmers.md
- Reproducibility and Raw Data: benchmarks/reference/reproducibility.md
- Corpora: benchmarks/reference/corpora.md - Corpora: benchmarks/reference/corpora.md
- Environment and Reports: benchmarks/reference/environment.md - Environment and Reports: benchmarks/reference/environment.md
- English Dictionary Coverage: benchmarks/reference/english-coverage.md - English Dictionary Coverage: benchmarks/reference/english-coverage.md
@@ -96,5 +99,6 @@ nav:
- Quality and Operations: - Quality and Operations:
- Quality and Operations: quality-and-operations.md - Quality and Operations: quality-and-operations.md
- Stemming Quality: stemming-quality.md
- Reports: reports.md - Reports: reports.md
- Test taxonomy and execution filtering: test-taxonomy-and-filtering.md - Test taxonomy and execution filtering: test-taxonomy-and-filtering.md

View File

@@ -254,8 +254,10 @@ public class HunspellStemmerComparisonBenchmarkQuality {
outputs[inputIndex] = termAttribute.toString(); outputs[inputIndex] = termAttribute.toString();
recordedForPosition = true; recordedForPosition = true;
} }
if (blackhole != null) {
blackhole.consume(termAttribute); blackhole.consume(termAttribute);
} }
}
output.end(); output.end();
output.close(); output.close();
@@ -285,6 +287,53 @@ public class HunspellStemmerComparisonBenchmarkQuality {
} }
} }
/**
* Stems one analytical batch through the exact Hunspell quality-benchmark path.
*
* @param languageCase declared Hunspell language case
* @param tokens original dictionary forms
* @return first Hunspell output per input form
* @throws IOException if dictionary parsing or token streaming fails
*/
static String[] stemForQuality(final HunspellLanguageCase languageCase, final String[] tokens) throws IOException {
try {
return firstHunspellOutputs(tokens, loadDictionary(languageCase), null);
} catch (ParseException exception) {
throw new IOException("Unable to parse the JMH Hunspell dictionary for " + languageCase + ".", exception);
}
}
/** Returns all distinct Hunspell stems per token through the quality-benchmark dictionary. */
static List<List<String>> stemCandidatesForQuality(final HunspellLanguageCase languageCase,
final String[] tokens) throws IOException {
try {
final Dictionary dictionary = loadDictionary(languageCase);
final List<java.util.LinkedHashSet<String>> candidates = new java.util.ArrayList<>(tokens.length);
for (int index = 0; index < tokens.length; index++) { candidates.add(new java.util.LinkedHashSet<>()); }
final BenchmarkTokenStream input = new BenchmarkTokenStream(tokens);
final TokenStream output = new HunspellStemFilter(new LowerCaseFilter(input), dictionary, true);
final CharTermAttribute term = output.addAttribute(CharTermAttribute.class);
final PositionIncrementAttribute position = output.addAttribute(PositionIncrementAttribute.class);
int inputIndex = -1;
output.reset();
while (output.incrementToken()) {
if (position.getPositionIncrement() > 0) { inputIndex += position.getPositionIncrement(); }
if (inputIndex >= 0 && inputIndex < candidates.size()) { candidates.get(inputIndex).add(term.toString()); }
}
output.end();
output.close();
final String[] primary = firstHunspellOutputs(tokens, dictionary, null);
final List<List<String>> result = new java.util.ArrayList<>(tokens.length);
for (int index = 0; index < tokens.length; index++) {
candidates.get(index).add(primary[index]);
result.add(List.copyOf(candidates.get(index)));
}
return List.copyOf(result);
} catch (ParseException exception) {
throw new IOException("Unable to parse the JMH Hunspell dictionary for " + languageCase + ".", exception);
}
}
/** /**
* Opens a required classpath resource. * Opens a required classpath resource.
* *
@@ -303,7 +352,7 @@ public class HunspellStemmerComparisonBenchmarkQuality {
/** /**
* Benchmark language mapping. * Benchmark language mapping.
*/ */
private enum HunspellLanguageCase { enum HunspellLanguageCase {
/** /**
* English Hunspell dictionary over the Radixor English corpus. * English Hunspell dictionary over the Radixor English corpus.

View File

@@ -0,0 +1,133 @@
package org.egothor.stemmer.benchmark;
import java.io.IOException;
import java.util.Arrays;
import java.util.List;
import java.util.Objects;
import java.util.ArrayList;
import java.util.EnumSet;
import org.egothor.stemmer.StemmerPatchTrieLoader.Language;
/** Authoritative analytical view of the candidate matrix defined by the JMH quality benchmark. */
public final class QualityStemmerMatrix {
/** Utility class. */
private QualityStemmerMatrix() {
throw new AssertionError("No instances.");
}
/**
* Returns every currently registered JMH quality candidate in declaration order.
* The returned list is immutable and is derived directly from the benchmark enum.
*
* @return complete immutable candidate list
*/
public static List<Candidate> candidates() {
final List<Candidate> candidates = new ArrayList<>();
Arrays.stream(StemmerComparisonBenchmarkQuality.QualityCandidate.values())
.map(candidate -> new Candidate(candidate.name(), candidate.radixorLanguage(),
() -> adapt(candidate.createStemmer())))
.forEach(candidates::add);
final EnumSet<Language> registeredRadixorLanguages = candidates.stream()
.filter(candidate -> candidate.name().endsWith("_RADIXOR"))
.map(Candidate::language).collect(() -> EnumSet.noneOf(Language.class), EnumSet::add, EnumSet::addAll);
Arrays.stream(Language.values()).filter(language -> !registeredRadixorLanguages.contains(language))
.map(language -> new Candidate(language.name() + "_RADIXOR", language,
() -> adapt(StemmerComparisonBenchmarkQuality.createRadixorQualityStemmer(language))))
.forEach(candidates::add);
Arrays.stream(HunspellStemmerComparisonBenchmarkQuality.HunspellLanguageCase.values())
.map(languageCase -> new Candidate("HUNSPELL_" + languageCase.name() + "_LUCENE_FILTER",
languageCase.radixorLanguage(),
() -> new BatchStemmer() {
/** {@inheritDoc} */
@Override public String[] stem(final String[] forms) throws IOException {
return HunspellStemmerComparisonBenchmarkQuality.stemForQuality(languageCase, forms);
}
/** {@inheritDoc} */
@Override public List<List<String>> stemCandidates(final String[] forms) throws IOException {
return HunspellStemmerComparisonBenchmarkQuality.stemCandidatesForQuality(languageCase, forms);
}
/** {@inheritDoc} */
@Override public boolean supportsMultipleOutputs() { return true; }
}))
.forEach(candidates::add);
return List.copyOf(candidates);
}
/** Adapts one authoritative general-matrix stemmer without changing capability semantics. */
private static BatchStemmer adapt(final StemmerComparisonBenchmarkQuality.CandidateStemmer stemmer) {
return new BatchStemmer() {
/** {@inheritDoc} */
@Override public String[] stem(final String[] forms) throws IOException { return stemmer.stem(forms); }
/** {@inheritDoc} */
@Override public List<List<String>> stemCandidates(final String[] forms) throws IOException {
return stemmer.stemCandidates(forms);
}
/** {@inheritDoc} */
@Override public boolean supportsMultipleOutputs() { return stemmer.supportsMultipleOutputs(); }
};
}
/** One JMH candidate and its authoritative dictionary-language mapping. */
public static final class Candidate {
private final String name;
private final Language language;
private final StemmerFactory factory;
/** Creates an immutable facade over one benchmark candidate. */
private Candidate(final String name, final Language language,
final StemmerFactory factory) {
this.name = Objects.requireNonNull(name, "name");
this.language = Objects.requireNonNull(language, "language");
this.factory = Objects.requireNonNull(factory, "factory");
}
/** @return stable JMH candidate name */
public String name() {
return this.name;
}
/** @return registered Radixor gold-standard dictionary language */
public Language language() {
return this.language;
}
/**
* Creates a scenario-confined adapter using exactly the JMH factory and preprocessing path.
*
* @return sequential batch stemmer
* @throws IOException if benchmark-only resources cannot be loaded
*/
public BatchStemmer createStemmer() throws IOException {
return this.factory.create();
}
}
/** Internal checked factory shared by the JMH quality registries. */
@FunctionalInterface
private interface StemmerFactory {
/** @return a scenario-confined adapter @throws IOException if resources fail */
BatchStemmer create() throws IOException;
}
/** Sequential, scenario-confined batch stemmer contract. */
@FunctionalInterface
public interface BatchStemmer {
/**
* Stems all supplied forms in order.
*
* @param forms input forms, never {@code null}
* @return one non-null output per form
* @throws IOException when the JMH adapter fails
*/
String[] stem(String[] forms) throws IOException;
/** Returns complete candidate sets; single-output adapters return singleton sets. */
default List<List<String>> stemCandidates(final String[] forms) throws IOException {
return Arrays.stream(stem(forms)).map(List::of).toList();
}
/** @return whether this adapter exposes genuine alternative outputs */
default boolean supportsMultipleOutputs() { return false; }
}
}

View File

@@ -30,7 +30,10 @@
******************************************************************************/ ******************************************************************************/
package org.egothor.stemmer.benchmark; package org.egothor.stemmer.benchmark;
import java.util.LinkedHashSet;
import java.util.List;
import java.util.Objects; import java.util.Objects;
import java.util.Set;
import org.egothor.stemmer.CompiledPatchCommand; import org.egothor.stemmer.CompiledPatchCommand;
import org.egothor.stemmer.FrequencyTrie; import org.egothor.stemmer.FrequencyTrie;
@@ -79,4 +82,22 @@ final class RadixorBenchmarkStemmer {
} }
return patch.apply(token); return patch.apply(token);
} }
/**
* Returns every distinct candidate stem from the ranked {@code getAll} path,
* always including the deterministic primary output.
*
* @param token original input token
* @return immutable candidate list in deterministic ranked order
*/
List<String> stemAll(final String token) {
final String primary = stem(token);
final Set<String> candidates = new LinkedHashSet<>();
candidates.add(primary);
final CompiledPatchCommand[] patches = this.trie.getAll(token);
for (CompiledPatchCommand patch : patches) {
candidates.add(patch.preservesAllSources() ? token : patch.apply(token));
}
return List.copyOf(candidates);
}
} }

View File

@@ -313,7 +313,7 @@ public class StemmerComparisonBenchmarkQuality {
/** /**
* Candidate stemmers that can be evaluated against a Radixor resource. * Candidate stemmers that can be evaluated against a Radixor resource.
*/ */
private enum QualityCandidate { enum QualityCandidate {
ENGLISH_RADIXOR(StemmerPatchTrieLoader.Language.US_UK), ENGLISH_RADIXOR(StemmerPatchTrieLoader.Language.US_UK),
ENGLISH_SNOWBALL_ORIGINAL_PORTER(StemmerPatchTrieLoader.Language.US_UK), ENGLISH_SNOWBALL_ORIGINAL_PORTER(StemmerPatchTrieLoader.Language.US_UK),
ENGLISH_SNOWBALL_PORTER2(StemmerPatchTrieLoader.Language.US_UK), ENGLISH_SNOWBALL_PORTER2(StemmerPatchTrieLoader.Language.US_UK),
@@ -444,9 +444,9 @@ public class StemmerComparisonBenchmarkQuality {
* @return quality evaluator * @return quality evaluator
* @throws IOException if stemmer resources cannot be loaded * @throws IOException if stemmer resources cannot be loaded
*/ */
QualityEvaluator createEvaluator() throws IOException { CandidateStemmer createStemmer() throws IOException {
if (name().endsWith("_RADIXOR")) { if (name().endsWith("_RADIXOR")) {
return direct(createRadixorStemmer(this.radixorLanguage)); return radixor(createRadixorStemmer(this.radixorLanguage));
} }
if (name().endsWith("_DIRECT") && this.snowballLanguageCase != null) { if (name().endsWith("_DIRECT") && this.snowballLanguageCase != null) {
return direct(this.snowballLanguageCase.createDirectStemmer()::stem); return direct(this.snowballLanguageCase.createDirectStemmer()::stem);
@@ -508,7 +508,7 @@ public class StemmerComparisonBenchmarkQuality {
} }
case POLISH_LUCENE_STEMPEL_FILTER -> case POLISH_LUCENE_STEMPEL_FILTER ->
tokenFilter(input -> new StempelFilter(input, new StempelStemmer(PolishAnalyzer.getDefaultTable()))); tokenFilter(input -> new StempelFilter(input, new StempelStemmer(PolishAnalyzer.getDefaultTable())));
case POLISH_LUCENE_MORFOLOGIK_FILTER -> tokenFilter(MorfologikFilter::new); case POLISH_LUCENE_MORFOLOGIK_FILTER -> tokenFilter(MorfologikFilter::new, true);
case PORTUGUESE_LUCENE_PORTUGUESE_STEM_FILTER -> case PORTUGUESE_LUCENE_PORTUGUESE_STEM_FILTER ->
tokenFilter(input -> new PortugueseStemFilter(lowercase(input))); tokenFilter(input -> new PortugueseStemFilter(lowercase(input)));
case PORTUGUESE_LUCENE_PORTUGUESE_LIGHT_STEM_FILTER -> case PORTUGUESE_LUCENE_PORTUGUESE_LIGHT_STEM_FILTER ->
@@ -523,15 +523,20 @@ public class StemmerComparisonBenchmarkQuality {
tokenFilter(input -> new SwedishMinimalStemFilter(lowercase(input))); tokenFilter(input -> new SwedishMinimalStemFilter(lowercase(input)));
case UKRAINIAN_MORFOLOGIK_DIRECT -> { case UKRAINIAN_MORFOLOGIK_DIRECT -> {
final DictionaryLookup lookup = new DictionaryLookup(loadUkrainianMorfologikDictionary()); final DictionaryLookup lookup = new DictionaryLookup(loadUkrainianMorfologikDictionary());
yield direct(token -> firstMorfologikStem(token, lookup)); yield morphologik(lookup);
} }
case UKRAINIAN_LUCENE_MORFOLOGIK_FILTER -> { case UKRAINIAN_LUCENE_MORFOLOGIK_FILTER -> {
final Dictionary dictionary = loadUkrainianMorfologikDictionary(); final Dictionary dictionary = loadUkrainianMorfologikDictionary();
yield tokenFilter(input -> new MorfologikFilter(input, dictionary)); yield tokenFilter(input -> new MorfologikFilter(input, dictionary), true);
} }
default -> throw new IllegalStateException("No evaluator for " + this + "."); default -> throw new IllegalStateException("No evaluator for " + this + ".");
}; };
} }
/** Creates the exact-root evaluator used by the JMH quality benchmark. */
QualityEvaluator createEvaluator() throws IOException {
return exactRootEvaluator(createStemmer());
}
} }
/** /**
@@ -575,6 +580,68 @@ public class StemmerComparisonBenchmarkQuality {
QualityResult evaluate(LanguageBenchmarkCorpus.Corpus corpus, Blackhole blackhole) throws IOException; QualityResult evaluate(LanguageBenchmarkCorpus.Corpus corpus, Blackhole blackhole) throws IOException;
} }
/** Stateful candidate adapter confined to one sequential evaluation scenario. */
@FunctionalInterface
interface CandidateStemmer {
/**
* Stems a deterministic batch through the authoritative JMH invocation path.
*
* @param tokens input tokens, never {@code null}
* @return one non-null output for every input token
* @throws IOException if a token-stream implementation fails
*/
String[] stem(String[] tokens) throws IOException;
/**
* Returns complete distinct candidate sets, each containing its primary output.
* Single-output adapters expose singleton lists.
*
* @param tokens input tokens
* @return immutable candidate list for every token
* @throws IOException if adapter processing fails
*/
default List<List<String>> stemCandidates(final String[] tokens) throws IOException {
final String[] primary = stem(tokens);
return java.util.Arrays.stream(primary).map(List::of).toList();
}
/** @return whether the adapter exposes genuine alternative outputs */
default boolean supportsMultipleOutputs() {
return false;
}
}
/** Creates the candidate-capable Radixor adapter backed by ranked {@code getAll}. */
private static CandidateStemmer radixor(final RadixorBenchmarkStemmer stemmer) {
return new CandidateStemmer() {
/** {@inheritDoc} */
@Override public String[] stem(final String[] tokens) {
final String[] outputs = new String[tokens.length];
for (int index = 0; index < tokens.length; index++) { outputs[index] = stemmer.stem(tokens[index]); }
return outputs;
}
/** {@inheritDoc} */
@Override public List<List<String>> stemCandidates(final String[] tokens) {
return java.util.Arrays.stream(tokens).map(stemmer::stemAll).toList();
}
/** {@inheritDoc} */
@Override public boolean supportsMultipleOutputs() { return true; }
};
}
/**
* Creates the authoritative multi-output Radixor adapter for a validated dictionary language.
*
* @param language bundled Radixor language
* @return scenario-confined adapter using the JMH invocation path
* @throws IOException if the compiled dictionary cannot be loaded
*/
static CandidateStemmer createRadixorQualityStemmer(final StemmerPatchTrieLoader.Language language)
throws IOException {
return radixor(createRadixorStemmer(language));
}
/** /**
* Exact-root agreement counters for one quality operation. * Exact-root agreement counters for one quality operation.
* *
@@ -590,57 +657,79 @@ public class StemmerComparisonBenchmarkQuality {
} }
/** /**
* Creates a direct evaluator. * Creates a direct candidate adapter.
* *
* @param stemmer direct stemmer * @param stemmer direct stemmer
* @return quality evaluator * @return sequential batch adapter
*/ */
private static QualityEvaluator direct(final Stemmer stemmer) { private static CandidateStemmer direct(final Stemmer stemmer) {
Objects.requireNonNull(stemmer, "stemmer"); Objects.requireNonNull(stemmer, "stemmer");
return (corpus, blackhole) -> { return tokens -> {
int correct = 0; final String[] outputs = new String[tokens.length];
int changedCorrect = 0;
int changedEvaluated = 0;
int rootPreserved = 0;
int rootEvaluated = 0;
final String[] tokens = corpus.tokens();
final String[] expectedRoots = corpus.expectedRoots();
for (int index = 0; index < tokens.length; index++) { for (int index = 0; index < tokens.length; index++) {
final String token = tokens[index]; outputs[index] = stemmer.stem(tokens[index]);
final String expectedRoot = expectedRoots[index];
final String actual = stemmer.stem(token);
blackhole.consume(actual);
final boolean exact = Objects.equals(expectedRoot, actual);
if (exact) {
correct++;
} }
if (Objects.equals(token, expectedRoot)) { return outputs;
rootEvaluated++;
if (exact) {
rootPreserved++;
}
} else {
changedEvaluated++;
if (exact) {
changedCorrect++;
}
}
}
return new QualityResult(correct, tokens.length, changedCorrect, changedEvaluated, rootPreserved,
rootEvaluated);
}; };
} }
/** /**
* Creates a TokenFilter evaluator. * Creates a TokenFilter candidate adapter.
* *
* @param factory token stream factory * @param factory token stream factory
* @return quality evaluator * @return sequential batch adapter
*/ */
private static QualityEvaluator tokenFilter(final Function<TokenStream, TokenStream> factory) { private static CandidateStemmer tokenFilter(final Function<TokenStream, TokenStream> factory) {
Objects.requireNonNull(factory, "factory"); Objects.requireNonNull(factory, "factory");
return tokens -> firstTokenFilterOutputs(tokens, factory, null);
}
/** Creates a TokenFilter adapter that preserves all terms emitted per position. */
private static CandidateStemmer tokenFilter(final Function<TokenStream, TokenStream> factory,
final boolean multipleOutputs) {
if (!multipleOutputs) { return tokenFilter(factory); }
return new CandidateStemmer() {
/** {@inheritDoc} */
@Override public String[] stem(final String[] tokens) throws IOException {
return firstTokenFilterOutputs(tokens, factory, null);
}
/** {@inheritDoc} */
@Override public List<List<String>> stemCandidates(final String[] tokens) throws IOException {
return allTokenFilterOutputs(tokens, factory);
}
/** {@inheritDoc} */
@Override public boolean supportsMultipleOutputs() { return true; }
};
}
/** Creates a multi-analysis Morphologik direct adapter. */
private static CandidateStemmer morphologik(final DictionaryLookup lookup) {
return new CandidateStemmer() {
/** {@inheritDoc} */
@Override public String[] stem(final String[] tokens) {
final String[] outputs = new String[tokens.length];
for (int index = 0; index < tokens.length; index++) { outputs[index] = firstMorfologikStem(tokens[index], lookup); }
return outputs;
}
/** {@inheritDoc} */
@Override public List<List<String>> stemCandidates(final String[] tokens) {
return java.util.Arrays.stream(tokens).map(token -> allMorfologikStems(token, lookup)).toList();
}
/** {@inheritDoc} */
@Override public boolean supportsMultipleOutputs() { return true; }
};
}
/**
* Creates exact-root accounting around an authoritative candidate adapter.
*
* @param stemmer candidate adapter
* @return JMH exact-root evaluator
*/
private static QualityEvaluator exactRootEvaluator(final CandidateStemmer stemmer) {
Objects.requireNonNull(stemmer, "stemmer");
return (corpus, blackhole) -> { return (corpus, blackhole) -> {
final String[] actualStems = firstTokenFilterOutputs(corpus.tokens(), factory, blackhole); final String[] actualStems = stemmer.stem(corpus.tokens());
final String[] expectedRoots = corpus.expectedRoots(); final String[] expectedRoots = corpus.expectedRoots();
final String[] tokens = corpus.tokens(); final String[] tokens = corpus.tokens();
int correct = 0; int correct = 0;
@@ -651,6 +740,7 @@ public class StemmerComparisonBenchmarkQuality {
for (int index = 0; index < actualStems.length; index++) { for (int index = 0; index < actualStems.length; index++) {
final String token = tokens[index]; final String token = tokens[index];
final String expectedRoot = expectedRoots[index]; final String expectedRoot = expectedRoots[index];
blackhole.consume(actualStems[index]);
final boolean exact = Objects.equals(expectedRoot, actualStems[index]); final boolean exact = Objects.equals(expectedRoot, actualStems[index]);
if (exact) { if (exact) {
correct++; correct++;
@@ -679,10 +769,9 @@ public class StemmerComparisonBenchmarkQuality {
* @return direct stemmer * @return direct stemmer
* @throws IOException if the trie cannot be loaded * @throws IOException if the trie cannot be loaded
*/ */
private static Stemmer createRadixorStemmer(final StemmerPatchTrieLoader.Language language) throws IOException { private static RadixorBenchmarkStemmer createRadixorStemmer(final StemmerPatchTrieLoader.Language language) throws IOException {
final RadixorBenchmarkStemmer stemmer = new RadixorBenchmarkStemmer(StemmerPatchTrieLoader.loadCompiled( return new RadixorBenchmarkStemmer(StemmerPatchTrieLoader.loadCompiled(
language, true, ReductionMode.MERGE_SUBTREES_WITH_EQUIVALENT_RANKED_GET_ALL_RESULTS)); language, true, ReductionMode.MERGE_SUBTREES_WITH_EQUIVALENT_RANKED_GET_ALL_RESULTS));
return stemmer::stem;
} }
/** /**
@@ -715,6 +804,14 @@ public class StemmerComparisonBenchmarkQuality {
return analyses.get(0).getStem().toString(); return analyses.get(0).getStem().toString();
} }
/** Returns all distinct Morphologik lemma strings and always includes the primary output. */
private static List<String> allMorfologikStems(final String token, final DictionaryLookup lookup) {
final java.util.LinkedHashSet<String> stems = new java.util.LinkedHashSet<>();
stems.add(firstMorfologikStem(token, lookup));
for (WordData analysis : lookup.lookup(token)) { stems.add(analysis.getStem().toString()); }
return List.copyOf(stems);
}
/** /**
* Extracts the first emitted term for each input token from a TokenFilter * Extracts the first emitted term for each input token from a TokenFilter
* pipeline. * pipeline.
@@ -746,8 +843,10 @@ public class StemmerComparisonBenchmarkQuality {
outputs[inputIndex] = termAttribute.toString(); outputs[inputIndex] = termAttribute.toString();
recordedForPosition = true; recordedForPosition = true;
} }
if (blackhole != null) {
blackhole.consume(termAttribute); blackhole.consume(termAttribute);
} }
}
output.end(); output.end();
output.close(); output.close();
@@ -759,6 +858,35 @@ public class StemmerComparisonBenchmarkQuality {
return outputs; return outputs;
} }
/**
* Extracts every distinct emitted term for each input position and includes the
* deterministic primary output even when a filter omits it.
*/
private static List<List<String>> allTokenFilterOutputs(final String[] tokens,
final Function<TokenStream, TokenStream> factory) throws IOException {
final List<java.util.LinkedHashSet<String>> candidates = new java.util.ArrayList<>(tokens.length);
for (int index = 0; index < tokens.length; index++) { candidates.add(new java.util.LinkedHashSet<>()); }
final BenchmarkTokenStream input = new BenchmarkTokenStream(tokens);
final TokenStream output = factory.apply(input);
final CharTermAttribute term = output.addAttribute(CharTermAttribute.class);
final PositionIncrementAttribute position = output.addAttribute(PositionIncrementAttribute.class);
int inputIndex = -1;
output.reset();
while (output.incrementToken()) {
if (position.getPositionIncrement() > 0) { inputIndex += position.getPositionIncrement(); }
if (inputIndex >= 0 && inputIndex < candidates.size()) { candidates.get(inputIndex).add(term.toString()); }
}
output.end();
output.close();
final String[] primary = firstTokenFilterOutputs(tokens, factory, null);
final List<List<String>> result = new java.util.ArrayList<>(tokens.length);
for (int index = 0; index < tokens.length; index++) {
candidates.get(index).add(primary[index]);
result.add(List.copyOf(candidates.get(index)));
}
return List.copyOf(result);
}
/** /**
* Adds Lucene lower-case normalization. * Adds Lucene lower-case normalization.
* *

View File

@@ -58,10 +58,10 @@ final class PaiceHuskLancasterStemmerTest {
*/ */
private static final String[][] SAMPLE_STEMS = { private static final String[][] SAMPLE_STEMS = {
{ "running", "run" }, { "running", "run" },
{ "caresses", "cares" }, { "caresses", "caress" },
{ "happiness", "happi" }, { "happiness", "happy" },
{ "connected", "connect" }, { "connected", "connect" },
{ "dancing", "danc" }, { "dancing", "dant" },
{ "happy", "happy" } { "happy", "happy" }
}; };
@@ -116,7 +116,7 @@ final class PaiceHuskLancasterStemmerTest {
final Object stemmer = createStemmer(); final Object stemmer = createStemmer();
final Method stemMethod = stemMethod(); final Method stemMethod = stemMethod();
assertEquals("running", stemMethod.invoke(stemmer, "running")); assertEquals("run", stemMethod.invoke(stemmer, "running"));
assertEquals(null, stemMethod.invoke(stemmer, new Object[] { null })); assertEquals(null, stemMethod.invoke(stemmer, new Object[] { null }));
assertNotNull(stemMethod.invoke(stemmer, "connected")); assertNotNull(stemMethod.invoke(stemmer, "connected"));
} }

View File

@@ -0,0 +1,58 @@
package org.egothor.stemmer.benchmark.quality;
import java.io.BufferedReader;
import java.io.IOException;
import java.io.InputStream;
import java.io.InputStreamReader;
import java.nio.charset.StandardCharsets;
import java.util.ArrayList;
import java.util.Arrays;
import java.util.List;
import java.util.Objects;
import java.util.zip.GZIPInputStream;
import org.egothor.stemmer.CaseProcessingMode;
import org.egothor.stemmer.StemmerDictionaryParser;
import org.egothor.stemmer.StemmerPatchTrieLoader.Language;
/** Loads gold-standard groups from authoritative bundled dictionary resources. */
public final class BundledGoldStandardLoader {
/** Utility class. */
private BundledGoldStandardLoader() { throw new AssertionError("No instances."); }
/**
* Parses one compressed UTF-8 dictionary with case preserved.
* @param language registered bundled language
* @return immutable groups in source-row order
* @throws IOException if the resource is absent, malformed, or unreadable
*/
public static List<GoldStandardGroup> load(final Language language) throws IOException {
Objects.requireNonNull(language, "language");
final String resource = language.resourcePath();
final List<GoldStandardGroup> groups = new ArrayList<>();
try (InputStream raw = openResource(language, resource); InputStream gzip = new GZIPInputStream(raw);
BufferedReader reader = new BufferedReader(new InputStreamReader(gzip, StandardCharsets.UTF_8))) {
StemmerDictionaryParser.parse(reader, resource, CaseProcessingMode.AS_IS, (stem, variants, row) -> {
final List<String> forms = new ArrayList<>(variants.length + 1);
forms.add(stem);
forms.addAll(Arrays.asList(variants));
try {
groups.add(new GoldStandardGroup(row, forms));
} catch (IllegalArgumentException exception) {
throw new IOException("Invalid dictionary group for language " + language + ", resource "
+ resource + ", row " + row + ": " + exception.getMessage(), exception);
}
});
}
return List.copyOf(groups);
}
/** Opens one required classpath resource with a precise language diagnostic. */
private static InputStream openResource(final Language language, final String resource) throws IOException {
final InputStream input = Thread.currentThread().getContextClassLoader().getResourceAsStream(resource);
if (input == null) {
throw new IOException("Dictionary resource is missing for language " + language + ": " + resource + ".");
}
return input;
}
}

View File

@@ -0,0 +1,165 @@
package org.egothor.stemmer.benchmark.quality;
import java.io.IOException;
import java.util.ArrayList;
import java.util.HashMap;
import java.util.HashSet;
import java.util.List;
import java.util.Map;
import java.util.Set;
import java.util.TreeSet;
import org.egothor.stemmer.benchmark.QualityStemmerMatrix.BatchStemmer;
/**
* Calculates exact candidate-intersection pair metrics from canonical candidate-set signatures.
* Candidate-aware output defines an overlap relation rather than a partition. The algorithm
* aggregates signature frequencies and uses an inverted candidate index; it never enumerates
* complete dictionary word pairs. All pair arithmetic is checked.
*/
final class CandidateAwareEvaluator {
/** Utility class. */
private CandidateAwareEvaluator() { throw new AssertionError("No instances."); }
/** Evaluates one genuinely multi-output scenario through its authoritative JMH adapter. */
static QualityResult evaluate(final String stemmerName, final String language, final ProcessingMode mode,
final OutputPolicy policy, final List<GoldStandardGroup> groups, final BatchStemmer stemmer) throws IOException {
if (policy == OutputPolicy.PRIMARY_OUTPUT) {
throw new IllegalArgumentException("Candidate-aware evaluation requires ANY_CANDIDATE or ALL_CANDIDATES.");
}
final List<GoldStandardGroup> includedGroups = groups.stream().filter(group -> mode.includes(group.forms())).toList();
final List<String> forms = new ArrayList<>();
final List<Integer> groupIndexes = new ArrayList<>();
long singletonRows = 0;
long pairRows = 0;
long underPossible = 0;
for (int groupIndex = 0; groupIndex < includedGroups.size(); groupIndex++) {
final GoldStandardGroup group = includedGroups.get(groupIndex);
if (group.forms().size() == 1) { singletonRows = add(singletonRows, 1, "singleton rows"); }
else { pairRows = add(pairRows, 1, "rows contributing under-stemming pairs"); }
underPossible = add(underPossible, QualityEvaluator.chooseTwo(group.forms().size()), "under denominator");
for (String form : group.forms()) { forms.add(form); groupIndexes.add(groupIndex); }
}
final String[] input = forms.toArray(String[]::new);
final String[] primary = stemmer.stem(input);
final List<List<String>> rawCandidates = stemmer.stemCandidates(input);
if (primary == null || primary.length != input.length || rawCandidates == null || rawCandidates.size() != input.length) {
throw failure(stemmerName, language, mode, policy, "the adapter returned an invalid output batch");
}
final Map<Signature, SignatureCount> counts = new HashMap<>();
final Set<String> distinctCandidates = new HashSet<>();
long oneCandidate = 0;
long multipleCandidates = 0;
long maximumCandidates = 0;
long assignments = 0;
for (int index = 0; index < input.length; index++) {
final Signature signature = signature(rawCandidates.get(index), primary[index], stemmerName, language,
mode, policy, includedGroups.get(groupIndexes.get(index)).rowNumber(), input[index]);
final int size = signature.candidates().size();
if (size == 1) { oneCandidate = add(oneCandidate, 1, "single-candidate forms"); }
else { multipleCandidates = add(multipleCandidates, 1, "multi-candidate forms"); }
maximumCandidates = Math.max(maximumCandidates, size);
assignments = add(assignments, size, "candidate assignments");
distinctCandidates.addAll(signature.candidates());
counts.computeIfAbsent(signature, ignored -> new SignatureCount()).increment(groupIndexes.get(index));
}
final List<Map.Entry<Signature, SignatureCount>> signatures = new ArrayList<>(counts.entrySet());
signatures.sort(Map.Entry.comparingByKey());
long sameGroupRelated = 0;
long crossGroupRelated = 0;
final Map<String, List<Integer>> inverted = new HashMap<>();
for (int index = 0; index < signatures.size(); index++) {
final Map.Entry<Signature, SignatureCount> entry = signatures.get(index);
long sameWithin = 0;
for (long groupCount : entry.getValue().byGroup().values()) {
sameWithin = add(sameWithin, QualityEvaluator.chooseTwo(groupCount), "same-signature group pairs");
}
sameGroupRelated = add(sameGroupRelated, sameWithin, "same-group related pairs");
if (policy == OutputPolicy.ALL_CANDIDATES || entry.getKey().candidates().size() == 1) {
crossGroupRelated = add(crossGroupRelated,
subtract(QualityEvaluator.chooseTwo(entry.getValue().total()), sameWithin, "same-signature cross pairs"),
"cross-group related pairs");
}
for (String candidate : entry.getKey().candidates()) {
inverted.computeIfAbsent(candidate, ignored -> new ArrayList<>()).add(index);
}
}
final Set<SignaturePair> relatedSignaturePairs = new HashSet<>();
for (List<Integer> indexes : inverted.values()) {
for (int left = 0; left < indexes.size(); left++) {
for (int right = left + 1; right < indexes.size(); right++) {
relatedSignaturePairs.add(new SignaturePair(indexes.get(left), indexes.get(right)));
}
}
}
for (SignaturePair pair : relatedSignaturePairs) {
final SignatureCount left = signatures.get(pair.left()).getValue();
final SignatureCount right = signatures.get(pair.right()).getValue();
long same = 0;
for (Map.Entry<Integer, Long> group : left.byGroup().entrySet()) {
same = add(same, multiply(group.getValue(), right.byGroup().getOrDefault(group.getKey(), 0L),
"different-signature same-group pairs"), "same-group related pairs");
}
final long total = multiply(left.total(), right.total(), "different-signature pairs");
sameGroupRelated = add(sameGroupRelated, same, "same-group related pairs");
if (policy == OutputPolicy.ALL_CANDIDATES) {
crossGroupRelated = add(crossGroupRelated, subtract(total, same, "different-signature cross pairs"),
"cross-group related pairs");
}
}
final long wordCount = input.length;
final long overPossible = subtract(QualityEvaluator.chooseTwo(wordCount), underPossible, "over denominator");
final long underError = subtract(underPossible, sameGroupRelated, "candidate under errors");
return new QualityResult(stemmerName, language, mode, policy,
includedGroups.size(), wordCount, singletonRows, pairRows, oneCandidate, multipleCandidates,
maximumCandidates, assignments, distinctCandidates.size(), crossGroupRelated, overPossible,
underError, underPossible, null);
}
/** Canonicalizes and validates one adapter candidate collection. */
private static Signature signature(final List<String> raw, final String primary, final String stemmer,
final String language, final ProcessingMode mode, final OutputPolicy policy,
final int row, final String form) throws IOException {
if (primary == null) { throw failure(stemmer, language, mode, policy, "null primary output at row " + row + " for '" + form + "'"); }
if (raw == null || raw.isEmpty()) { throw failure(stemmer, language, mode, policy, "null or empty candidate collection at row " + row + " for '" + form + "'"); }
final TreeSet<String> candidates = new TreeSet<>();
for (String candidate : raw) {
if (candidate == null) { throw failure(stemmer, language, mode, policy, "null candidate at row " + row + " for '" + form + "'"); }
candidates.add(candidate);
}
if (!candidates.contains(primary)) { throw failure(stemmer, language, mode, policy, "candidate set omits primary output '" + primary + "' at row " + row + " for '" + form + "'"); }
return new Signature(List.copyOf(candidates));
}
/** Checked addition with diagnostic context. */
private static long add(final long left, final long right, final String context) { try { return Math.addExact(left, right); } catch (ArithmeticException exception) { throw new IllegalStateException("Arithmetic overflow in " + context + ".", exception); } }
/** Checked subtraction with diagnostic context. */
private static long subtract(final long left, final long right, final String context) { try { return Math.subtractExact(left, right); } catch (ArithmeticException exception) { throw new IllegalStateException("Arithmetic overflow in " + context + ".", exception); } }
/** Checked multiplication with diagnostic context. */
private static long multiply(final long left, final long right, final String context) { try { return Math.multiplyExact(left, right); } catch (ArithmeticException exception) { throw new IllegalStateException("Arithmetic overflow in " + context + ".", exception); } }
/** Creates one scenario-qualified adapter failure. */
private static IOException failure(final String stemmer, final String language, final ProcessingMode mode,
final OutputPolicy policy, final String reason) { return new IOException("Candidate-aware evaluation failed for stemmer " + stemmer + ", language " + language + ", dictionary mode " + mode + ", and output policy " + policy + ": " + reason + "."); }
/** Deterministic immutable candidate-set signature. */
private record Signature(List<String> candidates) implements Comparable<Signature> {
/** Orders signatures lexicographically without depending on map iteration. */
@Override public int compareTo(final Signature other) {
final int common = Math.min(candidates.size(), other.candidates.size());
for (int index = 0; index < common; index++) { final int compared = candidates.get(index).compareTo(other.candidates.get(index)); if (compared != 0) { return compared; } }
return Integer.compare(candidates.size(), other.candidates.size());
}
}
/** Aggregated global and per-group frequency of one signature. */
private static final class SignatureCount {
private long total;
private final Map<Integer, Long> byGroup = new HashMap<>();
/** Adds one word occurrence. */ private void increment(final int group) { total = add(total, 1, "signature frequency"); byGroup.merge(group, 1L, (left, right) -> add(left, right, "signature group frequency")); }
/** @return global signature frequency */ private long total() { return total; }
/** @return mutable internally owned per-group frequencies */ private Map<Integer, Long> byGroup() { return byGroup; }
}
/** Unordered pair of distinct canonical signature indexes. */
private record SignaturePair(int left, int right) { }
}

View File

@@ -0,0 +1,199 @@
package org.egothor.stemmer.benchmark.quality;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertThrows;
import static org.junit.jupiter.api.Assertions.assertTrue;
import java.io.IOException;
import java.util.ArrayList;
import java.util.HashMap;
import java.util.LinkedHashSet;
import java.util.List;
import java.util.Map;
import java.util.Random;
import java.util.Set;
import org.egothor.stemmer.benchmark.QualityStemmerMatrix.BatchStemmer;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Tag;
import org.junit.jupiter.api.Test;
/** Exact candidate-relation tests, including an independent quadratic oracle. */
@Tag("unit")
@DisplayName("Candidate-aware pairwise stemming quality")
final class CandidateAwareEvaluatorTest {
/** Verifies intersections repair under-stemming while several shared candidates count once. */
@Test @DisplayName("Candidate intersections repair primary under-stemming and count each pair once")
void intersectionsRepairUnderStemming() throws IOException {
final List<GoldStandardGroup> groups = List.of(new GoldStandardGroup(1, List.of("a", "b", "c")));
final Map<String, String> primary = Map.of("a", "y", "b", "x", "c", "z");
final Map<String, List<String>> candidates = Map.of("a", List.of("y", "x", "x"),
"b", List.of("x", "shared"), "c", List.of("z", "x", "shared"));
final QualityResult primaryResult = QualityEvaluator.evaluateBatch("Synthetic", "MULTI",
ProcessingMode.ALL_WORDS, groups, adapter(primary, candidates));
final QualityResult candidateResult = CandidateAwareEvaluator.evaluate("Synthetic", "MULTI",
ProcessingMode.ALL_WORDS, OutputPolicy.ALL_CANDIDATES, groups, adapter(primary, candidates));
assertEquals(3, primaryResult.underErrorPairs());
assertEquals(0, candidateResult.underErrorPairs());
assertEquals(7, candidateResult.totalCandidateAssignments(), "Duplicate candidates must be removed per word.");
assertTrue(candidateResult.underErrorPairs() <= primaryResult.underErrorPairs());
}
/** Verifies exact within-row disconnections and cross-row candidate collisions. */
@Test @DisplayName("Disjoint sets and cross-group intersections produce exact candidate-aware counts")
void disjointAndCollidingSets() throws IOException {
final List<GoldStandardGroup> groups = List.of(
new GoldStandardGroup(1, List.of("a", "b")), new GoldStandardGroup(2, List.of("c", "d")));
final Map<String, String> primary = Map.of("a", "a", "b", "b", "c", "c", "d", "d");
final Map<String, List<String>> candidates = Map.of("a", List.of("a", "collision"), "b", List.of("b"),
"c", List.of("c", "collision", "other"), "d", List.of("d", "other"));
final QualityResult result = CandidateAwareEvaluator.evaluate("Synthetic", "MULTI",
ProcessingMode.ALL_WORDS, OutputPolicy.ALL_CANDIDATES, groups, adapter(primary, candidates));
assertEquals(1, result.underErrorPairs());
assertEquals(2, result.underPossiblePairs());
assertEquals(1, result.overErrorPairs(), "Only the cross-group a-c pair shares a candidate.");
assertEquals(4, result.overPossiblePairs());
}
/** Verifies optimistic and all-active cross-group semantics for canonical examples. */
@Test @DisplayName("ANY_CANDIDATE and ALL_CANDIDATES apply their distinct over-stemming relations")
void policySpecificOverStemming() throws IOException {
assertPolicyOver(List.of("x"), List.of("x"), 1, 1);
assertPolicyOver(List.of("x"), List.of("y"), 0, 0);
assertPolicyOver(List.of("x"), List.of("x", "y"), 0, 1);
assertPolicyOver(List.of("x", "y"), List.of("x", "y"), 0, 1);
assertPolicyOver(List.of("x", "y"), List.of("x", "z"), 0, 1);
}
/** Verifies both candidate policies have identical same-group under-stemming. */
@Test @DisplayName("Candidate policies share the exact same within-group intersection rule")
void candidatePoliciesShareUnderStemming() throws IOException {
final List<GoldStandardGroup> groups = List.of(new GoldStandardGroup(1, List.of("a", "b", "c")));
final Map<String, String> primary = Map.of("a", "x", "b", "y", "c", "z");
final Map<String, List<String>> candidates = Map.of("a", List.of("x", "shared"),
"b", List.of("y", "shared"), "c", List.of("z"));
final QualityResult any = CandidateAwareEvaluator.evaluate("Synthetic", "MULTI", ProcessingMode.ALL_WORDS,
OutputPolicy.ANY_CANDIDATE, groups, adapter(primary, candidates));
final QualityResult all = CandidateAwareEvaluator.evaluate("Synthetic", "MULTI", ProcessingMode.ALL_WORDS,
OutputPolicy.ALL_CANDIDATES, groups, adapter(primary, candidates));
assertEquals(2, any.underErrorPairs()); assertEquals(any.underErrorPairs(), all.underErrorPairs());
}
/** Compares the optimized signature algorithm with an independent fixed-seed oracle. */
@Test @DisplayName("Optimized candidate metrics equal a deterministic randomized brute-force oracle")
void randomizedOracleAgreement() throws IOException {
final Random random = new Random(0x5EEDC0DEL);
for (int trial = 0; trial < 150; trial++) {
final int groupCount = 1 + random.nextInt(5);
final List<GoldStandardGroup> groups = new ArrayList<>();
final Map<String, String> primary = new HashMap<>();
final Map<String, List<String>> candidates = new HashMap<>();
int word = 0;
for (int group = 0; group < groupCount; group++) {
final List<String> forms = new ArrayList<>();
for (int member = 0; member < 1 + random.nextInt(5); member++) {
final String form = "w" + word++;
forms.add(form);
final String primaryStem = "s" + random.nextInt(7);
primary.put(form, primaryStem);
final List<String> raw = new ArrayList<>();
raw.add(primaryStem);
for (int candidate = 0; candidate < random.nextInt(4); candidate++) {
raw.add("s" + random.nextInt(7));
}
candidates.put(form, raw);
}
groups.add(new GoldStandardGroup(group + 1, forms));
}
final QualityResult optimized = CandidateAwareEvaluator.evaluate("Random", "MULTI",
ProcessingMode.ALL_WORDS, OutputPolicy.ALL_CANDIDATES, groups, adapter(primary, candidates));
final QualityResult any = CandidateAwareEvaluator.evaluate("Random", "MULTI",
ProcessingMode.ALL_WORDS, OutputPolicy.ANY_CANDIDATE, groups, adapter(primary, candidates));
final QualityResult primaryResult = QualityEvaluator.evaluateBatch("Random", "MULTI",
ProcessingMode.ALL_WORDS, groups, adapter(primary, candidates));
final long[] oracle = oracle(groups, candidates);
assertEquals(oracle[0], optimized.underErrorPairs(), "Under errors differ in trial " + trial);
assertEquals(oracle[1], optimized.underPossiblePairs(), "Under denominator differs in trial " + trial);
assertEquals(oracle[2], optimized.overErrorPairs(), "Over errors differ in trial " + trial);
assertEquals(oracle[3], optimized.overPossiblePairs(), "Over denominator differs in trial " + trial);
assertEquals(oracle[4], any.overErrorPairs(), "Optimistic over errors differ in trial " + trial);
assertEquals(any.underErrorPairs(), optimized.underErrorPairs());
assertTrue(any.underErrorPairs() <= primaryResult.underErrorPairs());
assertTrue(any.overErrorPairs() <= primaryResult.overErrorPairs());
assertTrue(optimized.overErrorPairs() >= primaryResult.overErrorPairs());
}
}
/** Verifies candidate contract violations fail with scenario and word context. */
@Test @DisplayName("Invalid candidate collections fail with precise contextual diagnostics")
void invalidCandidateOutput() {
final List<GoldStandardGroup> groups = List.of(new GoldStandardGroup(7, List.of("žluťoučký")));
final BatchStemmer invalid = adapter(Map.of("žluťoučký", "stem"), Map.of("žluťoučký", List.of("other")));
final IOException exception = assertThrows(IOException.class, () -> CandidateAwareEvaluator.evaluate(
"Invalid", "CS_CZ", ProcessingMode.ALL_WORDS, OutputPolicy.ALL_CANDIDATES, groups, invalid));
assertTrue(exception.getMessage().contains("row 7"));
assertTrue(exception.getMessage().contains("žluťoučký"));
assertTrue(exception.getMessage().contains("omits primary output"));
}
/** Creates a deterministic multi-output adapter from per-form fixtures. */
private static BatchStemmer adapter(final Map<String, String> primary,
final Map<String, List<String>> candidates) {
return new BatchStemmer() {
/** {@inheritDoc} */
@Override public String[] stem(final String[] forms) {
final String[] outputs = new String[forms.length];
for (int index = 0; index < forms.length; index++) { outputs[index] = primary.get(forms[index]); }
return outputs;
}
/** {@inheritDoc} */
@Override public List<List<String>> stemCandidates(final String[] forms) {
final List<List<String>> outputs = new ArrayList<>();
for (String form : forms) { outputs.add(candidates.get(form)); }
return outputs;
}
/** {@inheritDoc} */
@Override public boolean supportsMultipleOutputs() { return true; }
};
}
/** Evaluates one two-row example and checks both policy numerators. */
private static void assertPolicyOver(final List<String> left, final List<String> right,
final long expectedAny, final long expectedAll) throws IOException {
final List<GoldStandardGroup> groups = List.of(new GoldStandardGroup(1, List.of("a")),
new GoldStandardGroup(2, List.of("b")));
final Map<String, String> primary = Map.of("a", left.get(0), "b", right.get(0));
final Map<String, List<String>> candidates = Map.of("a", left, "b", right);
final QualityResult any = CandidateAwareEvaluator.evaluate("Synthetic", "MULTI", ProcessingMode.ALL_WORDS,
OutputPolicy.ANY_CANDIDATE, groups, adapter(primary, candidates));
final QualityResult all = CandidateAwareEvaluator.evaluate("Synthetic", "MULTI", ProcessingMode.ALL_WORDS,
OutputPolicy.ALL_CANDIDATES, groups, adapter(primary, candidates));
assertEquals(expectedAny, any.overErrorPairs()); assertEquals(expectedAll, all.overErrorPairs());
}
/** Enumerates small word pairs independently and returns under error/possible and over error/possible counts. */
private static long[] oracle(final List<GoldStandardGroup> groups,
final Map<String, List<String>> candidates) {
final List<String> forms = new ArrayList<>();
final List<Integer> labels = new ArrayList<>();
for (int group = 0; group < groups.size(); group++) {
for (String form : groups.get(group).forms()) { forms.add(form); labels.add(group); }
}
long underError = 0; long underPossible = 0; long overError = 0; long overPossible = 0; long anyOverError = 0;
for (int left = 0; left < forms.size(); left++) {
for (int right = left + 1; right < forms.size(); right++) {
final Set<String> intersection = new LinkedHashSet<>(candidates.get(forms.get(left)));
intersection.retainAll(new LinkedHashSet<>(candidates.get(forms.get(right))));
if (labels.get(left).equals(labels.get(right))) {
underPossible++; if (intersection.isEmpty()) { underError++; }
} else {
overPossible++; if (!intersection.isEmpty()) { overError++; }
final Set<String> leftSet = new LinkedHashSet<>(candidates.get(forms.get(left)));
final Set<String> rightSet = new LinkedHashSet<>(candidates.get(forms.get(right)));
if (leftSet.size() == 1 && leftSet.equals(rightSet)) { anyOverError++; }
}
}
}
return new long[] {underError, underPossible, overError, overPossible, anyOverError};
}
}

View File

@@ -0,0 +1,121 @@
package org.egothor.stemmer.benchmark.quality;
import java.io.IOException;
import java.nio.charset.StandardCharsets;
import java.nio.file.Files;
import java.nio.file.Path;
import java.nio.file.StandardOpenOption;
import java.util.ArrayList;
import java.util.Comparator;
import java.util.HashMap;
import java.util.HashSet;
import java.util.List;
import java.util.Map;
import java.util.Set;
import java.util.TreeSet;
import org.egothor.stemmer.benchmark.QualityStemmerMatrix.BatchStemmer;
import org.egothor.stemmer.benchmark.QualityStemmerMatrix.Candidate;
/** Produces deterministic word-level diagnostics for genuinely multi-output adapters. */
final class CandidateQualityAudit {
/** Utility class. */
private CandidateQualityAudit() { throw new AssertionError("No instances."); }
/** Evaluates candidate output and retains the largest candidate sets for reproducible inspection. */
static Scenario evaluate(final Candidate candidate, final ProcessingMode mode,
final List<GoldStandardGroup> groups, final QualityResult primary, final QualityResult any,
final int limit) throws IOException {
final List<String> forms = new ArrayList<>();
final List<Integer> groupIndexes = new ArrayList<>();
final List<Integer> rows = new ArrayList<>();
for (int group = 0; group < groups.size(); group++) {
final GoldStandardGroup item = groups.get(group);
if (!mode.includes(item.forms())) { continue; }
for (String form : item.forms()) { forms.add(form); groupIndexes.add(group); rows.add(item.rowNumber()); }
}
final BatchStemmer stemmer = candidate.createStemmer();
final String[] primaryOutputs = stemmer.stem(forms.toArray(String[]::new));
final List<List<String>> rawCandidates = stemmer.stemCandidates(forms.toArray(String[]::new));
final Map<String, List<Integer>> inverted = new HashMap<>();
final Map<Integer, Long> candidateCountDistribution = new java.util.TreeMap<>();
final List<List<String>> candidateSets = new ArrayList<>();
for (int index = 0; index < forms.size(); index++) {
final TreeSet<String> canonical = new TreeSet<>(rawCandidates.get(index));
canonical.add(primaryOutputs[index]);
final List<String> set = List.copyOf(canonical);
candidateSets.add(set);
candidateCountDistribution.merge(set.size(), 1L, Math::addExact);
for (String value : set) { inverted.computeIfAbsent(value, ignored -> new ArrayList<>()).add(index); }
}
final QualityResult candidateResult = CandidateAwareEvaluator.evaluate(candidate.name(), candidate.language().name(),
mode, OutputPolicy.ALL_CANDIDATES, groups, candidate.createStemmer());
final List<Integer> selected = new ArrayList<>();
for (int index = 0; index < forms.size(); index++) { if (candidateSets.get(index).size() > 1) { selected.add(index); } }
selected.sort(Comparator.<Integer>comparingInt(index -> candidateSets.get(index).size()).reversed()
.thenComparing(index -> forms.get(index)).thenComparingInt(index -> rows.get(index)));
final List<Word> words = new ArrayList<>();
for (int index : selected.subList(0, Math.min(limit, selected.size()))) {
final Set<Integer> partners = new HashSet<>();
for (String value : candidateSets.get(index)) { partners.addAll(inverted.get(value)); }
partners.remove(index);
long repaired = 0; long introduced = 0;
for (int partner : partners) {
final boolean primaryRelated = primaryOutputs[index].equals(primaryOutputs[partner]);
if (groupIndexes.get(index).equals(groupIndexes.get(partner)) && !primaryRelated) { repaired++; }
if (!groupIndexes.get(index).equals(groupIndexes.get(partner)) && !primaryRelated) { introduced++; }
}
words.add(new Word(rows.get(index), forms.get(index), primaryOutputs[index], candidateSets.get(index),
repaired, introduced));
}
return new Scenario(primary, any, candidateResult, Map.copyOf(candidateCountDistribution), List.copyOf(words));
}
/** Appends candidate diagnostics to the freshly generated audit report. */
static void append(final Path path, final List<Scenario> scenarios) throws IOException {
if (scenarios.isEmpty()) { return; }
final StringBuilder text = new StringBuilder(4096);
text.append("\n# Candidate-aware audit\n\nWord-level sections below retain original Unicode forms. Per-word repaired and introduced counts describe relations involving that word and are diagnostic, not additive scenario totals.\n\n");
for (Scenario scenario : scenarios.stream().sorted(Comparator.comparing(Scenario::candidate, QualityResult.ORDER)).toList()) {
final QualityResult primary = scenario.primary();
final QualityResult any = scenario.any();
final QualityResult candidate = scenario.candidate();
text.append("## ").append(candidate.stemmer()).append(" / ").append(candidate.language()).append(" / ")
.append(candidate.processingMode()).append(" / ALL_CANDIDATES\n\n")
.append("- Primary under-stemming pairs: ").append(primary.underErrorPairs()).append(" / ").append(primary.underPossiblePairs()).append("\n")
.append("- ANY_CANDIDATE under-stemming pairs: ").append(any.underErrorPairs()).append(" / ").append(any.underPossiblePairs()).append("\n")
.append("- ALL_CANDIDATES under-stemming pairs: ").append(candidate.underErrorPairs()).append(" / ").append(candidate.underPossiblePairs()).append("\n")
.append("- Under-stemming pairs repaired by alternatives: ").append(primary.underErrorPairs() - candidate.underErrorPairs()).append("\n")
.append("- Primary over-stemming pairs: ").append(primary.overErrorPairs()).append(" / ").append(primary.overPossiblePairs()).append("\n")
.append("- ANY_CANDIDATE over-stemming pairs: ").append(any.overErrorPairs()).append(" / ").append(any.overPossiblePairs()).append("\n")
.append("- Best-case over-stemming pairs avoided: ").append(primary.overErrorPairs() - any.overErrorPairs()).append("\n")
.append("- ALL_CANDIDATES over-stemming pairs: ").append(candidate.overErrorPairs()).append(" / ").append(candidate.overPossiblePairs()).append("\n")
.append("- Additional candidate collision pairs: ").append(candidate.overErrorPairs() - primary.overErrorPairs()).append("\n")
.append("- Forms with multiple candidates: ").append(candidate.formsWithMultipleCandidates()).append("\n")
.append("- Maximum candidates for one word: ").append(candidate.maximumCandidatesForOneWord()).append("\n\n")
.append("- Candidate-count distribution: ").append(new java.util.TreeMap<>(scenario.candidateCountDistribution())).append("\n\n")
.append("### Forms with the largest candidate sets\n\n");
for (Word word : scenario.words()) {
text.append("- Row ").append(word.row()).append(", form `").append(escape(word.form()))
.append("`, primary `").append(escape(word.primary())).append("`, candidates ")
.append(word.candidates().stream().map(value -> "`" + escape(value) + "`").toList())
.append(", repaired same-group relations ").append(word.repairedUnderRelations())
.append(", introduced cross-group relations ").append(word.introducedOverRelations()).append(".\n");
}
text.append('\n');
}
Files.writeString(path, text.toString(), StandardCharsets.UTF_8, StandardOpenOption.APPEND);
}
/** Escapes Markdown code-span delimiters without altering linguistic content. */
private static String escape(final String value) { return value.replace("`", "\\`"); }
/** Immutable candidate-aware audit scenario. */
record Scenario(QualityResult primary, QualityResult any, QualityResult candidate,
Map<Integer, Long> candidateCountDistribution,
List<Word> words) { }
/** Immutable word-level candidate diagnostic. */
record Word(int row, String form, String primary, List<String> candidates,
long repairedUnderRelations, long introducedOverRelations) { }
}

View File

@@ -0,0 +1,35 @@
package org.egothor.stemmer.benchmark.quality;
import java.util.LinkedHashSet;
import java.util.List;
import java.util.Objects;
import java.util.Set;
/** Immutable gold-standard equivalence class originating from one dictionary row. */
public record GoldStandardGroup(int rowNumber, List<String> forms) {
private static final int FIRST_ROW_NUMBER = 1;
/**
* Creates a group while removing exact duplicates within this row.
*
* @param rowNumber positive physical dictionary row number
* @param forms supplied forms; encounter order has no metric significance
* @throws IllegalArgumentException if the row or forms are invalid
*/
public GoldStandardGroup {
if (rowNumber < FIRST_ROW_NUMBER) {
throw new IllegalArgumentException("Dictionary row number must be positive.");
}
Objects.requireNonNull(forms, "forms");
final Set<String> distinct = new LinkedHashSet<>();
for (String form : forms) {
if (form == null || form.isEmpty()) {
throw new IllegalArgumentException("Dictionary group forms must be non-empty strings.");
}
distinct.add(form);
}
if (distinct.isEmpty()) {
throw new IllegalArgumentException("A dictionary group must contain at least one usable form.");
}
forms = List.copyOf(distinct);
}
}

View File

@@ -0,0 +1,53 @@
package org.egothor.stemmer.benchmark.quality;
import java.io.IOException;
import java.nio.file.Files;
import java.nio.file.Path;
import java.util.ArrayList;
import java.util.EnumMap;
import java.util.HashMap;
import java.util.List;
import java.util.Map;
import java.util.Set;
import java.util.TreeSet;
import org.egothor.stemmer.StemmerPatchTrieLoader.Language;
/** Reconciles bundled dictionary resources with every production language enumeration value. */
record LanguageUniverse(Map<Language, Path> dictionaries, List<String> resourceDirectories,
List<String> enumerationValues) {
/** Discovers and validates a one-to-one resource mapping without silent exclusions. */
static LanguageUniverse discover(final Path resourcesDirectory) throws IOException {
final Map<String, Path> resources = new HashMap<>();
try (java.util.stream.Stream<Path> paths = Files.list(resourcesDirectory)) {
for (Path directory : paths.filter(Files::isDirectory).toList()) {
final Path dictionary = directory.resolve("stemmer.gz");
if (Files.isRegularFile(dictionary)) {
final Path previous = resources.put(directory.getFileName().toString(), dictionary);
if (previous != null) { throw new IOException("Two dictionary resources map to directory " + directory + "."); }
}
}
}
final Map<Language, Path> mappings = new EnumMap<>(Language.class);
final Set<String> mappedDirectories = new TreeSet<>();
for (Language language : Language.values()) {
final Path dictionary = resources.get(language.resourceDirectory());
if (dictionary == null) {
throw new IOException("Enumeration language " + language + " has no stemmer.gz dictionary under "
+ resourcesDirectory + ".");
}
mappings.put(language, dictionary);
mappedDirectories.add(language.resourceDirectory());
}
final Set<String> unmatched = new TreeSet<>(resources.keySet());
unmatched.removeAll(mappedDirectories);
if (!unmatched.isEmpty()) {
throw new IOException("Dictionary resource directories have no StemmerPatchTrieLoader.Language mapping: "
+ unmatched + ".");
}
final List<String> enumValues = new ArrayList<>();
for (Language language : Language.values()) { enumValues.add(language.name()); }
return new LanguageUniverse(Map.copyOf(mappings), List.copyOf(new TreeSet<>(resources.keySet())),
List.copyOf(enumValues));
}
}

View File

@@ -0,0 +1,55 @@
package org.egothor.stemmer.benchmark.quality;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertThrows;
import static org.junit.jupiter.api.Assertions.assertTrue;
import java.io.IOException;
import java.nio.file.Files;
import java.nio.file.Path;
import org.egothor.stemmer.StemmerPatchTrieLoader.Language;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Tag;
import org.junit.jupiter.api.Test;
import org.junit.jupiter.api.io.TempDir;
/** Regression tests for independent dictionary-resource and enumeration reconciliation. */
@Tag("integration")
@DisplayName("Authoritative Radixor language universe")
final class LanguageUniverseTest {
/** Temporary resource tree. */ @TempDir Path temporaryDirectory;
/** Verifies every production enumeration value has exactly one bundled dictionary. */
@Test @DisplayName("Production resources reconcile with every language enumeration value")
void productionResourcesReconcile() throws IOException {
final LanguageUniverse universe = LanguageUniverse.discover(Path.of("src/main/resources"));
assertEquals(Language.values().length, universe.dictionaries().size());
assertTrue(universe.dictionaries().containsKey(Language.DA_DK));
assertTrue(universe.dictionaries().containsKey(Language.YI));
}
/** Verifies a missing enumerated resource produces an exact diagnostic. */
@Test @DisplayName("Missing enumeration resources fail validation")
void missingResourceFails() throws IOException {
final Path first = this.temporaryDirectory.resolve(Language.CS_CZ.resourceDirectory());
Files.createDirectories(first); Files.createFile(first.resolve("stemmer.gz"));
final IOException exception = assertThrows(IOException.class,
() -> LanguageUniverse.discover(this.temporaryDirectory));
assertTrue(exception.getMessage().contains("DA_DK"));
}
/** Verifies an unenumerated dictionary directory is rejected. */
@Test @DisplayName("Unmapped dictionary directories fail validation")
void extraResourceFails() throws IOException {
for (Language language : Language.values()) {
final Path directory = this.temporaryDirectory.resolve(language.resourceDirectory());
Files.createDirectories(directory); Files.createFile(directory.resolve("stemmer.gz"));
}
final Path extra = this.temporaryDirectory.resolve("unmapped_language");
Files.createDirectories(extra); Files.createFile(extra.resolve("stemmer.gz"));
final IOException exception = assertThrows(IOException.class,
() -> LanguageUniverse.discover(this.temporaryDirectory));
assertTrue(exception.getMessage().contains("unmapped_language"));
}
}

View File

@@ -0,0 +1,108 @@
package org.egothor.stemmer.benchmark.quality;
import java.io.IOException;
import java.nio.charset.StandardCharsets;
import java.nio.file.Files;
import java.nio.file.Path;
import java.util.ArrayList;
import java.util.Comparator;
import java.util.LinkedHashMap;
import java.util.List;
import java.util.Map;
import java.util.OptionalDouble;
import java.util.function.Function;
/** Writes deterministic Pearson and tied-rank Spearman correlations within compatible cohorts. */
final class MetricCorrelationWriter {
/** Stable metric extractors used for correlation analysis. */
private static final Map<String, Function<QualityResult, OptionalDouble>> METRICS = metrics();
/** Utility class. */
private MetricCorrelationWriter() { throw new AssertionError("No instances."); }
/** Writes both correlation reports from unrounded per-language scenario values. */
static void write(final Path pearson, final Path spearman, final List<QualityResult> results) throws IOException {
writeOne(pearson, results, false); writeOne(spearman, results, true);
}
/** Writes one correlation method with explicit missing-value reasons. */
private static void writeOne(final Path path, final List<QualityResult> results, final boolean ranks) throws IOException {
final StringBuilder output = new StringBuilder("Aggregation,Dictionary mode,Output policy,Metric A,Metric B,Observation count,Correlation,Missing-value reason\n");
for (ProcessingMode mode : ProcessingMode.values()) {
for (OutputPolicy policy : OutputPolicy.values()) {
final List<QualityResult> cohort = results.stream().filter(row -> row.processingMode() == mode
&& row.outputPolicy() == policy).toList();
final List<String> names = new ArrayList<>(METRICS.keySet());
if (policy != OutputPolicy.PRIMARY_OUTPUT) { names.remove("Adjusted Rand Index"); }
for (int left = 0; left < names.size(); left++) {
for (int right = left; right < names.size(); right++) {
append(output, mode, policy, names.get(left), names.get(right), cohort, ranks);
}
}
}
}
final Path parent = path.toAbsolutePath().getParent(); if (parent != null) { Files.createDirectories(parent); }
Files.writeString(path, output.toString(), StandardCharsets.UTF_8);
}
/** Appends one coefficient after pairwise removal of undefined observations. */
private static void append(final StringBuilder output, final ProcessingMode mode, final OutputPolicy policy,
final String leftName, final String rightName, final List<QualityResult> cohort, final boolean ranks) {
final List<Double> left = new ArrayList<>(); final List<Double> right = new ArrayList<>();
for (QualityResult row : cohort) {
final OptionalDouble a = METRICS.get(leftName).apply(row); final OptionalDouble b = METRICS.get(rightName).apply(row);
if (a.isPresent() && b.isPresent()) { left.add(a.getAsDouble()); right.add(b.getAsDouble()); }
}
String value = ""; String reason = "";
if (left.size() < 3) { reason = "Fewer than three defined observations."; }
else {
final double[] a = ranks ? ranks(left) : values(left); final double[] b = ranks ? ranks(right) : values(right);
final OptionalDouble correlation = pearson(a, b);
if (correlation.isEmpty()) { reason = "At least one metric has zero variance."; }
else { value = String.format(java.util.Locale.ROOT, "%.12f", correlation.getAsDouble()); }
}
output.append("Per-language scenario,").append(mode).append(',').append(policy).append(',')
.append(csv(leftName)).append(',').append(csv(rightName)).append(',').append(left.size()).append(',')
.append(value).append(',').append(csv(reason)).append('\n');
}
/** Calculates Pearson correlation with an empty result for zero variance. */
private static OptionalDouble pearson(final double[] left, final double[] right) {
double leftMean = 0.0; double rightMean = 0.0;
for (int index = 0; index < left.length; index++) { leftMean += left[index]; rightMean += right[index]; }
leftMean /= left.length; rightMean /= right.length;
double covariance = 0.0; double leftVariance = 0.0; double rightVariance = 0.0;
for (int index = 0; index < left.length; index++) {
final double a = left[index] - leftMean; final double b = right[index] - rightMean;
covariance += a * b; leftVariance += a * a; rightVariance += b * b;
}
return leftVariance == 0.0 || rightVariance == 0.0 ? OptionalDouble.empty()
: OptionalDouble.of(covariance / Math.sqrt(leftVariance * rightVariance));
}
/** Assigns deterministic average ranks to tied values. */
private static double[] ranks(final List<Double> input) {
final List<Integer> order = new ArrayList<>(); for (int index = 0; index < input.size(); index++) { order.add(index); }
order.sort(Comparator.comparingDouble(input::get)); final double[] ranks = new double[input.size()];
int start = 0; while (start < order.size()) {
int end = start + 1; while (end < order.size() && input.get(order.get(start)).equals(input.get(order.get(end)))) { end++; }
final double rank = (start + 1 + end) / 2.0; for (int index = start; index < end; index++) { ranks[order.get(index)] = rank; }
start = end;
}
return ranks;
}
/** Copies boxed values into a primitive array. */
private static double[] values(final List<Double> values) { final double[] result = new double[values.size()]; for (int index = 0; index < result.length; index++) { result[index] = values.get(index); } return result; }
/** Defines stable metric names and unrounded extractors. */
private static Map<String, Function<QualityResult, OptionalDouble>> metrics() {
final Map<String, Function<QualityResult, OptionalDouble>> values = new LinkedHashMap<>();
values.put("Pairwise F0.5", row -> row.pairwiseMetrics().f05()); values.put("Pairwise F1", row -> row.pairwiseMetrics().f1());
values.put("Pairwise F2", row -> row.pairwiseMetrics().f2()); values.put("Jaccard", row -> row.pairwiseMetrics().jaccard());
values.put("Fowlkes-Mallows", row -> row.pairwiseMetrics().fowlkesMallows());
values.put("Matthews correlation coefficient", row -> row.pairwiseMetrics().matthewsCorrelationCoefficient());
values.put("Balanced accuracy", row -> row.pairwiseMetrics().balancedAccuracy());
values.put("Adjusted Rand Index", row -> row.partitionMetrics() == null ? OptionalDouble.empty() : OptionalDouble.of(row.partitionMetrics().adjustedRandIndex()));
return java.util.Collections.unmodifiableMap(values);
}
/** Quotes one CSV field. */
private static String csv(final String value) { return '"' + value.replace("\"", "\"\"") + '"'; }
}

View File

@@ -0,0 +1,11 @@
package org.egothor.stemmer.benchmark.quality;
/** Defines which outputs of a JMH stemmer adapter establish the measured relation. */
enum OutputPolicy {
/** Uses only the deterministic output selected by the existing JMH comparison. */
PRIMARY_OUTPUT,
/** Uses an optimistic pair-specific choice from the complete candidate sets. */
ANY_CANDIDATE,
/** Treats all candidates as active and uses the complete intersection relation. */
ALL_CANDIDATES
}

View File

@@ -0,0 +1,68 @@
package org.egothor.stemmer.benchmark.quality;
import java.util.OptionalDouble;
/**
* Derives scientifically labelled pairwise confusion metrics from unrounded raw counts.
* Undefined ratios are represented by empty optionals; no method returns NaN or infinity.
*/
record PairwiseMetrics(long truePositivePairs, long falsePositivePairs, long falseNegativePairs,
long trueNegativePairs) {
/** Creates checked confusion counts from one quality result. */
static PairwiseMetrics from(final QualityResult result) {
return new PairwiseMetrics(Math.subtractExact(result.underPossiblePairs(), result.underErrorPairs()),
result.overErrorPairs(), result.underErrorPairs(),
Math.subtractExact(result.overPossiblePairs(), result.overErrorPairs()));
}
/** @return pairwise precision */ OptionalDouble precision() { return ratio(truePositivePairs, Math.addExact(truePositivePairs, falsePositivePairs)); }
/** @return pairwise recall */ OptionalDouble recall() { return ratio(truePositivePairs, Math.addExact(truePositivePairs, falseNegativePairs)); }
/** @return pairwise specificity */ OptionalDouble specificity() { return ratio(trueNegativePairs, Math.addExact(trueNegativePairs, falsePositivePairs)); }
/** @return pairwise accuracy, potentially dominated by true negatives */
OptionalDouble accuracy() { return ratio(Math.addExact(truePositivePairs, trueNegativePairs), total()); }
/** @return arithmetic mean of recall and specificity */
OptionalDouble balancedAccuracy() { return mean(recall(), specificity()); }
/** @return pairwise F0.5 */ OptionalDouble f05() { return fBeta(0.25); }
/** @return pairwise F1 */ OptionalDouble f1() { return fBeta(1.0); }
/** @return pairwise F2 */ OptionalDouble f2() { return fBeta(4.0); }
/** @return Jaccard index */
OptionalDouble jaccard() { return ratio(truePositivePairs, Math.addExact(Math.addExact(truePositivePairs, falsePositivePairs), falseNegativePairs)); }
/** @return Fowlkes-Mallows index */
OptionalDouble fowlkesMallows() {
final OptionalDouble precisionValue = precision(); final OptionalDouble recallValue = recall();
return precisionValue.isEmpty() || recallValue.isEmpty() ? OptionalDouble.empty()
: OptionalDouble.of(Math.sqrt(precisionValue.getAsDouble() * recallValue.getAsDouble()));
}
/** @return Matthews correlation coefficient using scaled double arithmetic */
OptionalDouble matthewsCorrelationCoefficient() {
final double a = (double) truePositivePairs + falsePositivePairs;
final double b = (double) truePositivePairs + falseNegativePairs;
final double c = (double) trueNegativePairs + falsePositivePairs;
final double d = (double) trueNegativePairs + falseNegativePairs;
final double denominator = Math.sqrt(a * b * c * d);
if (denominator == 0.0) { return OptionalDouble.empty(); }
final double numerator = (double) truePositivePairs * trueNegativePairs
- (double) falsePositivePairs * falseNegativePairs;
return OptionalDouble.of(numerator / denominator);
}
/** @return pairwise error rate */
OptionalDouble errorRate() { return ratio(Math.addExact(falsePositivePairs, falseNegativePairs), total()); }
/** Calculates F-beta directly from raw counts. */
private OptionalDouble fBeta(final double betaSquared) {
final double numerator = (1.0 + betaSquared) * truePositivePairs;
final double denominator = numerator + betaSquared * falseNegativePairs + falsePositivePairs;
return denominator == 0.0 ? OptionalDouble.empty() : OptionalDouble.of(numerator / denominator);
}
/** Returns the checked total pair population. */
private long total() { return Math.addExact(Math.addExact(truePositivePairs, falsePositivePairs), Math.addExact(falseNegativePairs, trueNegativePairs)); }
/** Calculates one ratio with explicit zero-denominator handling. */
private static OptionalDouble ratio(final long numerator, final long denominator) {
return denominator == 0 ? OptionalDouble.empty() : OptionalDouble.of((double) numerator / denominator);
}
/** Averages two defined ratios. */
private static OptionalDouble mean(final OptionalDouble left, final OptionalDouble right) {
return left.isEmpty() || right.isEmpty() ? OptionalDouble.empty()
: OptionalDouble.of((left.getAsDouble() + right.getAsDouble()) / 2.0);
}
}

View File

@@ -0,0 +1,43 @@
package org.egothor.stemmer.benchmark.quality;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertTrue;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Tag;
import org.junit.jupiter.api.Test;
/** Formula and degenerate-case tests for aggregate pairwise metrics. */
@Tag("unit")
@DisplayName("Pairwise aggregate metrics")
final class PairwiseMetricsTest {
/** Verifies all formulas use the supplied raw confusion counts. */
@Test @DisplayName("Aggregate metrics are calculated from raw confusion counts")
void formulas() {
final PairwiseMetrics metrics = new PairwiseMetrics(8, 2, 4, 16);
assertEquals(0.8, metrics.precision().orElseThrow(), 1.0e-12);
assertEquals(8.0 / 12.0, metrics.recall().orElseThrow(), 1.0e-12);
assertEquals(16.0 / 18.0, metrics.specificity().orElseThrow(), 1.0e-12);
assertEquals(24.0 / 30.0, metrics.accuracy().orElseThrow(), 1.0e-12);
assertEquals(8.0 / 14.0, metrics.jaccard().orElseThrow(), 1.0e-12);
assertEquals(6.0 / 30.0, metrics.errorRate().orElseThrow(), 1.0e-12);
assertTrue(metrics.f05().orElseThrow() > metrics.f2().orElseThrow());
}
/** Verifies a perfect nondegenerate relation reaches every applicable maximum. */
@Test @DisplayName("Perfect confusion counts produce maximum defined scores")
void perfect() {
final PairwiseMetrics metrics = new PairwiseMetrics(10, 0, 0, 20);
assertEquals(1.0, metrics.f05().orElseThrow()); assertEquals(1.0, metrics.f1().orElseThrow());
assertEquals(1.0, metrics.f2().orElseThrow()); assertEquals(1.0, metrics.matthewsCorrelationCoefficient().orElseThrow());
assertEquals(1.0, metrics.balancedAccuracy().orElseThrow());
}
/** Verifies undefined denominators remain explicit missing values. */
@Test @DisplayName("Degenerate zero denominators remain undefined")
void undefined() {
final PairwiseMetrics metrics = new PairwiseMetrics(0, 0, 0, 0);
assertTrue(metrics.precision().isEmpty()); assertTrue(metrics.recall().isEmpty());
assertTrue(metrics.matthewsCorrelationCoefficient().isEmpty());
}
}

View File

@@ -0,0 +1,8 @@
package org.egothor.stemmer.benchmark.quality;
/**
* Immutable strict-partition comparison metrics. Values use the arithmetic-mean
* normalization for normalized mutual information and are applicable only to primary output.
*/
record PartitionMetrics(double adjustedRandIndex, double homogeneity, double completeness,
double vMeasure, double normalizedMutualInformation) { }

View File

@@ -0,0 +1,32 @@
package org.egothor.stemmer.benchmark.quality;
/** Selects the gold-standard groups included in a stemming-quality scenario. */
public enum ProcessingMode {
/** Includes every parsed dictionary group. */
ALL_WORDS,
/** Includes only groups containing no uppercase or titlecase Unicode code point. */
LOWERCASE_GROUPS_ONLY;
/**
* Tests whether a group is eligible for this mode.
*
* @param forms distinct word forms in the group, never {@code null}
* @return {@code true} when the complete group is eligible
*/
public boolean includes(final Iterable<String> forms) {
if (this == ALL_WORDS) {
return true;
}
for (String form : forms) {
int offset = 0;
while (offset < form.length()) {
final int codePoint = form.codePointAt(offset);
if (Character.isUpperCase(codePoint) || Character.isTitleCase(codePoint)) {
return false;
}
offset += Character.charCount(codePoint);
}
}
return true;
}
}

View File

@@ -0,0 +1,157 @@
package org.egothor.stemmer.benchmark.quality;
import java.io.IOException;
import java.nio.charset.StandardCharsets;
import java.nio.file.Files;
import java.nio.file.Path;
import java.util.ArrayList;
import java.util.Comparator;
import java.util.LinkedHashMap;
import java.util.List;
import java.util.Locale;
import java.util.Map;
import org.egothor.stemmer.benchmark.QualityStemmerMatrix.Candidate;
/** Produces deterministic scenario and group-contribution diagnostics for audit runs. */
final class QualityAudit {
/** Utility class. */
private QualityAudit() {
throw new AssertionError("No instances.");
}
/**
* Evaluates one scenario and retains its highest under-stemming contributors.
*
* @param candidate authoritative JMH candidate
* @param mode processing mode
* @param groups parsed dictionary groups
* @param limit maximum listed contributors
* @return immutable audited scenario
* @throws IOException if the candidate adapter fails
*/
static Scenario evaluate(final Candidate candidate, final ProcessingMode mode,
final List<GoldStandardGroup> groups, final int limit) throws IOException {
final List<GoldStandardGroup> includedGroups = groups.stream().filter(group -> mode.includes(group.forms())).toList();
final List<String> forms = new ArrayList<>();
for (GoldStandardGroup group : includedGroups) {
forms.addAll(group.forms());
}
final String[] outputs = candidate.createStemmer().stem(forms.toArray(String[]::new));
if (outputs.length != forms.size()) {
throw new IOException("Invalid audit output count for stemmer " + candidate.name() + ", language "
+ candidate.language() + ", and processing mode " + mode + ".");
}
final int[] outputIndex = {0};
final QualityResult result = QualityEvaluator.evaluate(candidate.name(), candidate.language().name(), mode,
groups, word -> outputs[outputIndex[0]++]);
final List<Contributor> contributors = new ArrayList<>();
long exactMatches = 0;
int offset = 0;
final List<Integer> sizes = new ArrayList<>();
for (GoldStandardGroup group : includedGroups) {
final Map<String, List<String>> formsByStem = new LinkedHashMap<>();
final String expected = group.forms().get(0);
long mergedPairs = 0;
for (String form : group.forms()) {
final String output = outputs[offset++];
formsByStem.computeIfAbsent(output, ignored -> new ArrayList<>()).add(form);
if (expected.equals(output)) {
exactMatches++;
}
}
for (List<String> stemForms : formsByStem.values()) {
mergedPairs = Math.addExact(mergedPairs, QualityEvaluator.chooseTwo(stemForms.size()));
}
final long possible = QualityEvaluator.chooseTwo(group.forms().size());
final long errors = Math.subtractExact(possible, mergedPairs);
sizes.add(group.forms().size());
if (errors > 0) {
contributors.add(new Contributor(group.rowNumber(), group.forms().size(), formsByStem, errors, possible));
}
}
contributors.sort(Comparator.comparingLong(Contributor::errorPairs).reversed()
.thenComparingInt(Contributor::rowNumber));
final long contributionSum = contributors.stream().mapToLong(Contributor::errorPairs).reduce(0L, Math::addExact);
if (contributionSum != result.underErrorPairs()) {
throw new IOException("The summed group contributions do not equal the optimized under-stemming total for "
+ candidate.name() + ", " + candidate.language() + ", and " + mode + ".");
}
sizes.sort(Integer::compareTo);
final double mean = sizes.stream().mapToInt(Integer::intValue).average().orElse(0.0);
final double median = median(sizes);
return new Scenario(result, candidate.language().resourcePath(), exactMatches, forms.size(),
sizes.isEmpty() ? 0 : sizes.get(0), sizes.isEmpty() ? 0 : sizes.get(sizes.size() - 1), mean, median,
List.copyOf(contributors.subList(0, Math.min(limit, contributors.size()))), contributionSum);
}
/** Writes all audited scenarios to a fresh UTF-8 Markdown file. */
static void write(final Path path, final List<Scenario> scenarios) throws IOException {
final StringBuilder text = new StringBuilder(8192);
text.append("# Stemming-quality audit\n\nThis report uses original dictionary forms and the exact JMH candidate adapters. Exact-output counts compare outputs with the first parsed field of each group; they are not interchangeable with the existing JMH exact-root counters when that corpus lowercases dictionary fields.\n\n");
for (Scenario scenario : scenarios.stream().sorted(Comparator.comparing(item -> item.result(), QualityResult.ORDER)).toList()) {
final QualityResult result = scenario.result();
text.append("## ").append(result.stemmer()).append(" / ").append(result.language()).append(" / ")
.append(result.processingMode()).append("\n\n")
.append("- Dictionary source: `").append(scenario.dictionarySource()).append("`\n")
.append("- Processed dictionary rows: ").append(result.appliedDictionaryRows()).append("\n")
.append("- Processed unique word forms: ").append(result.processedWordForms()).append("\n")
.append("- Singleton dictionary rows: ").append(result.singletonDictionaryRows()).append("\n")
.append("- Dictionary rows contributing under-stemming pairs: ").append(result.dictionaryRowsContributingUnderPairs()).append("\n")
.append("- Group size minimum / maximum / mean / median: ").append(scenario.minimumGroupSize()).append(" / ")
.append(scenario.maximumGroupSize()).append(" / ").append(String.format(Locale.ROOT, "%.6f", scenario.meanGroupSize()))
.append(" / ").append(String.format(Locale.ROOT, "%.6f", scenario.medianGroupSize())).append("\n")
.append("- Exact first-field matches: ").append(scenario.exactMatches()).append(" / ").append(scenario.exactDenominator()).append("\n")
.append("- Under-stemming pairs: ").append(result.underErrorPairs()).append(" / ").append(result.underPossiblePairs()).append("\n")
.append("- Over-stemming pairs: ").append(result.overErrorPairs()).append(" / ").append(result.overPossiblePairs()).append("\n")
.append("- Independently summed under-stemming contributions: ").append(scenario.contributionSum()).append("\n\n")
.append("### Highest under-stemming contributors\n\n");
for (Contributor contributor : scenario.contributors()) {
text.append("#### Dictionary row ").append(contributor.rowNumber()).append("\n\n")
.append("Unique forms: ").append(contributor.groupSize()).append("; distinct predicted stems: ")
.append(contributor.formsByStem().size()).append("; contribution: ").append(contributor.errorPairs())
.append(" / ").append(contributor.possiblePairs()).append(" pairs.\n\n");
for (Map.Entry<String, List<String>> entry : contributor.formsByStem().entrySet()) {
text.append("- Predicted stem `").append(escape(entry.getKey())).append("` (").append(entry.getValue().size())
.append("): ").append(entry.getValue().stream().map(QualityAudit::quoted).toList()).append("\n");
}
text.append('\n');
}
}
final Path parent = path.toAbsolutePath().getParent();
if (parent != null) {
Files.createDirectories(parent);
}
Files.writeString(path, text.toString(), StandardCharsets.UTF_8);
}
/** Calculates the conventional median of a sorted integer list. */
private static double median(final List<Integer> sorted) {
if (sorted.isEmpty()) {
return 0.0;
}
final int middle = sorted.size() / 2;
return sorted.size() % 2 == 0 ? (sorted.get(middle - 1) + sorted.get(middle)) / 2.0 : sorted.get(middle);
}
/** Escapes Markdown code-span delimiters. */
private static String escape(final String value) {
return value.replace("`", "\\`");
}
/** Quotes one original dictionary form for Markdown diagnostics. */
private static String quoted(final String value) {
return "`" + escape(value) + "`";
}
/** Immutable complete audit summary for one scenario. */
record Scenario(QualityResult result, String dictionarySource, long exactMatches, long exactDenominator,
int minimumGroupSize, int maximumGroupSize, double meanGroupSize, double medianGroupSize,
List<Contributor> contributors, long contributionSum) {
}
/** Immutable contribution of one gold-standard group. */
record Contributor(int rowNumber, int groupSize, Map<String, List<String>> formsByStem,
long errorPairs, long possiblePairs) {
}
}

View File

@@ -0,0 +1,197 @@
package org.egothor.stemmer.benchmark.quality;
import java.io.IOException;
import java.util.HashMap;
import java.util.HashSet;
import java.util.List;
import java.util.Map;
import java.util.Objects;
import java.util.Set;
import java.util.ArrayList;
import org.egothor.stemmer.benchmark.QualityStemmerMatrix.BatchStemmer;
/** Evaluates pairwise partition agreement using aggregated frequencies, never explicit pairs. */
public final class QualityEvaluator {
/** Utility class. */
private QualityEvaluator() { throw new AssertionError("No instances."); }
/**
* Evaluates one scenario in time proportional to forms and group-to-stem associations.
* All combinatorial arithmetic is checked and overflow is reported.
*
* @param stemmerName stable stemmer name
* @param language stable language identifier
* @param mode processing mode
* @param groups parsed gold-standard groups
* @param stemmer stemmer implementation
* @return immutable metric result
*/
public static QualityResult evaluate(final String stemmerName, final String language, final ProcessingMode mode,
final Iterable<GoldStandardGroup> groups, final StemmerFunction stemmer) {
Objects.requireNonNull(groups, "groups");
Objects.requireNonNull(stemmer, "stemmer");
final Map<String, Long> global = new HashMap<>();
long rows = 0;
long words = 0;
long singletonRows = 0;
long pairRows = 0;
long underPossible = 0;
long withinSameStem = 0;
final Set<String> stems = new HashSet<>();
final Map<String, Long> local = new HashMap<>();
final List<Map<String, Long>> contingency = new ArrayList<>();
final List<Long> groupSizes = new ArrayList<>();
for (GoldStandardGroup group : groups) {
final List<String> forms = group.forms();
if (!mode.includes(forms)) {
continue;
}
rows = add(rows, 1, "applied dictionary rows");
words = add(words, forms.size(), "processed word forms");
if (forms.size() == 1) {
singletonRows = add(singletonRows, 1, "singleton dictionary rows");
} else {
pairRows = add(pairRows, 1, "dictionary rows contributing under-stemming pairs");
}
underPossible = add(underPossible, chooseTwo(forms.size()), "under-stemming possible pairs");
local.clear();
for (String form : forms) {
final String output;
try {
output = stemmer.stem(form);
} catch (IOException exception) {
throw failure(stemmerName, language, mode, group.rowNumber(), form,
"the stemmer threw an exception", exception);
}
if (output == null) {
throw failure(stemmerName, language, mode, group.rowNumber(), form,
"the stemmer returned null", null);
}
local.merge(output, 1L, (left, right) -> add(left, right, "group-to-stem frequency"));
global.merge(output, 1L, (left, right) -> add(left, right, "global stem frequency"));
stems.add(output);
}
for (long frequency : local.values()) {
withinSameStem = add(withinSameStem, chooseTwo(frequency), "within-group merged pairs");
}
contingency.add(Map.copyOf(local));
groupSizes.add((long) forms.size());
}
long allPairs = chooseTwo(words);
long overPossible = subtract(allPairs, underPossible, "over-stemming possible pairs");
long allSameStem = 0;
for (long frequency : global.values()) {
allSameStem = add(allSameStem, chooseTwo(frequency), "same-stem pairs");
}
final long underError = subtract(underPossible, withinSameStem, "under-stemming error pairs");
final long overError = subtract(allSameStem, withinSameStem, "over-stemming error pairs");
final PartitionMetrics partition = partitionMetrics(words, underPossible, allSameStem,
withinSameStem, groupSizes, global, contingency);
return new QualityResult(stemmerName, language, mode, OutputPolicy.PRIMARY_OUTPUT,
rows, words, singletonRows, pairRows, words, 0, words == 0 ? 0 : 1, words, stems.size(), overError, overPossible,
underError, underPossible, partition);
}
/** Calculates strict-partition metrics from the exact contingency table. */
private static PartitionMetrics partitionMetrics(final long words, final long rowPairs, final long columnPairs,
final long indexPairs, final List<Long> groupSizes, final Map<String, Long> global,
final List<Map<String, Long>> contingency) {
if (words == 0) { return new PartitionMetrics(0.0, 0.0, 0.0, 0.0, 0.0); }
final double totalPairs = chooseTwo(words);
final double expected = totalPairs == 0.0 ? 0.0 : (double) rowPairs * columnPairs / totalPairs;
final double maximum = (rowPairs + (double) columnPairs) / 2.0;
final double adjustedRand = maximum == expected ? 1.0 : (indexPairs - expected) / (maximum - expected);
final double goldEntropy = entropy(words, groupSizes);
final double predictedEntropy = entropy(words, global.values());
double mutualInformation = 0.0;
for (int group = 0; group < contingency.size(); group++) {
final long groupSize = groupSizes.get(group);
for (Map.Entry<String, Long> cell : contingency.get(group).entrySet()) {
final double frequency = cell.getValue();
mutualInformation += frequency / words * Math.log(frequency * words
/ (groupSize * (double) global.get(cell.getKey())));
}
}
final double homogeneity = goldEntropy == 0.0 ? 1.0 : mutualInformation / goldEntropy;
final double completeness = predictedEntropy == 0.0 ? 1.0 : mutualInformation / predictedEntropy;
final double vMeasure = homogeneity + completeness == 0.0 ? 0.0
: 2.0 * homogeneity * completeness / (homogeneity + completeness);
final double nmiDenominator = (goldEntropy + predictedEntropy) / 2.0;
final double nmi = nmiDenominator == 0.0 ? 1.0 : mutualInformation / nmiDenominator;
return new PartitionMetrics(adjustedRand, homogeneity, completeness, vMeasure, nmi);
}
/** Calculates natural-log entropy from category frequencies. */
private static double entropy(final long total, final Iterable<Long> frequencies) {
double weightedLogs = 0.0;
for (long frequency : frequencies) { weightedLogs += frequency * Math.log(frequency); }
return Math.log(total) - weightedLogs / total;
}
/**
* Evaluates one scenario through an authoritative JMH batch adapter.
* The temporary input and output arrays are required to preserve TokenStream
* lifecycle and preprocessing semantics used by the JMH comparison.
*
* @param stemmerName stable JMH candidate name
* @param language registered dictionary language
* @param mode processing mode
* @param groups parsed gold-standard groups
* @param stemmer scenario-confined batch adapter
* @return immutable pairwise result
* @throws IOException when the benchmark adapter fails
*/
public static QualityResult evaluateBatch(final String stemmerName, final String language,
final ProcessingMode mode, final List<GoldStandardGroup> groups, final BatchStemmer stemmer)
throws IOException {
final List<String> included = new ArrayList<>();
for (GoldStandardGroup group : groups) {
if (mode.includes(group.forms())) {
included.addAll(group.forms());
}
}
final String[] outputs = stemmer.stem(included.toArray(String[]::new));
if (outputs == null || outputs.length != included.size()) {
throw new IOException("JMH stemmer " + stemmerName + " returned an invalid output batch for language "
+ language + " and processing mode " + mode + ".");
}
final int[] index = {0};
return evaluate(stemmerName, language, mode, groups, word -> {
final String output = outputs[index[0]++];
if (output == null) {
throw new IOException("JMH stemmer " + stemmerName + " returned null for language " + language
+ ", processing mode " + mode + ", and word form '" + word + "'.");
}
return output;
});
}
/** Calculates C2(n) with checked arithmetic. */
/* default */ static long chooseTwo(final long value) {
if (value < 0) { throw new IllegalArgumentException("Pair population must not be negative."); }
try {
return value % 2 == 0 ? Math.multiplyExact(value / 2, value - 1)
: Math.multiplyExact(value, (value - 1) / 2);
} catch (ArithmeticException exception) {
throw new IllegalStateException("Arithmetic overflow while calculating unordered word-form pairs.", exception);
}
}
/** Checked addition with metric context. */
private static long add(final long left, final long right, final String context) {
try { return Math.addExact(left, right); }
catch (ArithmeticException exception) { throw new IllegalStateException("Arithmetic overflow in " + context + ".", exception); }
}
/** Checked subtraction with metric context. */
private static long subtract(final long left, final long right, final String context) {
try { return Math.subtractExact(left, right); }
catch (ArithmeticException exception) { throw new IllegalStateException("Arithmetic overflow in " + context + ".", exception); }
}
/** Builds a contextual failure without producing a partial result. */
private static IllegalStateException failure(final String stemmer, final String language,
final ProcessingMode mode, final int row, final String form, final String reason, final Exception cause) {
final String message = "Quality evaluation failed for stemmer " + stemmer + ", language " + language
+ ", processing mode " + mode + ", dictionary row " + row + ", word form '" + form + "': " + reason + ".";
return cause == null ? new IllegalStateException(message) : new IllegalStateException(message, cause);
}
}

View File

@@ -0,0 +1,178 @@
package org.egothor.stemmer.benchmark.quality;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertFalse;
import static org.junit.jupiter.api.Assertions.assertThrows;
import static org.junit.jupiter.api.Assertions.assertTrue;
import java.util.List;
import java.util.Map;
import java.io.IOException;
import java.util.ArrayList;
import java.util.HashMap;
import java.util.Random;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Tag;
import org.junit.jupiter.api.Test;
/** Mathematical and filtering tests for pairwise quality evaluation. */
@Tag("unit")
@DisplayName("Pairwise stemming-quality evaluator")
final class QualityEvaluatorTest {
/** Verifies a perfect partition. */
@Test @DisplayName("A perfect predicted partition has no errors")
void perfectPartition() {
final QualityResult result = evaluate(List.of(group(1, "a", "b"), group(2, "c", "d")),
Map.of("a", "x", "b", "x", "c", "y", "d", "y"));
assertEquals(0, result.overErrorPairs()); assertEquals(4, result.overPossiblePairs());
assertEquals(0, result.underErrorPairs()); assertEquals(2, result.underPossiblePairs());
assertEquals(2, result.distinctOutputStems());
assertEquals(1.0, result.partitionMetrics().adjustedRandIndex(), 1.0e-12);
assertEquals(1.0, result.partitionMetrics().homogeneity(), 1.0e-12);
assertEquals(1.0, result.partitionMetrics().completeness(), 1.0e-12);
assertEquals(1.0, result.partitionMetrics().vMeasure(), 1.0e-12);
assertEquals(1.0, result.partitionMetrics().normalizedMutualInformation(), 1.0e-12);
}
/** Verifies partial merge and pure under-stemming pair counts. */
@Test @DisplayName("A partial within-group merge is counted by pairs")
void partialMerge() {
final QualityResult result = evaluate(List.of(group(1, "a", "b", "c"), group(2, "d")),
Map.of("a", "x", "b", "x", "c", "z", "d", "q"));
assertEquals(2, result.underErrorPairs()); assertEquals(3, result.underPossiblePairs());
assertEquals(0, result.overErrorPairs()); assertEquals(3, result.overPossiblePairs());
}
/** Verifies multi-group over-stemming combinatorics. */
@Test @DisplayName("Several gold groups colliding in one stem count every cross-group pair")
void multipleGroupsCollide() {
final QualityResult result = evaluate(List.of(group(1, "a", "b"), group(2, "c"), group(3, "d", "e", "f")),
Map.of("a", "x", "b", "x", "c", "x", "d", "x", "e", "x", "f", "x"));
assertEquals(11, result.overErrorPairs()); assertEquals(11, result.overPossiblePairs());
assertEquals(0, result.underErrorPairs());
}
/** Verifies combined split and collision counts. */
@Test @DisplayName("Combined over-stemming and under-stemming are independent")
void combinedErrors() {
final QualityResult result = evaluate(List.of(group(1, "a", "b", "c"), group(2, "d", "e")),
Map.of("a", "x", "b", "x", "c", "y", "d", "y", "e", "y"));
assertEquals(2, result.overErrorPairs()); assertEquals(6, result.overPossiblePairs());
assertEquals(2, result.underErrorPairs()); assertEquals(4, result.underPossiblePairs());
}
/** Verifies duplicate scope and singleton undefined denominator. */
@Test @DisplayName("Duplicates are removed only within a group and singleton under-stemming is undefined")
void duplicateScope() {
final QualityResult result = evaluate(List.of(group(1, "same", "same"), group(2, "same")), Map.of("same", "x"));
assertEquals(2, result.processedWordForms()); assertEquals(1, result.overErrorPairs());
assertTrue(result.underPercentage().isEmpty());
}
/** Verifies the zero over-stemming denominator. */
@Test @DisplayName("One gold group has an undefined over-stemming percentage")
void zeroOverDenominator() {
final QualityResult result = evaluate(List.of(group(1, "a", "b")), Map.of("a", "x", "b", "y"));
assertTrue(result.overPercentage().isEmpty()); assertFalse(result.underPercentage().isEmpty());
}
/** Verifies Unicode code-point filtering and uncased data. */
@Test @DisplayName("Lowercase filtering detects uppercase and titlecase code points without excluding uncased symbols")
void lowercaseFiltering() {
assertTrue(ProcessingMode.LOWERCASE_GROUPS_ONLY.includes(List.of("žluťoučký-123", "தமிழ்")));
assertFalse(ProcessingMode.LOWERCASE_GROUPS_ONLY.includes(List.of("Upper")));
assertFalse(ProcessingMode.LOWERCASE_GROUPS_ONLY.includes(List.of("Džungla")));
assertFalse(ProcessingMode.LOWERCASE_GROUPS_ONLY.includes(List.of("a\uD801\uDC00")));
}
/** Verifies contextual stemmer failures. */
@Test @DisplayName("Stemmer exceptions contain complete scenario context")
void stemmerFailure() {
final IllegalStateException exception = assertThrows(IllegalStateException.class,
() -> QualityEvaluator.evaluate("Broken", "TEST", ProcessingMode.ALL_WORDS,
List.of(group(7, "word")), word -> { throw new IOException("failure"); }));
assertTrue(exception.getMessage().contains("dictionary row 7")); assertTrue(exception.getMessage().contains("word form 'word'"));
}
/** Verifies the largest safe and first overflowing combinatorial values. */
@Test @DisplayName("Pair calculation detects arithmetic overflow")
void arithmeticBoundary() {
assertEquals(4_611_686_013_944_624_251L, QualityEvaluator.chooseTwo(3_037_000_499L));
assertThrows(IllegalStateException.class, () -> QualityEvaluator.chooseTwo(Long.MAX_VALUE));
}
/** Demonstrates the documented denominator difference from exact accuracy. */
@Test @DisplayName("Ninety-nine percent exact accuracy can coexist with sixteen percent pairwise under-stemming")
void exactAccuracyAndPairwiseRateUseDifferentDenominators() {
final List<GoldStandardGroup> groups = new ArrayList<>();
final Map<String, String> stems = new HashMap<>();
for (int index = 0; index < 88; index++) {
final String form = "singleton-" + index;
groups.add(group(index + 1, form));
stems.put(form, form);
}
final String[] largeGroup = new String[12];
for (int index = 0; index < largeGroup.length; index++) {
largeGroup[index] = "form-" + index;
stems.put(largeGroup[index], index == 11 ? "different" : "shared");
}
groups.add(group(89, largeGroup));
final QualityResult result = evaluate(groups, stems);
assertEquals(100, result.processedWordForms());
assertEquals(66, result.underPossiblePairs());
assertEquals(11, result.underErrorPairs());
assertEquals(16.666666666666668, result.underPercentage().orElseThrow(), 0.000000000000001);
}
/** Compares the optimized accumulator with an independent explicit pair oracle. */
@Test @DisplayName("Deterministic randomized partitions agree with a brute-force pair oracle")
void randomizedOracleAgreement() {
final Random random = new Random(0x52414449584f52L);
for (int trial = 0; trial < 250; trial++) {
final List<GoldStandardGroup> groups = new ArrayList<>();
final Map<String, String> stems = new HashMap<>();
final Map<String, Integer> gold = new HashMap<>();
final int groupCount = 1 + random.nextInt(7);
int formIndex = 0;
for (int groupIndex = 0; groupIndex < groupCount; groupIndex++) {
final int size = 1 + random.nextInt(6);
final String[] forms = new String[size];
for (int index = 0; index < size; index++) {
final String form = "t" + trial + "-f" + formIndex++;
forms[index] = form;
gold.put(form, groupIndex);
stems.put(form, "s" + random.nextInt(6));
}
groups.add(group(groupIndex + 1, forms));
}
final QualityResult optimized = evaluate(groups, stems);
final long[] oracle = bruteForce(new ArrayList<>(gold.keySet()), gold, stems);
assertEquals(oracle[0], optimized.underErrorPairs(), "Under-stemming errors differed in trial " + trial);
assertEquals(oracle[1], optimized.underPossiblePairs(), "Under-stemming denominator differed in trial " + trial);
assertEquals(oracle[2], optimized.overErrorPairs(), "Over-stemming errors differed in trial " + trial);
assertEquals(oracle[3], optimized.overPossiblePairs(), "Over-stemming denominator differed in trial " + trial);
}
}
/** Explicit quadratic oracle used only for small controlled test data. */
private static long[] bruteForce(final List<String> forms, final Map<String, Integer> gold,
final Map<String, String> stems) {
long underError = 0;
long underPossible = 0;
long overError = 0;
long overPossible = 0;
for (int left = 0; left < forms.size(); left++) {
for (int right = left + 1; right < forms.size(); right++) {
final boolean sameGold = gold.get(forms.get(left)).equals(gold.get(forms.get(right)));
final boolean sameStem = stems.get(forms.get(left)).equals(stems.get(forms.get(right)));
if (sameGold) {
underPossible++;
if (!sameStem) { underError++; }
} else {
overPossible++;
if (sameStem) { overError++; }
}
}
}
return new long[] {underError, underPossible, overError, overPossible};
}
/** Builds a group. */
private static GoldStandardGroup group(final int row, final String... forms) { return new GoldStandardGroup(row, List.of(forms)); }
/** Runs the common synthetic evaluator. */
private static QualityResult evaluate(final List<GoldStandardGroup> groups, final Map<String, String> stems) {
return QualityEvaluator.evaluate("Synthetic", "TEST", ProcessingMode.ALL_WORDS, groups, stems::get);
}
}

View File

@@ -0,0 +1,256 @@
package org.egothor.stemmer.benchmark.quality;
import java.io.IOException;
import java.nio.charset.StandardCharsets;
import java.nio.file.Files;
import java.nio.file.Path;
import java.util.ArrayList;
import java.util.Comparator;
import java.util.List;
import java.util.Locale;
import java.util.OptionalDouble;
import java.util.Set;
import java.util.TreeSet;
import org.egothor.stemmer.benchmark.QualityStemmerMatrix.Candidate;
/** Writes deterministic UTF-8 Markdown and CSV quality reports. */
public final class QualityReportWriter {
private static final String TABLE_DELIMITER = " | ";
/** Utility class. */
private QualityReportWriter() { throw new AssertionError("No instances."); }
/** Writes a report without external coverage metadata for focused formatting tests. */
public static void writeMarkdown(final Path path, final Iterable<QualityResult> input,
final boolean filtered) throws IOException {
writeMarkdown(path, input, filtered, new LanguageUniverse(java.util.Map.of(), List.of(), List.of()),
List.of(), sorted(input).size(), "PAIRWISE_F05");
}
/** Writes the human-readable report with methodology and required table columns. */
public static void writeMarkdown(final Path path, final Iterable<QualityResult> input, final boolean filtered,
final LanguageUniverse universe, final List<Candidate> candidates, final int expectedRows,
final String rankMetric) throws IOException {
final List<QualityResult> rows = sorted(input);
final StringBuilder text = new StringBuilder(4096);
text.append("# Stemming quality\n\n");
if (filtered) {
text.append("> This is a filtered analytical report and is not the complete JMH candidate matrix.\n\n");
}
text.append("## Methodology\n\nEach parsed multilingual dictionary row is a gold-standard equivalence class. Exact duplicates are removed only within that row. `PRIMARY_OUTPUT` is the deterministic JMH partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: within-row sets must intersect, while a cross-row error occurs only for two equal singleton sets. `ALL_CANDIDATES` activates the complete overlap relation: within-row disjoint sets are false negatives and cross-row intersections are false positives. A shared pair is counted once. Candidate policies need not define partitions.\n\nTP is a related within-row pair, FN is an unrelated within-row pair, FP is a related cross-row pair, and TN is an unrelated cross-row pair. Under-stemming is FN/(TP+FN); over-stemming is FP/(TN+FP), so their denominators differ. F0.5 emphasizes precision, F1 balances precision and recall, and F2 emphasizes recall. Undefined values are `n/a`. Percentages and scores use `Locale.ROOT`.\n\n| Stemmer | Language | Dictionary mode | Output policy | Applied dictionary rows | Processed word forms | Distinct output stems | Over-stemming | Under-stemming | Pairwise F0.5 | Pairwise F1 | Pairwise F2 |\n|---|---|---|---|---:|---:|---:|---:|---:|---:|---:|---:|\n");
for (QualityResult row : rows) {
text.append("| ").append(escapeMarkdown(row.stemmer())).append(TABLE_DELIMITER)
.append(escapeMarkdown(row.language())).append(TABLE_DELIMITER).append(row.processingMode()).append(TABLE_DELIMITER)
.append(row.outputPolicy()).append(TABLE_DELIMITER).append(row.appliedDictionaryRows()).append(TABLE_DELIMITER)
.append(row.processedWordForms()).append(TABLE_DELIMITER).append(row.distinctOutputStems()).append(TABLE_DELIMITER)
.append(humanMetric(row.overErrorPairs(), row.overPossiblePairs(), row.overPercentage())).append(TABLE_DELIMITER)
.append(humanMetric(row.underErrorPairs(), row.underPossiblePairs(), row.underPercentage())).append(TABLE_DELIMITER)
.append(score(row.pairwiseMetrics().f05())).append(TABLE_DELIMITER)
.append(score(row.pairwiseMetrics().f1())).append(TABLE_DELIMITER)
.append(score(row.pairwiseMetrics().f2())).append(" |\n");
}
appendComparisons(text, rows);
appendCoverage(text, universe, candidates, expectedRows, rows.size());
appendRankings(text, rows, rankMetric);
appendSummaries(text, rows);
text.append("\n## Reproducibility environment\n\n- JDK: `").append(System.getProperty("java.version"))
.append("`\n- Operating system: `").append(System.getProperty("os.name")).append(' ')
.append(System.getProperty("os.version")).append("`\n");
write(path, text.toString());
}
/** Writes machine-readable counts and separate percentage fields. */
public static void writeCsv(final Path path, final Iterable<QualityResult> input) throws IOException {
final StringBuilder text = new StringBuilder(4096);
text.append("Stemmer,Language,Dictionary mode,Output policy,Applied dictionary rows,Processed word forms,Singleton dictionary rows,Forms with one candidate,Forms with multiple candidates,Maximum candidates for one form,Total candidate assignments,Distinct output stems,True-positive pairs,False-positive pairs,False-negative pairs,True-negative pairs,Over-stemming error pairs,Over-stemming possible pairs,Over-stemming percentage,Under-stemming error pairs,Under-stemming possible pairs,Under-stemming percentage,Pairwise precision,Pairwise recall,Pairwise specificity,Pairwise accuracy,Balanced accuracy,Pairwise F0.5,Pairwise F1,Pairwise F2,Jaccard index,Fowlkes-Mallows index,Matthews correlation coefficient,Pairwise error rate,Adjusted Rand Index,Homogeneity,Completeness,V-measure,Normalized mutual information\n");
for (QualityResult row : sorted(input)) {
final PairwiseMetrics metrics = row.pairwiseMetrics();
appendCsv(text, row.stemmer()); appendCsv(text, row.language()); appendCsv(text, row.processingMode().name());
appendCsv(text, row.outputPolicy().name());
appendCsv(text, Long.toString(row.appliedDictionaryRows())); appendCsv(text, Long.toString(row.processedWordForms()));
appendCsv(text, Long.toString(row.singletonDictionaryRows()));
appendCsv(text, Long.toString(row.formsWithOneCandidate()));
appendCsv(text, Long.toString(row.formsWithMultipleCandidates()));
appendCsv(text, Long.toString(row.maximumCandidatesForOneWord()));
appendCsv(text, Long.toString(row.totalCandidateAssignments()));
appendCsv(text, Long.toString(row.distinctOutputStems()));
appendCsv(text, Long.toString(metrics.truePositivePairs())); appendCsv(text, Long.toString(metrics.falsePositivePairs()));
appendCsv(text, Long.toString(metrics.falseNegativePairs())); appendCsv(text, Long.toString(metrics.trueNegativePairs()));
appendCsv(text, Long.toString(row.overErrorPairs()));
appendCsv(text, Long.toString(row.overPossiblePairs())); appendCsv(text, machinePercent(row.overPercentage()));
appendCsv(text, Long.toString(row.underErrorPairs())); appendCsv(text, Long.toString(row.underPossiblePairs()));
appendCsv(text, machinePercent(row.underPercentage()));
appendCsv(text, machineScore(metrics.precision())); appendCsv(text, machineScore(metrics.recall()));
appendCsv(text, machineScore(metrics.specificity())); appendCsv(text, machineScore(metrics.accuracy()));
appendCsv(text, machineScore(metrics.balancedAccuracy())); appendCsv(text, machineScore(metrics.f05()));
appendCsv(text, machineScore(metrics.f1())); appendCsv(text, machineScore(metrics.f2()));
appendCsv(text, machineScore(metrics.jaccard())); appendCsv(text, machineScore(metrics.fowlkesMallows()));
appendCsv(text, machineScore(metrics.matthewsCorrelationCoefficient())); appendCsv(text, machineScore(metrics.errorRate()));
final PartitionMetrics partition = row.partitionMetrics();
appendCsv(text, partition == null ? "" : format(partition.adjustedRandIndex()));
appendCsv(text, partition == null ? "" : format(partition.homogeneity()));
appendCsv(text, partition == null ? "" : format(partition.completeness()));
appendCsv(text, partition == null ? "" : format(partition.vMeasure()));
appendCsv(text, partition == null ? "" : format(partition.normalizedMutualInformation()));
text.setLength(text.length() - 1); text.append('\n');
}
write(path, text.toString());
}
/** Appends deterministic primary-versus-candidate trade-off rows for multi-output scenarios. */
private static void appendComparisons(final StringBuilder text, final List<QualityResult> rows) {
text.append("\n## Primary-versus-candidate comparison\n\n")
.append("| Stemmer | Language | Dictionary mode | Primary under | Any under | All under | Repaired under | Primary over | Any over | Best-case avoided over | All over | Additional all-candidate over | Multi-candidate forms | Multi-candidate percent | Maximum candidates | Candidate assignments |\n")
.append("|---|---|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|\n");
for (QualityResult candidate : rows) {
if (candidate.outputPolicy() != OutputPolicy.ANY_CANDIDATE) { continue; }
final QualityResult primary = rows.stream().filter(row -> row.stemmer().equals(candidate.stemmer())
&& row.language().equals(candidate.language()) && row.processingMode() == candidate.processingMode()
&& row.outputPolicy() == OutputPolicy.PRIMARY_OUTPUT).findFirst().orElse(null);
final QualityResult all = rows.stream().filter(row -> row.stemmer().equals(candidate.stemmer())
&& row.language().equals(candidate.language()) && row.processingMode() == candidate.processingMode()
&& row.outputPolicy() == OutputPolicy.ALL_CANDIDATES).findFirst().orElse(null);
if (primary == null || all == null) { continue; }
text.append("| ").append(escapeMarkdown(candidate.stemmer())).append(TABLE_DELIMITER)
.append(escapeMarkdown(candidate.language())).append(TABLE_DELIMITER).append(candidate.processingMode()).append(TABLE_DELIMITER)
.append(primary.underErrorPairs()).append(TABLE_DELIMITER).append(candidate.underErrorPairs()).append(TABLE_DELIMITER)
.append(all.underErrorPairs()).append(TABLE_DELIMITER)
.append(primary.underErrorPairs() - candidate.underErrorPairs()).append(TABLE_DELIMITER)
.append(primary.overErrorPairs()).append(TABLE_DELIMITER).append(candidate.overErrorPairs()).append(TABLE_DELIMITER)
.append(primary.overErrorPairs() - candidate.overErrorPairs()).append(TABLE_DELIMITER)
.append(all.overErrorPairs()).append(TABLE_DELIMITER)
.append(all.overErrorPairs() - primary.overErrorPairs()).append(TABLE_DELIMITER)
.append(candidate.formsWithMultipleCandidates()).append(TABLE_DELIMITER)
.append(String.format(Locale.ROOT, "%.6f%%", 100.0 * candidate.formsWithMultipleCandidates()
/ candidate.processedWordForms())).append(TABLE_DELIMITER)
.append(candidate.maximumCandidatesForOneWord()).append(TABLE_DELIMITER)
.append(candidate.totalCandidateAssignments()).append(" |\n");
}
}
/** Appends validated language, adapter, policy, and row-count coverage. */
private static void appendCoverage(final StringBuilder text, final LanguageUniverse universe,
final List<Candidate> candidates, final int expectedRows, final int actualRows) {
text.append("\n## Matrix coverage\n\n- Discovered dictionary languages: ").append(universe.resourceDirectories()).append("\n")
.append("- Discovered `StemmerPatchTrieLoader.Language` values: ").append(universe.enumerationValues()).append("\n")
.append("- Reconciled mappings: ").append(universe.dictionaries().entrySet().stream()
.sorted(java.util.Map.Entry.comparingByKey()).map(entry -> entry.getKey() + " -> " + entry.getValue().getFileName()).toList()).append("\n")
.append("- Discovered adapter-language mappings: ").append(candidates.size()).append("\n")
.append("- Expected result rows: ").append(expectedRows).append("\n")
.append("- Actual result rows: ").append(actualRows).append("\n\n")
.append("Unsupported third-party combinations are excluded because their authoritative JMH adapter metadata declares no mapping for that language. They are not emitted as zero-valued rows. Radixor is independently registered for every reconciled dictionary language.\n");
final java.util.Map<String, Set<String>> support = new java.util.TreeMap<>();
for (Candidate candidate : candidates) { support.computeIfAbsent(candidate.name(), ignored -> new TreeSet<>()).add(candidate.language().name()); }
text.append("\n| Adapter | Supported language count | Supported languages |\n|---|---:|---|\n");
support.forEach((name, languages) -> text.append("| ").append(escapeMarkdown(name)).append(TABLE_DELIMITER)
.append(languages.size()).append(TABLE_DELIMITER).append(languages).append(" |\n"));
}
/** Appends policy-separated rankings for the requested navigation metric and all required alternatives. */
private static void appendRankings(final StringBuilder text, final List<QualityResult> rows, final String selectedMetric) {
text.append("\n## Rankings\n\nThe default or selected ranking metric (`").append(selectedMetric)
.append("`) is a navigation choice, not a declaration of universal scientific superiority. Policies are ranked separately. Full-coverage and common-language comparisons must not be conflated.\n");
final List<String> metricNames = List.of("Pairwise F0.5", "Pairwise F1", "Pairwise F2", "Jaccard index",
"Fowlkes-Mallows index", "Matthews correlation coefficient", "Balanced accuracy", "Adjusted Rand Index");
for (String metric : metricNames) {
text.append("\n### ").append(metric).append("\n\n| Output policy | Stemmer | Language | Dictionary mode | Score |\n|---|---|---|---|---:|\n");
rows.stream().filter(row -> !metric.equals("Adjusted Rand Index") || row.outputPolicy() == OutputPolicy.PRIMARY_OUTPUT)
.sorted(Comparator.comparingDouble((QualityResult row) -> rankingValue(row, metric)).reversed()
.thenComparingDouble(row -> row.overPercentage().orElse(Double.POSITIVE_INFINITY))
.thenComparingLong(QualityResult::overErrorPairs)
.thenComparingDouble(row -> row.underPercentage().orElse(Double.POSITIVE_INFINITY))
.thenComparing(QualityResult::stemmer).thenComparing(QualityResult::language))
.limit(25).forEach(row -> text.append("| ").append(row.outputPolicy()).append(TABLE_DELIMITER)
.append(escapeMarkdown(row.stemmer())).append(TABLE_DELIMITER).append(row.language()).append(TABLE_DELIMITER)
.append(row.processingMode()).append(TABLE_DELIMITER).append(score(metricValue(row, metric))).append(" |\n"));
}
}
/** Returns one optional ranking metric. */
private static OptionalDouble metricValue(final QualityResult row, final String metric) {
return switch (metric) {
case "Pairwise F0.5" -> row.pairwiseMetrics().f05(); case "Pairwise F1" -> row.pairwiseMetrics().f1();
case "Pairwise F2" -> row.pairwiseMetrics().f2(); case "Jaccard index" -> row.pairwiseMetrics().jaccard();
case "Fowlkes-Mallows index" -> row.pairwiseMetrics().fowlkesMallows();
case "Matthews correlation coefficient" -> row.pairwiseMetrics().matthewsCorrelationCoefficient();
case "Balanced accuracy" -> row.pairwiseMetrics().balancedAccuracy();
case "Adjusted Rand Index" -> row.partitionMetrics() == null ? OptionalDouble.empty()
: OptionalDouble.of(row.partitionMetrics().adjustedRandIndex());
default -> OptionalDouble.empty();
};
}
/** Appends full-coverage micro and macro summaries with explicit coverage. */
private static void appendSummaries(final StringBuilder text, final List<QualityResult> rows) {
text.append("\n## Aggregate summaries\n\nMicro values sum raw confusion counts before calculation. Macro values average defined per-language F1 values.\n\n")
.append("| Stemmer | Dictionary mode | Output policy | Languages | Micro F0.5 | Micro F1 | Micro F2 | Macro F1 | Macro contributing languages |\n")
.append("|---|---|---|---:|---:|---:|---:|---:|---:|\n");
final java.util.Map<String, List<QualityResult>> groups = new java.util.TreeMap<>();
for (QualityResult row : rows) {
groups.computeIfAbsent(row.stemmer() + "\u0000" + row.processingMode() + "\u0000" + row.outputPolicy(),
ignored -> new ArrayList<>()).add(row);
}
for (List<QualityResult> group : groups.values()) {
final QualityResult first = group.get(0); long tp = 0; long fp = 0; long fn = 0; long tn = 0;
double macroF1 = 0.0; int macroCount = 0; final Set<String> languages = new TreeSet<>();
for (QualityResult row : group) {
final PairwiseMetrics metrics = row.pairwiseMetrics();
tp = Math.addExact(tp, metrics.truePositivePairs()); fp = Math.addExact(fp, metrics.falsePositivePairs());
fn = Math.addExact(fn, metrics.falseNegativePairs()); tn = Math.addExact(tn, metrics.trueNegativePairs());
if (metrics.f1().isPresent()) { macroF1 += metrics.f1().getAsDouble(); macroCount++; }
languages.add(row.language());
}
final PairwiseMetrics micro = new PairwiseMetrics(tp, fp, fn, tn);
text.append("| ").append(escapeMarkdown(first.stemmer())).append(TABLE_DELIMITER).append(first.processingMode())
.append(TABLE_DELIMITER).append(first.outputPolicy()).append(TABLE_DELIMITER).append(languages.size())
.append(TABLE_DELIMITER).append(score(micro.f05())).append(TABLE_DELIMITER).append(score(micro.f1()))
.append(TABLE_DELIMITER).append(score(micro.f2())).append(TABLE_DELIMITER)
.append(macroCount == 0 ? "n/a" : format(macroF1 / macroCount)).append(TABLE_DELIMITER)
.append(macroCount).append(" |\n");
}
Set<String> common = null;
final java.util.Map<String, Set<String>> byStemmer = new java.util.TreeMap<>();
for (QualityResult row : rows) { byStemmer.computeIfAbsent(row.stemmer(), ignored -> new TreeSet<>()).add(row.language()); }
for (Set<String> supported : byStemmer.values()) {
if (common == null) { common = new TreeSet<>(supported); } else { common.retainAll(supported); }
}
text.append("\n### Common-language comparison\n\nCommon language intersection across displayed stemmers: ")
.append(common == null ? Set.of() : common).append(". Unsupported languages are not assigned zero scores.\n");
}
/** Converts an undefined metric to negative infinity for descending navigation order. */
private static double rankingValue(final QualityResult row, final String metric) { return metricValue(row, metric).orElse(Double.NEGATIVE_INFINITY); }
/** Returns a sorted defensive list for deterministic output. */
private static List<QualityResult> sorted(final Iterable<QualityResult> input) {
final List<QualityResult> rows = new ArrayList<>();
input.forEach(rows::add); rows.sort(QualityResult.ORDER); return rows;
}
/** Formats one human-readable ratio. */
private static String humanMetric(final long errors, final long possible, final OptionalDouble percentage) {
if (percentage.isEmpty()) { return errors + " / " + possible + " (n/a)"; }
return String.format(Locale.ROOT, "%d / %d (%.6f%%)", errors, possible, percentage.getAsDouble());
}
/** Formats one optional machine-readable percentage. */
private static String machinePercent(final OptionalDouble percentage) {
return percentage.isEmpty() ? "" : String.format(Locale.ROOT, "%.6f", percentage.getAsDouble());
}
/** Formats one bounded or signed score for Markdown. */
private static String score(final OptionalDouble value) { return value.isEmpty() ? "n/a" : format(value.getAsDouble()); }
/** Formats one optional score for machine-readable output. */
private static String machineScore(final OptionalDouble value) { return value.isEmpty() ? "" : format(value.getAsDouble()); }
/** Formats an unrounded calculation deterministically with scientific precision. */
private static String format(final double value) { return String.format(Locale.ROOT, "%.12f", value); }
/** Appends one correctly quoted CSV field and delimiter. */
private static void appendCsv(final StringBuilder output, final String value) {
output.append('"').append(value.replace("\"", "\"\"")).append("\",");
}
/** Escapes Markdown table delimiters. */
private static String escapeMarkdown(final String value) { return value.replace("|", "\\|"); }
/** Creates the parent directory and atomically delegates UTF-8 file writing. */
private static void write(final Path path, final String content) throws IOException {
final Path parent = path.toAbsolutePath().getParent();
if (parent != null) { Files.createDirectories(parent); }
Files.writeString(path, content, StandardCharsets.UTF_8);
}
}

View File

@@ -0,0 +1,74 @@
package org.egothor.stemmer.benchmark.quality;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertThrows;
import static org.junit.jupiter.api.Assertions.assertTrue;
import java.io.IOException;
import java.nio.charset.StandardCharsets;
import java.nio.file.Files;
import java.nio.file.Path;
import java.util.List;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Tag;
import org.junit.jupiter.api.Test;
import org.junit.jupiter.api.io.TempDir;
/** UTF-8, formatting, ordering, escaping, and write-failure tests for reports. */
@Tag("unit")
@DisplayName("Stemming-quality report writer")
final class QualityReportWriterTest {
/** Temporary output directory owned by JUnit. */
@TempDir Path temporaryDirectory;
/** Verifies required Markdown semantics and deterministic row ordering. */
@Test @DisplayName("Markdown contains the required English columns and deterministic metrics")
void markdownFormat() throws IOException {
final Path report = this.temporaryDirectory.resolve("report.md");
QualityReportWriter.writeMarkdown(report, List.of(result("Zulu", "B", 1, 2), result("Alpha|Stemmer", "A", 0, 0)), false);
final String text = Files.readString(report, StandardCharsets.UTF_8);
assertTrue(text.contains("| Stemmer | Language | Dictionary mode | Output policy | Applied dictionary rows | Processed word forms | Distinct output stems | Over-stemming | Under-stemming | Pairwise F0.5 | Pairwise F1 | Pairwise F2 |"));
assertTrue(text.contains("0 / 0 (n/a)"));
assertTrue(text.contains("1 / 2 (50.000000%)"));
assertTrue(text.indexOf("Alpha\\|Stemmer") < text.indexOf("Zulu"));
assertEquals(text, new String(Files.readAllBytes(report), StandardCharsets.UTF_8));
}
/** Verifies CSV headers, separate missing fields, ordering, and quoting. */
@Test @DisplayName("CSV uses separate English columns, correct quoting, and empty undefined percentages")
void csvFormat() throws IOException {
final Path report = this.temporaryDirectory.resolve("report.csv");
QualityReportWriter.writeCsv(report, List.of(result("Stemmer, \"quoted\"", "A", 0, 0)));
final String text = Files.readString(report, StandardCharsets.UTF_8);
assertTrue(text.startsWith("Stemmer,Language,Dictionary mode,Output policy,Applied dictionary rows,Processed word forms,Singleton dictionary rows,Forms with one candidate,"));
assertTrue(text.contains("\"Stemmer, \"\"quoted\"\"\""));
assertTrue(text.contains("Adjusted Rand Index,Homogeneity,Completeness,V-measure,Normalized mutual information"));
}
/** Verifies that filesystem failures are propagated. */
@Test @DisplayName("A report write failure is propagated")
void writeFailure() throws IOException {
final Path file = this.temporaryDirectory.resolve("parent-file");
Files.writeString(file, "occupied", StandardCharsets.UTF_8);
assertThrows(IOException.class, () -> QualityReportWriter.writeMarkdown(file.resolve("report.md"), List.of(), false));
}
/** Verifies that a second generation replaces stale content instead of appending. */
@Test @DisplayName("Report generation replaces stale content")
void reportReplacement() throws IOException {
final Path report = this.temporaryDirectory.resolve("replacement.md");
QualityReportWriter.writeMarkdown(report, List.of(result("Old", "A", 0, 1)), false);
QualityReportWriter.writeMarkdown(report, List.of(result("New", "B", 0, 1)), false);
final String text = Files.readString(report, StandardCharsets.UTF_8);
assertTrue(text.contains("New"));
assertTrue(!text.contains("Old"));
}
/** Creates a compact valid result for formatting tests. */
private static QualityResult result(final String stemmer, final String language, final long errors, final long possible) {
return new QualityResult(stemmer, language, ProcessingMode.ALL_WORDS, OutputPolicy.PRIMARY_OUTPUT,
1, 1, 1, 0, 1, 0, 1, 1, 1, errors, possible, 0, 0,
new PartitionMetrics(1.0, 1.0, 1.0, 1.0, 1.0));
}
}

View File

@@ -0,0 +1,53 @@
package org.egothor.stemmer.benchmark.quality;
import java.util.Comparator;
import java.util.Objects;
import java.util.OptionalDouble;
/** Immutable pairwise stemming-quality result; all pair quantities are counts. */
public record QualityResult(String stemmer, String language, ProcessingMode processingMode,
OutputPolicy outputPolicy,
long appliedDictionaryRows, long processedWordForms, long singletonDictionaryRows,
long dictionaryRowsContributingUnderPairs, long formsWithOneCandidate, long formsWithMultipleCandidates,
long maximumCandidatesForOneWord, long totalCandidateAssignments, long distinctOutputStems,
long overErrorPairs, long overPossiblePairs, long underErrorPairs, long underPossiblePairs,
PartitionMetrics partitionMetrics) {
/** Stable report ordering by stemmer, language, and processing mode. */
public static final Comparator<QualityResult> ORDER = Comparator.comparing(QualityResult::stemmer)
.thenComparing(QualityResult::language).thenComparing(QualityResult::processingMode)
.thenComparing(QualityResult::outputPolicy);
/** Validates non-null labels, non-negative counts, and bounded errors. */
public QualityResult {
Objects.requireNonNull(stemmer, "stemmer");
Objects.requireNonNull(language, "language");
Objects.requireNonNull(processingMode, "processingMode");
Objects.requireNonNull(outputPolicy, "outputPolicy");
final long[] counts = {appliedDictionaryRows, processedWordForms, singletonDictionaryRows,
dictionaryRowsContributingUnderPairs, formsWithOneCandidate, formsWithMultipleCandidates,
maximumCandidatesForOneWord, totalCandidateAssignments, distinctOutputStems,
overErrorPairs, overPossiblePairs, underErrorPairs, underPossiblePairs};
for (long count : counts) {
if (count < 0) {
throw new IllegalArgumentException("Quality-result counts must not be negative.");
}
}
if (overErrorPairs > overPossiblePairs || underErrorPairs > underPossiblePairs) {
throw new IllegalArgumentException("Error-pair counts must not exceed possible-pair counts.");
}
if (outputPolicy != OutputPolicy.PRIMARY_OUTPUT && partitionMetrics != null) {
throw new IllegalArgumentException("Partition metrics apply only to PRIMARY_OUTPUT.");
}
}
/** @return over-stemming percentage, or empty when its denominator is zero */
public OptionalDouble overPercentage() { return percentage(overErrorPairs, overPossiblePairs); }
/** @return under-stemming percentage, or empty when its denominator is zero */
public OptionalDouble underPercentage() { return percentage(underErrorPairs, underPossiblePairs); }
/** @return aggregate pairwise metrics derived from raw confusion counts */
public PairwiseMetrics pairwiseMetrics() { return PairwiseMetrics.from(this); }
/** Calculates a percentage without manufacturing a value for a zero denominator. */
private static OptionalDouble percentage(final long errors, final long possible) {
return possible == 0 ? OptionalDouble.empty() : OptionalDouble.of(100.0 * errors / possible);
}
}

View File

@@ -0,0 +1,56 @@
package org.egothor.stemmer.benchmark.quality;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertTrue;
import java.nio.charset.StandardCharsets;
import java.nio.file.Files;
import java.nio.file.Path;
import java.util.ArrayList;
import java.util.List;
import org.egothor.stemmer.benchmark.QualityStemmerMatrix;
import org.egothor.stemmer.benchmark.QualityStemmerMatrix.Candidate;
import org.junit.jupiter.api.DisplayName;
import org.junit.jupiter.api.Tag;
import org.junit.jupiter.api.Test;
import org.junit.jupiter.api.io.TempDir;
/** Integration checks binding report coverage to the authoritative JMH candidate registry. */
@Tag("integration")
@DisplayName("JMH stemming-quality candidate matrix")
final class QualityStemmerMatrixTest {
/** Temporary report location. */
@TempDir Path temporaryDirectory;
/** Verifies discovery includes the complete current benchmark enum rather than Radixor alone. */
@Test @DisplayName("Candidate discovery is derived from every JMH quality candidate")
void discoversEveryCandidate() {
final List<Candidate> candidates = QualityStemmerMatrix.candidates();
assertEquals(92, candidates.size(), "The current adapter-language matrix size changed; report coverage must be reviewed.");
assertTrue(candidates.stream().anyMatch(candidate -> !candidate.name().endsWith("_RADIXOR")));
assertTrue(candidates.stream().anyMatch(candidate -> candidate.name().equals("DA_DK_RADIXOR")));
assertTrue(candidates.stream().anyMatch(candidate -> candidate.name().equals("YI_RADIXOR")));
}
/** Verifies a complete report row exists for both modes of every discovered candidate. */
@Test @DisplayName("Report rendering includes both modes for every discovered candidate")
void reportContainsCompleteMatrix() throws Exception {
final List<QualityResult> rows = new ArrayList<>();
for (Candidate candidate : QualityStemmerMatrix.candidates()) {
for (ProcessingMode mode : ProcessingMode.values()) {
rows.add(new QualityResult(candidate.name(), candidate.language().name(), mode,
OutputPolicy.PRIMARY_OUTPUT, 1, 1, 1, 0, 1, 0, 1, 1, 1, 0, 0, 0, 0,
new PartitionMetrics(1.0, 1.0, 1.0, 1.0, 1.0)));
}
}
final Path report = this.temporaryDirectory.resolve("matrix.csv");
QualityReportWriter.writeCsv(report, rows);
final String text = Files.readString(report, StandardCharsets.UTF_8);
assertEquals(185, text.lines().count());
for (Candidate candidate : QualityStemmerMatrix.candidates()) {
assertTrue(text.contains("\"" + candidate.name() + "\",\"" + candidate.language() + "\",\"ALL_WORDS\""));
assertTrue(text.contains("\"" + candidate.name() + "\",\"" + candidate.language() + "\",\"LOWERCASE_GROUPS_ONLY\""));
}
}
}

View File

@@ -0,0 +1,15 @@
package org.egothor.stemmer.benchmark.quality;
import java.io.IOException;
/** Contract used to apply one production stemmer during quality evaluation. */
@FunctionalInterface
public interface StemmerFunction {
/**
* Stems one word form without test-specific post-processing.
* @param word input form, never {@code null}
* @return output stem, never {@code null}
* @throws IOException when an adapted production stemmer fails
*/
String stem(String word) throws IOException;
}

View File

@@ -0,0 +1,245 @@
package org.egothor.stemmer.benchmark.quality;
import java.io.IOException;
import java.nio.file.Path;
import java.util.ArrayList;
import java.util.EnumMap;
import java.util.EnumSet;
import java.util.List;
import java.util.Locale;
import java.util.Map;
import java.util.HashMap;
import java.util.HashSet;
import java.util.Set;
import java.util.logging.Level;
import java.util.logging.Logger;
import org.egothor.stemmer.StemmerPatchTrieLoader.Language;
import org.egothor.stemmer.benchmark.QualityStemmerMatrix;
import org.egothor.stemmer.benchmark.QualityStemmerMatrix.Candidate;
/** Command-line entry point for JMH-backed pairwise stemming-quality reports. */
public final class StemmingQualityApplication {
private static final int ARGUMENT_COUNT = 9;
private static final Logger LOGGER = Logger.getLogger(StemmingQualityApplication.class.getName());
/** Utility class. */
private StemmingQualityApplication() {
throw new AssertionError("No instances.");
}
/**
* Generates a complete report or an explicitly labelled filtered report.
*
* @param arguments output directory, language filter, candidate filter, mode
* filter, output-policy filter, audit flag, and audit contributor limit
* @throws IOException if dictionary, JMH adapter, or report processing fails
*/
public static void main(final String[] arguments) throws IOException {
if (arguments.length != ARGUMENT_COUNT) {
throw new IllegalArgumentException("Expected output directory, resource directory, language filter, stemmer filter, dictionary-mode filter, output-policy filter, ranking metric, audit flag, and audit limit.");
}
final Path directory = Path.of(arguments[0]);
final LanguageUniverse universe = LanguageUniverse.discover(Path.of(arguments[1]));
final Set<Language> languages = parseLanguages(arguments[2]);
final Set<ProcessingMode> modes = parseModes(arguments[4]);
final Set<OutputPolicy> policies = parsePolicies(arguments[5]);
final String stemmerFilter = arguments[3].strip();
final String rankMetric = arguments[6].strip();
final boolean audit = Boolean.parseBoolean(arguments[7]);
final int auditLimit = parseAuditLimit(arguments[8]);
final boolean filtered = !arguments[2].isBlank() || !stemmerFilter.isBlank()
|| !arguments[4].isBlank() || !arguments[5].isBlank();
final List<Candidate> candidates = selectCandidates(languages, stemmerFilter);
if (candidates.isEmpty()) {
throw new IllegalArgumentException("The supplied filters select no JMH stemming-quality candidates.");
}
if (!filtered && !languages.equals(universe.dictionaries().keySet())) {
throw new IllegalStateException("The complete evaluation language selection differs from the reconciled dictionary universe.");
}
final Map<Candidate, Boolean> multiOutput = new HashMap<>();
final Set<ResultKey> expected = new HashSet<>();
for (Candidate candidate : candidates) {
final boolean multiple = candidate.createStemmer().supportsMultipleOutputs();
multiOutput.put(candidate, multiple);
for (ProcessingMode mode : modes) {
for (OutputPolicy policy : policies) {
if (policy == OutputPolicy.PRIMARY_OUTPUT || multiple) {
expected.add(new ResultKey(candidate.name(), candidate.language().name(), mode, policy));
}
}
}
}
LOGGER.log(Level.INFO, filtered ? "Starting a filtered stemming-quality report."
: "Starting the complete stemming-quality report.");
final Map<Language, List<GoldStandardGroup>> dictionaries = new EnumMap<>(Language.class);
final List<QualityResult> results = new ArrayList<>();
final List<QualityAudit.Scenario> audits = new ArrayList<>();
final List<CandidateQualityAudit.Scenario> candidateAudits = new ArrayList<>();
for (Candidate candidate : candidates) {
List<GoldStandardGroup> groups = dictionaries.get(candidate.language());
if (groups == null) {
groups = BundledGoldStandardLoader.load(candidate.language());
dictionaries.put(candidate.language(), groups);
}
for (ProcessingMode mode : modes) {
final QualityStemmerMatrix.BatchStemmer primaryStemmer = candidate.createStemmer();
final QualityResult primary;
if (audit && policies.contains(OutputPolicy.PRIMARY_OUTPUT)) {
final QualityAudit.Scenario scenario = QualityAudit.evaluate(candidate, mode, groups, auditLimit);
audits.add(scenario);
primary = scenario.result();
} else {
primary = QualityEvaluator.evaluateBatch(candidate.name(), candidate.language().name(),
mode, groups, primaryStemmer);
}
if (policies.contains(OutputPolicy.PRIMARY_OUTPUT)) {
results.add(primary);
logScenario(candidate, mode, OutputPolicy.PRIMARY_OUTPUT);
}
if (multiOutput.get(candidate)) {
final QualityResult anyCandidate = CandidateAwareEvaluator.evaluate(candidate.name(),
candidate.language().name(), mode, OutputPolicy.ANY_CANDIDATE, groups, candidate.createStemmer());
final QualityResult allCandidates;
if (audit) {
final CandidateQualityAudit.Scenario scenario = CandidateQualityAudit.evaluate(
candidate, mode, groups, primary, anyCandidate, auditLimit);
candidateAudits.add(scenario);
allCandidates = scenario.candidate();
} else {
allCandidates = CandidateAwareEvaluator.evaluate(candidate.name(), candidate.language().name(),
mode, OutputPolicy.ALL_CANDIDATES, groups, candidate.createStemmer());
}
verifyPolicyInvariants(primary, anyCandidate, allCandidates);
if (policies.contains(OutputPolicy.ANY_CANDIDATE)) {
results.add(anyCandidate); logScenario(candidate, mode, OutputPolicy.ANY_CANDIDATE);
}
if (policies.contains(OutputPolicy.ALL_CANDIDATES)) {
results.add(allCandidates); logScenario(candidate, mode, OutputPolicy.ALL_CANDIDATES);
}
}
}
}
validateMatrix(expected, results);
final String suffix = filtered ? "-filtered" : "";
final Path markdown = directory.resolve("stemming-quality" + suffix + ".md");
final Path csv = directory.resolve("stemming-quality" + suffix + ".csv");
QualityReportWriter.writeMarkdown(markdown, results, filtered, universe, candidates, expected.size(), rankMetric);
QualityReportWriter.writeCsv(csv, results);
final Path pearson = directory.resolve("metric-correlations-pearson" + suffix + ".csv");
final Path spearman = directory.resolve("metric-correlations-spearman" + suffix + ".csv");
MetricCorrelationWriter.write(pearson, spearman, results);
System.out.println("Stemming-quality Markdown report: " + markdown.toAbsolutePath());
System.out.println("Stemming-quality CSV report: " + csv.toAbsolutePath());
System.out.println("Pearson metric-correlation report: " + pearson.toAbsolutePath());
System.out.println("Spearman metric-correlation report: " + spearman.toAbsolutePath());
if (audit) {
final Path auditPath = directory.resolve("stemming-quality-audit" + suffix + ".md");
QualityAudit.write(auditPath, audits);
CandidateQualityAudit.append(auditPath, candidateAudits);
System.out.println("Stemming-quality audit report: " + auditPath.toAbsolutePath());
}
LOGGER.log(Level.INFO, "Completed the stemming-quality report with {0} evaluated scenarios.", results.size());
}
/** Selects candidates directly from the authoritative JMH matrix. */
private static List<Candidate> selectCandidates(final Set<Language> languages, final String filter) {
return QualityStemmerMatrix.candidates().stream()
.filter(candidate -> languages.contains(candidate.language()))
.filter(candidate -> filter.isBlank() || candidate.name().equalsIgnoreCase(filter)
|| candidate.name().toUpperCase(Locale.ROOT).endsWith("_" + filter.toUpperCase(Locale.ROOT)))
.toList();
}
/** Parses a comma-separated language filter or selects every language. */
private static Set<Language> parseLanguages(final String filter) {
if (filter.isBlank()) {
return EnumSet.allOf(Language.class);
}
final Set<Language> selected = EnumSet.noneOf(Language.class);
for (String item : filter.split(",")) {
selected.add(Language.valueOf(item.strip().toUpperCase(Locale.ROOT)));
}
return selected;
}
/** Parses a comma-separated mode filter or selects both processing modes. */
private static Set<ProcessingMode> parseModes(final String filter) {
if (filter.isBlank()) {
return EnumSet.allOf(ProcessingMode.class);
}
final Set<ProcessingMode> selected = EnumSet.noneOf(ProcessingMode.class);
for (String item : filter.split(",")) {
selected.add(ProcessingMode.valueOf(item.strip().toUpperCase(Locale.ROOT)));
}
return selected;
}
/** Parses a comma-separated output-policy filter or selects both policies. */
private static Set<OutputPolicy> parsePolicies(final String filter) {
if (filter.isBlank()) { return EnumSet.allOf(OutputPolicy.class); }
final Set<OutputPolicy> selected = EnumSet.noneOf(OutputPolicy.class);
for (String item : filter.split(",")) {
selected.add(OutputPolicy.valueOf(item.strip().toUpperCase(Locale.ROOT)));
}
return selected;
}
/** Enforces the mathematical monotonicity guaranteed by primary-output inclusion. */
private static void verifyPolicyInvariants(final QualityResult primary, final QualityResult any,
final QualityResult all) {
if (any.underErrorPairs() > primary.underErrorPairs() || all.underErrorPairs() > primary.underErrorPairs()
|| any.underErrorPairs() != all.underErrorPairs() || any.overErrorPairs() > primary.overErrorPairs()
|| all.overErrorPairs() < primary.overErrorPairs()) {
throw new IllegalStateException("Output-policy invariants failed for stemmer " + primary.stemmer()
+ ", language " + primary.language() + ", dictionary mode " + primary.processingMode()
+ ": PRIMARY_OUTPUT under/over=" + primary.underErrorPairs() + "/" + primary.overErrorPairs()
+ ", ANY_CANDIDATE under/over=" + any.underErrorPairs() + "/" + any.overErrorPairs()
+ ", ALL_CANDIDATES under/over=" + all.underErrorPairs() + "/" + all.overErrorPairs() + ".");
}
}
/** Validates exact expected and actual result keys, including duplicates. */
private static void validateMatrix(final Set<ResultKey> expected, final List<QualityResult> results) {
final Set<ResultKey> actual = new HashSet<>();
for (QualityResult result : results) {
final ResultKey key = ResultKey.from(result);
if (!actual.add(key)) { throw new IllegalStateException("Duplicate stemming-quality result key: " + key + "."); }
}
if (!expected.equals(actual)) {
final Set<ResultKey> missing = new HashSet<>(expected); missing.removeAll(actual);
final Set<ResultKey> unexpected = new HashSet<>(actual); unexpected.removeAll(expected);
throw new IllegalStateException("Stemming-quality result matrix mismatch. Missing rows: " + missing
+ "; unexpected rows: " + unexpected + ".");
}
}
/** Immutable expected-matrix key. */
private record ResultKey(String stemmer, String language, ProcessingMode mode, OutputPolicy policy) {
/** Creates a key from one immutable result. */
private static ResultKey from(final QualityResult result) {
return new ResultKey(result.stemmer(), result.language(), result.processingMode(), result.outputPolicy());
}
}
/** Parses and validates the deterministic audit contributor limit. */
private static int parseAuditLimit(final String value) {
final int limit = Integer.parseInt(value);
if (limit < 1) {
throw new IllegalArgumentException("The audit contributor limit must be positive.");
}
return limit;
}
/** Logs one completed scenario without per-word noise. */
private static void logScenario(final Candidate candidate, final ProcessingMode mode, final OutputPolicy policy) {
if (LOGGER.isLoggable(Level.INFO)) {
LOGGER.log(Level.INFO, "Completed stemming-quality evaluation for stemmer {0}, language {1}, dictionary mode {2}, and output policy {3}.",
new Object[] {candidate.name(), candidate.language(), mode, policy});
}
}
}

View File

@@ -0,0 +1,743 @@
package org.egothor.stemmer.benchmark.quality;
import java.io.IOException;
import java.nio.charset.StandardCharsets;
import java.nio.file.Files;
import java.nio.file.Path;
import java.nio.file.StandardCopyOption;
import java.security.MessageDigest;
import java.security.NoSuchAlgorithmException;
import java.util.ArrayList;
import java.util.Comparator;
import java.util.HashMap;
import java.util.HashSet;
import java.util.LinkedHashMap;
import java.util.List;
import java.util.Locale;
import java.util.Map;
import java.util.Set;
import java.util.regex.Matcher;
import java.util.regex.Pattern;
/**
* Publishes validated stemming-quality CSV results into marked sections of the
* existing language benchmark pages. This test-source utility never modifies
* performance benchmark content outside its markers.
*/
public final class StemmingQualityDocumentationPublisher {
private static final String START = "<!-- STEMMING-QUALITY:START -->";
private static final String END = "<!-- STEMMING-QUALITY:END -->";
private static final String OVERVIEW_START = "<!-- STEMMING-QUALITY-OVERVIEW:START -->";
private static final String OVERVIEW_END = "<!-- STEMMING-QUALITY-OVERVIEW:END -->";
private static final List<String> MODES = List.of("ALL_WORDS", "LOWERCASE_GROUPS_ONLY");
private static final Map<String, Integer> POLICY_ORDER = Map.of("PRIMARY_OUTPUT", 0, "ANY_CANDIDATE", 1, "ALL_CANDIDATES", 2);
private static final Pattern PAGE_ROW = Pattern.compile("^\\|[^|]+\\| `([^`]+)` \\| \\[([^]]+)]\\(([^)]+\\.md)\\) \\|$");
private static final Pattern BUILT_IN_LANGUAGE_ROW = Pattern.compile("^\\|[^|]+\\| `([^`]+)` \\|.*$");
/** Prevents construction of this command-line utility. */
private StemmingQualityDocumentationPublisher() { }
/**
* Updates or verifies the documentation from one complete source CSV.
*
* @param arguments source CSV, documentation root, and either {@code update} or {@code verify}
* @throws IOException when source or documentation access fails
*/
public static void main(final String[] arguments) throws IOException {
if (arguments.length != 3) {
throw new IllegalArgumentException("Expected arguments: source CSV, documentation root, and update or verify mode.");
}
final Path source = Path.of(arguments[0]);
final Path documentationRoot = Path.of(arguments[1]);
final boolean update = switch (arguments[2]) {
case "update" -> true;
case "verify" -> false;
default -> throw new IllegalArgumentException("Documentation mode must be update or verify.");
};
publish(source, documentationRoot, update);
}
/**
* Validates the complete result set and updates or verifies every mapped page.
*
* @param source authoritative complete CSV
* @param documentationRoot repository documentation directory
* @param update whether files may be replaced
* @throws IOException when files cannot be read or written
*/
static void publish(final Path source, final Path documentationRoot, final boolean update) throws IOException {
if (!Files.isRegularFile(source) || source.getFileName().toString().contains("filtered")) {
throw new IllegalArgumentException("The documentation source must be an existing complete, unfiltered CSV report: " + source);
}
final List<ResultRow> rows = readRows(source);
final Map<String, Page> pages = readPages(documentationRoot.resolve("benchmarks/languages/index.md"));
final Set<String> languageUniverse = readLanguageUniverse(documentationRoot.resolve("built-in-languages.md"));
validate(rows, pages.keySet(), languageUniverse);
final String checksum = sha256(source);
if (!update) {
final Path checksumFile = documentationRoot.resolve("benchmarks/data/stemming-quality.sha256");
final String recorded = Files.readString(checksumFile, StandardCharsets.UTF_8).strip();
if (!recorded.equals(checksum + " stemming-quality.csv")) {
throw new IllegalStateException("The published stemming-quality checksum does not match the authoritative CSV.");
}
}
for (Page page : pages.values()) {
final List<ResultRow> languageRows = rows.stream().filter(row -> row.language().equals(page.language())).toList();
final String section = render(page, languageRows, checksum);
final Path path = documentationRoot.resolve("benchmarks/languages").resolve(page.file());
final String original = Files.readString(path, StandardCharsets.UTF_8);
final String expected = replaceSection(original, section);
if (update) {
Files.writeString(path, expected, StandardCharsets.UTF_8);
} else if (!original.equals(expected)) {
throw new IllegalStateException("Stemming-quality documentation is stale or manually altered: " + path);
}
}
final Path overviewPath = documentationRoot.resolve("benchmarks/index.md");
final String overview = Files.readString(overviewPath, StandardCharsets.UTF_8);
final String expectedOverview = replaceMarkedSection(overview, renderOverview(pages, rows, checksum),
OVERVIEW_START, OVERVIEW_END);
if (update) {
Files.writeString(overviewPath, expectedOverview, StandardCharsets.UTF_8);
} else if (!overview.equals(expectedOverview)) {
throw new IllegalStateException("The generated benchmark quality overview is stale or manually altered: " + overviewPath);
}
if (update) {
final Path publishedSource = documentationRoot.resolve("benchmarks/data/stemming-quality.csv");
Files.createDirectories(publishedSource.getParent());
Files.copy(source, publishedSource, StandardCopyOption.REPLACE_EXISTING);
Files.writeString(documentationRoot.resolve("benchmarks/data/stemming-quality.sha256"), checksum + " stemming-quality.csv\n", StandardCharsets.UTF_8);
}
System.out.printf(Locale.ROOT, "%s stemming-quality documentation for %d languages from %d validated rows.%n",
update ? "Updated" : "Verified", pages.size(), rows.stream().filter(row -> pages.containsKey(row.language())).count());
}
/** Reads the authoritative built-in language identifiers from the existing registry table. */
private static Set<String> readLanguageUniverse(final Path builtInLanguages) throws IOException {
final Set<String> languages = new HashSet<>();
for (String line : Files.readAllLines(builtInLanguages, StandardCharsets.UTF_8)) {
final Matcher matcher = BUILT_IN_LANGUAGE_ROW.matcher(line);
if (matcher.matches()) {
languages.add(matcher.group(1));
}
}
if (languages.isEmpty()) {
throw new IllegalStateException("No authoritative built-in languages were discovered in " + builtInLanguages);
}
return Set.copyOf(languages);
}
/** Reads the language-code-to-page mapping from the existing documentation index. */
private static Map<String, Page> readPages(final Path index) throws IOException {
final Map<String, Page> pages = new LinkedHashMap<>();
for (String line : Files.readAllLines(index, StandardCharsets.UTF_8)) {
final Matcher matcher = PAGE_ROW.matcher(line);
if (matcher.matches()) {
final Page previous = pages.put(matcher.group(1), new Page(matcher.group(1), matcher.group(2), matcher.group(3)));
if (previous != null) {
throw new IllegalStateException("Duplicate language mapping in benchmark index: " + matcher.group(1));
}
}
}
if (pages.isEmpty()) {
throw new IllegalStateException("No language benchmark pages were discovered in " + index);
}
return pages;
}
/** Reads and schema-validates the quoted UTF-8 CSV. */
private static List<ResultRow> readRows(final Path source) throws IOException {
final List<String> lines = Files.readAllLines(source, StandardCharsets.UTF_8);
if (lines.isEmpty()) {
throw new IllegalStateException("The stemming-quality CSV is empty.");
}
final List<String> header = parseCsv(lines.getFirst());
final List<String> required = List.of("Stemmer", "Language", "Dictionary mode", "Output policy", "Applied dictionary rows",
"Processed word forms", "Forms with multiple candidates", "Maximum candidates for one form", "Total candidate assignments",
"True-positive pairs", "False-positive pairs", "False-negative pairs", "True-negative pairs",
"Over-stemming error pairs", "Over-stemming possible pairs", "Over-stemming percentage", "Under-stemming error pairs",
"Under-stemming possible pairs", "Under-stemming percentage", "Pairwise precision", "Pairwise recall", "Pairwise specificity",
"Pairwise accuracy", "Balanced accuracy", "Pairwise F0.5", "Pairwise F1", "Pairwise F2", "Jaccard index",
"Fowlkes-Mallows index", "Matthews correlation coefficient", "Pairwise error rate", "Adjusted Rand Index", "Homogeneity",
"Completeness", "V-measure", "Normalized mutual information");
if (!header.containsAll(required)) {
throw new IllegalStateException("The stemming-quality CSV does not contain the required publication schema.");
}
final Map<String, Integer> indexes = new HashMap<>();
for (int index = 0; index < header.size(); index++) {
indexes.put(header.get(index), index);
}
final List<ResultRow> rows = new ArrayList<>();
for (int line = 1; line < lines.size(); line++) {
final List<String> values = parseCsv(lines.get(line));
if (values.size() != header.size()) {
throw new IllegalStateException("CSV column count differs from the header at logical row " + (line + 1));
}
rows.add(new ResultRow(values, indexes));
}
return List.copyOf(rows);
}
/** Parses one RFC-4180-compatible line emitted by the quality report writer. */
private static List<String> parseCsv(final String line) {
final List<String> values = new ArrayList<>();
final StringBuilder value = new StringBuilder();
boolean quoted = false;
for (int index = 0; index < line.length(); index++) {
final char character = line.charAt(index);
if (character == '"') {
if (quoted && index + 1 < line.length() && line.charAt(index + 1) == '"') {
value.append('"');
index++;
} else {
quoted = !quoted;
}
} else if (character == ',' && !quoted) {
values.add(value.toString());
value.setLength(0);
} else {
value.append(character);
}
}
if (quoted) {
throw new IllegalStateException("Unterminated quoted CSV value.");
}
values.add(value.toString());
return values;
}
/** Validates uniqueness, coverage, raw arithmetic, metrics, and policy invariants. */
private static void validate(final List<ResultRow> rows, final Set<String> documentedLanguages,
final Set<String> languageUniverse) {
final Set<String> keys = new HashSet<>();
for (ResultRow row : rows) {
if (!keys.add(row.key())) {
throw new IllegalStateException("Duplicate stemming-quality result key: " + row.key());
}
row.validate();
}
final Set<String> resultLanguages = new HashSet<>();
rows.forEach(row -> resultLanguages.add(row.language()));
if (!resultLanguages.equals(languageUniverse)) {
throw new IllegalStateException("Complete-report language coverage differs from the authoritative built-in universe. Results: "
+ resultLanguages + "; authoritative languages: " + languageUniverse);
}
for (String language : languageUniverse) {
for (String mode : MODES) {
for (String policy : POLICY_ORDER.keySet()) {
final boolean present = rows.stream().anyMatch(row -> row.language().equals(language) && row.mode().equals(mode)
&& row.policy().equals(policy) && row.stemmer().endsWith("_RADIXOR"));
if (!present) {
throw new IllegalStateException("The complete report omits Radixor result " + language + "/" + mode + "/" + policy);
}
}
}
}
for (String language : documentedLanguages) {
final List<ResultRow> languageRows = rows.stream().filter(row -> row.language().equals(language)).toList();
if (languageRows.isEmpty()) {
throw new IllegalStateException("No stemming-quality results exist for documented language " + language);
}
for (String mode : MODES) {
if (languageRows.stream().noneMatch(row -> row.mode().equals(mode))) {
throw new IllegalStateException("Missing dictionary mode " + mode + " for documented language " + language);
}
}
validatePolicies(languageRows);
}
if (!documentedLanguages.contains("DA_DK") || !documentedLanguages.contains("YI")) {
throw new IllegalStateException("The documentation mapping must contain DA_DK and YI.");
}
}
/** Validates policy monotonicity for each multi-output scenario. */
private static void validatePolicies(final List<ResultRow> rows) {
final Map<String, Map<String, ResultRow>> scenarios = new HashMap<>();
for (ResultRow row : rows) {
scenarios.computeIfAbsent(row.stemmer() + "\u0000" + row.mode(), ignored -> new HashMap<>()).put(row.policy(), row);
}
for (Map<String, ResultRow> policies : scenarios.values()) {
final ResultRow primary = policies.get("PRIMARY_OUTPUT");
if (primary == null) {
throw new IllegalStateException("Every documented stemmer scenario must contain PRIMARY_OUTPUT.");
}
if (policies.containsKey("ANY_CANDIDATE") || policies.containsKey("ALL_CANDIDATES")) {
final ResultRow any = policies.get("ANY_CANDIDATE");
final ResultRow all = policies.get("ALL_CANDIDATES");
if (any == null || all == null || any.fn() > primary.fn() || all.fn() != any.fn()
|| any.fp() > primary.fp() || all.fp() < primary.fp()) {
throw new IllegalStateException("Output-policy invariants fail for " + primary.key());
}
}
}
}
/** Renders one complete generated section for a language page. */
private static String render(final Page page, final List<ResultRow> rows, final String checksum) {
final StringBuilder output = new StringBuilder(32768);
output.append(START).append("\n\n## Stemming Quality\n\n")
.append("Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `")
.append(page.language()).append("` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.\n\n")
.append("`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).\n\n")
.append("### Evaluation Scope and Key Findings\n\n")
.append("The dictionary resource is `src/main/resources/").append(page.language().toLowerCase(Locale.ROOT)).append("/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.\n\n");
for (String mode : MODES) {
appendFinding(output, rows, mode);
}
for (String mode : MODES) {
final List<ResultRow> selected = rows.stream().filter(row -> row.mode().equals(mode)).sorted(resultOrder()).toList();
final long stemmers = selected.stream().map(ResultRow::stemmer).distinct().count();
final long policies = selected.stream().map(ResultRow::policy).distinct().count();
output.append("### `").append(mode).append("`\n\n")
.append("This mode contains **").append(selected.size()).append(" result rows**, **").append(stemmers)
.append(" evaluated stemmers**, and **").append(policies).append(" output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.\n\n");
for (String policy : List.of("PRIMARY_OUTPUT", "ANY_CANDIDATE", "ALL_CANDIDATES")) {
final List<ResultRow> policyRows = selected.stream().filter(row -> row.policy().equals(policy)).toList();
if (!policyRows.isEmpty()) {
output.append("#### `").append(policy).append("` ranking\n\n");
renderPrimaryTable(output, policyRows);
renderDetailedTables(output, policyRows);
}
}
renderCandidateAnalysis(output, selected);
}
appendMethodology(output);
output.append("### Provenance\n\n")
.append("- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`\n")
.append("- Source SHA-256: `").append(checksum).append("`\n")
.append("- Evaluation command: `./gradlew stemmingQuality`\n")
.append("- Dictionary language: `").append(page.language()).append("`\n")
.append("- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`\n")
.append("- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`\n")
.append("- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV\n\n")
.append(END).append('\n');
return output.toString();
}
/** Appends one deterministic primary-output winner and runner-up statement. */
private static void appendFinding(final StringBuilder output, final List<ResultRow> rows, final String mode) {
final List<ResultRow> primary = rows.stream().filter(row -> row.mode().equals(mode) && row.policy().equals("PRIMARY_OUTPUT"))
.sorted(resultOrder()).toList();
final ResultRow winner = primary.getFirst();
final ResultRow runnerUp = primary.size() > 1 ? primary.get(1) : null;
output.append("- **").append(mode).append(":** `").append(displayStemmer(winner.stemmer())).append("` ranks first by balanced accuracy at **")
.append(metric(winner, "Balanced accuracy")).append("** among ").append(primary.size()).append(" deterministic stemmers");
if (runnerUp == null) {
output.append("; no same-language competitor was available");
} else {
final double difference = winner.number("Balanced accuracy") - runnerUp.number("Balanced accuracy");
output.append(". The runner-up is `").append(displayStemmer(runnerUp.stemmer())).append("` at ")
.append(metric(runnerUp, "Balanced accuracy")).append(", a difference of ")
.append(String.format(Locale.ROOT, "%.6f", difference));
if (difference == 0.0) {
output.append(" (an exact tie before formatting)");
}
}
output.append(". This rank does not imply leadership in throughput or every secondary metric.\n");
}
/** Renders the compact primary ranking table in an accessible scroll region. */
private static void renderPrimaryTable(final StringBuilder output, final List<ResultRow> rows) {
output.append("<div class=\"quality-table quality-table--compact\" role=\"region\" aria-label=\"Compact stemming-quality ranking; scroll horizontally for additional columns\" tabindex=\"0\" markdown=\"1\">\n\n")
.append("| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |\n")
.append("|---:|---|---|---:|---:|---:|---:|---:|---:|\n");
for (int index = 0; index < rows.size(); index++) {
final ResultRow row = rows.get(index);
output.append('|').append(index + 1).append('|').append(displayStemmer(row.stemmer())).append('|').append(row.policy()).append('|')
.append(metric(row, "Balanced accuracy")).append('|')
.append(pair(row, "Over-stemming error pairs", "Over-stemming possible pairs", "Over-stemming percentage")).append('|')
.append(pair(row, "Under-stemming error pairs", "Under-stemming possible pairs", "Under-stemming percentage")).append('|')
.append(metric(row, "Pairwise F0.5")).append('|').append(metric(row, "Pairwise F1")).append('|')
.append(metric(row, "Matthews correlation coefficient")).append("|\n");
}
output.append("\n</div>\n\n");
}
/** Renders classification, relation, partition, and raw-count tables with repeated identities. */
private static void renderDetailedTables(final StringBuilder output, final List<ResultRow> rows) {
output.append("<details class=\"quality-details\" markdown=\"1\"><summary>Classification metrics</summary>\n\n")
.append("| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |\n")
.append("|---:|---|---|---:|---:|---:|---:|---:|---:|\n");
for (int index = 0; index < rows.size(); index++) {
final ResultRow row = rows.get(index);
output.append(identity(index, row)).append(metric(row, "Pairwise precision")).append('|').append(metric(row, "Pairwise recall")).append('|')
.append(metric(row, "Pairwise specificity")).append('|').append(metric(row, "Balanced accuracy")).append('|')
.append(metric(row, "Pairwise accuracy")).append('|').append(metric(row, "Pairwise error rate")).append("|\n");
}
output.append("\n</details>\n\n<details class=\"quality-details\" markdown=\"1\"><summary>Pair-relation metrics</summary>\n\n")
.append("| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | FowlkesMallows | MCC |\n")
.append("|---:|---|---|---:|---:|---:|---:|---:|---:|\n");
for (int index = 0; index < rows.size(); index++) {
final ResultRow row = rows.get(index);
output.append(identity(index, row)).append(metric(row, "Pairwise F0.5")).append('|').append(metric(row, "Pairwise F1")).append('|')
.append(metric(row, "Pairwise F2")).append('|').append(metric(row, "Jaccard index")).append('|')
.append(metric(row, "Fowlkes-Mallows index")).append('|').append(metric(row, "Matthews correlation coefficient")).append("|\n");
}
output.append("\n</details>\n\n<details class=\"quality-details\" markdown=\"1\"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>\n\n")
.append("| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |\n")
.append("|---:|---|---|---:|---:|---:|---:|---:|\n");
for (int index = 0; index < rows.size(); index++) {
final ResultRow row = rows.get(index);
output.append(identity(index, row)).append(metric(row, "Adjusted Rand Index")).append('|').append(metric(row, "Homogeneity")).append('|')
.append(metric(row, "Completeness")).append('|').append(metric(row, "V-measure")).append('|')
.append(metric(row, "Normalized mutual information")).append("|\n");
}
output.append("\n</details>\n\n<details class=\"quality-details\" markdown=\"1\"><summary>Raw pair counts</summary>\n\n")
.append("| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |\n")
.append("|---:|---|---|---:|---:|---:|---:|---:|---:|\n");
for (int index = 0; index < rows.size(); index++) {
final ResultRow row = rows.get(index);
output.append(identity(index, row)).append(row.value("True-positive pairs")).append('|').append(row.value("False-positive pairs"))
.append('|').append(row.value("False-negative pairs")).append('|').append(row.value("True-negative pairs")).append('|')
.append(row.value("Over-stemming error pairs")).append(" / ").append(row.value("Over-stemming possible pairs")).append('|')
.append(row.value("Under-stemming error pairs")).append(" / ").append(row.value("Under-stemming possible pairs")).append("|\n");
}
output.append("\n</details>\n\n");
}
/** Renders the candidate-policy trade-off for every genuinely multi-output adapter. */
private static void renderCandidateAnalysis(final StringBuilder output, final List<ResultRow> rows) {
final Map<String, Map<String, ResultRow>> byStemmer = new LinkedHashMap<>();
rows.forEach(row -> byStemmer.computeIfAbsent(row.stemmer(), ignored -> new HashMap<>()).put(row.policy(), row));
final List<Map.Entry<String, Map<String, ResultRow>>> multi = byStemmer.entrySet().stream()
.filter(entry -> entry.getValue().containsKey("ANY_CANDIDATE")).sorted(Map.Entry.comparingByKey()).toList();
if (multi.isEmpty()) {
return;
}
output.append("#### Multi-output analysis\n\nAlternative candidates are capability analyses, not replacements for the deterministic comparison.\n\n")
.append("| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |\n")
.append("|---|---:|---:|---:|---:|---:|---:|---:|\n");
for (Map.Entry<String, Map<String, ResultRow>> entry : multi) {
final ResultRow primary = entry.getValue().get("PRIMARY_OUTPUT");
final ResultRow any = entry.getValue().get("ANY_CANDIDATE");
final ResultRow all = entry.getValue().get("ALL_CANDIDATES");
final long forms = any.longValue("Processed word forms");
final long multiple = any.longValue("Forms with multiple candidates");
output.append('|').append(displayStemmer(entry.getKey())).append('|').append(primary.fn() - any.fn()).append('|')
.append(primary.fp() - any.fp()).append('|').append(all.fp() - primary.fp()).append('|').append(multiple).append('|')
.append(String.format(Locale.ROOT, "%.6f%%", 100.0 * multiple / forms)).append('|')
.append(any.value("Maximum candidates for one form")).append('|').append(any.value("Total candidate assignments")).append("|\n");
}
output.append('\n');
}
/** Returns the repeated rank, stemmer, and policy prefix for a detailed table row. */
private static String identity(final int index, final ResultRow row) {
return "|" + (index + 1) + "|" + displayStemmer(row.stemmer()) + "|" + row.policy() + "|";
}
/** Converts authoritative adapter identifiers into a stable readable label without merging competitors. */
private static String displayStemmer(final String identifier) {
return identifier.endsWith("_RADIXOR") ? "Radixor" : identifier.replace('_', ' ');
}
/** Appends the self-contained policy, confusion-matrix, and metric definitions. */
private static void appendMethodology(final StringBuilder output) {
output.append("### Output Policies and Metric Definitions\n\n")
.append("`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.\n\n")
.append("For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.\n\n")
.append("- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.\n")
.append("- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.\n")
.append("- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.\n")
.append("- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.\n")
.append("- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.\n")
.append("- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.\n")
.append("- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.\n")
.append("- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.\n")
.append("- Jaccard index: `TP / (TP + FP + FN)`.\n")
.append("- FowlkesMallows index: `sqrt(precision * recall)`.\n")
.append("- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.\n")
.append("- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.\n\n")
.append("Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.\n\n");
}
/** Renders the generated executive findings, winner matrix, and Radixor aggregates. */
private static String renderOverview(final Map<String, Page> pages, final List<ResultRow> rows, final String checksum) {
final StringBuilder output = new StringBuilder(16384);
output.append(OVERVIEW_START).append("\n\n## Pairwise Quality Findings\n\n")
.append("The validated snapshot is a broad multilingual comparison covering the complete 20-language Radixor dictionary universe; 19 languages have existing benchmark pages. The direct ranking below uses only deterministic `PRIMARY_OUTPUT` rows over identical per-language inputs. Candidate-aware rows are intentionally excluded from this claim.\n\n");
int radixorWins = 0;
int comparisons = 0;
for (String mode : MODES) {
for (String language : pages.keySet()) {
final List<ResultRow> ranked = primaryRows(rows, language, mode);
comparisons++;
if (ranked.getFirst().stemmer().endsWith("_RADIXOR")) {
radixorWins++;
}
}
}
if (radixorWins == comparisons) {
output.append("!!! success \"Evidence-based primary-output result\"\n Radixor achieved the highest balanced accuracy among the evaluated deterministic stemmers for every documented language in both `ALL_WORDS` and `LOWERCASE_GROUPS_ONLY`: **")
.append(radixorWins).append(" wins in ").append(comparisons).append(" language-mode comparisons, with no exact first-place ties**. This statement is limited to the evaluated implementations, versions, dictionaries, adapters, and balanced-accuracy metric; it is not a universal claim about every stemming use case.\n\n");
} else {
output.append("Radixor ranks first in **").append(radixorWins).append(" of ").append(comparisons)
.append("** documented primary-output language-mode comparisons.\n\n");
}
output.append("### Per-language winner matrix\n\n| Language | Dictionary mode | Winner | Balanced accuracy | Runner-up | Difference | Exact tie | Deterministic stemmers |\n")
.append("|---|---|---|---:|---|---:|---|---:|\n");
for (Page page : pages.values()) {
for (String mode : MODES) {
final List<ResultRow> ranked = primaryRows(rows, page.language(), mode);
final ResultRow winner = ranked.getFirst();
final ResultRow runner = ranked.size() > 1 ? ranked.get(1) : null;
final double difference = runner == null ? Double.NaN : winner.number("Balanced accuracy") - runner.number("Balanced accuracy");
output.append('|').append(page.displayName()).append(" (`").append(page.language()).append("`)|").append(mode).append('|')
.append(displayStemmer(winner.stemmer())).append('|').append(metric(winner, "Balanced accuracy")).append('|')
.append(runner == null ? "n/a" : displayStemmer(runner.stemmer())).append('|')
.append(runner == null ? "n/a" : String.format(Locale.ROOT, "%.9f", difference)).append('|')
.append(runner != null && difference == 0.0 ? "yes" : "no").append('|').append(ranked.size()).append("|\n");
}
}
renderSecondaryLeaders(output, pages, rows);
output.append("\n### Win, tie, and placement summary\n\nCounts use `PRIMARY_OUTPUT` only and retain each adapter configuration as a separate stemmer except that language-specific Radixor identifiers are combined as Radixor. Coverage is displayed explicitly; unsupported languages are absent, not losses.\n\n");
for (String mode : MODES) {
renderPlacementSummary(output, pages, rows, mode);
}
output.append("\n### Radixor full-coverage aggregates\n\nThese aggregates cover all 19 documented languages. Macro balanced accuracy gives each language equal weight. Micro metrics first sum raw pair counts across languages. Unsupported third-party languages are never inserted as zero results, so this full-coverage table is not presented as a cross-stemmer common-language ranking.\n\n")
.append("| Dictionary mode | Languages | Macro balanced accuracy | Micro balanced accuracy | Micro precision | Micro recall | Micro F1 |\n")
.append("|---|---:|---:|---:|---:|---:|---:|\n");
for (String mode : MODES) {
final List<ResultRow> radixor = rows.stream().filter(row -> pages.containsKey(row.language()) && row.mode().equals(mode)
&& row.policy().equals("PRIMARY_OUTPUT") && row.stemmer().endsWith("_RADIXOR")).toList();
final double macroBalanced = radixor.stream().mapToDouble(row -> row.number("Balanced accuracy")).average().orElseThrow();
long tp = 0;
long fp = 0;
long fn = 0;
long tn = 0;
for (ResultRow row : radixor) {
tp = Math.addExact(tp, row.longValue("True-positive pairs"));
fp = Math.addExact(fp, row.fp());
fn = Math.addExact(fn, row.fn());
tn = Math.addExact(tn, row.longValue("True-negative pairs"));
}
final double precision = (double) tp / Math.addExact(tp, fp);
final double recall = (double) tp / Math.addExact(tp, fn);
final double specificity = (double) tn / Math.addExact(tn, fp);
final double f1 = 2.0 * tp / (2.0 * tp + fp + fn);
output.append('|').append(mode).append('|').append(radixor.size()).append('|').append(format(macroBalanced)).append('|')
.append(format((recall + specificity) / 2.0)).append('|').append(format(precision)).append('|')
.append(format(recall)).append('|').append(format(f1)).append("|\n");
}
output.append("\n### Reproducible data\n\n- [Machine-readable quality snapshot](data/stemming-quality.csv)\n")
.append("- SHA-256: `").append(checksum).append("`\n")
.append("- [Linguistic quality methodology](reference/linguistic-quality.md)\n")
.append("- [Tested stemmer inventory](reference/tested-stemmers.md)\n")
.append("- [Reproducibility and raw data](reference/reproducibility.md)\n")
.append("- Pearson and Spearman correlation files are generated under `build/reports/stemming-quality/`; they are separated by dictionary mode and output policy. Correlation does not establish metric equivalence.\n\n")
.append(OVERVIEW_END).append('\n');
return output.toString();
}
/** Publishes every deterministic secondary-metric case led by a non-Radixor adapter. */
private static void renderSecondaryLeaders(final StringBuilder output, final Map<String, Page> pages,
final List<ResultRow> rows) {
final Map<String, Boolean> metrics = new LinkedHashMap<>();
metrics.put("Pairwise precision", true);
metrics.put("Pairwise recall", true);
metrics.put("Pairwise F0.5", true);
metrics.put("Pairwise F1", true);
metrics.put("Pairwise F2", true);
metrics.put("Matthews correlation coefficient", true);
metrics.put("Over-stemming percentage", false);
metrics.put("Under-stemming percentage", false);
final StringBuilder cases = new StringBuilder();
int count = 0;
for (Page page : pages.values()) {
for (String mode : MODES) {
final List<ResultRow> primary = primaryRows(rows, page.language(), mode);
for (Map.Entry<String, Boolean> metric : metrics.entrySet()) {
final Comparator<ResultRow> comparator = Comparator.comparingDouble(row -> row.number(metric.getKey()));
final ResultRow leader = metric.getValue() ? primary.stream().max(comparator).orElseThrow()
: primary.stream().min(comparator).orElseThrow();
if (!leader.stemmer().endsWith("_RADIXOR")) {
count++;
cases.append('|').append(page.displayName()).append('|').append(mode).append('|').append(metric.getKey()).append('|')
.append(displayStemmer(leader.stemmer())).append('|').append(metric(leader, metric.getKey())).append("|\n");
}
}
}
}
output.append("\n### Secondary-metric trade-offs\n\nBalanced-accuracy leadership does not imply leadership on every error trade-off. The table below lists all **")
.append(count).append("** deterministic primary-output language-mode-metric cases where a non-Radixor adapter has the best displayed value. Equal values are resolved by the authoritative row ordering and should be read as ties when the unrounded values are equal. Throughput leadership remains in the separate performance tables.\n\n")
.append("<details class=\"quality-details\" markdown=\"1\"><summary>Non-Radixor secondary-metric leaders</summary>\n\n")
.append("| Language | Dictionary mode | Metric | Leader | Value |\n|---|---|---|---|---:|\n")
.append(cases).append("\n</details>\n");
}
/** Renders coverage-aware placement statistics for one dictionary mode. */
private static void renderPlacementSummary(final StringBuilder output, final Map<String, Page> pages,
final List<ResultRow> rows, final String mode) {
final Map<String, List<Integer>> ranks = new HashMap<>();
final Map<String, Integer> wins = new HashMap<>();
final Map<String, Integer> ties = new HashMap<>();
final Map<String, Integer> topThree = new HashMap<>();
for (String language : pages.keySet()) {
final List<ResultRow> ranked = primaryRows(rows, language, mode);
final double leading = ranked.getFirst().number("Balanced accuracy");
final long leaders = ranked.stream().filter(row -> row.number("Balanced accuracy") == leading).count();
for (int index = 0; index < ranked.size(); index++) {
final ResultRow row = ranked.get(index);
final String name = displayStemmer(row.stemmer());
ranks.computeIfAbsent(name, ignored -> new ArrayList<>()).add(index + 1);
if (row.number("Balanced accuracy") == leading) {
wins.merge(name, 1, Integer::sum);
if (leaders > 1) {
ties.merge(name, 1, Integer::sum);
}
}
if (index < 3) {
topThree.merge(name, 1, Integer::sum);
}
}
}
output.append("<details class=\"quality-details\" markdown=\"1\"><summary>").append(mode).append(" placements</summary>\n\n")
.append("| Stemmer | Evaluated languages | Wins | Exact first-place ties | Top-three placements | Average rank | Median rank |\n")
.append("|---|---:|---:|---:|---:|---:|---:|\n");
final List<String> names = ranks.keySet().stream().sorted(Comparator
.comparingInt((String name) -> wins.getOrDefault(name, 0)).reversed()
.thenComparing(Comparator.comparingInt((String name) -> ranks.get(name).size()).reversed())
.thenComparing(name -> name)).toList();
for (String name : names) {
final List<Integer> placements = ranks.get(name).stream().sorted().toList();
final double average = placements.stream().mapToInt(Integer::intValue).average().orElseThrow();
final int middle = placements.size() / 2;
final double median = placements.size() % 2 == 0
? (placements.get(middle - 1) + placements.get(middle)) / 2.0 : placements.get(middle);
output.append('|').append(name).append('|').append(placements.size()).append('|').append(wins.getOrDefault(name, 0)).append('|')
.append(ties.getOrDefault(name, 0)).append('|').append(topThree.getOrDefault(name, 0)).append('|')
.append(String.format(Locale.ROOT, "%.3f", average)).append('|').append(String.format(Locale.ROOT, "%.3f", median)).append("|\n");
}
output.append("\n</details>\n\n");
}
/** Returns deterministically ranked primary-output rows for one language and mode. */
private static List<ResultRow> primaryRows(final List<ResultRow> rows, final String language, final String mode) {
return rows.stream().filter(row -> row.language().equals(language) && row.mode().equals(mode)
&& row.policy().equals("PRIMARY_OUTPUT")).sorted(resultOrder()).toList();
}
/** Formats an aggregate metric at the publication precision. */
private static String format(final double value) {
return String.format(Locale.ROOT, "%.6f", value);
}
/** Returns the deterministic publication order based on unrounded source values. */
private static Comparator<ResultRow> resultOrder() {
return Comparator.comparingDouble((ResultRow row) -> row.number("Balanced accuracy")).reversed()
.thenComparing(Comparator.comparingDouble((ResultRow row) -> row.number("Matthews correlation coefficient")).reversed())
.thenComparing(Comparator.comparingDouble((ResultRow row) -> row.number("Pairwise F1")).reversed())
.thenComparingDouble(row -> row.number("Over-stemming percentage"))
.thenComparingLong(row -> row.longValue("Over-stemming error pairs"))
.thenComparingDouble(row -> row.number("Under-stemming percentage"))
.thenComparing(ResultRow::stemmer).thenComparingInt(row -> POLICY_ORDER.get(row.policy()));
}
/** Formats a score to the publication-wide six-decimal precision. */
private static String metric(final ResultRow row, final String name) {
final String value = row.value(name);
return value.isEmpty() ? "n/a" : String.format(Locale.ROOT, "%.6f", Double.parseDouble(value));
}
/** Formats one raw error numerator, denominator, and percentage. */
private static String pair(final ResultRow row, final String error, final String possible, final String percentage) {
final String rate = row.value(percentage);
return row.value(error) + " / " + row.value(possible) + " (" + (rate.isEmpty() ? "n/a" : String.format(Locale.ROOT, "%.6f%%", Double.parseDouble(rate))) + ")";
}
/** Replaces an existing marked section or appends the first generated section. */
private static String replaceSection(final String original, final String section) {
return replaceMarkedSection(original, section, START, END);
}
/** Replaces or appends a section delimited by the supplied deterministic markers. */
private static String replaceMarkedSection(final String original, final String section, final String startMarker,
final String endMarker) {
final int start = original.indexOf(startMarker);
final int end = original.indexOf(endMarker);
if ((start < 0) != (end < 0) || (start >= 0 && end < start)) {
throw new IllegalStateException("Malformed stemming-quality generated-section markers.");
}
if (start < 0) {
return original.stripTrailing() + "\n\n" + section;
}
return original.substring(0, start) + section + original.substring(end + endMarker.length()).stripLeading();
}
/** Calculates a lowercase hexadecimal SHA-256 checksum. */
private static String sha256(final Path source) throws IOException {
try {
final byte[] digest = MessageDigest.getInstance("SHA-256").digest(Files.readAllBytes(source));
final StringBuilder text = new StringBuilder(digest.length * 2);
for (byte value : digest) {
text.append(String.format(Locale.ROOT, "%02x", value & 0xff));
}
return text.toString();
} catch (NoSuchAlgorithmException exception) {
throw new IllegalStateException("The required SHA-256 algorithm is unavailable.", exception);
}
}
/** Immutable mapping from a language identifier to its existing page. */
private record Page(String language, String displayName, String file) { }
/** Immutable view of one authoritative CSV row. */
private record ResultRow(List<String> values, Map<String, Integer> indexes) {
/** Creates and validates an immutable row view. */
private ResultRow {
values = List.copyOf(values);
indexes = Map.copyOf(indexes);
}
/** Returns a field by its exact English header. */
private String value(final String name) { return this.values.get(this.indexes.get(name)); }
/** Returns the stemmer identifier. */
private String stemmer() { return value("Stemmer"); }
/** Returns the language identifier. */
private String language() { return value("Language"); }
/** Returns the dictionary-processing mode. */
private String mode() { return value("Dictionary mode"); }
/** Returns the output policy. */
private String policy() { return value("Output policy"); }
/** Returns a unique scenario key. */
private String key() { return stemmer() + "/" + language() + "/" + mode() + "/" + policy(); }
/** Parses a required long field. */
private long longValue(final String name) { return Long.parseLong(value(name)); }
/** Parses a numeric field, placing undefined values last during sorting. */
private double number(final String name) { return value(name).isEmpty() ? Double.NEGATIVE_INFINITY : Double.parseDouble(value(name)); }
/** Returns false-negative pairs. */
private long fn() { return longValue("False-negative pairs"); }
/** Returns false-positive pairs. */
private long fp() { return longValue("False-positive pairs"); }
/** Validates raw confusion counts and the published balanced accuracy. */
private void validate() {
final long tp = longValue("True-positive pairs");
final long fp = fp();
final long fn = fn();
final long tn = longValue("True-negative pairs");
if (fn != longValue("Under-stemming error pairs") || fp != longValue("Over-stemming error pairs")
|| Math.addExact(tp, fn) != longValue("Under-stemming possible pairs")
|| Math.addExact(tn, fp) != longValue("Over-stemming possible pairs")) {
throw new IllegalStateException("Raw pair-count invariants fail for " + key());
}
final double recall = ratio(tp, Math.addExact(tp, fn));
final double specificity = ratio(tn, Math.addExact(tn, fp));
final double expected = (recall + specificity) / 2.0;
if (Math.abs(expected - number("Balanced accuracy")) > 0.0000000000015) {
throw new IllegalStateException("Balanced accuracy is inconsistent with raw counts for " + key());
}
if (!policy().equals("PRIMARY_OUTPUT") && !value("Adjusted Rand Index").isEmpty()) {
throw new IllegalStateException("Partition-only metrics are present for a candidate relation: " + key());
}
}
/** Divides raw counts with explicit zero-denominator handling. */
private static double ratio(final long numerator, final long denominator) {
if (denominator == 0) {
throw new IllegalStateException("A balanced-accuracy component is undefined in a published result row.");
}
return (double) numerator / (double) denominator;
}
}
}