Compare commits
8 Commits
release@3.
...
model/fa-i
| Author | SHA1 | Date | |
|---|---|---|---|
|
62be4c9127
|
|||
|
e7800b29c9
|
|||
|
9c5b9e331b
|
|||
|
05f3855b99
|
|||
|
6d35f01303
|
|||
|
049f44e697
|
|||
|
a52e82933f
|
|||
|
5a65de21d9
|
2
.gitattributes
vendored
2
.gitattributes
vendored
@@ -9,4 +9,4 @@
|
|||||||
|
|
||||||
# Binary files should be left untouched
|
# Binary files should be left untouched
|
||||||
*.jar binary
|
*.jar binary
|
||||||
|
*.gz binary
|
||||||
|
|||||||
4
.github/workflows/benchmarks.yml
vendored
4
.github/workflows/benchmarks.yml
vendored
@@ -10,6 +10,8 @@ on:
|
|||||||
paths:
|
paths:
|
||||||
- 'src/main/**'
|
- 'src/main/**'
|
||||||
- 'src/jmh/**'
|
- 'src/jmh/**'
|
||||||
|
- 'models/**'
|
||||||
|
- 'build-logic/**'
|
||||||
- 'build.gradle'
|
- 'build.gradle'
|
||||||
- 'gradle.properties'
|
- 'gradle.properties'
|
||||||
- 'gradle.lockfile'
|
- 'gradle.lockfile'
|
||||||
@@ -56,7 +58,7 @@ jobs:
|
|||||||
test -f gradle/verification-metadata.xml
|
test -f gradle/verification-metadata.xml
|
||||||
|
|
||||||
- name: Run JMH benchmarks
|
- name: Run JMH benchmarks
|
||||||
run: ./gradlew clean jmh -Pjmh.includes='.*StemmerComparisonBenchmark.*' --no-daemon
|
run: ./gradlew clean jmh -Pjmh.includes='.*EnglishStemmerComparisonBenchmark.*' --no-daemon
|
||||||
|
|
||||||
- name: Upload JMH reports
|
- name: Upload JMH reports
|
||||||
uses: actions/upload-artifact@v4
|
uses: actions/upload-artifact@v4
|
||||||
|
|||||||
26
.github/workflows/build.yml
vendored
26
.github/workflows/build.yml
vendored
@@ -51,7 +51,7 @@ jobs:
|
|||||||
test -f gradle/verification-metadata.xml
|
test -f gradle/verification-metadata.xml
|
||||||
|
|
||||||
- name: Execute build, tests, PMD, coverage, Javadoc, distribution packaging, and SBOM generation
|
- name: Execute build, tests, PMD, coverage, Javadoc, distribution packaging, and SBOM generation
|
||||||
run: ./gradlew --no-daemon clean ciRelease distZip pmdMain javadoc jacocoCiReleaseReport cyclonedxBom
|
run: ./gradlew --no-daemon clean ciRelease distZip pmdMain javadoc jacocoCiReleaseReport :cyclonedxDirectBom
|
||||||
|
|
||||||
- name: Upload SBOM
|
- name: Upload SBOM
|
||||||
if: always()
|
if: always()
|
||||||
@@ -156,11 +156,14 @@ jobs:
|
|||||||
test -f gradle.properties
|
test -f gradle.properties
|
||||||
test -f gradle/verification-metadata.xml
|
test -f gradle/verification-metadata.xml
|
||||||
|
|
||||||
|
- name: Validate exact core release tag
|
||||||
|
run: ./tools/parse-model-release-tag.sh "${GITHUB_REF_NAME}" .
|
||||||
|
|
||||||
- name: Build release inputs, signed Maven bundle, and SBOM
|
- name: Build release inputs, signed Maven bundle, and SBOM
|
||||||
env:
|
env:
|
||||||
SIGNING_KEY: ${{ secrets.SIGNING_KEY }}
|
SIGNING_KEY: ${{ secrets.SIGNING_KEY }}
|
||||||
SIGNING_PASSWORD: ${{ secrets.SIGNING_PASSWORD }}
|
SIGNING_PASSWORD: ${{ secrets.SIGNING_PASSWORD }}
|
||||||
run: ./gradlew --no-daemon clean ciRelease distZip pmdMain javadoc jacocoCiReleaseReport cyclonedxBom centralBundle
|
run: ./gradlew --no-daemon clean ciRelease distZip pmdMain javadoc jacocoCiReleaseReport :cyclonedxDirectBom centralBundle
|
||||||
|
|
||||||
- name: Generate release changelog
|
- name: Generate release changelog
|
||||||
shell: bash
|
shell: bash
|
||||||
@@ -177,24 +180,7 @@ jobs:
|
|||||||
shell: bash
|
shell: bash
|
||||||
env:
|
env:
|
||||||
CENTRAL_BEARER_TOKEN: ${{ secrets.CENTRAL_BEARER_TOKEN }}
|
CENTRAL_BEARER_TOKEN: ${{ secrets.CENTRAL_BEARER_TOKEN }}
|
||||||
run: |
|
run: ./tools/publish-central-bundle.sh "$(ls build/central-bundle/*.zip)" "org.egothor:radixor:${GITHUB_REF_NAME#release@}"
|
||||||
set -euo pipefail
|
|
||||||
echo "::add-mask::$CENTRAL_BEARER_TOKEN"
|
|
||||||
|
|
||||||
BUNDLE="$(ls build/central-bundle/*.zip)"
|
|
||||||
HEADER_FILE="$(mktemp)"
|
|
||||||
trap 'rm -f "$HEADER_FILE"' EXIT
|
|
||||||
printf 'Authorization: Bearer %s\n' "$CENTRAL_BEARER_TOKEN" > "$HEADER_FILE"
|
|
||||||
|
|
||||||
curl \
|
|
||||||
--fail \
|
|
||||||
--silent \
|
|
||||||
--show-error \
|
|
||||||
--request POST \
|
|
||||||
--header @"$HEADER_FILE" \
|
|
||||||
--form "bundle=@${BUNDLE}" \
|
|
||||||
--form "name=org.egothor:radixor:${GITHUB_REF_NAME#release@}" \
|
|
||||||
"https://central.sonatype.com/api/v1/publisher/upload?publishingType=AUTOMATIC"
|
|
||||||
|
|
||||||
- name: Publish GitHub release assets
|
- name: Publish GitHub release assets
|
||||||
uses: softprops/action-gh-release@v2
|
uses: softprops/action-gh-release@v2
|
||||||
|
|||||||
37
.github/workflows/catalog-release.yml
vendored
Normal file
37
.github/workflows/catalog-release.yml
vendored
Normal file
@@ -0,0 +1,37 @@
|
|||||||
|
name: Model Catalog Release
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
tags:
|
||||||
|
- 'models-catalog@*'
|
||||||
|
|
||||||
|
permissions:
|
||||||
|
contents: read
|
||||||
|
|
||||||
|
concurrency:
|
||||||
|
group: model-catalog-${{ github.ref_name }}
|
||||||
|
cancel-in-progress: false
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
catalog:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
environment: maven-central
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
- uses: gradle/actions/wrapper-validation@v4
|
||||||
|
- uses: actions/setup-java@v4
|
||||||
|
with:
|
||||||
|
distribution: temurin
|
||||||
|
java-version: '21'
|
||||||
|
- uses: gradle/actions/setup-gradle@v4
|
||||||
|
- name: Validate catalog tag
|
||||||
|
run: ./tools/parse-model-release-tag.sh "${GITHUB_REF_NAME}" .
|
||||||
|
- name: Build only signed catalog metadata
|
||||||
|
env:
|
||||||
|
SIGNING_KEY: ${{ secrets.SIGNING_KEY }}
|
||||||
|
SIGNING_PASSWORD: ${{ secrets.SIGNING_PASSWORD }}
|
||||||
|
run: ./gradlew --no-daemon verifyModelCatalogReleaseCandidate
|
||||||
|
- name: Publish only catalog metadata
|
||||||
|
env:
|
||||||
|
CENTRAL_BEARER_TOKEN: ${{ secrets.CENTRAL_BEARER_TOKEN }}
|
||||||
|
run: ./tools/publish-central-bundle.sh "build/model-catalog-release-candidate/radixor-models-catalog-${GITHUB_REF_NAME#models-catalog@}-central-bundle.zip" "org.egothor:radixor-models-catalog:${GITHUB_REF_NAME#models-catalog@}"
|
||||||
147
.github/workflows/model-release.yml
vendored
Normal file
147
.github/workflows/model-release.yml
vendored
Normal file
@@ -0,0 +1,147 @@
|
|||||||
|
name: Model Release
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
tags:
|
||||||
|
- 'model/*@*'
|
||||||
|
workflow_dispatch:
|
||||||
|
inputs:
|
||||||
|
tag:
|
||||||
|
description: Model tag to validate without publishing
|
||||||
|
required: true
|
||||||
|
type: string
|
||||||
|
|
||||||
|
permissions:
|
||||||
|
contents: read
|
||||||
|
|
||||||
|
concurrency:
|
||||||
|
group: model-release-${{ github.event_name == 'push' && github.ref_name || inputs.tag }}
|
||||||
|
cancel-in-progress: false
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
validate:
|
||||||
|
name: Validate selected model
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
outputs:
|
||||||
|
model_id: ${{ steps.release.outputs.MODEL_ID }}
|
||||||
|
model_version: ${{ steps.release.outputs.MODEL_VERSION }}
|
||||||
|
gradle_project: ${{ steps.release.outputs.GRADLE_PROJECT }}
|
||||||
|
|
||||||
|
steps:
|
||||||
|
- name: Check out repository
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
with:
|
||||||
|
fetch-depth: 0
|
||||||
|
|
||||||
|
- name: Validate Gradle wrapper
|
||||||
|
uses: gradle/actions/wrapper-validation@v4
|
||||||
|
|
||||||
|
- name: Set up Temurin JDK 21
|
||||||
|
uses: actions/setup-java@v4
|
||||||
|
with:
|
||||||
|
distribution: temurin
|
||||||
|
java-version: '21'
|
||||||
|
|
||||||
|
- name: Set up Gradle caching and instrumentation
|
||||||
|
uses: gradle/actions/setup-gradle@v4
|
||||||
|
|
||||||
|
- name: Verify reproducibility inputs
|
||||||
|
shell: bash
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
test -f gradle.lockfile
|
||||||
|
test -f gradle.properties
|
||||||
|
test -f gradle/verification-metadata.xml
|
||||||
|
|
||||||
|
- name: Validate and select exactly one model
|
||||||
|
id: release
|
||||||
|
shell: bash
|
||||||
|
env:
|
||||||
|
REQUESTED_TAG: ${{ inputs.tag }}
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
if [[ "${GITHUB_EVENT_NAME}" == "push" ]]; then
|
||||||
|
tag="${GITHUB_REF_NAME}"
|
||||||
|
else
|
||||||
|
tag="${REQUESTED_TAG}"
|
||||||
|
fi
|
||||||
|
|
||||||
|
./tools/parse-model-release-tag.sh "${tag}" . >> "${GITHUB_OUTPUT}"
|
||||||
|
git merge-base --is-ancestor "${GITHUB_SHA}" origin/main
|
||||||
|
|
||||||
|
- name: Validate one model
|
||||||
|
shell: bash
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
project="${{ steps.release.outputs.GRADLE_PROJECT }}"
|
||||||
|
version="${{ steps.release.outputs.MODEL_VERSION }}"
|
||||||
|
|
||||||
|
./gradlew --no-daemon "${project}:clean"
|
||||||
|
./gradlew --no-daemon "${project}:check"
|
||||||
|
./gradlew --no-daemon \
|
||||||
|
"${project}:validateModelRelease" \
|
||||||
|
-PmodelReleaseVersion="${version}"
|
||||||
|
|
||||||
|
publish:
|
||||||
|
name: Publish selected model
|
||||||
|
if: github.event_name == 'push'
|
||||||
|
needs: validate
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
environment: maven-central
|
||||||
|
|
||||||
|
steps:
|
||||||
|
- name: Check out repository
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
with:
|
||||||
|
fetch-depth: 0
|
||||||
|
|
||||||
|
- name: Validate Gradle wrapper
|
||||||
|
uses: gradle/actions/wrapper-validation@v4
|
||||||
|
|
||||||
|
- name: Set up Temurin JDK 21
|
||||||
|
uses: actions/setup-java@v4
|
||||||
|
with:
|
||||||
|
distribution: temurin
|
||||||
|
java-version: '21'
|
||||||
|
|
||||||
|
- name: Set up Gradle caching and instrumentation
|
||||||
|
uses: gradle/actions/setup-gradle@v4
|
||||||
|
|
||||||
|
- name: Verify reproducibility inputs
|
||||||
|
shell: bash
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
test -f gradle.lockfile
|
||||||
|
test -f gradle.properties
|
||||||
|
test -f gradle/verification-metadata.xml
|
||||||
|
|
||||||
|
- name: Build signed model release candidate
|
||||||
|
shell: bash
|
||||||
|
env:
|
||||||
|
SIGNING_KEY: ${{ secrets.SIGNING_KEY }}
|
||||||
|
SIGNING_PASSWORD: ${{ secrets.SIGNING_PASSWORD }}
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
project="${{ needs.validate.outputs.gradle_project }}"
|
||||||
|
version="${{ needs.validate.outputs.model_version }}"
|
||||||
|
|
||||||
|
./gradlew --no-daemon \
|
||||||
|
"${project}:packageModelReleaseCandidate" \
|
||||||
|
-PmodelReleaseVersion="${version}"
|
||||||
|
|
||||||
|
- name: Publish one model
|
||||||
|
shell: bash
|
||||||
|
env:
|
||||||
|
CENTRAL_BEARER_TOKEN: ${{ secrets.CENTRAL_BEARER_TOKEN }}
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
model_id="${{ needs.validate.outputs.model_id }}"
|
||||||
|
version="${{ needs.validate.outputs.model_version }}"
|
||||||
|
|
||||||
|
./tools/publish-central-bundle.sh \
|
||||||
|
"models/${model_id}/build/model-release-candidate/central-bundle.zip" \
|
||||||
|
"org.egothor:radixor-model-${model_id}:${version}"
|
||||||
33
.github/workflows/pages.yml
vendored
33
.github/workflows/pages.yml
vendored
@@ -10,6 +10,8 @@ on:
|
|||||||
- 'src/main/**'
|
- 'src/main/**'
|
||||||
- 'src/test/**'
|
- 'src/test/**'
|
||||||
- 'src/jmh/**'
|
- 'src/jmh/**'
|
||||||
|
- 'models/**'
|
||||||
|
- 'build-logic/**'
|
||||||
- 'build.gradle'
|
- 'build.gradle'
|
||||||
- 'gradle.properties'
|
- 'gradle.properties'
|
||||||
- 'gradle.lockfile'
|
- 'gradle.lockfile'
|
||||||
@@ -70,7 +72,7 @@ jobs:
|
|||||||
test -f gradle/verification-metadata.xml
|
test -f gradle/verification-metadata.xml
|
||||||
|
|
||||||
- name: Build reports for publication
|
- name: Build reports for publication
|
||||||
run: ./gradlew --no-daemon clean ciRelease pmdMain javadoc jacocoCiReleaseReport pitest jmh -Pjmh.includes='.*StemmerComparisonBenchmark.*' cyclonedxBom
|
run: ./gradlew --no-daemon clean ciRelease pmdMain javadoc jacocoCiReleaseReport pitest jmh -Pjmh.includes='.*EnglishStemmerComparisonBenchmark.*' :cyclonedxDirectBom
|
||||||
|
|
||||||
- name: Prepare gh-pages worktree
|
- name: Prepare gh-pages worktree
|
||||||
shell: bash
|
shell: bash
|
||||||
@@ -88,6 +90,9 @@ jobs:
|
|||||||
cd ..
|
cd ..
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
- name: Prepare staged MkDocs source
|
||||||
|
run: ./gradlew --no-daemon prepareMkDocsSource verifyModelCatalogDocumentation
|
||||||
|
|
||||||
- name: Stage published reports
|
- name: Stage published reports
|
||||||
shell: bash
|
shell: bash
|
||||||
run: |
|
run: |
|
||||||
@@ -246,7 +251,7 @@ jobs:
|
|||||||
|
|
||||||
cp "${RUN_DIR}/index.html" "${LATEST_DIR}/index.html"
|
cp "${RUN_DIR}/index.html" "${LATEST_DIR}/index.html"
|
||||||
|
|
||||||
cat > docs/reports.md <<EOF
|
cat > build/mkdocs-source/reports.md <<EOF
|
||||||
# CI Reports
|
# CI Reports
|
||||||
|
|
||||||
Radixor publishes durable CI artifacts to GitHub Pages on every qualifying run of \`.github/workflows/pages.yml\`.
|
Radixor publishes durable CI artifacts to GitHub Pages on every qualifying run of \`.github/workflows/pages.yml\`.
|
||||||
@@ -275,11 +280,26 @@ jobs:
|
|||||||
- [Browse historical build reports](https://leogalambos.github.io/Radixor/builds/)
|
- [Browse historical build reports](https://leogalambos.github.io/Radixor/builds/)
|
||||||
EOF
|
EOF
|
||||||
|
|
||||||
|
# Retain only the 10 most recent numbered builds to stay within
|
||||||
|
# GitHub Pages capacity limits. The "latest" alias is kept separately.
|
||||||
|
mapfile -t EXPIRED_BUILDS < <(
|
||||||
|
find "${SITE_DIR}/builds" -mindepth 1 -maxdepth 1 -type d -printf '%P\n' \
|
||||||
|
| grep -E '^[0-9]+$' \
|
||||||
|
| sort -r -n \
|
||||||
|
| tail -n +11
|
||||||
|
)
|
||||||
|
|
||||||
|
for build in "${EXPIRED_BUILDS[@]}"; do
|
||||||
|
rm -rf "${SITE_DIR}/builds/${build}"
|
||||||
|
done
|
||||||
|
|
||||||
{
|
{
|
||||||
echo "# Historical Build Reports"
|
echo "# Historical Build Reports"
|
||||||
echo
|
echo
|
||||||
echo "The following build report sets are currently published on GitHub Pages."
|
echo "The following build report sets are currently published on GitHub Pages."
|
||||||
echo
|
echo
|
||||||
|
echo "To stay within GitHub Pages capacity limits, only the 10 most recent build report sets are retained."
|
||||||
|
echo
|
||||||
echo "| Build | Published | Link |"
|
echo "| Build | Published | Link |"
|
||||||
echo "|---:|---|---|"
|
echo "|---:|---|---|"
|
||||||
|
|
||||||
@@ -299,19 +319,18 @@ jobs:
|
|||||||
| while IFS=$'\t' read -r _ts build published; do
|
| while IFS=$'\t' read -r _ts build published; do
|
||||||
echo "| ${build} | ${published} | [Open](../builds/${build}/) |"
|
echo "| ${build} | ${published} | [Open](../builds/${build}/) |"
|
||||||
done
|
done
|
||||||
} > docs/builds.md
|
} > build/mkdocs-source/builds.md
|
||||||
|
|
||||||
- name: Build documentation site (MkDocs Material)
|
- name: Build documentation site (MkDocs Material)
|
||||||
shell: bash
|
shell: bash
|
||||||
run: |
|
run: |
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
mkdocs build --strict --site-dir .mkdocs-site
|
mkdocs build --strict --config-file build/mkdocs/mkdocs.yml
|
||||||
rsync -a --delete --exclude '.git' --exclude '.git/' --exclude 'builds/' .mkdocs-site/ .gh-pages/
|
rsync -a --delete --exclude '.git' --exclude '.git/' --exclude 'builds/' build/mkdocs-site/ .gh-pages/
|
||||||
mkdir -p .gh-pages/builds
|
mkdir -p .gh-pages/builds
|
||||||
cp .mkdocs-site/builds/index.html .gh-pages/builds/index.html
|
cp build/mkdocs-site/builds/index.html .gh-pages/builds/index.html
|
||||||
cat > .gh-pages/.nojekyll <<EOF
|
cat > .gh-pages/.nojekyll <<EOF
|
||||||
EOF
|
EOF
|
||||||
rm -rf .mkdocs-site
|
|
||||||
|
|
||||||
- name: Commit and push gh-pages
|
- name: Commit and push gh-pages
|
||||||
shell: bash
|
shell: bash
|
||||||
|
|||||||
15
.gitignore
vendored
15
.gitignore
vendored
@@ -37,6 +37,7 @@ local.properties
|
|||||||
.settings/
|
.settings/
|
||||||
.loadpath
|
.loadpath
|
||||||
.recommenders
|
.recommenders
|
||||||
|
.classpath
|
||||||
|
|
||||||
# External tool builders
|
# External tool builders
|
||||||
.externalToolBuilders/
|
.externalToolBuilders/
|
||||||
@@ -94,19 +95,17 @@ local.properties
|
|||||||
.jqwik-database
|
.jqwik-database
|
||||||
|
|
||||||
##---------------------------------------------------------------------------------------- Gradle
|
##---------------------------------------------------------------------------------------- Gradle
|
||||||
.gradle
|
.gradle/
|
||||||
**/build/
|
**/build/
|
||||||
!src/**/build/
|
|
||||||
|
# MkDocs generated site
|
||||||
|
/site/
|
||||||
|
|
||||||
# Ignore Gradle GUI config
|
# Ignore Gradle GUI config
|
||||||
gradle-app.setting
|
gradle-app.setting
|
||||||
|
|
||||||
# Avoid ignoring Gradle wrapper jar file (.jar files are usually ignored)
|
# Avoid ignoring the Gradle Wrapper JAR
|
||||||
!gradle-wrapper.jar
|
!gradle-wrapper.jar
|
||||||
|
|
||||||
# Cache of project
|
# Gradle task-name cache
|
||||||
.gradletasknamecache
|
.gradletasknamecache
|
||||||
|
|
||||||
|
|
||||||
# Ignore Gradle build output directory
|
|
||||||
build
|
|
||||||
|
|||||||
61
README.md
61
README.md
@@ -22,6 +22,40 @@ It is particularly well suited to systems that need stemming which is:
|
|||||||
|
|
||||||
It also retains the operational advantages of a compiled artifact model: predictable runtime behavior, direct binary loading, and clear separation between preparation-time compilation and live request processing.
|
It also retains the operational advantages of a compiled artifact model: predictable runtime behavior, direct binary loading, and clear separation between preparation-time compilation and live request processing.
|
||||||
|
|
||||||
|
## Add Radixor and a model
|
||||||
|
|
||||||
|
The core artifact contains the algorithm and registry, but no language dictionary. Add either one minimal model or the optional standard default pack:
|
||||||
|
|
||||||
|
```groovy
|
||||||
|
dependencies {
|
||||||
|
implementation 'org.egothor:radixor:<radixor-version>'
|
||||||
|
runtimeOnly 'org.egothor:radixor-model-pl-pl-unimorph:1.0.0'
|
||||||
|
// Or: runtimeOnly 'org.egothor:radixor-models-standard:<catalog-version>'
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
```java
|
||||||
|
final FrequencyTrie<CompiledPatchCommand> polish =
|
||||||
|
StemmerPatchTrieLoader.loadCompiled(
|
||||||
|
StemmerPatchTrieLoader.Language.PL_PL,
|
||||||
|
true,
|
||||||
|
ReductionMode.MERGE_SUBTREES_WITH_EQUIVALENT_RANKED_GET_ALL_RESULTS);
|
||||||
|
```
|
||||||
|
|
||||||
|
`Language.PL_PL` selects the documented default `pl-pl-unimorph`. The optional `pl-pl-polimorf` model requires its own runtime artifact and explicit selection; adding it does not change the default. See [Model Selection and Loading](docs/model-selection-and-loading.md) for complete executable examples and [Stemmer Models](docs/stemmer-models.md) for artifact concepts.
|
||||||
|
|
||||||
|
`radixor-models-standard` is a POM-only runtime aggregate: it brings the 20 default model JARs transitively but publishes no empty aggregate JAR. `radixor-models-bom` is the separate POM-only Maven dependency BOM for version management; importing it alone adds no model. The root CycloneDX SBOM report is unrelated to that dependency BOM.
|
||||||
|
|
||||||
|
```java
|
||||||
|
final FrequencyTrie<CompiledPatchCommand> polimorf =
|
||||||
|
StemmerPatchTrieLoader.loadCompiled(
|
||||||
|
"pl-pl-polimorf",
|
||||||
|
true,
|
||||||
|
ReductionMode.MERGE_SUBTREES_WITH_EQUIVALENT_RANKED_GET_ALL_RESULTS);
|
||||||
|
```
|
||||||
|
|
||||||
|
Complete PoliMorf construction is supported but unusually memory-intensive: the dedicated verification task uses a 6 GiB maximum heap. Applications should load and retain the resulting immutable trie during startup rather than rebuilding it per request.
|
||||||
|
|
||||||
## Table of Contents
|
## Table of Contents
|
||||||
|
|
||||||
- [Why Radixor](#why-radixor)
|
- [Why Radixor](#why-radixor)
|
||||||
@@ -138,7 +172,7 @@ Compared with the historical baseline, Radixor emphasizes:
|
|||||||
- Compressed binary persistence
|
- Compressed binary persistence
|
||||||
- Programmatic compilation and loading
|
- Programmatic compilation and loading
|
||||||
- CLI compilation tool
|
- CLI compilation tool
|
||||||
- Bundled language resources
|
- Independently versioned language-model resources
|
||||||
- Support for extending compiled stemmer tables
|
- Support for extending compiled stemmer tables
|
||||||
- Reproducible and auditable engineering posture
|
- Reproducible and auditable engineering posture
|
||||||
|
|
||||||
@@ -149,16 +183,16 @@ The repository keeps the front page concise and places detailed documentation un
|
|||||||
### Getting Started
|
### Getting Started
|
||||||
|
|
||||||
- [Fast Track](docs/fast-track.md)
|
- [Fast Track](docs/fast-track.md)
|
||||||
The shortest path from adding the dependency to getting a first stem from a bundled dictionary.
|
The shortest path from adding core plus a model artifact to getting a first stem.
|
||||||
|
|
||||||
- [Quick Start](docs/quick-start.md)
|
- [Quick Start](docs/quick-start.md)
|
||||||
A broader developer walkthrough covering loading options, querying, extension, persistence, and metadata.
|
A broader developer walkthrough covering loading options, querying, extension, persistence, and metadata.
|
||||||
|
|
||||||
- [Integration Deep Dive](docs/integration-deep-dive.md)
|
- [Integration Deep Dive](docs/integration-deep-dive.md)
|
||||||
Dependency setup, bundled dictionary selection, production lifecycle, search-pipeline guidance, and operational checklist.
|
Dependency setup, model selection, production lifecycle, search-pipeline guidance, and operational checklist.
|
||||||
|
|
||||||
- [Built-in Languages](docs/built-in-languages.md)
|
- [Built-in Languages](docs/built-in-languages.md)
|
||||||
Overview of bundled language resources such as `US_UK`.
|
Language enum values, default model IDs, artifacts, and optional variants.
|
||||||
|
|
||||||
- [Dictionary Format](docs/dictionary-format.md)
|
- [Dictionary Format](docs/dictionary-format.md)
|
||||||
How to write and normalize stemming dictionaries.
|
How to write and normalize stemming dictionaries.
|
||||||
@@ -171,6 +205,9 @@ The repository keeps the front page concise and places detailed documentation un
|
|||||||
- [Programmatic Usage Overview](docs/programmatic-usage.md)
|
- [Programmatic Usage Overview](docs/programmatic-usage.md)
|
||||||
Entry point to the Java API and the overall usage model.
|
Entry point to the Java API and the overall usage model.
|
||||||
|
|
||||||
|
- [Model Selection and Loading](docs/model-selection-and-loading.md)
|
||||||
|
Default, explicit, dual-model, ClassLoader, dependency, and troubleshooting examples.
|
||||||
|
|
||||||
- [Loading and Building Stemmers](docs/programmatic-loading-and-building.md)
|
- [Loading and Building Stemmers](docs/programmatic-loading-and-building.md)
|
||||||
Loading bundled resources, textual dictionaries, binary artifacts, and direct builder usage.
|
Loading bundled resources, textual dictionaries, binary artifacts, and direct builder usage.
|
||||||
|
|
||||||
@@ -245,3 +282,19 @@ The goal is to keep the Egothor/Stempel lineage useful as a serious contemporary
|
|||||||
## Historical note
|
## Historical note
|
||||||
|
|
||||||
Egothor showed that stemming could be both algorithmic and compact. Stempel proved that the approach was practical enough to survive inside major search ecosystems. Radixor continues that tradition with a modernized implementation focused on production use, maintainability, and controlled evolution.
|
Egothor showed that stemming could be both algorithmic and compact. Stempel proved that the approach was practical enough to survive inside major search ecosystems. Radixor continues that tradition with a modernized implementation focused on production use, maintainability, and controlled evolution.
|
||||||
|
# Radixor 4 artifact architecture
|
||||||
|
|
||||||
|
The established `org.egothor:radixor` artifact remains the algorithmic core and contains no language-model data. From version 4 onward, applications explicitly add individual `org.egothor:radixor-model-<model-id>` runtime artifacts or the optional metadata-only `org.egothor:radixor-models-standard` aggregate. Polish defaults to `pl-pl-unimorph`; `pl-pl-polimorf` is opt-in. See [Stemmer Models](docs/stemmer-models.md) and [Migration and Backward Compatibility](docs/migration-and-backward-compatibility.md).
|
||||||
|
|
||||||
|
Radixor Java software remains licensed under BSD-3-Clause. UniMorph-derived model data is
|
||||||
|
distributed under CC BY-SA 3.0, with upstream attribution, the canonical license URI, Radixor
|
||||||
|
transformations, and Leo Galambos's limited contribution notice carried by each model artifact.
|
||||||
|
PoliMorf model data retains its separate BSD-2-Clause license. There is no project-wide CC license
|
||||||
|
directory because the root artifact contains no model data.
|
||||||
|
|
||||||
|
```groovy
|
||||||
|
dependencies {
|
||||||
|
implementation 'org.egothor:radixor:4.0.0'
|
||||||
|
runtimeOnly 'org.egothor:radixor-model-pl-pl-polimorf:1.0.0'
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|||||||
25
build-logic/build.gradle
Normal file
25
build-logic/build.gradle
Normal file
@@ -0,0 +1,25 @@
|
|||||||
|
plugins {
|
||||||
|
id 'groovy-gradle-plugin'
|
||||||
|
}
|
||||||
|
|
||||||
|
dependencies {
|
||||||
|
testImplementation 'org.junit.jupiter:junit-jupiter:5.14.3'
|
||||||
|
testRuntimeOnly 'org.junit.platform:junit-platform-launcher:1.14.3'
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.named('test') {
|
||||||
|
useJUnitPlatform()
|
||||||
|
}
|
||||||
|
|
||||||
|
gradlePlugin {
|
||||||
|
plugins {
|
||||||
|
radixorModel {
|
||||||
|
id = 'org.egothor.radixor.model'
|
||||||
|
implementationClass = 'org.egothor.radixor.RadixorModelPlugin'
|
||||||
|
}
|
||||||
|
radixorBuildSupport {
|
||||||
|
id = 'org.egothor.radixor.build-support'
|
||||||
|
implementationClass = 'org.egothor.radixor.RadixorBuildSupportPlugin'
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
8
build-logic/settings.gradle
Normal file
8
build-logic/settings.gradle
Normal file
@@ -0,0 +1,8 @@
|
|||||||
|
rootProject.name = 'radixor-build-logic'
|
||||||
|
|
||||||
|
dependencyResolutionManagement {
|
||||||
|
repositories {
|
||||||
|
gradlePluginPortal()
|
||||||
|
mavenCentral()
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,21 @@
|
|||||||
|
package org.egothor.radixor
|
||||||
|
|
||||||
|
import org.gradle.api.file.ConfigurableFileCollection
|
||||||
|
import org.gradle.api.tasks.Classpath
|
||||||
|
import org.gradle.process.CommandLineArgumentProvider
|
||||||
|
|
||||||
|
import javax.inject.Inject
|
||||||
|
|
||||||
|
abstract class MockitoAgentArgumentProvider implements CommandLineArgumentProvider {
|
||||||
|
@Classpath
|
||||||
|
abstract ConfigurableFileCollection getAgentClasspath()
|
||||||
|
|
||||||
|
@Inject
|
||||||
|
MockitoAgentArgumentProvider() {
|
||||||
|
}
|
||||||
|
|
||||||
|
@Override
|
||||||
|
Iterable<String> asArguments() {
|
||||||
|
return ["-javaagent:${agentClasspath.singleFile.absolutePath}"]
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,105 @@
|
|||||||
|
package org.egothor.radixor
|
||||||
|
|
||||||
|
import org.gradle.api.DefaultTask
|
||||||
|
import org.gradle.api.GradleException
|
||||||
|
import org.gradle.api.file.ConfigurableFileCollection
|
||||||
|
import org.gradle.api.file.DirectoryProperty
|
||||||
|
import org.gradle.api.file.RegularFileProperty
|
||||||
|
import org.gradle.api.provider.MapProperty
|
||||||
|
import org.gradle.api.provider.Property
|
||||||
|
import org.gradle.api.tasks.Input
|
||||||
|
import org.gradle.api.tasks.InputFile
|
||||||
|
import org.gradle.api.tasks.InputFiles
|
||||||
|
import org.gradle.api.tasks.OutputDirectory
|
||||||
|
import org.gradle.api.tasks.PathSensitive
|
||||||
|
import org.gradle.api.tasks.PathSensitivity
|
||||||
|
import org.gradle.api.tasks.TaskAction
|
||||||
|
|
||||||
|
import java.nio.file.Files
|
||||||
|
import java.nio.file.Path
|
||||||
|
import java.nio.file.StandardCopyOption
|
||||||
|
import java.util.stream.Stream
|
||||||
|
|
||||||
|
/** Builds the isolated Maven-layout repository used by consumer resolution tests. */
|
||||||
|
abstract class PrepareModelConsumerRepositoryTask extends DefaultTask {
|
||||||
|
@Input abstract Property<String> getCoreVersion()
|
||||||
|
@Input abstract Property<String> getCatalogVersion()
|
||||||
|
@Input abstract MapProperty<String, String> getModelVersions()
|
||||||
|
|
||||||
|
@InputFile @PathSensitive(PathSensitivity.RELATIVE)
|
||||||
|
abstract RegularFileProperty getCorePom()
|
||||||
|
|
||||||
|
@InputFile @PathSensitive(PathSensitivity.RELATIVE)
|
||||||
|
abstract RegularFileProperty getCoreJar()
|
||||||
|
|
||||||
|
@InputFiles @PathSensitive(PathSensitivity.RELATIVE)
|
||||||
|
abstract ConfigurableFileCollection getModelPoms()
|
||||||
|
|
||||||
|
@InputFiles @PathSensitive(PathSensitivity.RELATIVE)
|
||||||
|
abstract ConfigurableFileCollection getModelJars()
|
||||||
|
|
||||||
|
@InputFile @PathSensitive(PathSensitivity.RELATIVE)
|
||||||
|
abstract RegularFileProperty getStandardPom()
|
||||||
|
|
||||||
|
@InputFile @PathSensitive(PathSensitivity.RELATIVE)
|
||||||
|
abstract RegularFileProperty getBomPom()
|
||||||
|
|
||||||
|
@OutputDirectory
|
||||||
|
abstract DirectoryProperty getRepositoryDirectory()
|
||||||
|
|
||||||
|
/** Creates the repository using only declared task state and Java file APIs. */
|
||||||
|
@TaskAction
|
||||||
|
void prepareRepository() {
|
||||||
|
final Path repository = repositoryDirectory.get().asFile.toPath()
|
||||||
|
deleteTree(repository)
|
||||||
|
Files.createDirectories(repository)
|
||||||
|
install(repository, 'radixor', coreVersion.get(), corePom.get().asFile.toPath(), coreJar.get().asFile.toPath())
|
||||||
|
|
||||||
|
final Map<String, Path> pomsByModel = indexModelFiles(modelPoms.files)
|
||||||
|
final Map<String, Path> jarsByModel = indexModelFiles(modelJars.files)
|
||||||
|
modelVersions.get().toSorted().each { String modelId, String modelVersion ->
|
||||||
|
final Path pom = pomsByModel.get(modelId)
|
||||||
|
final Path jar = jarsByModel.get(modelId)
|
||||||
|
if (pom == null || jar == null) {
|
||||||
|
throw new GradleException("Missing generated publication input for model ${modelId}.")
|
||||||
|
}
|
||||||
|
PrepareModelConsumerRepositoryTask.install(
|
||||||
|
repository, "radixor-model-${modelId}", modelVersion, pom, jar)
|
||||||
|
}
|
||||||
|
install(repository, 'radixor-models-standard', catalogVersion.get(), standardPom.get().asFile.toPath(), null)
|
||||||
|
install(repository, 'radixor-models-bom', catalogVersion.get(), bomPom.get().asFile.toPath(), null)
|
||||||
|
}
|
||||||
|
|
||||||
|
private static Map<String, Path> indexModelFiles(final Set<File> files) {
|
||||||
|
final Map<String, Path> indexed = [:]
|
||||||
|
files.each { File file ->
|
||||||
|
Path cursor = file.toPath().toAbsolutePath().parent
|
||||||
|
while (cursor != null && cursor.fileName.toString() != 'build') cursor = cursor.parent
|
||||||
|
if (cursor == null || cursor.parent == null) {
|
||||||
|
throw new GradleException("Cannot determine model ID from generated input ${file}.")
|
||||||
|
}
|
||||||
|
final String modelId = cursor.parent.fileName.toString()
|
||||||
|
if (indexed.put(modelId, file.toPath()) != null) {
|
||||||
|
throw new GradleException("Duplicate generated publication input for model ${modelId}.")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return indexed
|
||||||
|
}
|
||||||
|
|
||||||
|
private static void install(final Path repository, final String artifactId, final String version,
|
||||||
|
final Path pom, final Path jar) {
|
||||||
|
final Path module = repository.resolve("org/egothor/${artifactId}/${version}")
|
||||||
|
Files.createDirectories(module)
|
||||||
|
Files.copy(pom, module.resolve("${artifactId}-${version}.pom"), StandardCopyOption.REPLACE_EXISTING)
|
||||||
|
if (jar != null) {
|
||||||
|
Files.copy(jar, module.resolve("${artifactId}-${version}.jar"), StandardCopyOption.REPLACE_EXISTING)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static void deleteTree(final Path directory) {
|
||||||
|
if (!Files.exists(directory)) return
|
||||||
|
Files.walk(directory).withCloseable { Stream<Path> paths ->
|
||||||
|
paths.sorted(Comparator.reverseOrder()).forEach(Files::delete)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,108 @@
|
|||||||
|
package org.egothor.radixor
|
||||||
|
|
||||||
|
import org.gradle.api.DefaultTask
|
||||||
|
import org.gradle.api.file.DirectoryProperty
|
||||||
|
import org.gradle.api.file.RegularFileProperty
|
||||||
|
import org.gradle.api.provider.MapProperty
|
||||||
|
import org.gradle.api.provider.Property
|
||||||
|
import org.gradle.api.tasks.Input
|
||||||
|
import org.gradle.api.tasks.InputFile
|
||||||
|
import org.gradle.api.tasks.Optional
|
||||||
|
import org.gradle.api.tasks.OutputDirectory
|
||||||
|
import org.gradle.api.tasks.PathSensitive
|
||||||
|
import org.gradle.api.tasks.PathSensitivity
|
||||||
|
import org.gradle.api.tasks.TaskAction
|
||||||
|
|
||||||
|
import java.nio.file.Files
|
||||||
|
import java.nio.file.Path
|
||||||
|
import java.nio.file.StandardCopyOption
|
||||||
|
import java.security.MessageDigest
|
||||||
|
import java.util.stream.Stream
|
||||||
|
|
||||||
|
/** Generates one model's deterministic resource tree without retaining Project state. */
|
||||||
|
abstract class PrepareModelResourcesTask extends DefaultTask {
|
||||||
|
@InputFile @PathSensitive(PathSensitivity.RELATIVE) abstract RegularFileProperty getDictionaryFile()
|
||||||
|
@InputFile @PathSensitive(PathSensitivity.RELATIVE) abstract RegularFileProperty getVersionFile()
|
||||||
|
@Optional @InputFile @PathSensitive(PathSensitivity.RELATIVE) abstract RegularFileProperty getLicenseFile()
|
||||||
|
@Optional @InputFile @PathSensitive(PathSensitivity.RELATIVE) abstract RegularFileProperty getNoticeFile()
|
||||||
|
@Input abstract Property<Boolean> getShareAlike()
|
||||||
|
@Input abstract MapProperty<String, String> getDescriptorValues()
|
||||||
|
@OutputDirectory abstract DirectoryProperty getGeneratedDirectory()
|
||||||
|
|
||||||
|
/** Copies bounded inputs and writes descriptor and index files. */
|
||||||
|
@TaskAction
|
||||||
|
void prepareResources() {
|
||||||
|
final Path generated = generatedDirectory.get().asFile.toPath()
|
||||||
|
deleteTree(generated)
|
||||||
|
final Map<String, String> values = descriptorValues.get()
|
||||||
|
final String id = values['model.id']
|
||||||
|
final String resource = "org/egothor/stemmer/models/${id}/stemmer.gz"
|
||||||
|
final Path dictionaryTarget = generated.resolve(resource)
|
||||||
|
Files.createDirectories(dictionaryTarget.parent)
|
||||||
|
Files.copy(dictionaryFile.get().asFile.toPath(), dictionaryTarget, StandardCopyOption.REPLACE_EXISTING)
|
||||||
|
|
||||||
|
final Path descriptor = generated.resolve("META-INF/radixor/models/${id}.properties")
|
||||||
|
Files.createDirectories(descriptor.parent)
|
||||||
|
Files.writeString(descriptor, descriptorText(values,
|
||||||
|
versionFile.get().asFile.getText('UTF-8').trim(), resource, sha256(dictionaryFile.get().asFile)))
|
||||||
|
final Path index = generated.resolve('META-INF/radixor/models.index')
|
||||||
|
Files.createDirectories(index.parent)
|
||||||
|
Files.writeString(index, "META-INF/radixor/models/${id}.properties\n")
|
||||||
|
|
||||||
|
if (shareAlike.get()) {
|
||||||
|
final Path notice = generated.resolve("META-INF/NOTICE/${id}-data.txt")
|
||||||
|
Files.createDirectories(notice.parent)
|
||||||
|
Files.copy(noticeFile.get().asFile.toPath(), notice, StandardCopyOption.REPLACE_EXISTING)
|
||||||
|
} else {
|
||||||
|
final Path license = generated.resolve('META-INF/LICENSES/PoliMorf-BSD-2-Clause.txt')
|
||||||
|
Files.createDirectories(license.parent)
|
||||||
|
Files.copy(licenseFile.get().asFile.toPath(), license, StandardCopyOption.REPLACE_EXISTING)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static String descriptorText(final Map<String, String> value, final String version,
|
||||||
|
final String resource, final String checksum) {
|
||||||
|
return """model.id=${value['model.id']}
|
||||||
|
model.version=${version}
|
||||||
|
model.language=${value['model.language']}
|
||||||
|
model.displayName=${value['model.displayName']}
|
||||||
|
model.resource=${resource}
|
||||||
|
model.default=${value['model.default']}
|
||||||
|
model.format=radixor-dictionary-tsv-gzip
|
||||||
|
model.formatVersion=1
|
||||||
|
model.sha256=${checksum}
|
||||||
|
model.rightToLeft=${['FA_IR', 'HE_IL', 'YI'].contains(value['model.language'])}
|
||||||
|
model.caseProcessing=LOWERCASE_WITH_LOCALE_ROOT
|
||||||
|
model.diacriticProcessing=AS_IS
|
||||||
|
model.storeOriginal=true
|
||||||
|
source.name=${value['source.name']}
|
||||||
|
source.version=${value['source.version']}
|
||||||
|
source.project=${value['source.project']}
|
||||||
|
source.repository=${value['source.repository']}
|
||||||
|
source.dataset=${value['source.dataset']}
|
||||||
|
source.revision=${value['source.revision']}
|
||||||
|
source.revisionStatus=${value['source.revisionStatus']}
|
||||||
|
source.license=${value['source.license']}
|
||||||
|
source.licenseUri=${value['source.licenseUri']}
|
||||||
|
source.attribution=${value['source.attribution']}
|
||||||
|
source.verificationDate=${value['source.verificationDate']}
|
||||||
|
transformations.summary=${value['transformations.summary']}
|
||||||
|
compiler.radixorVersion=3.x
|
||||||
|
compiler.radixorCommit=unavailable
|
||||||
|
statistics.groups=unavailable
|
||||||
|
statistics.forms=unavailable
|
||||||
|
"""
|
||||||
|
}
|
||||||
|
|
||||||
|
private static String sha256(final File file) {
|
||||||
|
return MessageDigest.getInstance('SHA-256').digest(file.bytes)
|
||||||
|
.collect { byte value -> String.format('%02x', value & 0xff) }.join()
|
||||||
|
}
|
||||||
|
|
||||||
|
private static void deleteTree(final Path directory) {
|
||||||
|
if (!Files.exists(directory)) return
|
||||||
|
Files.walk(directory).withCloseable { Stream<Path> paths ->
|
||||||
|
paths.sorted(Comparator.reverseOrder()).forEach(Files::delete)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,16 @@
|
|||||||
|
package org.egothor.radixor
|
||||||
|
|
||||||
|
import org.gradle.api.Plugin
|
||||||
|
import org.gradle.api.Project
|
||||||
|
|
||||||
|
/** Exposes typed repository build-support tasks to the root build. */
|
||||||
|
final class RadixorBuildSupportPlugin implements Plugin<Project> {
|
||||||
|
/** Registers build-support tasks without inspecting project state during execution. */
|
||||||
|
@Override
|
||||||
|
void apply(final Project project) {
|
||||||
|
project.tasks.register('prepareModelConsumerTestRepository', PrepareModelConsumerRepositoryTask) {
|
||||||
|
group = 'verification'
|
||||||
|
description = 'Creates an isolated local Maven repository for model dependency-resolution integration tests.'
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,73 @@
|
|||||||
|
package org.egothor.radixor
|
||||||
|
|
||||||
|
import org.gradle.api.model.ObjectFactory
|
||||||
|
import org.gradle.api.provider.Property
|
||||||
|
|
||||||
|
import javax.inject.Inject
|
||||||
|
|
||||||
|
/** Declarative configuration for one independently published Radixor model. */
|
||||||
|
abstract class RadixorModelExtension {
|
||||||
|
/** Stable model identifier. */
|
||||||
|
abstract Property<String> getModelId()
|
||||||
|
|
||||||
|
/** Radixor language enum constant. */
|
||||||
|
abstract Property<String> getLanguage()
|
||||||
|
|
||||||
|
/** Human-readable model name. */
|
||||||
|
abstract Property<String> getDisplayName()
|
||||||
|
|
||||||
|
/** Whether this is the documented default for its language. */
|
||||||
|
abstract Property<Boolean> getDefaultModel()
|
||||||
|
|
||||||
|
/** Source dictionary name. */
|
||||||
|
abstract Property<String> getSourceName()
|
||||||
|
|
||||||
|
/** Source dictionary version or explicit unavailable marker. */
|
||||||
|
abstract Property<String> getSourceVersion()
|
||||||
|
|
||||||
|
/** Exact upstream revision or the explicit legacy-import sentinel. */
|
||||||
|
abstract Property<String> getSourceRevision()
|
||||||
|
|
||||||
|
/** Upstream source project. */
|
||||||
|
abstract Property<String> getSourceProject()
|
||||||
|
|
||||||
|
/** Official upstream repository URL. */
|
||||||
|
abstract Property<String> getSourceRepository()
|
||||||
|
|
||||||
|
/** Upstream dataset identity. */
|
||||||
|
abstract Property<String> getSourceDataset()
|
||||||
|
|
||||||
|
/** Whether the source revision is recorded or was not recorded by a legacy import. */
|
||||||
|
abstract Property<String> getSourceRevisionStatus()
|
||||||
|
|
||||||
|
/** SPDX license identifier. */
|
||||||
|
abstract Property<String> getSourceLicense()
|
||||||
|
|
||||||
|
/** Canonical URI for the source-data license. */
|
||||||
|
abstract Property<String> getSourceLicenseUri()
|
||||||
|
|
||||||
|
/** Upstream attribution supplied with the source data. */
|
||||||
|
abstract Property<String> getSourceAttribution()
|
||||||
|
|
||||||
|
/** Date on which the upstream metadata was verified. */
|
||||||
|
abstract Property<String> getSourceVerificationDate()
|
||||||
|
|
||||||
|
/** Material transformations applied by Radixor. */
|
||||||
|
abstract Property<String> getTransformationsSummary()
|
||||||
|
|
||||||
|
/** Model-specific data notice input file name, when required. */
|
||||||
|
abstract Property<String> getNoticeFileName()
|
||||||
|
|
||||||
|
/** License input file name. */
|
||||||
|
abstract Property<String> getLicenseFileName()
|
||||||
|
|
||||||
|
/** Creates the extension. */
|
||||||
|
@Inject
|
||||||
|
RadixorModelExtension(final ObjectFactory objects) {
|
||||||
|
defaultModel.convention(false)
|
||||||
|
sourceVersion.convention('unavailable')
|
||||||
|
sourceLicense.convention('LicenseRef-Radixor-Stemmer-Data')
|
||||||
|
licenseFileName.convention('LICENSE-stemmer-data.txt')
|
||||||
|
noticeFileName.convention('NOTICE-model-data.txt')
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,505 @@
|
|||||||
|
package org.egothor.radixor
|
||||||
|
|
||||||
|
import org.gradle.api.GradleException
|
||||||
|
import org.gradle.api.Plugin
|
||||||
|
import org.gradle.api.Project
|
||||||
|
import org.gradle.api.file.DuplicatesStrategy
|
||||||
|
import org.gradle.api.plugins.JavaPlugin
|
||||||
|
import org.gradle.api.publish.PublishingExtension
|
||||||
|
import org.gradle.api.publish.maven.MavenPublication
|
||||||
|
import org.gradle.api.tasks.Copy
|
||||||
|
import org.gradle.api.tasks.bundling.Jar
|
||||||
|
import org.gradle.api.tasks.bundling.Zip
|
||||||
|
import org.gradle.plugins.signing.SigningExtension
|
||||||
|
|
||||||
|
import java.nio.charset.CodingErrorAction
|
||||||
|
import java.nio.charset.StandardCharsets
|
||||||
|
import java.nio.file.Files
|
||||||
|
import java.security.MessageDigest
|
||||||
|
import java.util.zip.GZIPInputStream
|
||||||
|
|
||||||
|
/** Configures validation, generation, packaging, and publication for one model artifact. */
|
||||||
|
final class RadixorModelPlugin implements Plugin<Project> {
|
||||||
|
/** Applies the model convention to a project. */
|
||||||
|
@Override
|
||||||
|
void apply(final Project project) {
|
||||||
|
project.pluginManager.apply(JavaPlugin)
|
||||||
|
project.pluginManager.apply('maven-publish')
|
||||||
|
project.pluginManager.apply('signing')
|
||||||
|
project.java {
|
||||||
|
withSourcesJar()
|
||||||
|
withJavadocJar()
|
||||||
|
sourceCompatibility = org.gradle.api.JavaVersion.VERSION_21
|
||||||
|
targetCompatibility = org.gradle.api.JavaVersion.VERSION_21
|
||||||
|
}
|
||||||
|
final RadixorModelExtension model = project.extensions.create('radixorModel', RadixorModelExtension)
|
||||||
|
project.group = 'org.egothor'
|
||||||
|
project.version = project.providers.gradleProperty('modelReleaseVersion')
|
||||||
|
.orElse(project.providers.fileContents(project.layout.projectDirectory.file('model-version.txt')).asText.map(String::trim))
|
||||||
|
.get()
|
||||||
|
|
||||||
|
final File input = project.file('src/modelInput/stemmer.gz')
|
||||||
|
final File generated = project.layout.buildDirectory.dir('generated/modelResources').get().asFile
|
||||||
|
project.sourceSets.main.resources.setSrcDirs([generated])
|
||||||
|
|
||||||
|
final def validate = project.tasks.register('validateModelInput', ValidateModelInputTask) {
|
||||||
|
group = 'verification'
|
||||||
|
description = 'Validates the immutable source dictionary, metadata, version, and model-specific licensing material.'
|
||||||
|
dictionaryFile = project.layout.projectDirectory.file('src/modelInput/stemmer.gz')
|
||||||
|
versionFile = project.layout.projectDirectory.file('model-version.txt')
|
||||||
|
modelId = model.modelId
|
||||||
|
moduleName = project.name
|
||||||
|
shareAlike = model.sourceLicense.map { String license -> license == 'CC-BY-SA-3.0' }
|
||||||
|
metadata.put('source.project', model.sourceProject)
|
||||||
|
metadata.put('source.repository', model.sourceRepository)
|
||||||
|
metadata.put('source.dataset', model.sourceDataset)
|
||||||
|
metadata.put('source.revision', model.sourceRevision)
|
||||||
|
metadata.put('source.revisionStatus', model.sourceRevisionStatus)
|
||||||
|
metadata.put('source.license', model.sourceLicense)
|
||||||
|
metadata.put('source.licenseUri', model.sourceLicenseUri)
|
||||||
|
metadata.put('source.attribution', model.sourceAttribution)
|
||||||
|
metadata.put('source.verificationDate', model.sourceVerificationDate)
|
||||||
|
metadata.put('transformations.summary', model.transformationsSummary)
|
||||||
|
}
|
||||||
|
|
||||||
|
final def prepare = project.tasks.register('prepareModelResources', PrepareModelResourcesTask) {
|
||||||
|
group = 'build'
|
||||||
|
description = 'Copies validated dictionary bytes and generates the immutable model descriptor and index.'
|
||||||
|
dependsOn(validate)
|
||||||
|
dictionaryFile = project.layout.projectDirectory.file('src/modelInput/stemmer.gz')
|
||||||
|
versionFile = project.layout.projectDirectory.file('model-version.txt')
|
||||||
|
shareAlike = model.sourceLicense.map { String license -> license == 'CC-BY-SA-3.0' }
|
||||||
|
generatedDirectory = project.layout.buildDirectory.dir('generated/modelResources')
|
||||||
|
descriptorValues.put('model.id', model.modelId)
|
||||||
|
descriptorValues.put('model.language', model.language)
|
||||||
|
descriptorValues.put('model.displayName', model.displayName)
|
||||||
|
descriptorValues.put('model.default', model.defaultModel.map(String::valueOf))
|
||||||
|
descriptorValues.put('source.name', model.sourceName)
|
||||||
|
descriptorValues.put('source.version', model.sourceVersion)
|
||||||
|
descriptorValues.put('source.project', model.sourceProject)
|
||||||
|
descriptorValues.put('source.repository', model.sourceRepository)
|
||||||
|
descriptorValues.put('source.dataset', model.sourceDataset)
|
||||||
|
descriptorValues.put('source.revision', model.sourceRevision)
|
||||||
|
descriptorValues.put('source.revisionStatus', model.sourceRevisionStatus)
|
||||||
|
descriptorValues.put('source.license', model.sourceLicense)
|
||||||
|
descriptorValues.put('source.licenseUri', model.sourceLicenseUri)
|
||||||
|
descriptorValues.put('source.attribution', model.sourceAttribution)
|
||||||
|
descriptorValues.put('source.verificationDate', model.sourceVerificationDate)
|
||||||
|
descriptorValues.put('transformations.summary', model.transformationsSummary)
|
||||||
|
}
|
||||||
|
project.afterEvaluate {
|
||||||
|
final boolean shareAlike = model.sourceLicense.get() == 'CC-BY-SA-3.0'
|
||||||
|
if (shareAlike) {
|
||||||
|
final def notice = project.layout.projectDirectory.file("src/modelInput/${model.noticeFileName.get()}")
|
||||||
|
validate.configure { noticeFile = notice }
|
||||||
|
prepare.configure { noticeFile = notice }
|
||||||
|
} else {
|
||||||
|
final def license = project.layout.projectDirectory.file("src/modelInput/${model.licenseFileName.get()}")
|
||||||
|
validate.configure { licenseFile = license }
|
||||||
|
prepare.configure { licenseFile = license }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
project.tasks.named('processResources', Copy).configure { dependsOn(prepare); duplicatesStrategy = DuplicatesStrategy.FAIL }
|
||||||
|
project.tasks.named('sourcesJar', Jar).configure { dependsOn(prepare); exclude('**/stemmer.gz') }
|
||||||
|
project.tasks.named('javadocJar', Jar).configure { exclude('**/stemmer.gz') }
|
||||||
|
project.tasks.named('jar', Jar).configure {
|
||||||
|
archiveBaseName.set("radixor-model-${project.name}")
|
||||||
|
preserveFileTimestamps = false
|
||||||
|
reproducibleFileOrder = true
|
||||||
|
}
|
||||||
|
final def verifyDescriptor = project.tasks.register('verifyModelDescriptor') {
|
||||||
|
group = 'verification'; description = 'Verifies generated descriptor identity and checksum.'; dependsOn(prepare)
|
||||||
|
doLast {
|
||||||
|
final Properties properties = new Properties()
|
||||||
|
new File(generated, "META-INF/radixor/models/${model.modelId.get()}.properties").withInputStream(properties::load)
|
||||||
|
if (properties.getProperty('model.sha256') != sha256(input)) {
|
||||||
|
throw new GradleException('Generated descriptor checksum does not match the immutable source input.')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
final def verifyJar = project.tasks.register('verifyModelJar') {
|
||||||
|
group = 'verification'; description = 'Verifies the model JAR checksum, layout, metadata, and dictionary-free documentation artifacts.'
|
||||||
|
dependsOn(project.tasks.named('jar'), project.tasks.named('sourcesJar'), project.tasks.named('javadocJar'))
|
||||||
|
doLast {
|
||||||
|
final File archive = project.tasks.named('jar', Jar).get().archiveFile.get().asFile
|
||||||
|
final List<String> names = []
|
||||||
|
final String resource = "org/egothor/stemmer/models/${model.modelId.get()}/stemmer.gz"
|
||||||
|
final boolean shareAlike = model.sourceLicense.get() == 'CC-BY-SA-3.0'
|
||||||
|
final String licenseResource = 'META-INF/LICENSES/PoliMorf-BSD-2-Clause.txt'
|
||||||
|
final File sourceLicense = shareAlike ? null : project.file("src/modelInput/${model.licenseFileName.get()}")
|
||||||
|
final File sourceNotice = shareAlike
|
||||||
|
? project.file("src/modelInput/${model.noticeFileName.get()}") : null
|
||||||
|
final String noticeResource = "META-INF/NOTICE/${model.modelId.get()}-data.txt"
|
||||||
|
String packagedChecksum
|
||||||
|
String packagedLicenseChecksum
|
||||||
|
String packagedNoticeChecksum
|
||||||
|
new java.util.zip.ZipFile(archive).withCloseable { zip ->
|
||||||
|
zip.entries().each { names.add(it.name) }
|
||||||
|
final def entry = zip.getEntry(resource)
|
||||||
|
if (entry != null) {
|
||||||
|
packagedChecksum = sha256(zip.getInputStream(entry).bytes)
|
||||||
|
}
|
||||||
|
final def licenseEntry = zip.getEntry(licenseResource)
|
||||||
|
if (licenseEntry != null) {
|
||||||
|
packagedLicenseChecksum = sha256(zip.getInputStream(licenseEntry).bytes)
|
||||||
|
}
|
||||||
|
final def noticeEntry = zip.getEntry(noticeResource)
|
||||||
|
if (noticeEntry != null) {
|
||||||
|
packagedNoticeChecksum = sha256(zip.getInputStream(noticeEntry).bytes)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (names.count { String name -> name.endsWith('/stemmer.gz') } != 1 || !names.contains(resource)) {
|
||||||
|
throw new GradleException("Model JAR must contain exactly one dictionary at ${resource}.")
|
||||||
|
}
|
||||||
|
if (packagedChecksum != sha256(input)) {
|
||||||
|
throw new GradleException("Packaged dictionary checksum does not match the immutable source input at ${resource}.")
|
||||||
|
}
|
||||||
|
if (shareAlike) {
|
||||||
|
requireMatchingChecksum('notice', noticeResource, sha256(sourceNotice), packagedNoticeChecksum)
|
||||||
|
validateUniMorphJarContents(names)
|
||||||
|
} else {
|
||||||
|
requireMatchingChecksum('license', licenseResource, sha256(sourceLicense), packagedLicenseChecksum)
|
||||||
|
validatePoliMorfJarContents(names)
|
||||||
|
}
|
||||||
|
['META-INF/radixor/models.index', "META-INF/radixor/models/${model.modelId.get()}.properties"].each { String name ->
|
||||||
|
if (!names.contains(name)) throw new GradleException("Model JAR is missing ${name}.")
|
||||||
|
}
|
||||||
|
[project.tasks.named('sourcesJar', Jar).get(), project.tasks.named('javadocJar', Jar).get()].each { Jar task ->
|
||||||
|
final File documentationArchive = task.archiveFile.get().asFile
|
||||||
|
new java.util.zip.ZipFile(documentationArchive).withCloseable { zip ->
|
||||||
|
if (zip.entries().any { entry -> entry.name.endsWith('/stemmer.gz') || entry.name == 'stemmer.gz' }) {
|
||||||
|
throw new GradleException("Documentation artifact ${documentationArchive.name} must not contain a model dictionary.")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
project.tasks.register('validateModelRelease') {
|
||||||
|
group = 'verification'; description = 'Validates a tag-supplied model release version.'; dependsOn(verifyDescriptor, verifyJar)
|
||||||
|
doLast {
|
||||||
|
if (!project.hasProperty('modelReleaseVersion')) throw new GradleException('Model release validation requires -PmodelReleaseVersion=<version>.')
|
||||||
|
final String recorded = project.file('model-version.txt').text.trim()
|
||||||
|
if (project.property('modelReleaseVersion').toString() != recorded) throw new GradleException("Release version does not match model-version.txt: ${recorded}")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
project.tasks.named('check').configure { dependsOn(verifyDescriptor, verifyJar) }
|
||||||
|
project.extensions.configure(PublishingExtension) { PublishingExtension publishing ->
|
||||||
|
publishing.publications.create('model', MavenPublication) { MavenPublication publication ->
|
||||||
|
publication.from(project.components.java)
|
||||||
|
publication.artifactId = "radixor-model-${project.name}"
|
||||||
|
publication.pom {
|
||||||
|
name.set("Radixor model ${project.name}")
|
||||||
|
description.set(model.displayName.zip(model.sourceLicense) { String displayName, String licenseId ->
|
||||||
|
final String material = licenseId == 'CC-BY-SA-3.0'
|
||||||
|
? 'See the packaged model-specific notice.'
|
||||||
|
: 'See the packaged model-data license.'
|
||||||
|
return "${displayName}. This artifact contains Radixor-derived model data licensed under ${licenseId}; "
|
||||||
|
.concat("Radixor software is licensed separately under BSD-3-Clause. ${material}")
|
||||||
|
})
|
||||||
|
url.set('https://github.com/leogalambos/Radixor')
|
||||||
|
licenses {
|
||||||
|
license {
|
||||||
|
name.set(model.sourceLicense)
|
||||||
|
url.set(model.sourceLicenseUri)
|
||||||
|
distribution.set('repo')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
developers {
|
||||||
|
developer {
|
||||||
|
id.set('egothor')
|
||||||
|
name.set('Leo Galambos')
|
||||||
|
email.set('egothor@gmail.com')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
scm {
|
||||||
|
url.set('https://github.com/leogalambos/Radixor')
|
||||||
|
connection.set('scm:git:https://github.com/leogalambos/Radixor.git')
|
||||||
|
developerConnection.set('scm:git:ssh://git@github.com/leogalambos/Radixor.git')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
publishing.repositories.maven {
|
||||||
|
name = 'modelStaging'
|
||||||
|
url = project.layout.buildDirectory.dir('model-staging-repository').get().asFile.toURI()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
final String signingKey = project.providers.environmentVariable('SIGNING_KEY').orNull
|
||||||
|
final String signingPassword = project.providers.environmentVariable('SIGNING_PASSWORD').orNull
|
||||||
|
project.extensions.configure(SigningExtension) { SigningExtension signing ->
|
||||||
|
signing.required = {
|
||||||
|
project.providers.environmentVariable('GITHUB_REF_TYPE').orNull == 'tag'
|
||||||
|
}
|
||||||
|
if (signingKey != null && !signingKey.isBlank()) {
|
||||||
|
signing.useInMemoryPgpKeys(signingKey, signingPassword)
|
||||||
|
signing.sign(project.extensions.getByType(PublishingExtension).publications.getByName('model'))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
final def checksums = project.tasks.register('createModelCentralChecksums') {
|
||||||
|
group = 'publishing'
|
||||||
|
description = 'Creates Maven Central checksums for this model staging repository.'
|
||||||
|
dependsOn(project.tasks.named('publishModelPublicationToModelStagingRepository'))
|
||||||
|
doLast {
|
||||||
|
final File repository = project.layout.buildDirectory.dir('model-staging-repository').get().asFile
|
||||||
|
repository.eachFileRecurse { File artifact ->
|
||||||
|
if (artifact.isFile() && !['.md5', '.sha1', '.sha256', '.sha512'].any {
|
||||||
|
String extension -> artifact.name.endsWith(extension)
|
||||||
|
}) {
|
||||||
|
new File(artifact.absolutePath + '.md5').setText(sha256WithAlgorithm(artifact, 'MD5'), 'US-ASCII')
|
||||||
|
new File(artifact.absolutePath + '.sha1').setText(sha256WithAlgorithm(artifact, 'SHA-1'), 'US-ASCII')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
project.tasks.register('packageModelReleaseCandidate', Zip) {
|
||||||
|
group = 'distribution'
|
||||||
|
description = 'Packages only this model publication as a Maven-layout local release candidate.'
|
||||||
|
dependsOn(checksums)
|
||||||
|
from(project.layout.buildDirectory.dir('model-staging-repository')) {
|
||||||
|
exclude('**/maven-metadata*.xml*')
|
||||||
|
}
|
||||||
|
destinationDirectory.set(project.layout.buildDirectory.dir('model-release-candidate'))
|
||||||
|
archiveFileName.set('central-bundle.zip')
|
||||||
|
doFirst {
|
||||||
|
if (project.providers.environmentVariable('GITHUB_REF_TYPE').orNull == 'tag'
|
||||||
|
&& (signingKey == null || signingKey.isBlank()
|
||||||
|
|| signingPassword == null || signingPassword.isBlank())) {
|
||||||
|
throw new GradleException('A tagged model release requires SIGNING_KEY and SIGNING_PASSWORD.')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Ensures a required file exists. */
|
||||||
|
static void requireFile(final File file, final String diagnostic) {
|
||||||
|
if (!file.isFile()) throw new GradleException(diagnostic)
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Rejects a missing or byte-different packaged licensing resource. */
|
||||||
|
static void requireMatchingChecksum(final String kind, final String resource,
|
||||||
|
final String sourceChecksum, final String packagedChecksum) {
|
||||||
|
if (packagedChecksum != sourceChecksum) {
|
||||||
|
throw new GradleException("Packaged ${kind} does not match the source ${kind} at ${resource}.")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Validates complete source, licensing, attribution, revision-status, and transformation metadata. */
|
||||||
|
private static void validateMetadata(final RadixorModelExtension model) {
|
||||||
|
final Map<String, String> required = [
|
||||||
|
'source.project': model.sourceProject.orNull,
|
||||||
|
'source.repository': model.sourceRepository.orNull,
|
||||||
|
'source.dataset': model.sourceDataset.orNull,
|
||||||
|
'source.revision': model.sourceRevision.orNull,
|
||||||
|
'source.revisionStatus': model.sourceRevisionStatus.orNull,
|
||||||
|
'source.license': model.sourceLicense.orNull,
|
||||||
|
'source.licenseUri': model.sourceLicenseUri.orNull,
|
||||||
|
'source.attribution': model.sourceAttribution.orNull,
|
||||||
|
'source.verificationDate': model.sourceVerificationDate.orNull,
|
||||||
|
'transformations.summary': model.transformationsSummary.orNull]
|
||||||
|
required.each { String key, String value ->
|
||||||
|
if (value == null || value.isBlank()) {
|
||||||
|
throw new GradleException("Required model metadata is missing: ${key}")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
validateRevisionMetadata(model.sourceRevision.get(), model.sourceRevisionStatus.get())
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Accepts an exact recorded revision or the explicit legacy-import sentinel, but never an absent status. */
|
||||||
|
static void validateRevisionMetadata(final String revision, final String status) {
|
||||||
|
if (revision == null || revision.isBlank()) {
|
||||||
|
throw new GradleException('Required model metadata is missing: source.revision')
|
||||||
|
}
|
||||||
|
if (status == null || status.isBlank()) {
|
||||||
|
throw new GradleException('Required model metadata is missing: source.revisionStatus')
|
||||||
|
}
|
||||||
|
final String sentinel = 'not-recorded-in-legacy-import'
|
||||||
|
if (revision == sentinel && status != sentinel) {
|
||||||
|
throw new GradleException('The legacy revision sentinel requires source.revisionStatus=not-recorded-in-legacy-import.')
|
||||||
|
}
|
||||||
|
if (revision != sentinel && status != 'recorded') {
|
||||||
|
throw new GradleException('An exact source revision requires source.revisionStatus=recorded.')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Validates the model-specific attribution and ShareAlike notice. */
|
||||||
|
static void validateShareAlikeNotice(final File notice, final RadixorModelExtension model) {
|
||||||
|
validateShareAlikeNoticeText(notice.getText('UTF-8'), notice.toString(), model.modelId.get(),
|
||||||
|
model.sourceRepository.get(), model.sourceLicenseUri.get(), model.sourceRevision.get(),
|
||||||
|
model.sourceRevisionStatus.get())
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Validates required content in one UniMorph model-data notice. */
|
||||||
|
static void validateShareAlikeNoticeText(final String text, final String noticeName,
|
||||||
|
final String modelId, final String repository, final String licenseUri,
|
||||||
|
final String revision, final String revisionStatus) {
|
||||||
|
final List<String> required = [
|
||||||
|
"Model ID: ${modelId}",
|
||||||
|
"Official repository: ${repository}",
|
||||||
|
'Attribution:',
|
||||||
|
'License:\nCreative Commons Attribution-ShareAlike 3.0 Unported',
|
||||||
|
"Canonical license URI: ${licenseUri}",
|
||||||
|
'Radixor modifications:',
|
||||||
|
"Revision status: ${revisionStatus}",
|
||||||
|
'Copyright (C) 2026, Leo Galambos.',
|
||||||
|
'Radixor-specific selection, verification, cleaning, normalization,',
|
||||||
|
'to the extent protected by applicable law.',
|
||||||
|
'The underlying morphological data remains attributed to UniMorph and',
|
||||||
|
"This derived model data, including Radixor's protectable contributions,",
|
||||||
|
'is distributed under Creative Commons Attribution-ShareAlike 3.0',
|
||||||
|
'Neither UniMorph nor any upstream contributor endorses Radixor.']
|
||||||
|
if (revision == 'not-recorded-in-legacy-import') {
|
||||||
|
required.add('The exact UniMorph commit used for the original Radixor import was not recorded.')
|
||||||
|
}
|
||||||
|
final List<String> missing = required.findAll { String value -> !text.contains(value) }
|
||||||
|
if (!missing.isEmpty()) {
|
||||||
|
throw new GradleException("Model notice ${noticeName} is missing required content: ${missing.join(', ')}")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Rejects generic license files and foreign notices in a UniMorph model artifact. */
|
||||||
|
static void validateUniMorphJarContents(final List<String> names) {
|
||||||
|
if (names.any { String name -> name.startsWith('META-INF/LICENSES/') }) {
|
||||||
|
throw new GradleException('A UniMorph model artifact must use only its model-specific notice for data licensing.')
|
||||||
|
}
|
||||||
|
if (names.count { String name -> name.startsWith('META-INF/NOTICE/') && !name.endsWith('/') } != 1) {
|
||||||
|
throw new GradleException('A UniMorph model artifact must contain exactly one model-specific notice.')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Rejects UniMorph licensing material in the separately licensed PoliMorf artifact. */
|
||||||
|
static void validatePoliMorfJarContents(final List<String> names) {
|
||||||
|
if (names.any { String name -> name.startsWith('META-INF/NOTICE/')
|
||||||
|
|| name.contains('CC-BY-SA') }) {
|
||||||
|
throw new GradleException('The PoliMorf artifact must not contain UniMorph CC BY-SA material.')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Memory-bounded validation statistics for one dictionary input. */
|
||||||
|
static final class DictionaryValidationResult {
|
||||||
|
final long acceptedGroupCount
|
||||||
|
final long acceptedFormCount
|
||||||
|
final long ignoredEmptyVariantCount
|
||||||
|
|
||||||
|
DictionaryValidationResult(final long acceptedGroupCount, final long acceptedFormCount,
|
||||||
|
final long ignoredEmptyVariantCount) {
|
||||||
|
this.acceptedGroupCount = acceptedGroupCount
|
||||||
|
this.acceptedFormCount = acceptedFormCount
|
||||||
|
this.ignoredEmptyVariantCount = ignoredEmptyVariantCount
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Validates GZip, strict UTF-8, and dictionary rows without retaining decompressed input. */
|
||||||
|
static DictionaryValidationResult validateDictionary(final File file) {
|
||||||
|
long acceptedGroups = 0L
|
||||||
|
long acceptedForms = 0L
|
||||||
|
long ignoredEmptyVariants = 0L
|
||||||
|
try {
|
||||||
|
final def decoder = StandardCharsets.UTF_8.newDecoder()
|
||||||
|
.onMalformedInput(CodingErrorAction.REPORT)
|
||||||
|
.onUnmappableCharacter(CodingErrorAction.REPORT)
|
||||||
|
Files.newInputStream(file.toPath()).withCloseable { InputStream source ->
|
||||||
|
new BufferedInputStream(source).withCloseable { BufferedInputStream bufferedInput ->
|
||||||
|
new GZIPInputStream(bufferedInput).withCloseable { GZIPInputStream gzipInput ->
|
||||||
|
new BufferedReader(new InputStreamReader(gzipInput, decoder)).withCloseable { BufferedReader reader ->
|
||||||
|
String line
|
||||||
|
long lineNumber = 0L
|
||||||
|
while ((line = reader.readLine()) != null) {
|
||||||
|
lineNumber++
|
||||||
|
final String trimmed = line.trim()
|
||||||
|
if (trimmed && !trimmed.startsWith('#') && !trimmed.startsWith('//')) {
|
||||||
|
final String[] columns = line.split('\\t', -1)
|
||||||
|
if (columns[0].isEmpty()) {
|
||||||
|
throw new GradleException("Invalid Radixor dictionary row ${lineNumber} in ${file}.")
|
||||||
|
}
|
||||||
|
if (containsUnicodeWhitespace(columns[0])) continue
|
||||||
|
long acceptedRowForms = 1L
|
||||||
|
for (int index = 1; index < columns.length; index++) {
|
||||||
|
final String variant = columns[index]
|
||||||
|
if (variant.isEmpty()) {
|
||||||
|
ignoredEmptyVariants++
|
||||||
|
} else if (!containsUnicodeWhitespace(variant)) {
|
||||||
|
acceptedRowForms++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
acceptedGroups++
|
||||||
|
acceptedForms += acceptedRowForms
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
} catch (GradleException exception) {
|
||||||
|
throw exception
|
||||||
|
} catch (Exception exception) {
|
||||||
|
throw new GradleException("Invalid GZip or UTF-8 model input: ${file}", exception)
|
||||||
|
}
|
||||||
|
if (acceptedGroups == 0L) throw new GradleException("Model dictionary contains no valid rows: ${file}")
|
||||||
|
if (ignoredEmptyVariants > 0L) {
|
||||||
|
println("Model validation warning: " + file + " contains " + ignoredEmptyVariants
|
||||||
|
+ " empty variant columns; the production parser intentionally ignores empty variants.")
|
||||||
|
}
|
||||||
|
return new DictionaryValidationResult(acceptedGroups, acceptedForms, ignoredEmptyVariants)
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Detects Unicode whitespace in one bounded dictionary field. */
|
||||||
|
private static boolean containsUnicodeWhitespace(final String value) {
|
||||||
|
for (int index = 0; index < value.length(); index++) {
|
||||||
|
if (Character.isWhitespace(value.charAt(index))) return true
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Builds deterministic descriptor text. */
|
||||||
|
private static String descriptorText(final RadixorModelExtension model, final String version,
|
||||||
|
final String resource, final String checksum) {
|
||||||
|
return """model.id=${model.modelId.get()}
|
||||||
|
model.version=${version}
|
||||||
|
model.language=${model.language.get()}
|
||||||
|
model.displayName=${model.displayName.get()}
|
||||||
|
model.resource=${resource}
|
||||||
|
model.default=${model.defaultModel.get()}
|
||||||
|
model.format=radixor-dictionary-tsv-gzip
|
||||||
|
model.formatVersion=1
|
||||||
|
model.sha256=${checksum}
|
||||||
|
model.rightToLeft=${['FA_IR', 'HE_IL', 'YI'].contains(model.language.get())}
|
||||||
|
model.caseProcessing=LOWERCASE_WITH_LOCALE_ROOT
|
||||||
|
model.diacriticProcessing=AS_IS
|
||||||
|
model.storeOriginal=true
|
||||||
|
source.name=${model.sourceName.get()}
|
||||||
|
source.version=${model.sourceVersion.get()}
|
||||||
|
source.project=${model.sourceProject.get()}
|
||||||
|
source.repository=${model.sourceRepository.get()}
|
||||||
|
source.dataset=${model.sourceDataset.get()}
|
||||||
|
source.revision=${model.sourceRevision.get()}
|
||||||
|
source.revisionStatus=${model.sourceRevisionStatus.get()}
|
||||||
|
source.license=${model.sourceLicense.get()}
|
||||||
|
source.licenseUri=${model.sourceLicenseUri.get()}
|
||||||
|
source.attribution=${model.sourceAttribution.get()}
|
||||||
|
source.verificationDate=${model.sourceVerificationDate.get()}
|
||||||
|
transformations.summary=${model.transformationsSummary.get()}
|
||||||
|
compiler.radixorVersion=3.x
|
||||||
|
compiler.radixorCommit=unavailable
|
||||||
|
statistics.groups=unavailable
|
||||||
|
statistics.forms=unavailable
|
||||||
|
"""
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Calculates the lowercase hexadecimal SHA-256 digest. */
|
||||||
|
private static String sha256(final File file) {
|
||||||
|
return sha256(file.bytes)
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Calculates the lowercase hexadecimal SHA-256 digest of bytes. */
|
||||||
|
private static String sha256(final byte[] bytes) {
|
||||||
|
return MessageDigest.getInstance('SHA-256').digest(bytes).collect { byte value -> String.format('%02x', value & 0xff) }.join()
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Calculates a lowercase hexadecimal digest using the requested algorithm. */
|
||||||
|
private static String sha256WithAlgorithm(final File file, final String algorithm) {
|
||||||
|
return MessageDigest.getInstance(algorithm).digest(file.bytes)
|
||||||
|
.collect { byte value -> String.format('%02x', value & 0xff) }.join()
|
||||||
|
}
|
||||||
|
|
||||||
|
}
|
||||||
@@ -0,0 +1,57 @@
|
|||||||
|
package org.egothor.radixor
|
||||||
|
|
||||||
|
import org.gradle.api.DefaultTask
|
||||||
|
import org.gradle.api.GradleException
|
||||||
|
import org.gradle.api.file.RegularFileProperty
|
||||||
|
import org.gradle.api.provider.MapProperty
|
||||||
|
import org.gradle.api.provider.Property
|
||||||
|
import org.gradle.api.tasks.Input
|
||||||
|
import org.gradle.api.tasks.InputFile
|
||||||
|
import org.gradle.api.tasks.Optional
|
||||||
|
import org.gradle.api.tasks.PathSensitive
|
||||||
|
import org.gradle.api.tasks.PathSensitivity
|
||||||
|
import org.gradle.api.tasks.TaskAction
|
||||||
|
|
||||||
|
/** Validates one immutable model input without retaining Project state. */
|
||||||
|
abstract class ValidateModelInputTask extends DefaultTask {
|
||||||
|
@InputFile @PathSensitive(PathSensitivity.RELATIVE) abstract RegularFileProperty getDictionaryFile()
|
||||||
|
@InputFile @PathSensitive(PathSensitivity.RELATIVE) abstract RegularFileProperty getVersionFile()
|
||||||
|
@Optional @InputFile @PathSensitive(PathSensitivity.RELATIVE) abstract RegularFileProperty getLicenseFile()
|
||||||
|
@Optional @InputFile @PathSensitive(PathSensitivity.RELATIVE) abstract RegularFileProperty getNoticeFile()
|
||||||
|
@Input abstract Property<String> getModelId()
|
||||||
|
@Input abstract Property<String> getModuleName()
|
||||||
|
@Input abstract Property<Boolean> getShareAlike()
|
||||||
|
@Input abstract MapProperty<String, String> getMetadata()
|
||||||
|
|
||||||
|
/** Performs deterministic metadata, licensing, and streaming dictionary validation. */
|
||||||
|
@TaskAction
|
||||||
|
void validateInput() {
|
||||||
|
final File dictionary = dictionaryFile.get().asFile
|
||||||
|
final String id = modelId.get()
|
||||||
|
final String version = versionFile.get().asFile.getText('UTF-8').trim()
|
||||||
|
if (id != moduleName.get() || !(id ==~ /[a-z]{2}(?:-[a-z]{2})?-[a-z0-9]+(?:-[a-z0-9]+)*/)) {
|
||||||
|
throw new GradleException("Model ID '${id}' must equal module '${moduleName.get()}' and use the safe model-ID syntax.")
|
||||||
|
}
|
||||||
|
if (!(version ==~ /[0-9]+\.[0-9]+\.[0-9]+(?:[-+][0-9A-Za-z.-]+)?/)) {
|
||||||
|
throw new GradleException("Invalid semantic model version '${version}'.")
|
||||||
|
}
|
||||||
|
final Map<String, String> values = metadata.get()
|
||||||
|
values.each { String key, String value ->
|
||||||
|
if (value == null || value.isBlank()) throw new GradleException("Required model metadata is missing: ${key}")
|
||||||
|
}
|
||||||
|
RadixorModelPlugin.validateRevisionMetadata(values['source.revision'], values['source.revisionStatus'])
|
||||||
|
if (shareAlike.get()) {
|
||||||
|
final File notice = noticeFile.get().asFile
|
||||||
|
RadixorModelPlugin.validateShareAlikeNoticeText(notice.getText('UTF-8'), notice.toString(), id,
|
||||||
|
values['source.repository'], values['source.licenseUri'], values['source.revision'],
|
||||||
|
values['source.revisionStatus'])
|
||||||
|
} else {
|
||||||
|
final String text = licenseFile.get().asFile.getText('UTF-8')
|
||||||
|
if (!text.contains('SPDX-License-Identifier: BSD-2-Clause')
|
||||||
|
|| !text.contains('Copyright (c) 2016, Marcin Miłkowski')) {
|
||||||
|
throw new GradleException('The PoliMorf license must contain the complete BSD-2-Clause text and upstream attribution.')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
RadixorModelPlugin.validateDictionary(dictionary)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,205 @@
|
|||||||
|
package org.egothor.radixor
|
||||||
|
|
||||||
|
import org.gradle.api.GradleException
|
||||||
|
import org.junit.jupiter.api.Test
|
||||||
|
import org.junit.jupiter.api.io.TempDir
|
||||||
|
|
||||||
|
import java.nio.charset.StandardCharsets
|
||||||
|
import java.nio.file.Files
|
||||||
|
import java.nio.file.Path
|
||||||
|
import java.util.zip.GZIPOutputStream
|
||||||
|
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertThrows
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue
|
||||||
|
|
||||||
|
/** Tests model licensing metadata and packaged-resource validation boundaries. */
|
||||||
|
final class RadixorModelPluginTest {
|
||||||
|
@TempDir
|
||||||
|
Path temporaryDirectory
|
||||||
|
|
||||||
|
/** Accepts a known exact source revision. */
|
||||||
|
@Test
|
||||||
|
void acceptsKnownExactRevision() {
|
||||||
|
RadixorModelPlugin.validateRevisionMetadata('6e63b53', 'recorded')
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Accepts the explicit legacy-import sentinel without fabricating a revision. */
|
||||||
|
@Test
|
||||||
|
void acceptsUnknownLegacyRevision() {
|
||||||
|
RadixorModelPlugin.validateRevisionMetadata(
|
||||||
|
'not-recorded-in-legacy-import', 'not-recorded-in-legacy-import')
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Rejects a missing revision-status declaration. */
|
||||||
|
@Test
|
||||||
|
void rejectsMissingRevisionStatus() {
|
||||||
|
assertThrows(GradleException) {
|
||||||
|
RadixorModelPlugin.validateRevisionMetadata('6e63b53', '')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Rejects a missing model-specific notice input. */
|
||||||
|
@Test
|
||||||
|
void rejectsMissingLicensingInputs() {
|
||||||
|
File missing = new File('build/nonexistent-model-licensing-input')
|
||||||
|
assertThrows(GradleException) {
|
||||||
|
RadixorModelPlugin.requireFile(missing, 'Required model notice is missing')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Accepts a complete model-specific UniMorph notice. */
|
||||||
|
@Test
|
||||||
|
void acceptsCompleteUniMorphNotice() {
|
||||||
|
validateNotice(validNotice())
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Rejects each independently required notice statement. */
|
||||||
|
@Test
|
||||||
|
void rejectsIncompleteUniMorphNotices() {
|
||||||
|
[
|
||||||
|
'Copyright (C) 2026, Leo Galambos.',
|
||||||
|
'Attribution:',
|
||||||
|
'Creative Commons Attribution-ShareAlike 3.0 Unported',
|
||||||
|
'Canonical license URI:',
|
||||||
|
"This derived model data, including Radixor's protectable contributions,",
|
||||||
|
'Radixor modifications:',
|
||||||
|
'Revision status:',
|
||||||
|
'Neither UniMorph nor any upstream contributor endorses Radixor.'
|
||||||
|
].each { String required ->
|
||||||
|
assertThrows(GradleException) {
|
||||||
|
validateNotice(validNotice().replace(required, 'omitted'))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Rejects packaged notice bytes that differ from their model-module source. */
|
||||||
|
@Test
|
||||||
|
void rejectsIncorrectPackagedNotice() {
|
||||||
|
assertThrows(GradleException) {
|
||||||
|
RadixorModelPlugin.requireMatchingChecksum(
|
||||||
|
'notice', 'META-INF/NOTICE/test-model-data.txt', 'source', 'different')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Rejects UniMorph CC material in the separately licensed PoliMorf artifact. */
|
||||||
|
@Test
|
||||||
|
void rejectsUniMorphMaterialInPoliMorf() {
|
||||||
|
assertThrows(GradleException) {
|
||||||
|
RadixorModelPlugin.validatePoliMorfJarContents(
|
||||||
|
['META-INF/LICENSES/PoliMorf-BSD-2-Clause.txt', 'META-INF/NOTICE/test-data.txt'])
|
||||||
|
}
|
||||||
|
assertThrows(GradleException) {
|
||||||
|
RadixorModelPlugin.validatePoliMorfJarContents(
|
||||||
|
['META-INF/LICENSES/PoliMorf-BSD-2-Clause.txt', 'META-INF/LICENSES/CC-BY-SA-3.0.txt'])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Streams a large dictionary while retaining only aggregate counters and the current row. */
|
||||||
|
@Test
|
||||||
|
void validatesLargeDictionaryWithBoundedState() {
|
||||||
|
final int groups = 250_000
|
||||||
|
final File dictionary = temporaryDirectory.resolve('large.gz').toFile()
|
||||||
|
writeGzip(dictionary) { BufferedWriter writer ->
|
||||||
|
for (int index = 0; index < groups; index++) {
|
||||||
|
writer.write("stem${index}\tvariant${index}\t\n")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
final RadixorModelPlugin.DictionaryValidationResult result =
|
||||||
|
RadixorModelPlugin.validateDictionary(dictionary)
|
||||||
|
|
||||||
|
assertEquals(groups, result.acceptedGroupCount)
|
||||||
|
assertEquals(groups * 2L, result.acceptedFormCount)
|
||||||
|
assertEquals(groups, result.ignoredEmptyVariantCount)
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Rejects a source that is not a GZip stream. */
|
||||||
|
@Test
|
||||||
|
void rejectsInvalidGzip() {
|
||||||
|
final File dictionary = temporaryDirectory.resolve('invalid.gz').toFile()
|
||||||
|
Files.writeString(dictionary.toPath(), 'not gzip', StandardCharsets.UTF_8)
|
||||||
|
assertThrows(GradleException) { RadixorModelPlugin.validateDictionary(dictionary) }
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Rejects malformed UTF-8 through the strict incremental decoder. */
|
||||||
|
@Test
|
||||||
|
void rejectsMalformedUtf8() {
|
||||||
|
final File dictionary = temporaryDirectory.resolve('malformed-utf8.gz').toFile()
|
||||||
|
new GZIPOutputStream(Files.newOutputStream(dictionary.toPath())).withCloseable { OutputStream output ->
|
||||||
|
output.write([0x73, 0x74, 0x65, 0x6d, 0x09, 0xc3, 0x28, 0x0a] as byte[])
|
||||||
|
}
|
||||||
|
assertThrows(GradleException) { RadixorModelPlugin.validateDictionary(dictionary) }
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Rejects structurally invalid rows with an empty stem. */
|
||||||
|
@Test
|
||||||
|
void rejectsInvalidRows() {
|
||||||
|
final File dictionary = temporaryDirectory.resolve('invalid-row.gz').toFile()
|
||||||
|
writeGzip(dictionary) { BufferedWriter writer -> writer.write("\tvariant\n") }
|
||||||
|
assertThrows(GradleException) { RadixorModelPlugin.validateDictionary(dictionary) }
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Preserves the production parser policy for Unicode-whitespace items. */
|
||||||
|
@Test
|
||||||
|
void rejectsUnicodeWhitespaceItemsWithoutRejectingTheSource() {
|
||||||
|
final File dictionary = temporaryDirectory.resolve('whitespace-items.gz').toFile()
|
||||||
|
writeGzip(dictionary) { BufferedWriter writer ->
|
||||||
|
writer.write("invalid stem\tvariant\n")
|
||||||
|
writer.write("valid\taccepted\tinvalid variant\n")
|
||||||
|
}
|
||||||
|
final RadixorModelPlugin.DictionaryValidationResult result =
|
||||||
|
RadixorModelPlugin.validateDictionary(dictionary)
|
||||||
|
assertEquals(1L, result.acceptedGroupCount)
|
||||||
|
assertEquals(2L, result.acceptedFormCount)
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Streams the complete maintained PoliMorf model input successfully. */
|
||||||
|
@Test
|
||||||
|
void validatesFullPoliMorfInput() {
|
||||||
|
final List<File> candidates = [
|
||||||
|
new File('models/pl-pl-polimorf/src/modelInput/stemmer.gz'),
|
||||||
|
new File('../models/pl-pl-polimorf/src/modelInput/stemmer.gz')]
|
||||||
|
final File dictionary = candidates.find { File candidate -> candidate.isFile() }
|
||||||
|
assertTrue(dictionary != null, 'The complete PoliMorf model input must be available to build-logic tests.')
|
||||||
|
|
||||||
|
final RadixorModelPlugin.DictionaryValidationResult result =
|
||||||
|
RadixorModelPlugin.validateDictionary(dictionary)
|
||||||
|
assertTrue(result.acceptedGroupCount > 0L)
|
||||||
|
assertTrue(result.acceptedFormCount > result.acceptedGroupCount)
|
||||||
|
}
|
||||||
|
|
||||||
|
private static void writeGzip(final File target, final Closure<Void> content) {
|
||||||
|
new GZIPOutputStream(Files.newOutputStream(target.toPath())).withCloseable { OutputStream gzip ->
|
||||||
|
new BufferedWriter(new OutputStreamWriter(gzip, StandardCharsets.UTF_8)).withCloseable {
|
||||||
|
BufferedWriter writer -> content.call(writer)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static void validateNotice(final String text) {
|
||||||
|
RadixorModelPlugin.validateShareAlikeNoticeText(text, 'test notice', 'test-model',
|
||||||
|
'https://github.com/unimorph/test', 'https://creativecommons.org/licenses/by-sa/3.0/',
|
||||||
|
'not-recorded-in-legacy-import', 'not-recorded-in-legacy-import')
|
||||||
|
}
|
||||||
|
|
||||||
|
private static String validNotice() {
|
||||||
|
return '''Model ID: test-model
|
||||||
|
Official repository: https://github.com/unimorph/test
|
||||||
|
Attribution: UniMorph and upstream contributors
|
||||||
|
License:
|
||||||
|
Creative Commons Attribution-ShareAlike 3.0 Unported
|
||||||
|
Canonical license URI: https://creativecommons.org/licenses/by-sa/3.0/
|
||||||
|
Radixor modifications: Cleaning and packaging.
|
||||||
|
Revision status: not-recorded-in-legacy-import
|
||||||
|
The exact UniMorph commit used for the original Radixor import was not recorded.
|
||||||
|
Copyright (C) 2026, Leo Galambos.
|
||||||
|
Radixor-specific selection, verification, cleaning, normalization,
|
||||||
|
to the extent protected by applicable law.
|
||||||
|
The underlying morphological data remains attributed to UniMorph and
|
||||||
|
This derived model data, including Radixor's protectable contributions,
|
||||||
|
is distributed under Creative Commons Attribution-ShareAlike 3.0
|
||||||
|
Neither UniMorph nor any upstream contributor endorses Radixor.
|
||||||
|
'''
|
||||||
|
}
|
||||||
|
}
|
||||||
642
build.gradle
642
build.gradle
@@ -1,4 +1,5 @@
|
|||||||
plugins {
|
plugins {
|
||||||
|
id 'org.egothor.radixor.build-support'
|
||||||
id 'java'
|
id 'java'
|
||||||
id 'eclipse'
|
id 'eclipse'
|
||||||
id 'application'
|
id 'application'
|
||||||
@@ -7,9 +8,9 @@ plugins {
|
|||||||
id 'pmd'
|
id 'pmd'
|
||||||
id 'jacoco'
|
id 'jacoco'
|
||||||
id 'info.solidsoft.pitest' version '1.19.0'
|
id 'info.solidsoft.pitest' version '1.19.0'
|
||||||
id 'me.champeau.jmh' version '0.7.2'
|
id 'me.champeau.jmh' version '0.7.3'
|
||||||
id 'org.owasp.dependencycheck' version '12.2.1'
|
id 'org.owasp.dependencycheck' version '12.2.1'
|
||||||
id 'org.cyclonedx.bom' version '3.2.4'
|
id 'org.cyclonedx.bom' version '3.3.0'
|
||||||
id 'com.palantir.git-version' version '4.0.0'
|
id 'com.palantir.git-version' version '4.0.0'
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -30,6 +31,11 @@ apply from: 'gradle/maven-pom.gradle'
|
|||||||
|
|
||||||
configurations {
|
configurations {
|
||||||
mockitoAgent
|
mockitoAgent
|
||||||
|
stemmingQualityJmhRuntime {
|
||||||
|
canBeConsumed = false
|
||||||
|
canBeResolved = true
|
||||||
|
extendsFrom(jmhImplementation, jmhRuntimeOnly)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
java {
|
java {
|
||||||
@@ -40,6 +46,10 @@ java {
|
|||||||
targetCompatibility = JavaVersion.VERSION_21
|
targetCompatibility = JavaVersion.VERSION_21
|
||||||
}
|
}
|
||||||
|
|
||||||
|
tasks.withType(JavaCompile).configureEach {
|
||||||
|
options.compilerArgs.addAll(['-Xlint:deprecation', '-Xlint:unchecked'])
|
||||||
|
}
|
||||||
|
|
||||||
tasks.withType(AbstractArchiveTask).configureEach {
|
tasks.withType(AbstractArchiveTask).configureEach {
|
||||||
preserveFileTimestamps = false
|
preserveFileTimestamps = false
|
||||||
reproducibleFileOrder = true
|
reproducibleFileOrder = true
|
||||||
@@ -65,6 +75,11 @@ dependencyLocking {
|
|||||||
dependencies {
|
dependencies {
|
||||||
jmhImplementation sourceSets.main.output
|
jmhImplementation sourceSets.main.output
|
||||||
|
|
||||||
|
modelProjects().each { Project modelProject ->
|
||||||
|
testRuntimeOnly project(modelProject.path)
|
||||||
|
jmhRuntimeOnly project(modelProject.path)
|
||||||
|
}
|
||||||
|
|
||||||
testImplementation platform(libs.junit.bom)
|
testImplementation platform(libs.junit.bom)
|
||||||
testImplementation libs.junit.jupiter
|
testImplementation libs.junit.jupiter
|
||||||
testRuntimeOnly libs.junit.platform.launcher
|
testRuntimeOnly libs.junit.platform.launcher
|
||||||
@@ -72,12 +87,55 @@ dependencies {
|
|||||||
testImplementation libs.mockito.core
|
testImplementation libs.mockito.core
|
||||||
testImplementation libs.mockito.junit.jupiter
|
testImplementation libs.mockito.junit.jupiter
|
||||||
testImplementation libs.jqwik
|
testImplementation libs.jqwik
|
||||||
|
testImplementation gradleTestKit()
|
||||||
|
|
||||||
mockitoAgent(libs.mockito.core) {
|
mockitoAgent(libs.mockito.core) {
|
||||||
transitive = false
|
transitive = false
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
def modelProjects() {
|
||||||
|
Properties topology = new Properties()
|
||||||
|
rootProject.file('models/model-projects.properties').withInputStream { InputStream input ->
|
||||||
|
topology.load(input)
|
||||||
|
}
|
||||||
|
return topology.stringPropertyNames().toList().sort().collect { String modelId ->
|
||||||
|
project(":models:${modelId}")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
def defaultModelProjects() {
|
||||||
|
Properties topology = new Properties()
|
||||||
|
rootProject.file('models/model-projects.properties').withInputStream { InputStream input ->
|
||||||
|
topology.load(input)
|
||||||
|
}
|
||||||
|
return topology.stringPropertyNames().findAll { String modelId ->
|
||||||
|
topology.getProperty(modelId) == 'default'
|
||||||
|
}.sort().collect { String modelId -> project(":models:${modelId}") }
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.named('projects') {
|
||||||
|
actions.clear()
|
||||||
|
doLast {
|
||||||
|
logger.lifecycle('Root project \'{}\'', rootProject.name)
|
||||||
|
rootProject.allprojects.findAll { Project candidate -> candidate != rootProject }
|
||||||
|
.sort { Project left, Project right -> left.path <=> right.path }
|
||||||
|
.each { Project candidate -> logger.lifecycle('+--- Project \'{}\'', candidate.path) }
|
||||||
|
gradle.includedBuilds.toList().sort { left, right -> left.name <=> right.name }
|
||||||
|
.each { includedBuild -> logger.lifecycle('Included build \'{}\'', includedBuild.name) }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
sourceSets.jmh.compileClasspath = sourceSets.jmh.compileClasspath - sourceSets.test.output
|
||||||
|
sourceSets.jmh.runtimeClasspath = sourceSets.jmh.runtimeClasspath - sourceSets.test.output
|
||||||
|
sourceSets.test.compileClasspath += sourceSets.jmh.output + configurations.jmhCompileClasspath
|
||||||
|
sourceSets.test.runtimeClasspath += sourceSets.jmh.output + configurations.jmhCompileClasspath
|
||||||
|
|
||||||
|
tasks.named('compileJmhJava', JavaCompile) {
|
||||||
|
classpath = classpath - sourceSets.test.output
|
||||||
|
setDependsOn([tasks.named('classes')])
|
||||||
|
}
|
||||||
|
|
||||||
dependencyCheck {
|
dependencyCheck {
|
||||||
failBuildOnCVSS = 7.0
|
failBuildOnCVSS = 7.0
|
||||||
failOnError = true
|
failOnError = true
|
||||||
@@ -123,9 +181,10 @@ def splitTagExpression = { String tagsExpr ->
|
|||||||
}
|
}
|
||||||
|
|
||||||
tasks.withType(Test).configureEach {
|
tasks.withType(Test).configureEach {
|
||||||
doFirst {
|
final def mockitoAgentArguments = objects.newInstance(
|
||||||
jvmArgs "-javaagent:${configurations.mockitoAgent.singleFile}"
|
org.egothor.radixor.MockitoAgentArgumentProvider)
|
||||||
}
|
mockitoAgentArguments.agentClasspath.from(configurations.mockitoAgent)
|
||||||
|
jvmArgumentProviders.add(mockitoAgentArguments)
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Bundled dictionary integration tests compile and reload large real-world
|
* Bundled dictionary integration tests compile and reload large real-world
|
||||||
@@ -141,6 +200,30 @@ tasks.withType(Test).configureEach {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
tasks.named('test', Test) {
|
||||||
|
dependsOn('prepareModelConsumerTestRepository')
|
||||||
|
systemProperty('radixor.consumer.repository',
|
||||||
|
layout.buildDirectory.dir('model-consumer-repository').get().asFile.absolutePath)
|
||||||
|
systemProperty('radixor.core.version', version.toString())
|
||||||
|
systemProperty('radixor.catalog.version', project(':models:standard').version.toString())
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.register('modelDependencyResolutionTest', Test) {
|
||||||
|
group = 'verification'
|
||||||
|
description = 'Verifies that published model coordinates resolve from the generated consumer repository.'
|
||||||
|
dependsOn(tasks.named('prepareModelConsumerTestRepository'))
|
||||||
|
testClassesDirs = sourceSets.test.output.classesDirs
|
||||||
|
classpath = sourceSets.test.runtimeClasspath
|
||||||
|
useJUnitPlatform()
|
||||||
|
filter {
|
||||||
|
includeTestsMatching('org.egothor.stemmer.ModelDependencyResolutionTest')
|
||||||
|
}
|
||||||
|
systemProperty('radixor.consumer.repository',
|
||||||
|
layout.buildDirectory.dir('model-consumer-repository').get().asFile.absolutePath)
|
||||||
|
systemProperty('radixor.core.version', version.toString())
|
||||||
|
systemProperty('radixor.catalog.version', project(':models:standard').version.toString())
|
||||||
|
}
|
||||||
|
|
||||||
def configureJUnitPlatformTags = { Test task, String includeTagsExpr, String excludeTagsExpr ->
|
def configureJUnitPlatformTags = { Test task, String includeTagsExpr, String excludeTagsExpr ->
|
||||||
task.useJUnitPlatform {
|
task.useJUnitPlatform {
|
||||||
final def includes = splitTagExpression(includeTagsExpr)
|
final def includes = splitTagExpression(includeTagsExpr)
|
||||||
@@ -158,11 +241,47 @@ def configureJUnitPlatformTags = { Test task, String includeTagsExpr, String exc
|
|||||||
tasks.named('test', Test) {
|
tasks.named('test', Test) {
|
||||||
final def requestedIncludes = splitTagExpression(cliIncludeTags)
|
final def requestedIncludes = splitTagExpression(cliIncludeTags)
|
||||||
final boolean slowExplicitlyIncluded = requestedIncludes.contains('slow')
|
final boolean slowExplicitlyIncluded = requestedIncludes.contains('slow')
|
||||||
final String defaultExcludeTags = cliExcludeTags ?: (slowExplicitlyIncluded ? null : 'slow')
|
final String defaultExcludeTags = cliExcludeTags ?: (slowExplicitlyIncluded ? 'large-model' : 'slow,large-model')
|
||||||
configureJUnitPlatformTags(it, cliIncludeTags, defaultExcludeTags)
|
configureJUnitPlatformTags(it, cliIncludeTags, defaultExcludeTags)
|
||||||
finalizedBy(tasks.named('jacocoTestReport'))
|
finalizedBy(tasks.named('jacocoTestReport'))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
def largeModelMaxHeap = providers.gradleProperty('radixorLargeModelMaxHeap').orElse('6g')
|
||||||
|
def runtimeModelId = providers.gradleProperty('modelId').orElse('pl-pl-polimorf')
|
||||||
|
def runtimeModelClasspath = configurations.testRuntimeClasspath.incoming.artifactView {
|
||||||
|
componentFilter { componentIdentifier ->
|
||||||
|
if (!(componentIdentifier instanceof org.gradle.api.artifacts.component.ProjectComponentIdentifier)) {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
final String projectPath = componentIdentifier.projectPath
|
||||||
|
return !projectPath.startsWith(':models:') || projectPath == ":models:${runtimeModelId.get()}"
|
||||||
|
}
|
||||||
|
}.files
|
||||||
|
|
||||||
|
tasks.register('runtimeModelIntegrationTest', Test) {
|
||||||
|
group = 'verification'
|
||||||
|
description = 'Constructs one complete selected runtime model in an isolated, memory-sized JVM.'
|
||||||
|
testClassesDirs = sourceSets.test.output.classesDirs
|
||||||
|
classpath = sourceSets.test.output + sourceSets.main.output + sourceSets.jmh.output + runtimeModelClasspath
|
||||||
|
dependsOn(tasks.named('compileTestJava'))
|
||||||
|
useJUnitPlatform {
|
||||||
|
includeTags('large-model')
|
||||||
|
}
|
||||||
|
systemProperty('radixor.test.modelId', runtimeModelId.get())
|
||||||
|
minHeapSize = '1g'
|
||||||
|
maxHeapSize = largeModelMaxHeap.get()
|
||||||
|
maxParallelForks = 1
|
||||||
|
forkEvery = 1
|
||||||
|
reports {
|
||||||
|
junitXml.required = true
|
||||||
|
html.required = true
|
||||||
|
}
|
||||||
|
doFirst {
|
||||||
|
logger.lifecycle("Runtime model integration uses model '{}' with maximum heap {}.",
|
||||||
|
systemProperties.get('radixor.test.modelId'), maxHeapSize)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
def configureTaggedTestProfile = { String taskName, String includeTagsExpr, String excludeTagsExpr = null,
|
def configureTaggedTestProfile = { String taskName, String includeTagsExpr, String excludeTagsExpr = null,
|
||||||
String taskDescription = null, String testNameExcludePatterns = null ->
|
String taskDescription = null, String testNameExcludePatterns = null ->
|
||||||
tasks.register(taskName, Test) {
|
tasks.register(taskName, Test) {
|
||||||
@@ -174,10 +293,6 @@ def configureTaggedTestProfile = { String taskName, String includeTagsExpr, Stri
|
|||||||
classpath = sourceSets.test.runtimeClasspath
|
classpath = sourceSets.test.runtimeClasspath
|
||||||
dependsOn(tasks.named('compileTestJava'))
|
dependsOn(tasks.named('compileTestJava'))
|
||||||
|
|
||||||
doFirst {
|
|
||||||
jvmArgs "-javaagent:${configurations.mockitoAgent.singleFile}"
|
|
||||||
}
|
|
||||||
|
|
||||||
if (testNameExcludePatterns != null && !testNameExcludePatterns.isBlank()) {
|
if (testNameExcludePatterns != null && !testNameExcludePatterns.isBlank()) {
|
||||||
filter {
|
filter {
|
||||||
testNameExcludePatterns.split(',').each { String pattern ->
|
testNameExcludePatterns.split(',').each { String pattern ->
|
||||||
@@ -238,11 +353,19 @@ configureTaggedTestProfile(
|
|||||||
configureTaggedTestProfile(
|
configureTaggedTestProfile(
|
||||||
'ciRelease',
|
'ciRelease',
|
||||||
null,
|
null,
|
||||||
'slow',
|
'slow,large-model',
|
||||||
'Release-profile validation of all non-slow tests.',
|
'Release-profile validation of all non-slow tests.',
|
||||||
'org.egothor.stemmer.CompileIntegrationTest*,org.egothor.stemmer.StemmerPatchTrieLoaderTest$BundledDictionaryTests*'
|
'org.egothor.stemmer.CompileIntegrationTest*,org.egothor.stemmer.StemmerPatchTrieLoaderTest$BundledDictionaryTests*'
|
||||||
)
|
)
|
||||||
|
|
||||||
|
tasks.named('ciRelease', Test) {
|
||||||
|
dependsOn('prepareModelConsumerTestRepository')
|
||||||
|
systemProperty('radixor.consumer.repository',
|
||||||
|
layout.buildDirectory.dir('model-consumer-repository').get().asFile.absolutePath)
|
||||||
|
systemProperty('radixor.core.version', version.toString())
|
||||||
|
systemProperty('radixor.catalog.version', project(':models:standard').version.toString())
|
||||||
|
}
|
||||||
|
|
||||||
configureTaggedTestProfile(
|
configureTaggedTestProfile(
|
||||||
'ciNightly',
|
'ciNightly',
|
||||||
'fuzz',
|
'fuzz',
|
||||||
@@ -318,25 +441,393 @@ tasks.named('check') {
|
|||||||
// no-default, only on-demand: dependsOn(tasks.named('dependencyCheckAnalyze'))
|
// no-default, only on-demand: dependsOn(tasks.named('dependencyCheckAnalyze'))
|
||||||
}
|
}
|
||||||
|
|
||||||
allprojects {
|
tasks.register('verifyCoreJarExcludesModels') {
|
||||||
tasks.matching { it.name == 'cyclonedxDirectBom' }.configureEach {
|
group = 'verification'
|
||||||
includeConfigs = ['runtimeClasspath', 'compileClasspath']
|
description = 'Verifies that the root Radixor JAR contains no language dictionary bytes.'
|
||||||
skipConfigs = ['testRuntimeClasspath', 'testCompileClasspath', 'jmh.*', 'mockitoAgent']
|
dependsOn(tasks.named('jar'))
|
||||||
includeBomSerialNumber = true
|
doLast {
|
||||||
includeLicenseText = false
|
File archive = tasks.named('jar', Jar).get().archiveFile.get().asFile
|
||||||
includeMetadataResolution = true
|
List<String> dictionaries = []
|
||||||
includeBuildSystem = true
|
new java.util.zip.ZipFile(archive).withCloseable { zip ->
|
||||||
|
zip.entries().each { entry -> if (entry.name.endsWith('/stemmer.gz')) dictionaries.add(entry.name) }
|
||||||
|
}
|
||||||
|
if (!dictionaries.isEmpty()) {
|
||||||
|
throw new GradleException('The org.egothor:radixor JAR must not contain model data: ' + dictionaries)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
tasks.named('cyclonedxBom') {
|
tasks.register('verifyJavaLicenseHeaders') {
|
||||||
|
group = 'verification'
|
||||||
|
description = 'Verifies deterministic license classification for every maintained Java source file.'
|
||||||
|
inputs.file(layout.projectDirectory.file('gradle/java-license-header.txt'))
|
||||||
|
inputs.files(fileTree('src/main/java') { include '**/*.java' })
|
||||||
|
inputs.files(fileTree('src/test/java') { include '**/*.java' })
|
||||||
|
inputs.files(fileTree('src/jmh/java') { include '**/*.java' })
|
||||||
|
outputs.file(layout.buildDirectory.file('reports/license/java-license-headers.txt'))
|
||||||
|
doLast {
|
||||||
|
String canonicalHeader = layout.projectDirectory.file('gradle/java-license-header.txt')
|
||||||
|
.asFile.getText('UTF-8')
|
||||||
|
File canonicalSource = file('src/main/java/org/egothor/stemmer/CaseProcessingMode.java')
|
||||||
|
if (!canonicalSource.getText('UTF-8').startsWith(canonicalHeader)) {
|
||||||
|
throw new GradleException('CaseProcessingMode.java does not begin with the canonical Radixor license template.')
|
||||||
|
}
|
||||||
|
|
||||||
|
List<File> maintainedSources = files(
|
||||||
|
fileTree('src/main/java') { include '**/*.java' },
|
||||||
|
fileTree('src/test/java') { include '**/*.java' },
|
||||||
|
fileTree('src/jmh/java') { include '**/*.java' })
|
||||||
|
.files.toList().sort { File left, File right ->
|
||||||
|
relativePath(left) <=> relativePath(right)
|
||||||
|
}
|
||||||
|
List<String> classifications = []
|
||||||
|
List<String> failures = []
|
||||||
|
maintainedSources.each { File sourceFile ->
|
||||||
|
String relative = relativePath(sourceFile)
|
||||||
|
String content = sourceFile.getText('UTF-8')
|
||||||
|
if (content.startsWith(canonicalHeader)) {
|
||||||
|
classifications.add("RADIXOR_CANONICAL_HEADER ${relative}")
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
String leadingNotice = ''
|
||||||
|
if (content.startsWith('/*')) {
|
||||||
|
int closingIndex = content.indexOf('*/')
|
||||||
|
if (closingIndex >= 0) {
|
||||||
|
leadingNotice = content.substring(0, closingIndex + 2)
|
||||||
|
}
|
||||||
|
} else if (content.startsWith('//')) {
|
||||||
|
leadingNotice = content.readLines().takeWhile { String line ->
|
||||||
|
line.startsWith('//') || line.isBlank()
|
||||||
|
}.join('\n')
|
||||||
|
}
|
||||||
|
String remainder = content.substring(leadingNotice.length()).stripLeading()
|
||||||
|
boolean duplicateNotice = !leadingNotice.isEmpty()
|
||||||
|
&& (remainder.startsWith('/*') || remainder.startsWith('//'))
|
||||||
|
boolean historicalRadixor = leadingNotice =~ /(?s)Copyright \(C\) \d{4}(?:-\d{4})?, Leo Galambos/
|
||||||
|
&& leadingNotice.contains('All rights reserved.')
|
||||||
|
&& leadingNotice.contains('Redistribution and use in source and binary forms')
|
||||||
|
&& leadingNotice.contains('THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS')
|
||||||
|
boolean thirdPartyOrProvenance = leadingNotice =~ /(?is)(SPDX-License-Identifier|Licensed under|MIT License|Apache License|Permission is hereby granted|Original source|Adapted from|Ported from|Source:\s*\S)/
|
||||||
|
if (duplicateNotice) {
|
||||||
|
classifications.add("AMBIGUOUS_AUTHORSHIP ${relative}")
|
||||||
|
failures.add("${relative}: duplicate leading comment blocks")
|
||||||
|
} else if (historicalRadixor) {
|
||||||
|
classifications.add("RADIXOR_HISTORICAL_HEADER ${relative}")
|
||||||
|
} else if (thirdPartyOrProvenance) {
|
||||||
|
classifications.add("THIRD_PARTY_OR_PROVENANCE_HEADER ${relative}")
|
||||||
|
} else {
|
||||||
|
classifications.add("AMBIGUOUS_AUTHORSHIP ${relative}")
|
||||||
|
failures.add("${relative}: no recognized governing license or provenance header")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
File report = layout.buildDirectory.file('reports/license/java-license-headers.txt').get().asFile
|
||||||
|
report.parentFile.mkdirs()
|
||||||
|
report.setText(classifications.join('\n') + '\n', 'UTF-8')
|
||||||
|
if (!failures.isEmpty()) {
|
||||||
|
throw new GradleException('Maintained Java license verification failed: '
|
||||||
|
+ failures.sort().join(', '))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.register('verifyAllDefaultModels') {
|
||||||
|
group = 'verification'
|
||||||
|
description = 'Verifies that every language default model project is configured.'
|
||||||
|
dependsOn(defaultModelProjects().collect { Project modelProject ->
|
||||||
|
modelProject.path + ':verifyModelDescriptor'
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.register('verifyAllModels') {
|
||||||
|
group = 'verification'
|
||||||
|
description = 'Runs complete validation and artifact verification for every independently versioned model module.'
|
||||||
|
dependsOn(modelProjects().collect { Project modelProject -> modelProject.tasks.named('check') })
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.register('verifyJmhModelClasspath') {
|
||||||
|
group = 'verification'
|
||||||
|
description = 'Verifies that JMH receives each individual model JAR exactly once and embeds no dictionary.'
|
||||||
|
dependsOn(tasks.named('jmhJar'))
|
||||||
|
dependsOn(modelProjects().collect { Project modelProject -> modelProject.tasks.named('jar') })
|
||||||
|
outputs.file(layout.buildDirectory.file('reports/models/jmh-model-classpath.txt'))
|
||||||
|
doLast {
|
||||||
|
List<File> modelJars = configurations.jmhRuntimeClasspath.files.findAll { File dependency ->
|
||||||
|
dependency.name.startsWith('radixor-model-') && dependency.name.endsWith('.jar')
|
||||||
|
}.sort { File left, File right -> left.name <=> right.name }
|
||||||
|
List<String> expectedPrefixes = modelProjects().collect { Project modelProject ->
|
||||||
|
"radixor-model-${modelProject.name}-"
|
||||||
|
}
|
||||||
|
expectedPrefixes.each { String prefix ->
|
||||||
|
List<File> matches = modelJars.findAll { File dependency -> dependency.name.startsWith(prefix) }
|
||||||
|
if (matches.size() != 1) {
|
||||||
|
throw new GradleException("JMH must resolve exactly one model JAR with prefix ${prefix}; resolved ${matches}.")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (modelJars.any { File dependency -> dependency.name.contains('benchmark-pack') }) {
|
||||||
|
throw new GradleException('JMH must not resolve a benchmark-pack artifact.')
|
||||||
|
}
|
||||||
|
File executable = tasks.named('jmhJar', Jar).get().archiveFile.get().asFile
|
||||||
|
if (!zipTree(executable).matching { include '**/stemmer.gz' }.isEmpty()) {
|
||||||
|
throw new GradleException('The JMH executable JAR must not embed model dictionaries.')
|
||||||
|
}
|
||||||
|
File report = layout.buildDirectory.file('reports/models/jmh-model-classpath.txt').get().asFile
|
||||||
|
report.parentFile.mkdirs()
|
||||||
|
report.setText(modelJars.collect { File dependency -> dependency.name }.join('\n') + '\n', 'UTF-8')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.named('prepareModelConsumerTestRepository') {
|
||||||
|
coreVersion = version.toString()
|
||||||
|
catalogVersion = project(':models:standard').version.toString()
|
||||||
|
modelVersions = modelProjects().collectEntries { Project modelProject ->
|
||||||
|
final String modelVersion = providers.gradleProperty('modelReleaseVersion')
|
||||||
|
.orElse(providers.fileContents(modelProject.layout.projectDirectory.file('model-version.txt'))
|
||||||
|
.asText.map(String::trim))
|
||||||
|
.get()
|
||||||
|
[(modelProject.name): modelVersion]
|
||||||
|
}
|
||||||
|
corePom = layout.file(tasks.named('generatePomFileForMavenJavaPublication').map { it.destination })
|
||||||
|
coreJar = tasks.named('jar', Jar).flatMap { it.archiveFile }
|
||||||
|
modelPoms.from(modelProjects().collect { Project modelProject ->
|
||||||
|
modelProject.tasks.named('generatePomFileForModelPublication').map { it.destination }
|
||||||
|
})
|
||||||
|
modelJars.from(modelProjects().collect { Project modelProject ->
|
||||||
|
modelProject.tasks.named('jar', Jar).flatMap { it.archiveFile }
|
||||||
|
})
|
||||||
|
standardPom = layout.file(project(':models:standard').tasks.named('generatePomFileForStandardPublication')
|
||||||
|
.map { it.destination })
|
||||||
|
bomPom = layout.file(project(':models:bom').tasks.named('generatePomFileForBomPublication')
|
||||||
|
.map { it.destination })
|
||||||
|
repositoryDirectory = layout.buildDirectory.dir('model-consumer-repository')
|
||||||
|
}
|
||||||
|
|
||||||
|
def cleanModelCatalogStaging = tasks.register('cleanModelCatalogStaging') {
|
||||||
|
group = 'publishing'
|
||||||
|
description = 'Cleans the isolated model catalog Maven staging repository.'
|
||||||
|
doLast {
|
||||||
|
project.delete(layout.buildDirectory.dir('model-catalog-staging-repository'))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
gradle.projectsEvaluated {
|
||||||
|
project(':models:standard').tasks.named('publishStandardPublicationToCatalogStagingRepository') {
|
||||||
|
dependsOn(cleanModelCatalogStaging)
|
||||||
|
}
|
||||||
|
project(':models:bom').tasks.named('publishBomPublicationToCatalogStagingRepository') {
|
||||||
|
dependsOn(cleanModelCatalogStaging)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.register('prepareModelCatalogReleaseCandidate') {
|
||||||
|
group = 'publishing'
|
||||||
|
description = 'Stages the POM-only standard aggregate and model BOM with Central checksums.'
|
||||||
|
dependsOn(project(':models:standard').tasks.named('check'))
|
||||||
|
dependsOn(project(':models:bom').tasks.named('check'))
|
||||||
|
dependsOn(':models:standard:publishStandardPublicationToCatalogStagingRepository')
|
||||||
|
dependsOn(':models:bom:publishBomPublicationToCatalogStagingRepository')
|
||||||
|
outputs.dir(layout.buildDirectory.dir('model-catalog-staging-repository'))
|
||||||
|
doLast {
|
||||||
|
File repository = layout.buildDirectory.dir('model-catalog-staging-repository').get().asFile
|
||||||
|
repository.eachFileRecurse { File artifact ->
|
||||||
|
if (artifact.isFile() && !['.md5', '.sha1', '.sha256', '.sha512'].any {
|
||||||
|
String extension -> artifact.name.endsWith(extension)
|
||||||
|
}) {
|
||||||
|
['MD5': 'md5', 'SHA-1': 'sha1'].each { String algorithm, String extension ->
|
||||||
|
String digest = java.security.MessageDigest.getInstance(algorithm)
|
||||||
|
.digest(artifact.bytes).encodeHex().toString()
|
||||||
|
new File(artifact.absolutePath + ".${extension}").setText(digest, 'US-ASCII')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.register('modelCatalogCentralBundle', Zip) {
|
||||||
|
group = 'publishing'
|
||||||
|
description = 'Builds the local POM-only model catalog bundle without remote publication.'
|
||||||
|
dependsOn(tasks.named('prepareModelCatalogReleaseCandidate'))
|
||||||
|
from(layout.buildDirectory.dir('model-catalog-staging-repository')) {
|
||||||
|
exclude('**/maven-metadata*.xml*', '**/*.module*')
|
||||||
|
}
|
||||||
|
destinationDirectory = layout.buildDirectory.dir('model-catalog-release-candidate')
|
||||||
|
archiveFileName = "radixor-models-catalog-${project(':models:standard').version}-central-bundle.zip"
|
||||||
|
doFirst {
|
||||||
|
if (providers.environmentVariable('GITHUB_REF_TYPE').orNull == 'tag'
|
||||||
|
&& (providers.environmentVariable('SIGNING_KEY').orNull?.isBlank() != false
|
||||||
|
|| providers.environmentVariable('SIGNING_PASSWORD').orNull?.isBlank() != false)) {
|
||||||
|
throw new GradleException('A tagged model catalog release requires SIGNING_KEY and SIGNING_PASSWORD.')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.register('verifyModelCatalogReleaseCandidate') {
|
||||||
|
group = 'verification'
|
||||||
|
description = 'Verifies that the local catalog bundle contains only two POM publications, signatures when configured, and checksums.'
|
||||||
|
dependsOn(tasks.named('modelCatalogCentralBundle'))
|
||||||
|
outputs.file(layout.buildDirectory.file('reports/models/catalog-release-candidate.txt'))
|
||||||
|
doLast {
|
||||||
|
File bundle = tasks.named('modelCatalogCentralBundle', Zip).get().archiveFile.get().asFile
|
||||||
|
List<String> entries = []
|
||||||
|
new java.util.zip.ZipFile(bundle).withCloseable { java.util.zip.ZipFile archive ->
|
||||||
|
archive.entries().each { java.util.zip.ZipEntry entry ->
|
||||||
|
if (!entry.directory) entries.add(entry.name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
entries.sort()
|
||||||
|
List<String> poms = entries.findAll { String entry -> entry.endsWith('.pom') }
|
||||||
|
if (poms.size() != 2 || entries.any { String entry ->
|
||||||
|
entry.endsWith('.jar') || entry.endsWith('/stemmer.gz') || entry.contains('benchmark-pack')
|
||||||
|
}) {
|
||||||
|
throw new GradleException('The model catalog bundle must contain only the standard and BOM POM publications.')
|
||||||
|
}
|
||||||
|
List<String> unsupported = entries.findAll { String entry ->
|
||||||
|
!(entry.endsWith('.pom') || entry.endsWith('.pom.md5') || entry.endsWith('.pom.sha1')
|
||||||
|
|| entry.endsWith('.pom.sha256') || entry.endsWith('.pom.sha512')
|
||||||
|
|| entry.endsWith('.pom.asc') || entry.endsWith('.pom.asc.md5')
|
||||||
|
|| entry.endsWith('.pom.asc.sha1') || entry.endsWith('.pom.asc.sha256')
|
||||||
|
|| entry.endsWith('.pom.asc.sha512'))
|
||||||
|
}
|
||||||
|
if (!unsupported.isEmpty()) {
|
||||||
|
throw new GradleException("The model catalog bundle contains unsupported files: ${unsupported}.")
|
||||||
|
}
|
||||||
|
poms.each { String pom ->
|
||||||
|
if (!entries.contains(pom + '.md5') || !entries.contains(pom + '.sha1')) {
|
||||||
|
throw new GradleException("The model catalog POM is missing Central checksums: ${pom}.")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
File report = layout.buildDirectory.file('reports/models/catalog-release-candidate.txt').get().asFile
|
||||||
|
report.parentFile.mkdirs()
|
||||||
|
report.setText("Bundle: ${bundle.name}\nBytes: ${bundle.length()}\n" + entries.join('\n') + '\n', 'UTF-8')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.register('verifyArtifactSizes') {
|
||||||
|
group = 'verification'
|
||||||
|
description = 'Reports artifact sizes and rejects dictionary bytes in core.'
|
||||||
|
dependsOn(tasks.named('verifyCoreJarExcludesModels'))
|
||||||
|
doLast {
|
||||||
|
File archive = tasks.named('jar', Jar).get().archiveFile.get().asFile
|
||||||
|
println('org.egothor:radixor:' + version + ' ' + archive.length() + ' bytes')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.named('check') {
|
||||||
|
dependsOn(tasks.named('verifyCoreJarExcludesModels'))
|
||||||
|
dependsOn(tasks.named('verifyAllDefaultModels'))
|
||||||
|
dependsOn(tasks.named('verifyJmhModelClasspath'))
|
||||||
|
dependsOn(tasks.named('verifyJavaLicenseHeaders'))
|
||||||
|
}
|
||||||
|
|
||||||
|
def modelCatalogText = providers.provider {
|
||||||
|
StringBuilder output = new StringBuilder()
|
||||||
|
output.append('| Model ID | Language | Default | Coordinates | Version | Source | Repository | Source version | Revision | Revision status | License | Attribution | SHA-256 | Bytes |\n')
|
||||||
|
output.append('|---|---|---:|---|---:|---|---|---|---|---|---|---|---|---:|\n')
|
||||||
|
modelProjects().each { Project modelProject ->
|
||||||
|
String script = modelProject.file('build.gradle').getText('UTF-8')
|
||||||
|
def value = { String key ->
|
||||||
|
def matcher = script =~ /(?m)^\s*${key}\s*=\s*(?:'([^']+)'|([^\s]+))\s*$/
|
||||||
|
if (!matcher.find()) return 'unavailable'
|
||||||
|
return matcher.group(1) != null ? matcher.group(1) : matcher.group(2)
|
||||||
|
}
|
||||||
|
File input = modelProject.file('src/modelInput/stemmer.gz')
|
||||||
|
String checksum = java.security.MessageDigest.getInstance('SHA-256').digest(input.bytes).encodeHex().toString()
|
||||||
|
output.append('| ').append(modelProject.name)
|
||||||
|
.append(' | ').append(value('language'))
|
||||||
|
.append(' | ').append(value('defaultModel'))
|
||||||
|
.append(' | org.egothor:radixor-model-').append(modelProject.name)
|
||||||
|
.append(' | ').append(modelProject.file('model-version.txt').text.trim())
|
||||||
|
.append(' | ').append(value('sourceName'))
|
||||||
|
.append(' | ').append(value('sourceRepository'))
|
||||||
|
.append(' | ').append(value('sourceVersion'))
|
||||||
|
.append(' | ').append(value('sourceRevision'))
|
||||||
|
.append(' | ').append(value('sourceRevisionStatus'))
|
||||||
|
.append(' | ').append(value('sourceLicense'))
|
||||||
|
.append(' | ').append(value('sourceAttribution'))
|
||||||
|
.append(' | ').append(checksum)
|
||||||
|
.append(' | ').append(input.length()).append(' |\n')
|
||||||
|
}
|
||||||
|
return output.toString()
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.register('generateModelCatalogDocumentation') {
|
||||||
|
group = 'documentation'
|
||||||
|
description = 'Generates the deterministic model catalog in the build documentation staging tree.'
|
||||||
|
outputs.file(layout.buildDirectory.file('mkdocs-source/stemmer-model-catalog.md'))
|
||||||
|
doLast {
|
||||||
|
File catalog = layout.buildDirectory.file('mkdocs-source/stemmer-model-catalog.md').get().asFile
|
||||||
|
catalog.parentFile.mkdirs()
|
||||||
|
catalog.setText('# Published Stemmer Model Catalog\n\n' + modelCatalogText.get(), 'UTF-8')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.register('prepareMkDocsSource', Sync) {
|
||||||
|
group = 'documentation'
|
||||||
|
description = 'Stages maintained documentation, generated catalog, and MkDocs configuration under build/.'
|
||||||
|
dependsOn(modelProjects().collect { Project modelProject -> modelProject.path + ':verifyModelDescriptor' })
|
||||||
|
into(layout.buildDirectory.dir('mkdocs-source'))
|
||||||
|
from(layout.projectDirectory.dir('docs'))
|
||||||
|
doLast {
|
||||||
|
File catalog = layout.buildDirectory.file('mkdocs-source/stemmer-model-catalog.md').get().asFile
|
||||||
|
catalog.setText('# Published Stemmer Model Catalog\n\n' + modelCatalogText.get(), 'UTF-8')
|
||||||
|
File buildsPage = layout.buildDirectory.file('mkdocs-source/builds.md').get().asFile
|
||||||
|
buildsPage.setText('# Historical Builds\n\nThe Pages publication workflow replaces this staging placeholder with the retained build index.\n', 'UTF-8')
|
||||||
|
File configuration = layout.buildDirectory.file('mkdocs/mkdocs.yml').get().asFile
|
||||||
|
configuration.parentFile.mkdirs()
|
||||||
|
configuration.setText(layout.projectDirectory.file('mkdocs.yml').asFile.getText('UTF-8')
|
||||||
|
+ '\ndocs_dir: ../mkdocs-source\nsite_dir: ../mkdocs-site\n', 'UTF-8')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.register('verifyModelCatalogDocumentation') {
|
||||||
|
group = 'verification'
|
||||||
|
description = 'Validates model metadata and the generated build-directory MkDocs catalog.'
|
||||||
|
dependsOn(tasks.named('prepareMkDocsSource'))
|
||||||
|
dependsOn(tasks.named('verifyAllDefaultModels'))
|
||||||
|
doLast {
|
||||||
|
File catalog = layout.buildDirectory.file('mkdocs-source/stemmer-model-catalog.md').get().asFile
|
||||||
|
String expected = '# Published Stemmer Model Catalog\n\n' + modelCatalogText.get()
|
||||||
|
if (!catalog.isFile() || catalog.getText('UTF-8') != expected) {
|
||||||
|
throw new GradleException('The staged model catalog is missing or nondeterministic.')
|
||||||
|
}
|
||||||
|
List<String> identifiers = modelProjects().collect { Project modelProject -> modelProject.name }
|
||||||
|
if (identifiers != identifiers.sort()) {
|
||||||
|
throw new GradleException('Published model projects are not in deterministic model-ID order.')
|
||||||
|
}
|
||||||
|
identifiers.each { String identifier ->
|
||||||
|
if (!expected.contains('org.egothor:radixor-model-' + identifier)) {
|
||||||
|
throw new GradleException('The staged model catalog omits published model ' + identifier + '.')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (!layout.buildDirectory.file('mkdocs/mkdocs.yml').get().asFile.isFile()) {
|
||||||
|
throw new GradleException('The staged MkDocs configuration is missing.')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.named('check') {
|
||||||
|
dependsOn(tasks.named('verifyModelCatalogDocumentation'))
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.named('cyclonedxDirectBom') {
|
||||||
|
includeConfigs = ['runtimeClasspath', 'compileClasspath']
|
||||||
|
skipConfigs = ['testRuntimeClasspath', 'testCompileClasspath', 'jmh.*', 'mockitoAgent']
|
||||||
includeBomSerialNumber = true
|
includeBomSerialNumber = true
|
||||||
includeLicenseText = false
|
includeLicenseText = false
|
||||||
|
includeMetadataResolution = true
|
||||||
includeBuildSystem = true
|
includeBuildSystem = true
|
||||||
jsonOutput.set(sbomReportsDirectory.map { it.file('radixor-sbom.json') })
|
jsonOutput.set(sbomReportsDirectory.map { it.file('radixor-sbom.json') })
|
||||||
xmlOutput.set(sbomReportsDirectory.map { it.file('radixor-sbom.xml') })
|
xmlOutput.set(sbomReportsDirectory.map { it.file('radixor-sbom.xml') })
|
||||||
}
|
}
|
||||||
|
|
||||||
|
subprojects {
|
||||||
|
tasks.matching { Task candidate -> candidate.name == 'cyclonedxDirectBom' }.configureEach {
|
||||||
|
enabled = false
|
||||||
|
description = 'Disabled because the root project exclusively owns CycloneDX SBOM generation.'
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
pitest {
|
pitest {
|
||||||
pitestVersion = '1.22.1'
|
pitestVersion = '1.22.1'
|
||||||
junit5PluginVersion = '1.2.3'
|
junit5PluginVersion = '1.2.3'
|
||||||
@@ -359,7 +850,10 @@ pitest {
|
|||||||
excludedTestClasses = [
|
excludedTestClasses = [
|
||||||
'org.egothor.stemmer.CompileIntegrationTest',
|
'org.egothor.stemmer.CompileIntegrationTest',
|
||||||
'org.egothor.stemmer.StemmerPatchTrieLoaderTest',
|
'org.egothor.stemmer.StemmerPatchTrieLoaderTest',
|
||||||
'org.egothor.stemmer.StemmerKnowledgeExperimentTest'
|
'org.egothor.stemmer.StemmerKnowledgeExperimentTest',
|
||||||
|
// These integration tests require dedicated Gradle task wiring and must not run in PIT worker JVMs.
|
||||||
|
'org.egothor.stemmer.FullRuntimeModelIntegrationTest',
|
||||||
|
'org.egothor.stemmer.ModelDependencyResolutionTest'
|
||||||
]
|
]
|
||||||
outputFormats = ['XML', 'HTML']
|
outputFormats = ['XML', 'HTML']
|
||||||
timestampedReports = false
|
timestampedReports = false
|
||||||
@@ -426,6 +920,7 @@ tasks.named('distTar') {
|
|||||||
|
|
||||||
jmh {
|
jmh {
|
||||||
jmhVersion = '1.37'
|
jmhVersion = '1.37'
|
||||||
|
includeTests = false
|
||||||
warmupIterations = 3
|
warmupIterations = 3
|
||||||
iterations = 5
|
iterations = 5
|
||||||
fork = 1
|
fork = 1
|
||||||
@@ -442,7 +937,14 @@ jmh {
|
|||||||
|
|
||||||
tasks.named('jmh') {
|
tasks.named('jmh') {
|
||||||
group = 'verification'
|
group = 'verification'
|
||||||
description = 'Runs JMH benchmarks for the Radixor algorithmic core and Snowball comparison suite.'
|
description = 'Runs JMH benchmarks for the Radixor algorithmic core and external stemmer comparison suites.'
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.named('jmhJar', Jar) {
|
||||||
|
exclude 'META-INF/radixor/models.index'
|
||||||
|
exclude 'META-INF/radixor/models/**'
|
||||||
|
exclude 'org/egothor/stemmer/models/**'
|
||||||
|
exclude 'META-INF/LICENSES/**'
|
||||||
}
|
}
|
||||||
|
|
||||||
apply from: 'gradle/lucene-benchmarks.gradle'
|
apply from: 'gradle/lucene-benchmarks.gradle'
|
||||||
@@ -468,6 +970,100 @@ tasks.register('regressionArtifactGenerator', JavaExec) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
tasks.register('stemmingQuality', JavaExec) {
|
||||||
|
group = 'verification'
|
||||||
|
description = 'Evaluates pairwise over-stemming and under-stemming against registered model dictionary groups.'
|
||||||
|
dependsOn(tasks.named('testClasses'))
|
||||||
|
dependsOn(tasks.named('jmhClasses'))
|
||||||
|
dependsOn(tasks.named('prepareBenchmarkModelInputs'))
|
||||||
|
classpath = files(sourceSets.test.runtimeClasspath, configurations.stemmingQualityJmhRuntime)
|
||||||
|
mainClass = 'org.egothor.stemmer.benchmark.quality.StemmingQualityApplication'
|
||||||
|
args layout.buildDirectory.dir('reports/stemming-quality').get().asFile.absolutePath,
|
||||||
|
layout.buildDirectory.dir('generated/benchmark-model-inputs').get().asFile.absolutePath,
|
||||||
|
providers.gradleProperty('stemmingQualityLanguage').getOrElse(''),
|
||||||
|
providers.gradleProperty('stemmingQualityStemmer').getOrElse(''),
|
||||||
|
providers.gradleProperty('stemmingQualityMode').getOrElse(''),
|
||||||
|
providers.gradleProperty('stemmingQualityOutputPolicy').getOrElse(''),
|
||||||
|
providers.gradleProperty('stemmingQualityRankMetric').getOrElse('PAIRWISE_F05'),
|
||||||
|
providers.gradleProperty('stemmingQualityAudit').getOrElse('false'),
|
||||||
|
providers.gradleProperty('stemmingQualityAuditLimit').getOrElse('25')
|
||||||
|
maxHeapSize = '6g'
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.register('prepareBenchmarkModelInputs', Sync) {
|
||||||
|
group = 'verification'
|
||||||
|
description = 'Prepares default model inputs for JMH and quality evaluation without changing source data.'
|
||||||
|
into(layout.buildDirectory.dir('generated/benchmark-model-inputs'))
|
||||||
|
defaultModelProjects().each { Project modelProject ->
|
||||||
|
String languageDirectory = modelProject.name == 'pl-pl-unimorph'
|
||||||
|
? 'pl_pl'
|
||||||
|
: modelProject.name.replace('-default', '').replace('-', '_')
|
||||||
|
from(modelProject.file('src/modelInput/stemmer.gz')) {
|
||||||
|
into(languageDirectory)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.register('publishStemmingQualityDocumentation', JavaExec) {
|
||||||
|
group = 'documentation'
|
||||||
|
description = 'Publishes validated complete stemming-quality results on the language benchmark pages.'
|
||||||
|
dependsOn(tasks.named('testClasses'))
|
||||||
|
classpath = sourceSets.test.runtimeClasspath
|
||||||
|
mainClass = 'org.egothor.stemmer.benchmark.quality.StemmingQualityDocumentationPublisher'
|
||||||
|
args layout.buildDirectory.file('reports/stemming-quality/stemming-quality.csv').get().asFile.absolutePath,
|
||||||
|
layout.projectDirectory.dir('docs').asFile.absolutePath,
|
||||||
|
'update'
|
||||||
|
doFirst {
|
||||||
|
if (!file("$buildDir/reports/stemming-quality/stemming-quality.csv").isFile()) {
|
||||||
|
throw new GradleException('A complete stemming-quality CSV is required. Run stemmingQuality only when no validated complete report is available.')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.register('verifyStemmingQualityDocumentation', JavaExec) {
|
||||||
|
group = 'verification'
|
||||||
|
description = 'Verifies published language-page quality tables against the checked-in authoritative CSV.'
|
||||||
|
dependsOn(tasks.named('testClasses'))
|
||||||
|
classpath = sourceSets.test.runtimeClasspath
|
||||||
|
mainClass = 'org.egothor.stemmer.benchmark.quality.StemmingQualityDocumentationPublisher'
|
||||||
|
args layout.projectDirectory.file('docs/benchmarks/data/stemming-quality.csv').asFile.absolutePath,
|
||||||
|
layout.projectDirectory.dir('docs').asFile.absolutePath,
|
||||||
|
'verify'
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.named('check') {
|
||||||
|
dependsOn(tasks.named('verifyStemmingQualityDocumentation'))
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.register('verifyStemmingQualitySourceSets') {
|
||||||
|
group = 'verification'
|
||||||
|
description = 'Verifies the production, JMH, and standard-test ownership of stemming-quality infrastructure.'
|
||||||
|
doLast {
|
||||||
|
if (sourceSets.findByName('stemmingQualityTest') != null || file('src/stemmingQualityTest').exists()) {
|
||||||
|
throw new GradleException('The obsolete stemmingQualityTest source set or directory still exists.')
|
||||||
|
}
|
||||||
|
if (!file('src/jmh/java/org/egothor/stemmer/benchmark/QualityStemmerMatrix.java').isFile()) {
|
||||||
|
throw new GradleException('The authoritative JMH stemmer matrix is not in src/jmh.')
|
||||||
|
}
|
||||||
|
if (!file('src/test/java/org/egothor/stemmer/benchmark/quality/StemmingQualityApplication.java').isFile()) {
|
||||||
|
throw new GradleException('The stemming-quality evaluator is not in the standard test source set.')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.register('verifyProductionJarExcludesStemmingQuality') {
|
||||||
|
group = 'verification'
|
||||||
|
description = 'Verifies that analytical stemming-quality classes are absent from the production JAR.'
|
||||||
|
dependsOn(tasks.named('jar'))
|
||||||
|
doLast {
|
||||||
|
final File archive = tasks.named('jar').get().archiveFile.get().asFile
|
||||||
|
final def forbidden = zipTree(archive).matching { include '**/benchmark/**' }.files
|
||||||
|
if (!forbidden.isEmpty()) {
|
||||||
|
throw new GradleException("Production JAR contains analytical stemming-quality classes: ${forbidden}")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
tasks.register('printDependencyCheckNvdConfig') {
|
tasks.register('printDependencyCheckNvdConfig') {
|
||||||
doLast {
|
doLast {
|
||||||
System.out.println("NVD API key present: " + (nvdApiKey != null && !nvdApiKey.isBlank()))
|
System.out.println("NVD API key present: " + (nvdApiKey != null && !nvdApiKey.isBlank()))
|
||||||
@@ -518,6 +1114,8 @@ javadoc {
|
|||||||
apply from: 'gradle/snowball-benchmarks.gradle'
|
apply from: 'gradle/snowball-benchmarks.gradle'
|
||||||
apply from: 'gradle/paicehusk-benchmarks.gradle'
|
apply from: 'gradle/paicehusk-benchmarks.gradle'
|
||||||
apply from: 'gradle/opennlp-benchmarks.gradle'
|
apply from: 'gradle/opennlp-benchmarks.gradle'
|
||||||
|
apply from: 'gradle/hunspell-benchmarks.gradle'
|
||||||
|
apply from: 'gradle/cistem-benchmarks.gradle'
|
||||||
|
|
||||||
gradle.taskGraph.whenReady { taskGraph ->
|
gradle.taskGraph.whenReady { taskGraph ->
|
||||||
def banner = """
|
def banner = """
|
||||||
|
|||||||
@@ -17,6 +17,10 @@ The build-time flow is:
|
|||||||
Dictionary -> Mutable trie -> Reduced trie -> Compiled trie
|
Dictionary -> Mutable trie -> Reduced trie -> Compiled trie
|
||||||
```
|
```
|
||||||
|
|
||||||
|
For registered models, the dictionary is an independently versioned GZip resource discovered through a descriptor and verified before this flow begins. The model resource is input to trie construction, not a precompiled trie. See [Model Selection and Loading](model-selection-and-loading.md) for discovery and [Architecture](architecture.md) for component and release boundaries.
|
||||||
|
|
||||||
|
Explicit descriptors and stable model IDs now use the same compiled-value path as language defaults. `loadCompiled(descriptor, ...)` and `loadCompiled(modelId, ...)` first build with serialized patch commands and then map those values to `CompiledPatchCommand` while preserving metadata, reduction semantics, and ranked `getAll` order. Very large inputs can have a high temporary construction peak; PoliMorf is verified in an isolated 6 GiB JVM rather than increasing ordinary test or Gradle daemon heaps.
|
||||||
|
|
||||||
At runtime, the compiled trie does not directly return the final stem string. It returns one or more stored patch commands for the addressed key, and those commands are then applied to the original input word.
|
At runtime, the compiled trie does not directly return the final stem string. It returns one or more stored patch commands for the addressed key, and those commands are then applied to the original input word.
|
||||||
|
|
||||||
## Why this matters
|
## Why this matters
|
||||||
@@ -50,3 +54,5 @@ For most readers, the best order is:
|
|||||||
- [Programmatic usage](programmatic-usage.md)
|
- [Programmatic usage](programmatic-usage.md)
|
||||||
- [CLI compilation](cli-compilation.md)
|
- [CLI compilation](cli-compilation.md)
|
||||||
- [Dictionary format](dictionary-format.md)
|
- [Dictionary format](dictionary-format.md)
|
||||||
|
- [Model selection and loading](model-selection-and-loading.md)
|
||||||
|
- [Stemmer models](stemmer-models.md)
|
||||||
|
|||||||
@@ -2,6 +2,92 @@
|
|||||||
|
|
||||||
This document explains the structural architecture of **Radixor**: what data is stored, how it flows through the build pipeline, and how runtime lookup works once a compiled trie has been produced.
|
This document explains the structural architecture of **Radixor**: what data is stored, how it flows through the build pipeline, and how runtime lookup works once a compiled trie has been produced.
|
||||||
|
|
||||||
|
## Component boundaries
|
||||||
|
|
||||||
|
| Component | Responsibility |
|
||||||
|
|---|---|
|
||||||
|
| Root Radixor core | Patch commands, dictionary parser, trie construction/lookup, descriptor and registry APIs, loaders; no language data |
|
||||||
|
| Individual model module | Immutable source input and license; publishes one independently versioned resource JAR |
|
||||||
|
| `StemmerModelRegistry` | Deterministic index/descriptor discovery and selection by model ID or language default |
|
||||||
|
| `StemmerModelDescriptor` | Immutable public view of validated runtime identity, format, resource, checksum, and source URL |
|
||||||
|
| Model convention plugin | Validates inputs and generates the resource namespace, descriptor, index, license, and publication |
|
||||||
|
| Standard aggregate | POM-only transitive runtime dependencies for one default per language |
|
||||||
|
| Verification classpaths | Direct individual-model dependencies for tests, quality evaluation, and JMH, including optional PoliMorf |
|
||||||
|
| Models BOM | POM-only recommended individual model versions in Maven dependency management |
|
||||||
|
| Documentation staging | Maintained `docs/` plus generated catalog under `build/mkdocs-source/` |
|
||||||
|
| Release workflows | Independent core, one-model, and catalog publication boundaries |
|
||||||
|
|
||||||
|
Read [Model Selection and Loading](model-selection-and-loading.md) for executable application examples and [Stemmer Models](stemmer-models.md) for artifact maintenance.
|
||||||
|
|
||||||
|
## Runtime model discovery and loading
|
||||||
|
|
||||||
|
The implemented sequence is:
|
||||||
|
|
||||||
|
1. use the thread context `ClassLoader`, or an explicit non-null loader;
|
||||||
|
2. enumerate every `META-INF/radixor/models.index` with `ClassLoader.getResources(...)`;
|
||||||
|
3. sort index URLs and validate every descriptor path;
|
||||||
|
4. read descriptor resources and required properties;
|
||||||
|
5. validate model ID, language, exact resource namespace, checksum syntax, format name, and format version;
|
||||||
|
6. sort descriptors by model ID and reject duplicate IDs;
|
||||||
|
7. resolve either `Language.defaultModelId()` or an exact explicit model ID;
|
||||||
|
8. open the declared model resource with the descriptor's discovering loader;
|
||||||
|
9. compare SHA-256 over the compressed bytes;
|
||||||
|
10. decompress GZip and parse UTF-8 Radixor dictionary rows;
|
||||||
|
11. build and reduce a `FrequencyTrie`;
|
||||||
|
12. optionally compile stored patch strings into `CompiledPatchCommand` values for the language-oriented compiled API.
|
||||||
|
|
||||||
|
Descriptor discovery verifies resource presence before selection. Byte-level checksum verification happens when the selected model is loaded. The registry never scans arbitrary JAR contents and never selects “the first model for a language.”
|
||||||
|
|
||||||
|
### Default Polish resolution
|
||||||
|
|
||||||
|
`StemmerPatchTrieLoader.Language.PL_PL` declares `pl-pl-unimorph` in the enum constructor. A language-oriented load creates a context-loader registry and calls `requireDefault(PL_PL)`. If that ID is absent, loading stops with `StemmerModelNotFoundException` naming `org.egothor:radixor-model-pl-pl-unimorph:<version>`.
|
||||||
|
|
||||||
|
### Explicit PoliMorf resolution
|
||||||
|
|
||||||
|
`registry.require("pl-pl-polimorf")` addresses the alternative directly. It neither changes nor consults the Polish default. Both descriptors may coexist; duplicate declarations of either same ID are rejected.
|
||||||
|
|
||||||
|
## Version axes
|
||||||
|
|
||||||
|
| Version | Owned by | Compatibility purpose |
|
||||||
|
|---|---|---|
|
||||||
|
| Core version | Root Git-derived release | Java implementation and public API |
|
||||||
|
| Model artifact version | Each `model-version.txt` | One independently published model JAR |
|
||||||
|
| Catalog version | `models/catalog-version.txt` | Standard aggregate and BOM recommendation set |
|
||||||
|
| Source dictionary version | Module provenance | Upstream lexical data lineage |
|
||||||
|
| Model format version | Descriptor and registry | Loader compatibility for packaged dictionary representation |
|
||||||
|
|
||||||
|
No equality relationship is implied between these values.
|
||||||
|
|
||||||
|
## Build topology and generated output
|
||||||
|
|
||||||
|
`models/model-projects.properties` is the single Gradle-readable topology list for the 21 individual model projects and their default or optional aggregate role. Per-model build scripts and generated descriptors remain authoritative for language, resource, provenance, checksum, and model-specific metadata. `settings.gradle`, root verification classpaths, the standard POM, and BOM constraints all derive membership from the topology list.
|
||||||
|
|
||||||
|
Gradle implicitly creates the lifecycle parent `:models` because child paths are nested. It has no build script, applied project plugin, Maven coordinate, publication, or archive. The root CycloneDX plugin exposes direct-task instances to subprojects internally; every subproject instance is disabled, so only root `:cyclonedxDirectBom` can generate an SBOM. The ignored path `models/build/` is generated output, not a module, and the supported build does not write reports there. Root aggregate reports, including `verifyJmhModelClasspath`, belong under `build/reports/models/`; each individual model retains its own outputs under `models/<model-id>/build/`.
|
||||||
|
|
||||||
|
`models/bom` is a Maven dependency BOM: it controls recommended dependency versions and adds no runtime artifacts. The root CycloneDX task produces a software bill of materials (SBOM) under `build/reports/sbom/`. These artifacts have different purposes and output locations.
|
||||||
|
|
||||||
|
## Build-time model packaging
|
||||||
|
|
||||||
|
The `org.egothor.radixor.model` convention plugin treats `src/modelInput` as immutable. `validateModelInput` checks the GZip stream, strict UTF-8, dictionary rows, ID, semantic model version, and license. `prepareModelResources` copies identical compressed bytes under `org/egothor/stemmer/models/<model-id>/stemmer.gz` and generates the descriptor, index, and packaged license under `build/`. `verifyModelDescriptor` checks the digest, while `verifyModelJar` checks the unique resource, packaged-byte digest, metadata, and dictionary-free documentation artifacts. The root `runtimeModelIntegrationTest` accepts `-PmodelId=<id>` and verifies transformation of a packaged resource into `FrequencyTrie<CompiledPatchCommand>`; PoliMorf release validation depends on this complete runtime test.
|
||||||
|
|
||||||
|
For UniMorph models, the convention validates and packages one model-specific attribution,
|
||||||
|
licensing, provenance, and contribution notice. Source and packaged notice bytes must match. The
|
||||||
|
notice identifies CC BY-SA 3.0 through its canonical URI; no project-wide CC license directory or
|
||||||
|
duplicated full legal text is used. Descriptors distinguish exact revisions from the explicit
|
||||||
|
legacy-import sentinel. UniMorph supplies morphological data; runtime patch commands and tries are
|
||||||
|
constructed by Radixor. The Java software remains BSD-3-Clause, while PoliMorf data remains under
|
||||||
|
its separately packaged BSD-2-Clause license.
|
||||||
|
|
||||||
|
## Release and security boundaries
|
||||||
|
|
||||||
|
| Tag | Publication boundary |
|
||||||
|
|---|---|
|
||||||
|
| `release@<core-version>` | Root `org.egothor:radixor` artifacts only; never model JARs |
|
||||||
|
| `model/<model-id>@<model-version>` | Exactly one matching model; never core, catalog, or other models |
|
||||||
|
| `models-catalog@<catalog-version>` | BOM and standard aggregate only; never model bytes |
|
||||||
|
|
||||||
|
License inclusion, strict metadata paths, resource presence, SHA-256 verification, unsupported-format rejection, and duplicate-ID rejection form the model integrity boundary. These checks detect packaging mistakes and corruption; model data remains non-executable dictionary input.
|
||||||
|
|
||||||
## The central idea
|
## The central idea
|
||||||
|
|
||||||
Radixor does not store final stems directly as a large flat lookup table. Instead, it stores **patch commands** that describe how a word form should be transformed into a canonical stem.
|
Radixor does not store final stems directly as a large flat lookup table. Instead, it stores **patch commands** that describe how a word form should be transformed into a canonical stem.
|
||||||
@@ -10,7 +96,7 @@ For example, if a dictionary states that `running` should reduce to `run`, the f
|
|||||||
|
|
||||||
That matters because many words share similar transformation patterns. Once those mappings are organized in a trie and compiled into a canonical structure, the result is much smaller and more reusable than a naive direct-output table.
|
That matters because many words share similar transformation patterns. Once those mappings are organized in a trie and compiled into a canonical structure, the result is much smaller and more reusable than a naive direct-output table.
|
||||||
|
|
||||||
## End-to-end build flow
|
## Trie construction flow
|
||||||
|
|
||||||
The full build-time flow is:
|
The full build-time flow is:
|
||||||
|
|
||||||
@@ -191,7 +277,7 @@ This is why a very large dictionary can still produce a manageable deployable ru
|
|||||||
|
|
||||||
The compactness of the final artifact should not be confused with the memory usage of preparation.
|
The compactness of the final artifact should not be confused with the memory usage of preparation.
|
||||||
|
|
||||||
Before reduction has completed, the mutable build-time structure must exist in memory. For large dictionaries, that temporary preparation cost can be noticeably higher than the size of the final persisted artifact or the loaded compiled trie.
|
Before reduction has completed, the mutable build-time structure must exist in memory. For large dictionaries, that temporary preparation cost can be noticeably higher than the size of the final persisted artifact or the loaded compiled trie. PoliMorf is the exceptional current case: two complete test constructions took 23.7 and 23.5 seconds, produced 358,993 canonical nodes, and used a task-specific 6 GiB maximum heap. The process peak does not establish the retained heap of the final trie, which is not currently measured separately.
|
||||||
|
|
||||||
That is why the preferred operational model is usually:
|
That is why the preferred operational model is usually:
|
||||||
|
|
||||||
@@ -218,3 +304,5 @@ Determinism matters not only for tests, but also for operational trust. It makes
|
|||||||
- [Reduction Semantics](reduction-semantics.md)
|
- [Reduction Semantics](reduction-semantics.md)
|
||||||
- [Programmatic usage](programmatic-usage.md)
|
- [Programmatic usage](programmatic-usage.md)
|
||||||
- [CLI compilation](cli-compilation.md)
|
- [CLI compilation](cli-compilation.md)
|
||||||
|
- [Model selection and loading](model-selection-and-loading.md)
|
||||||
|
- [Stemmer models](stemmer-models.md)
|
||||||
|
|||||||
@@ -67,6 +67,104 @@
|
|||||||
padding: 0.45rem 0.7rem;
|
padding: 0.45rem 0.7rem;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/* Publication-quality benchmark tables retain identity columns while scrolling. */
|
||||||
|
.quality-table {
|
||||||
|
max-width: 100%;
|
||||||
|
overflow-x: auto;
|
||||||
|
margin: 0.65rem 0 1rem;
|
||||||
|
border: 1px solid var(--md-default-fg-color--lightest);
|
||||||
|
border-radius: 0.2rem;
|
||||||
|
scrollbar-gutter: stable;
|
||||||
|
}
|
||||||
|
|
||||||
|
.quality-table:focus {
|
||||||
|
outline: 0.15rem solid var(--md-accent-fg-color);
|
||||||
|
outline-offset: 0.1rem;
|
||||||
|
}
|
||||||
|
|
||||||
|
.quality-table::before {
|
||||||
|
content: "Scrollable table: Rank, Stemmer, and Output policy remain visible.";
|
||||||
|
display: block;
|
||||||
|
padding: 0.35rem 0.55rem;
|
||||||
|
color: var(--md-default-fg-color--light);
|
||||||
|
font-size: 0.68rem;
|
||||||
|
}
|
||||||
|
|
||||||
|
.quality-table .md-typeset__table,
|
||||||
|
.quality-table table {
|
||||||
|
margin: 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
.quality-table table th:nth-child(1),
|
||||||
|
.quality-table table td:nth-child(1),
|
||||||
|
.quality-table table th:nth-child(2),
|
||||||
|
.quality-table table td:nth-child(2),
|
||||||
|
.quality-table table th:nth-child(3),
|
||||||
|
.quality-table table td:nth-child(3) {
|
||||||
|
position: sticky;
|
||||||
|
z-index: 2;
|
||||||
|
background: var(--md-default-bg-color);
|
||||||
|
background-clip: padding-box;
|
||||||
|
}
|
||||||
|
|
||||||
|
.quality-table table th:nth-child(1),
|
||||||
|
.quality-table table td:nth-child(1) {
|
||||||
|
left: 0;
|
||||||
|
min-width: 2.8rem;
|
||||||
|
}
|
||||||
|
|
||||||
|
.quality-table table th:nth-child(2),
|
||||||
|
.quality-table table td:nth-child(2) {
|
||||||
|
left: 2.8rem;
|
||||||
|
min-width: 13rem;
|
||||||
|
white-space: normal;
|
||||||
|
}
|
||||||
|
|
||||||
|
.quality-table table th:nth-child(3),
|
||||||
|
.quality-table table td:nth-child(3) {
|
||||||
|
left: 15.8rem;
|
||||||
|
min-width: 8.5rem;
|
||||||
|
box-shadow: 0.2rem 0 0.25rem rgb(0 0 0 / 8%);
|
||||||
|
}
|
||||||
|
|
||||||
|
.quality-details > summary {
|
||||||
|
font-weight: 600;
|
||||||
|
}
|
||||||
|
|
||||||
|
@media screen and (max-width: 44.99em) {
|
||||||
|
.quality-table table th:nth-child(2),
|
||||||
|
.quality-table table td:nth-child(2) {
|
||||||
|
min-width: 10rem;
|
||||||
|
}
|
||||||
|
|
||||||
|
.quality-table table th:nth-child(3),
|
||||||
|
.quality-table table td:nth-child(3) {
|
||||||
|
position: static;
|
||||||
|
min-width: 7.5rem;
|
||||||
|
box-shadow: none;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
@media print {
|
||||||
|
.quality-table {
|
||||||
|
overflow: visible;
|
||||||
|
border: 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
.quality-table::before {
|
||||||
|
display: none;
|
||||||
|
}
|
||||||
|
|
||||||
|
.quality-table table th,
|
||||||
|
.quality-table table td {
|
||||||
|
position: static !important;
|
||||||
|
}
|
||||||
|
|
||||||
|
.quality-details:not([open]) > *:not(summary) {
|
||||||
|
display: block;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/* Code blocks */
|
/* Code blocks */
|
||||||
.md-typeset pre > code {
|
.md-typeset pre > code {
|
||||||
font-size: 0.72rem;
|
font-size: 0.72rem;
|
||||||
|
|||||||
@@ -2,6 +2,10 @@
|
|||||||
|
|
||||||
Radixor contains internal trie microbenchmarks, a separate stemmer comparison suite, and a dictionary coverage benchmark for Radixor itself. Published stemmer comparison results must come only from benchmark classes matching `.*StemmerComparisonBenchmark.*`; internal `FrequencyTrie*` microbenchmarks are not part of those results.
|
Radixor contains internal trie microbenchmarks, a separate stemmer comparison suite, and a dictionary coverage benchmark for Radixor itself. Published stemmer comparison results must come only from benchmark classes matching `.*StemmerComparisonBenchmark.*`; internal `FrequencyTrie*` microbenchmarks are not part of those results.
|
||||||
|
|
||||||
|
Every current default Radixor benchmark scenario uses the model ID declared by its `Language.defaultModelId()`. The root JMH runtime configuration depends directly on all default model projects plus optional `pl-pl-polimorf`; no benchmark-pack project or artifact exists. These dependencies are benchmark-only and never enter the root published POM. A PoliMorf comparison must be labeled with model ID `pl-pl-polimorf`, while the default Polish row remains `pl-pl-unimorph`.
|
||||||
|
|
||||||
|
The optional model now has a verified complete compiled loading path. This does not alter existing benchmark rows or make PoliMorf part of the representative English JMH run. Any future full PoliMorf benchmark must provision its documented startup heap independently and record the exact model artifact version and checksum.
|
||||||
|
|
||||||
This page is the entry point for benchmark interpretation. Detailed tables and long reference material are split into focused subpages so that important points do not get buried.
|
This page is the entry point for benchmark interpretation. Detailed tables and long reference material are split into focused subpages so that important points do not get buried.
|
||||||
|
|
||||||
## Key Takeaways
|
## Key Takeaways
|
||||||
@@ -10,7 +14,7 @@ This page is the entry point for benchmark interpretation. Detailed tables and l
|
|||||||
- Radixor is the quality-oriented baseline in same-language comparisons. Its exact-root accuracy is often close to 100%, while many faster competitors are light, minimal, possessive, or aggressive rule-based stemmers with much lower root agreement.
|
- Radixor is the quality-oriented baseline in same-language comparisons. Its exact-root accuracy is often close to 100%, while many faster competitors are light, minimal, possessive, or aggressive rule-based stemmers with much lower root agreement.
|
||||||
- The measured Radixor cost buys dictionary-trained stemming precision. That precision improves search quality by mapping inflected forms to intended dictionary roots instead of approximate or over-reduced stems.
|
- The measured Radixor cost buys dictionary-trained stemming precision. That precision improves search quality by mapping inflected forms to intended dictionary roots instead of approximate or over-reduced stems.
|
||||||
- Speed benchmarks process changed dictionary tokens where the surface form differs from the expected root. Accuracy benchmarks process the complete dictionary.
|
- Speed benchmarks process changed dictionary tokens where the surface form differs from the expected root. Accuracy benchmarks process the complete dictionary.
|
||||||
- Accuracy-only benchmarks intentionally use one deterministic JMH measurement iteration without warmup because repeated precision passes would duplicate the same counters.
|
- Accuracy tables use deterministic auxiliary counters from the current JMH reports. Repeated measurement samples duplicate the same exact-root accounting and are not interpreted as timing results.
|
||||||
- The historical Porter performance badge is retired. Benchmark reporting now uses speed and quality tables rather than a single Porter ratio.
|
- The historical Porter performance badge is retired. Benchmark reporting now uses speed and quality tables rather than a single Porter ratio.
|
||||||
|
|
||||||
## Benchmark Documentation Map
|
## Benchmark Documentation Map
|
||||||
@@ -18,6 +22,9 @@ This page is the entry point for benchmark interpretation. Detailed tables and l
|
|||||||
| Page | Purpose |
|
| Page | Purpose |
|
||||||
| --- | --- |
|
| --- | --- |
|
||||||
| [Benchmark methodology](benchmarks/reference/methodology.md) | Workload design, speed pass, quality pass, normalization policy, and exact-root metrics. |
|
| [Benchmark methodology](benchmarks/reference/methodology.md) | Workload design, speed pass, quality pass, normalization policy, and exact-root metrics. |
|
||||||
|
| [Linguistic quality methodology](benchmarks/reference/linguistic-quality.md) | Pairwise gold standard, over/under-stemming, candidate policies, metrics, and ranking rules. |
|
||||||
|
| [Tested stemmers](benchmarks/reference/tested-stemmers.md) | Upstream attribution, tested versions, language coverage, adapter behaviour, and limitations. |
|
||||||
|
| [Reproducibility and raw data](benchmarks/reference/reproducibility.md) | Versioned quality snapshot, checksum, commands, reports, and provenance limitations. |
|
||||||
| [Benchmark corpora](benchmarks/reference/corpora.md) | Dictionary row counts, complete quality tokens, already-root tokens, changed speed tokens, and timing token counts. |
|
| [Benchmark corpora](benchmarks/reference/corpora.md) | Dictionary row counts, complete quality tokens, already-root tokens, changed speed tokens, and timing token counts. |
|
||||||
| [Benchmark environment and reports](benchmarks/reference/environment.md) | Hardware, OS, JVM, JMH settings, report files, and current badge/report policy. |
|
| [Benchmark environment and reports](benchmarks/reference/environment.md) | Hardware, OS, JVM, JMH settings, report files, and current badge/report policy. |
|
||||||
| [English dictionary coverage benchmark](benchmarks/reference/english-coverage.md) | The quality/speed operating curve for contracted Radixor tries built from 100% down to 10% of English dictionary rows. |
|
| [English dictionary coverage benchmark](benchmarks/reference/english-coverage.md) | The quality/speed operating curve for contracted Radixor tries built from 100% down to 10% of English dictionary rows. |
|
||||||
@@ -37,3 +44,4 @@ The [English dictionary coverage benchmark](benchmarks/reference/english-coverag
|
|||||||
The current measured language results are published in [Language Benchmark Pages](benchmarks/languages/index.md). Generated local report files for this benchmark update are listed in [Benchmark environment and reports](benchmarks/reference/environment.md).
|
The current measured language results are published in [Language Benchmark Pages](benchmarks/languages/index.md). Generated local report files for this benchmark update are listed in [Benchmark environment and reports](benchmarks/reference/environment.md).
|
||||||
|
|
||||||
JMH TXT and CSV reports are still published as benchmark artifacts. They are no longer converted into a Shields endpoint benchmark badge.
|
JMH TXT and CSV reports are still published as benchmark artifacts. They are no longer converted into a Shields endpoint benchmark badge.
|
||||||
|
Model IDs, independent artifact versions, and descriptor checksums identify inputs for future reproducibility. Historical snapshots remain tied to the model inputs used when measured; the optional PoliMorf model must not be retroactively attributed to results that predate it. See [Model Selection and Loading](model-selection-and-loading.md) and [Reproducibility](benchmarks/reference/reproducibility.md).
|
||||||
|
|||||||
309
docs/benchmarks/data/stemming-quality.csv
Normal file
309
docs/benchmarks/data/stemming-quality.csv
Normal file
@@ -0,0 +1,309 @@
|
|||||||
|
Stemmer,Language,Dictionary mode,Output policy,Applied dictionary rows,Processed word forms,Singleton dictionary rows,Forms with one candidate,Forms with multiple candidates,Maximum candidates for one form,Total candidate assignments,Distinct output stems,True-positive pairs,False-positive pairs,False-negative pairs,True-negative pairs,Over-stemming error pairs,Over-stemming possible pairs,Over-stemming percentage,Under-stemming error pairs,Under-stemming possible pairs,Under-stemming percentage,Pairwise precision,Pairwise recall,Pairwise specificity,Pairwise accuracy,Balanced accuracy,Pairwise F0.5,Pairwise F1,Pairwise F2,Jaccard index,Fowlkes-Mallows index,Matthews correlation coefficient,Pairwise error rate,Adjusted Rand Index,Homogeneity,Completeness,V-measure,Normalized mutual information
|
||||||
|
"CZECH_LUCENE_CZECH_STEM_FILTER","CS_CZ","ALL_WORDS","PRIMARY_OUTPUT","5113","51676","2","51676","0","1","51676","9647","177249","14480","124586","1334862335","14480","1334876815","0.001085","124586","301835","41.276194","0.924476735392","0.587238060530","0.999989152557","0.999895844650","0.793613606543","0.829234311828","0.718241200736","0.633453389361","0.560355974266","0.736809286788","0.736765291417","0.000104155350","0.718191706079","0.993800637348","0.944976928457","0.968774025802","0.968774025802"
|
||||||
|
"CZECH_LUCENE_CZECH_STEM_FILTER","CS_CZ","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","5038","50968","2","50968","0","1","50968","9558","174387","13950","124426","1298530265","13950","1298544215","0.001074","124426","298813","41.640089","0.925930645598","0.583599107134","0.999989257201","0.999893462107","0.791794182167","0.828708724235","0.715947860002","0.630197985095","0.557569149804","0.735100195918","0.735055396892","0.000106537893","0.715897321649","0.993897397445","0.944297456221","0.968462775946","0.968462775946"
|
||||||
|
"CZECH_RADIXOR","CS_CZ","ALL_WORDS","PRIMARY_OUTPUT","5113","51676","2","51676","0","1","51676","5162","299762","3867","2073","1334872948","3867","1334876815","0.000290","2073","301835","0.686799","0.987264062392","0.993132009210","0.999997103103","0.999995551157","0.996564556157","0.988432097845","0.990189342389","0.991952846154","0.980569312599","0.990193689085","0.990191466141","0.000004448843","0.990187117482","0.998733220675","0.998685552738","0.998709386137","0.998709386137"
|
||||||
|
"CZECH_RADIXOR","CS_CZ","ALL_WORDS","ANY_CANDIDATE","5113","51676","2","51080","596","4","52319","5166","301835","0","0","1334876815","0","1334876815","0.000000","0","301835","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"CZECH_RADIXOR","CS_CZ","ALL_WORDS","ALL_CANDIDATES","5113","51676","2","51080","596","4","52319","5166","301835","5850","0","1334870965","5850","1334876815","0.000438","0","301835","0.000000","0.980987048442","1.000000000000","0.999995617573","0.999995618564","0.999997808787","0.984731579205","0.990402283764","0.996138677580","0.980987048442","0.990447902942","0.990445732657","0.000004381436","","","","",""
|
||||||
|
"CZECH_RADIXOR","CS_CZ","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","5038","50968","2","50968","0","1","50968","5037","297104","3863","1709","1298540352","3863","1298544215","0.000297","1709","298813","0.571930","0.987164705765","0.994280703985","0.999997025130","0.999995710028","0.997138864558","0.988579745136","0.990709926973","0.992849308824","0.981590876052","0.990716315904","0.990714173387","0.000004289972","0.990707781520","0.998726091764","0.999029907266","0.998877976413","0.998877976413"
|
||||||
|
"CZECH_RADIXOR","CS_CZ","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","5038","50968","2","50428","540","4","51543","5040","298813","0","0","1298544215","0","1298544215","0.000000","0","298813","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"CZECH_RADIXOR","CS_CZ","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","5038","50968","2","50428","540","4","51543","5040","298813","5782","0","1298538433","5782","1298544215","0.000445","0","298813","0.000000","0.981017416570","1.000000000000","0.999995547321","0.999995548346","0.999997773661","0.984756059381","0.990417760454","0.996144940117","0.981017416570","0.990463233325","0.990461028216","0.000004451654","","","","",""
|
||||||
|
"DA_DK_RADIXOR","DA_DK","ALL_WORDS","PRIMARY_OUTPUT","4179","28079","32","28079","0","1","28079","4184","89188","1165","707","394110021","1165","394111186","0.000296","707","89895","0.786473","0.987106128186","0.992135268925","0.999997043981","0.999995251155","0.996066156453","0.988107873355","0.989614309174","0.991125345329","0.979442126071","0.989617503860","0.989615130363","0.000004748845","0.989611934224","0.998465862775","0.998718664384","0.998592247580","0.998592247580"
|
||||||
|
"DA_DK_RADIXOR","DA_DK","ALL_WORDS","ANY_CANDIDATE","4179","28079","32","27756","323","3","28405","4187","89895","0","0","394111186","0","394111186","0.000000","0","89895","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"DA_DK_RADIXOR","DA_DK","ALL_WORDS","ALL_CANDIDATES","4179","28079","32","27756","323","3","28405","4187","89895","1849","0","394109337","1849","394111186","0.000469","0","89895","0.000000","0.979846093478","1.000000000000","0.999995308431","0.999995309500","0.999997654215","0.983811622975","0.989820468071","0.995903164910","0.979846093478","0.989871756076","0.989869434047","0.000004690500","","","","",""
|
||||||
|
"DA_DK_RADIXOR","DA_DK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4173","28033","32","28033","0","1","28033","4170","89077","1165","663","392819623","1165","392820788","0.000297","663","89740","0.738801","0.987090268389","0.992611990194","0.999997034271","0.999995347541","0.996304512232","0.988189692661","0.989843428787","0.991502709249","0.979891095099","0.989847279032","0.989844954043","0.000004652459","0.989841102043","0.998463063294","0.998811876590","0.998637439483","0.998637439483"
|
||||||
|
"DA_DK_RADIXOR","DA_DK","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","4173","28033","32","27718","315","3","28351","4173","89740","0","0","392820788","0","392820788","0.000000","0","89740","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"DA_DK_RADIXOR","DA_DK","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","4173","28033","32","27718","315","3","28351","4173","89740","1849","0","392818939","1849","392820788","0.000471","0","89740","0.000000","0.979811986156","1.000000000000","0.999995293019","0.999995294094","0.999997646509","0.983784115625","0.989803065147","0.995896117847","0.979811986156","0.989854527774","0.989852198158","0.000004705906","","","","",""
|
||||||
|
"ENGLISH_LUCENE_KSTEM_FILTER","US_UK","ALL_WORDS","PRIMARY_OUTPUT","396939","607439","250964","607439","0","1","607439","371125","237565","1368501","76305","184489083270","1368501","184490451771","0.000742","76305","313870","24.311020","0.147917333410","0.756889795138","0.999992582267","0.999992168681","0.878441188702","0.176283968232","0.247471790726","0.415099040868","0.141208449266","0.334599940499","0.334597833111","0.000007831319","0.247469648794","0.980686838187","0.992107972963","0.986364345289","0.986364345289"
|
||||||
|
"ENGLISH_LUCENE_KSTEM_FILTER","US_UK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","374384","583910","228735","583910","0","1","583910","347624","237551","1367069","74340","170473473135","1367069","170474840204","0.000802","74340","311891","23.835250","0.148041904002","0.761647498645","0.999991980817","0.999991544756","0.880819739731","0.176476898525","0.247899438093","0.416437018089","0.141486991947","0.335791223646","0.335788961354","0.000008455244","0.247897133948","0.979822223835","0.991990504792","0.985868818390","0.985868818390"
|
||||||
|
"ENGLISH_LUCENE_MINIMAL_FILTER","US_UK","ALL_WORDS","PRIMARY_OUTPUT","396939","607439","250964","607439","0","1","607439","453328","137225","1122264","176645","184489329507","1122264","184490451771","0.000608","176645","313870","56.279670","0.108952916619","0.437203300730","0.999993916953","0.999992959490","0.718598608842","0.128203906480","0.174435713655","0.272816484020","0.095551668577","0.218253464509","0.218250987161","0.000007040510","0.174433464995","0.995202198233","0.981173943304","0.988138284715","0.988138284715"
|
||||||
|
"ENGLISH_LUCENE_MINIMAL_FILTER","US_UK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","374384","583910","228735","583910","0","1","583910","430129","136932","1120871","174959","170473719333","1120871","170474840204","0.000657","174959","311891","56.096200","0.108866014789","0.439037997249","0.999993425006","0.999992398716","0.719515711128","0.128139023335","0.174469673707","0.273277328232","0.095572048952","0.218623688336","0.218621020778","0.000007601284","0.174467253215","0.994993790771","0.980519680106","0.987703711280","0.987703711280"
|
||||||
|
"ENGLISH_LUCENE_PORTER_COPIED","US_UK","ALL_WORDS","PRIMARY_OUTPUT","396939","607439","250964","607439","0","1","607439","319968","285390","1557406","28480","184488894365","1557406","184490451771","0.000844","28480","313870","9.073820","0.154867928951","0.909261796285","0.999991558338","0.999991403982","0.954626677312","0.185678591198","0.264658505304","0.460562583837","0.152510906996","0.375253902399","0.375251973425","0.000008596018","0.264656367392","0.969648379409","0.997199419831","0.983230936080","0.983230936080"
|
||||||
|
"ENGLISH_LUCENE_PORTER_COPIED","US_UK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","374384","583910","228735","583910","0","1","583910","298779","283761","1552702","28130","170473287502","1552702","170474840204","0.000911","28130","311891","9.019177","0.154514956196","0.909808234287","0.999990891899","0.999990726907","0.954899563093","0.185277176317","0.264165961476","0.460049474275","0.152183881415","0.374938634269","0.374936557303","0.000009273093","0.264163659878","0.968644600847","0.997108656044","0.982670549062","0.982670549062"
|
||||||
|
"ENGLISH_LUCENE_PORTER_FILTER","US_UK","ALL_WORDS","PRIMARY_OUTPUT","396939","607439","250964","607439","0","1","607439","319968","285390","1557406","28480","184488894365","1557406","184490451771","0.000844","28480","313870","9.073820","0.154867928951","0.909261796285","0.999991558338","0.999991403982","0.954626677312","0.185678591198","0.264658505304","0.460562583837","0.152510906996","0.375253902399","0.375251973425","0.000008596018","0.264656367392","0.969648379409","0.997199419831","0.983230936080","0.983230936080"
|
||||||
|
"ENGLISH_LUCENE_PORTER_FILTER","US_UK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","374384","583910","228735","583910","0","1","583910","298779","283761","1552702","28130","170473287502","1552702","170474840204","0.000911","28130","311891","9.019177","0.154514956196","0.909808234287","0.999990891899","0.999990726907","0.954899563093","0.185277176317","0.264165961476","0.460049474275","0.152183881415","0.374938634269","0.374936557303","0.000009273093","0.264163659878","0.968644600847","0.997108656044","0.982670549062","0.982670549062"
|
||||||
|
"ENGLISH_LUCENE_POSSESSIVE_FILTER","US_UK","ALL_WORDS","PRIMARY_OUTPUT","396939","607439","250964","607439","0","1","607439","591899","7","1115154","313863","184489336617","1115154","184490451771","0.000604","313863","313870","99.997770","0.000006277121","0.000022302227","0.999993955492","0.999992254263","0.500008128860","0.000007330589","0.000009796848","0.000014763939","0.000004898448","0.000011831896","0.000008625150","0.000007745737","0.000007141644","0.995789196698","0.958018540631","0.976538780935","0.976538780935"
|
||||||
|
"ENGLISH_LUCENE_POSSESSIVE_FILTER","US_UK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","374384","583910","228735","583910","0","1","583910","568400","5","1113773","311886","170473726431","1113773","170474840204","0.000653","311886","311891","99.998397","0.000004489225","0.000016031242","0.999993466643","0.999991637145","0.500004748942","0.000005244385","0.000007014251","0.000010587200","0.000003507138","0.000008483387","0.000005026087","0.000008362855","0.000004155674","0.995605378040","0.956423154691","0.975621022465","0.975621022465"
|
||||||
|
"ENGLISH_OPENNLP_PORTER","US_UK","ALL_WORDS","PRIMARY_OUTPUT","396939","607439","250964","607439","0","1","607439","319968","285390","1557406","28480","184488894365","1557406","184490451771","0.000844","28480","313870","9.073820","0.154867928951","0.909261796285","0.999991558338","0.999991403982","0.954626677312","0.185678591198","0.264658505304","0.460562583837","0.152510906996","0.375253902399","0.375251973425","0.000008596018","0.264656367392","0.969648379409","0.997199419831","0.983230936080","0.983230936080"
|
||||||
|
"ENGLISH_OPENNLP_PORTER","US_UK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","374384","583910","228735","583910","0","1","583910","298779","283761","1552702","28130","170473287502","1552702","170474840204","0.000911","28130","311891","9.019177","0.154514956196","0.909808234287","0.999990891899","0.999990726907","0.954899563093","0.185277176317","0.264165961476","0.460049474275","0.152183881415","0.374938634269","0.374936557303","0.000009273093","0.264163659878","0.968644600847","0.997108656044","0.982670549062","0.982670549062"
|
||||||
|
"ENGLISH_PAICE_HUSK_LANCASTER","US_UK","ALL_WORDS","PRIMARY_OUTPUT","396939","607439","250964","607439","0","1","607439","268169","283991","3062661","29879","184487389110","3062661","184490451771","0.001660","29879","313870","9.519546","0.084858240415","0.904804536910","0.999983399352","0.999983237427","0.952393968131","0.103642734217","0.155164208820","0.308542866654","0.084107327905","0.277092260667","0.277089454298","0.000016762573","0.155161580693","0.937768073854","0.996599815184","0.966289292109","0.966289292109"
|
||||||
|
"ENGLISH_PAICE_HUSK_LANCASTER","US_UK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","374384","583910","228735","583910","0","1","583910","249411","282398","3045870","29493","170471794334","3045870","170474840204","0.001787","29493","311891","9.456188","0.084848335531","0.905438117804","0.999982133023","0.999981960051","0.952710125414","0.103632575002","0.155156958803","0.308575577075","0.084103067491","0.277173081705","0.277170064389","0.000018039949","0.155154132315","0.936076835754","0.996486960554","0.965337716341","0.965337716341"
|
||||||
|
"ENGLISH_RADIXOR","US_UK","ALL_WORDS","PRIMARY_OUTPUT","396939","607439","250964","607439","0","1","607439","390361","292001","1149886","21869","184489301885","1149886","184490451771","0.000623","21869","313870","6.967534","0.202513095686","0.930324656705","0.999993767233","0.999993648707","0.965159211969","0.240076409811","0.332621199859","0.541270431499","0.199487482886","0.434054059102","0.434052478080","0.000006351293","0.332619335001","0.994214506865","0.997769723414","0.995988942533","0.995988942533"
|
||||||
|
"ENGLISH_RADIXOR","US_UK","ALL_WORDS","ANY_CANDIDATE","396939","607439","250964","578231","29208","1355","2838145","397392","313855","12","15","184490451759","12","184490451771","0.000000","15","313870","0.004779","0.999961767245","0.999952209513","0.999999999935","0.999999999854","0.999976104724","0.999959855684","0.999956988357","0.999954121045","0.999913980413","0.999956988368","0.999956988295","0.000000000146","","","","",""
|
||||||
|
"ENGLISH_RADIXOR","US_UK","ALL_WORDS","ALL_CANDIDATES","396939","607439","250964","578231","29208","1355","2838145","397392","313855","11482166","15","184478969605","11482166","184490451771","0.006224","15","313870","0.004779","0.026606853277","0.999952209513","0.999937762817","0.999937762842","0.999944986165","0.033038791524","0.051834488023","0.120237128281","0.026606819443","0.163112175274","0.163107098882","0.000062237158","","","","",""
|
||||||
|
"ENGLISH_RADIXOR","US_UK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","374384","583910","228735","583910","0","1","583910","367590","290572","1148489","21319","170473691715","1148489","170474840204","0.000674","21319","311891","6.835401","0.201917778329","0.931645991709","0.999993263000","0.999993137956","0.965819627354","0.239424468968","0.331901731173","0.540775136091","0.198970131062","0.433723286019","0.433721583515","0.000006862044","0.331899721995","0.993959181482","0.997731171071","0.995841604460","0.995841604460"
|
||||||
|
"ENGLISH_RADIXOR","US_UK","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","374384","583910","228735","555084","28826","1355","2812871","374506","311891","0","0","170474840204","0","170474840204","0.000000","0","311891","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"ENGLISH_RADIXOR","US_UK","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","374384","583910","228735","555084","28826","1355","2812871","374506","311891","11470018","0","170463370186","11470018","170474840204","0.006728","0","311891","0.000000","0.026472025883","1.000000000000","0.999932717239","0.999932717362","0.999966358619","0.032872482055","0.051578660140","0.119686728696","0.026472025883","0.162702261457","0.162696787836","0.000067282638","","","","",""
|
||||||
|
"ENGLISH_SNOWBALL_ORIGINAL_PORTER","US_UK","ALL_WORDS","PRIMARY_OUTPUT","396939","607439","250964","607439","0","1","607439","321092","285304","1555293","28566","184488896478","1555293","184490451771","0.000843","28566","313870","9.101220","0.155006228957","0.908987797496","0.999991569791","0.999991414969","0.954489683644","0.185835337999","0.264848800190","0.460750814660","0.152637303435","0.375364850057","0.375362921954","0.000008585031","0.264846663203","0.969891477221","0.997192899073","0.983352728141","0.983352728141"
|
||||||
|
"ENGLISH_SNOWBALL_ORIGINAL_PORTER","US_UK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","374384","583910","228735","583910","0","1","583910","299877","283675","1550615","28216","170473289589","1550615","170474840204","0.000910","28216","311891","9.046750","0.154651118416","0.909532496930","0.999990904142","0.999990738644","0.954761700536","0.185431499934","0.264353286139","0.460234326480","0.152308234175","0.375046954242","0.375044878196","0.000009261356","0.264350985524","0.968893806180","0.997101844661","0.982795461434","0.982795461434"
|
||||||
|
"ENGLISH_SNOWBALL_PORTER2","US_UK","ALL_WORDS","PRIMARY_OUTPUT","396939","607439","250964","607439","0","1","607439","318385","285334","1566711","28536","184488885060","1566711","184490451771","0.000849","28536","313870","9.091662","0.154064291094","0.909083378469","0.999991507902","0.999991353242","0.954537443185","0.184752753479","0.263476636895","0.459101696688","0.151726514306","0.374242282819","0.374240346981","0.000008646758","0.263474493989","0.969037354042","0.997181597682","0.982908049045","0.982908049045"
|
||||||
|
"ENGLISH_SNOWBALL_PORTER2","US_UK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","374384","583910","228735","583910","0","1","583910","297220","283730","1561891","28161","170473278313","1561891","170474840204","0.000916","28161","311891","9.029116","0.153731454074","0.909708840589","0.999990837997","0.999990672823","0.954849839293","0.184374949232","0.263015918336","0.458637294569","0.151421029768","0.373966392672","0.373964308569","0.000009327177","0.263013611479","0.968019617024","0.997095706553","0.982342555079","0.982342555079"
|
||||||
|
"FINNISH_LUCENE_FINNISH_LIGHT_STEM_FILTER","FI_FI","ALL_WORDS","PRIMARY_OUTPUT","57027","1811717","292","1811717","0","1","1811717","439975","12355389","2223150","19168306","1641124591341","2223150","1641126814491","0.000135","19168306","31523695","60.806025","0.847505295284","0.391939745642","0.999998645351","0.999986965635","0.695969195497","0.687649407375","0.535999578676","0.439151826652","0.366119825424","0.576342788507","0.576337821084","0.000013034365","0.535993941880","0.988126027331","0.886473473160","0.934543630400","0.934543630400"
|
||||||
|
"FINNISH_LUCENE_FINNISH_LIGHT_STEM_FILTER","FI_FI","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","54762","1757055","274","1757055","0","1","1757055","431848","11988389","1806392","18825444","1543587637760","1806392","1543589444152","0.000117","18825444","30813833","61.094133","0.869052506162","0.389058673746","0.999998829746","0.999986634125","0.694528751746","0.697056446146","0.537492108587","0.437372459518","0.367513988637","0.581474346349","0.581469391800","0.000013365875","0.537486398327","0.989268269625","0.885293761089","0.934397488899","0.934397488899"
|
||||||
|
"FINNISH_RADIXOR","FI_FI","ALL_WORDS","PRIMARY_OUTPUT","57027","1811717","292","1811717","0","1","1811717","69091","30552427","731279","971268","1641126083212","731279","1641126814491","0.000045","971268","31523695","3.081073","0.976624284859","0.969189271753","0.999999554404","0.999998962594","0.984594413078","0.975128170336","0.972892573600","0.970667204156","0.947215985975","0.972899675927","0.972899157490","0.000001037406","0.972892054895","0.996084757586","0.993746341306","0.994914175412","0.994914175412"
|
||||||
|
"FINNISH_RADIXOR","FI_FI","ALL_WORDS","ANY_CANDIDATE","57027","1811717","292","1754389","57328","6","1876272","69769","31523695","0","0","1641126814491","0","1641126814491","0.000000","0","31523695","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"FINNISH_RADIXOR","FI_FI","ALL_WORDS","ALL_CANDIDATES","57027","1811717","292","1754389","57328","6","1876272","69769","31523695","1683575","0","1641125130916","1683575","1641126814491","0.000103","0","31523695","0.000000","0.949301011495","1.000000000000","0.999998974135","0.999998974154","0.999999487067","0.959025334376","0.973991195713","0.989431554710","0.949301011495","0.974320794962","0.974320295201","0.000001025846","","","","",""
|
||||||
|
"FINNISH_RADIXOR","FI_FI","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","54762","1757055","274","1757055","0","1","1757055","54633","30078528","730145","735305","1543588714007","730145","1543589444152","0.000047","735305","30813833","2.386282","0.976300667023","0.976137178390","0.999999526982","0.999999050641","0.988068352686","0.976267964916","0.976218915862","0.976169871736","0.953542638154","0.976218919284","0.976218444595","0.000000949359","0.976218441173","0.996000407428","0.996068984852","0.996034694959","0.996034694959"
|
||||||
|
"FINNISH_RADIXOR","FI_FI","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","54762","1757055","274","1712724","44331","6","1805864","54984","30813833","0","0","1543589444152","0","1543589444152","0.000000","0","30813833","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"FINNISH_RADIXOR","FI_FI","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","54762","1757055","274","1712724","44331","6","1805864","54984","30813833","1653320","0","1543587790832","1653320","1543589444152","0.000107","0","30813833","0.000000","0.949077148834","1.000000000000","0.999998928912","0.999998928933","0.999999464456","0.958842548108","0.973873352732","0.989382907677","0.949077148834","0.974205906795","0.974205385065","0.000001071067","","","","",""
|
||||||
|
"FRENCH_LUCENE_FRENCH_LIGHT_STEM_FILTER","FR_FR","ALL_WORDS","PRIMARY_OUTPUT","59240","425210","2301","425210","0","1","425210","245918","202782","276403","5251833","90395828427","276403","90396104830","0.000306","5251833","5454615","96.282377","0.423181026117","0.037176226003","0.999996942313","0.999938848002","0.518586584158","0.137547303040","0.068348107452","0.045471618191","0.035383242558","0.125428359900","0.125414592230","0.000061151998","0.068339028277","0.974109647704","0.812375827422","0.885921707253","0.885921707253"
|
||||||
|
"FRENCH_LUCENE_FRENCH_LIGHT_STEM_FILTER","FR_FR","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","57698","421231","2133","421231","0","1","421231","245182","200690","262689","5239869","88711863817","262689","88712126506","0.000296","5239869","5440559","96.311225","0.433101197939","0.036887753630","0.999997038860","0.999937976681","0.518442396245","0.137570562409","0.067985131280","0.045148356975","0.035188720533","0.126396717862","0.126383026150","0.000062023319","0.067976159358","0.975085555241","0.811143708698","0.885591261484","0.885591261484"
|
||||||
|
"FRENCH_LUCENE_FRENCH_MINIMAL_STEM_FILTER","FR_FR","ALL_WORDS","PRIMARY_OUTPUT","59240","425210","2301","425210","0","1","425210","269236","183612","160438","5271003","90395944392","160438","90396104830","0.000177","5271003","5454615","96.633823","0.533678244441","0.033661770812","0.999998225167","0.999939918724","0.516829997990","0.134399775137","0.063329059361","0.041424008382","0.032699958487","0.134031916915","0.134021061615","0.000060081276","0.063322352769","0.984019125555","0.810978546011","0.889158144694","0.889158144694"
|
||||||
|
"FRENCH_LUCENE_FRENCH_MINIMAL_STEM_FILTER","FR_FR","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","57698","421231","2133","421231","0","1","421231","268411","181686","147476","5258873","88711979030","147476","88712126506","0.000166","5258873","5440559","96.660527","0.551965293685","0.033394730211","0.999998337589","0.999939061122","0.516696533900","0.134438681544","0.062979128454","0.041121435592","0.032513396928","0.135767198057","0.135756528540","0.000060938878","0.062972571968","0.985086367216","0.809773733549","0.888868235590","0.888868235590"
|
||||||
|
"FRENCH_RADIXOR","FR_FR","ALL_WORDS","PRIMARY_OUTPUT","59240","425210","2301","425210","0","1","425210","60225","4985455","318767","469160","90395786063","318767","90396104830","0.000353","469160","5454615","8.601157","0.939903156391","0.913988429981","0.999996473664","0.999991284144","0.956992451823","0.934603310507","0.926764667966","0.919056419273","0.863524187383","0.926855226151","0.926850879168","0.000008715856","0.926760310630","0.988772235003","0.985214034569","0.986989927876","0.986989927876"
|
||||||
|
"FRENCH_RADIXOR","FR_FR","ALL_WORDS","ANY_CANDIDATE","59240","425210","2301","382170","43040","56","477024","60383","5454383","12","232","90396104818","12","90396104830","0.000000","232","5454615","0.004253","0.999997799939","0.999957467209","0.999999999867","0.999999997301","0.999978733538","0.999989733133","0.999977633167","0.999965533495","0.999955267335","0.999977633371","0.999977632021","0.000000002699","","","","",""
|
||||||
|
"FRENCH_RADIXOR","FR_FR","ALL_WORDS","ALL_CANDIDATES","59240","425210","2301","382170","43040","56","477024","60383","5454383","1056255","232","90395048575","1056255","90396104830","0.001168","232","5454615","0.004253","0.837764747479","0.999957467209","0.999988315260","0.999988313399","0.999972891234","0.865852951156","0.911703747510","0.962682080453","0.837734895644","0.915275431226","0.915270082203","0.000011686601","","","","",""
|
||||||
|
"FRENCH_RADIXOR","FR_FR","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","57698","421231","2133","421231","0","1","421231","58069","4975123","315266","465436","88711811240","315266","88712126506","0.000355","465436","5440559","8.554930","0.940407784758","0.914450702584","0.999996446190","0.999991200142","0.957223574387","0.935099145312","0.927247620620","0.919526848134","0.864363145162","0.927338427699","0.927334038919","0.000008799858","0.927243221287","0.988915897225","0.985549842615","0.987230000708","0.987230000708"
|
||||||
|
"FRENCH_RADIXOR","FR_FR","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","57698","421231","2133","380101","41130","56","468574","58208","5440559","0","0","88712126506","0","88712126506","0.000000","0","5440559","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"FRENCH_RADIXOR","FR_FR","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","57698","421231","2133","380101","41130","56","468574","58208","5440559","938985","0","88711187521","938985","88712126506","0.001058","0","5440559","0.000000","0.852813147774","1.000000000000","0.999989415370","0.999989416019","0.999994707685","0.878679151458","0.920560336911","0.966633773699","0.852813147774","0.923478829088","0.923473941734","0.000010583981","","","","",""
|
||||||
|
"GERMAN_CISTEM","DE_DE","ALL_WORDS","PRIMARY_OUTPUT","54092","296974","1474","296974","0","1","296974","59097","1053889","477122","329983","44094768857","477122","44095245979","0.001082","329983","1383872","23.844908","0.688361481400","0.761550923785","0.999989179741","0.999981696901","0.880770051763","0.701851885397","0.723108954973","0.745693871888","0.566304351331","0.724031989665","0.724022910459","0.000018303099","0.723099826442","0.974048119240","0.975147027686","0.974597263694","0.974597263694"
|
||||||
|
"GERMAN_CISTEM","DE_DE","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","16007","150098","228","150098","0","1","150098","23023","725447","156784","147964","11263599558","156784","11263756342","0.001392","147964","873411","16.940936","0.822286906717","0.830590638313","0.999986080665","0.999972946470","0.915288359489","0.823934343933","0.826417914358","0.828916502414","0.704184159310","0.826428343371","0.826414817348","0.000027053530","0.826404386881","0.985935685912","0.973569821618","0.979713735095","0.979713735095"
|
||||||
|
"GERMAN_LUCENE_GERMAN_LIGHT_STEM_FILTER","DE_DE","ALL_WORDS","PRIMARY_OUTPUT","54092","296974","1474","296974","0","1","296974","98357","709263","205740","674609","44095040239","205740","44095245979","0.000467","674609","1383872","48.747933","0.775148278202","0.512520666651","0.999995334191","0.999980035912","0.756258000421","0.703092101246","0.617052253820","0.549774428024","0.446186239158","0.630301128270","0.630292039259","0.000019964088","0.617042686770","0.980753120457","0.936533167951","0.958133203614","0.958133203614"
|
||||||
|
"GERMAN_LUCENE_GERMAN_LIGHT_STEM_FILTER","DE_DE","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","16007","150098","228","150098","0","1","150098","50335","471565","55477","401846","11263700865","55477","11263756342","0.000493","401846","873411","46.008809","0.894738939212","0.539911908597","0.999995074734","0.999959401861","0.769953491666","0.790797426464","0.673446377708","0.586423560557","0.507666155661","0.695039717114","0.695022690564","0.000040598139","0.673427319227","0.991320177896","0.915069562866","0.951669957566","0.951669957566"
|
||||||
|
"GERMAN_LUCENE_GERMAN_MINIMAL_STEM_FILTER","DE_DE","ALL_WORDS","PRIMARY_OUTPUT","54092","296974","1474","296974","0","1","296974","140505","271626","110840","1112246","44095135139","110840","44095245979","0.000251","1112246","1383872","80.372029","0.710196461908","0.196279713731","0.999997486350","0.999972263504","0.598138600041","0.466112921692","0.307558349534","0.229493166050","0.181724639931","0.373359288402","0.373350267608","0.000027736496","0.307548938689","0.983615403456","0.896263607272","0.937910029995","0.937910029995"
|
||||||
|
"GERMAN_LUCENE_GERMAN_MINIMAL_STEM_FILTER","DE_DE","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","16007","150098","228","150098","0","1","150098","80363","132221","21214","741190","11263735128","21214","11263756342","0.000188","741190","873411","84.861537","0.861739498811","0.151384628772","0.999998116614","0.999932318770","0.575691372693","0.444544636019","0.257528392768","0.181269722976","0.147794886125","0.361184321539","0.361168285320","0.000067681230","0.257511188319","0.992642524078","0.854402700840","0.918349417869","0.918349417869"
|
||||||
|
"GERMAN_LUCENE_GERMAN_STEM_FILTER","DE_DE","ALL_WORDS","PRIMARY_OUTPUT","54092","296974","1474","296974","0","1","296974","81085","619354","331871","764518","44094914108","331871","44095245979","0.000753","764518","1383872","55.244849","0.651111987174","0.447551507654","0.999992473769","0.999975136671","0.723771990712","0.596821367368","0.530473894660","0.477402037056","0.360982967729","0.539820480819","0.539808754751","0.000024863329","0.530461889452","0.975549631706","0.942889706548","0.958941664321","0.958941664321"
|
||||||
|
"GERMAN_LUCENE_GERMAN_STEM_FILTER","DE_DE","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","16007","150098","228","150098","0","1","150098","41574","378734","78723","494677","11263677619","78723","11263756342","0.000699","494677","873411","56.637368","0.827911694432","0.433626322545","0.999993010946","0.999949097306","0.716809666745","0.700518896035","0.569153364571","0.479276535831","0.397773842757","0.599169678345","0.599148958371","0.000050902694","0.569130398173","0.988583531594","0.918717729459","0.952371013508","0.952371013508"
|
||||||
|
"GERMAN_RADIXOR","DE_DE","ALL_WORDS","PRIMARY_OUTPUT","54092","296974","1474","296974","0","1","296974","68104","1128969","98192","254903","44095147787","98192","44095245979","0.000223","254903","1383872","18.419550","0.919984419322","0.815804496370","0.999997773184","0.999991992699","0.907901134777","0.897072808397","0.864768082211","0.834709150216","0.761754553110","0.866329859738","0.866325955582","0.000008007301","0.864764092865","0.989946248415","0.975085217969","0.982459538105","0.982459538105"
|
||||||
|
"GERMAN_RADIXOR","DE_DE","ALL_WORDS","ANY_CANDIDATE","54092","296974","1474","248400","48574","8","361016","70717","1272705","1375","111167","44095244604","1375","44095245979","0.000003","111167","1383872","8.033041","0.998920789903","0.919669593720","0.999999968818","0.999997447832","0.959834781269","0.981996366774","0.957658377578","0.934497606897","0.918756727140","0.958476435291","0.958475209548","0.000002552168","","","","",""
|
||||||
|
"GERMAN_RADIXOR","DE_DE","ALL_WORDS","ALL_CANDIDATES","54092","296974","1474","248400","48574","8","361016","70717","1272705","244817","111167","44095001162","244817","44095245979","0.000555","111167","1383872","8.033041","0.838673179038","0.919669593720","0.999994447996","0.999991927184","0.959832020858","0.853710645080","0.877305874349","0.902242446842","0.781429112618","0.878238135035","0.878234164088","0.000008072816","","","","",""
|
||||||
|
"GERMAN_RADIXOR","DE_DE","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","16007","150098","228","150098","0","1","150098","17264","814297","47898","59114","11263708444","47898","11263756342","0.000425","59114","873411","6.768177","0.944446441930","0.932318232768","0.999995747600","0.999990500176","0.966156990184","0.941995622128","0.938343149309","0.934718891125","0.883847872972","0.938362743125","0.938357995965","0.000009499824","0.938338399230","0.994062310308","0.990664418294","0.992360455671","0.992360455671"
|
||||||
|
"GERMAN_RADIXOR","DE_DE","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","16007","150098","228","135120","14978","8","167157","18366","873411","0","0","11263756342","0","11263756342","0.000000","0","873411","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"GERMAN_RADIXOR","DE_DE","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","16007","150098","228","135120","14978","8","167157","18366","873411","97544","0","11263658798","97544","11263756342","0.000866","0","873411","0.000000","0.899538083639","1.000000000000","0.999991340012","0.999991340683","0.999995670006","0.917982540684","0.947112449481","0.978151677228","0.899538083639","0.948439815507","0.948435708759","0.000008659317","","","","",""
|
||||||
|
"HE_IL_RADIXOR","HE_IL","ALL_WORDS","PRIMARY_OUTPUT","2358","58714","0","58714","0","1","58714","2358","688361","25234","19916","1722904030","25234","1722929264","0.001465","19916","708277","2.811894","0.964638205144","0.971881057835","0.999985354013","0.999973805398","0.985933205924","0.966078126522","0.968246086849","0.970423799230","0.938446730860","0.968252859146","0.968239762122","0.000026194602","0.968232984327","0.993166390361","0.993627509527","0.993396896433","0.993396896433"
|
||||||
|
"HE_IL_RADIXOR","HE_IL","ALL_WORDS","ANY_CANDIDATE","2358","58714","0","56674","2040","40","62376","2358","708277","0","0","1722929264","0","1722929264","0.000000","0","708277","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"HE_IL_RADIXOR","HE_IL","ALL_WORDS","ALL_CANDIDATES","2358","58714","0","56674","2040","40","62376","2358","708277","86628","0","1722842636","86628","1722929264","0.005028","0","708277","0.000000","0.891020939609","1.000000000000","0.999949720513","0.999949741174","0.999974860256","0.910874182109","0.942370251906","0.976122467036","0.891020939609","0.943939055029","0.943915324345","0.000050258826","","","","",""
|
||||||
|
"HE_IL_RADIXOR","HE_IL","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","2358","58714","0","58714","0","1","58714","2358","688361","25234","19916","1722904030","25234","1722929264","0.001465","19916","708277","2.811894","0.964638205144","0.971881057835","0.999985354013","0.999973805398","0.985933205924","0.966078126522","0.968246086849","0.970423799230","0.938446730860","0.968252859146","0.968239762122","0.000026194602","0.968232984327","0.993166390361","0.993627509527","0.993396896433","0.993396896433"
|
||||||
|
"HE_IL_RADIXOR","HE_IL","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","2358","58714","0","56674","2040","40","62376","2358","708277","0","0","1722929264","0","1722929264","0.000000","0","708277","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"HE_IL_RADIXOR","HE_IL","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","2358","58714","0","56674","2040","40","62376","2358","708277","86628","0","1722842636","86628","1722929264","0.005028","0","708277","0.000000","0.891020939609","1.000000000000","0.999949720513","0.999949741174","0.999974860256","0.910874182109","0.942370251906","0.976122467036","0.891020939609","0.943939055029","0.943915324345","0.000050258826","","","","",""
|
||||||
|
"HUNGARIAN_LUCENE_HUNGARIAN_LIGHT_STEM_FILTER","HU_HU","ALL_WORDS","PRIMARY_OUTPUT","19406","916344","1","916344","0","1","916344","94328","14036270","4132555","8125833","419816410338","4132555","419820542893","0.000984","8125833","22162103","36.665442","0.772546931351","0.633345580968","0.999990156377","0.999970802427","0.816667868673","0.740017627855","0.696054898613","0.657022705053","0.533806904809","0.699492090778","0.699477892426","0.000029197573","0.696040442258","0.982615378770","0.926771756762","0.953876941828","0.953876941828"
|
||||||
|
"HUNGARIAN_LUCENE_HUNGARIAN_LIGHT_STEM_FILTER","HU_HU","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","18360","878513","1","878513","0","1","878513","91516","13492703","3639046","7918708","385867055871","3639046","385870694917","0.000943","7918708","21411411","36.983588","0.787584676848","0.630164121365","0.999990569261","0.999970049260","0.815077345313","0.750107959995","0.700134758022","0.656404225003","0.538621031944","0.704491026122","0.704476576404","0.000029950740","0.700119966552","0.983686881315","0.925487153321","0.953699929950","0.953699929950"
|
||||||
|
"HUNGARIAN_RADIXOR","HU_HU","ALL_WORDS","PRIMARY_OUTPUT","19406","916344","1","916344","0","1","916344","20535","21962266","272900","199837","419820269993","272900","419820542893","0.000065","199837","22162103","0.901706","0.987726648859","0.990982940563","0.999999349960","0.999998874014","0.995491145262","0.988376194087","0.989352115329","0.990329965723","0.978928596533","0.989353455019","0.989352892139","0.000001125986","0.989351552308","0.998036093538","0.997808712909","0.997922390271","0.997922390271"
|
||||||
|
"HUNGARIAN_RADIXOR","HU_HU","ALL_WORDS","ANY_CANDIDATE","19406","916344","1","904024","12320","5","929326","20567","22162103","0","0","419820542893","0","419820542893","0.000000","0","22162103","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"HUNGARIAN_RADIXOR","HU_HU","ALL_WORDS","ALL_CANDIDATES","19406","916344","1","904024","12320","5","929326","20567","22162103","460158","0","419820082735","460158","419820542893","0.000110","0","22162103","0.000000","0.979659062372","1.000000000000","0.999998903917","0.999998903975","0.999999451959","0.983660778882","0.989725029923","0.995864516790","0.979659062372","0.989777279176","0.989776736737","0.000001096025","","","","",""
|
||||||
|
"HUNGARIAN_RADIXOR","HU_HU","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","18360","878513","1","878513","0","1","878513","18363","21247134","272775","164277","385870422142","272775","385870694917","0.000071","164277","21411411","0.767240","0.987324528185","0.992327595785","0.999999293092","0.999998867424","0.996163444439","0.988321101757","0.989819739994","0.991322930046","0.979844666523","0.989822900984","0.989822335019","0.000001132576","0.989819173678","0.997945135090","0.998273386381","0.998109233747","0.998109233747"
|
||||||
|
"HUNGARIAN_RADIXOR","HU_HU","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","18360","878513","1","867360","11153","5","890245","18375","21411411","0","0","385870694917","0","385870694917","0.000000","0","21411411","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"HUNGARIAN_RADIXOR","HU_HU","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","18360","878513","1","867360","11153","5","890245","18375","21411411","458462","0","385870236455","458462","385870694917","0.000119","0","21411411","0.000000","0.979036823854","1.000000000000","0.999998811877","0.999998811943","0.999999405938","0.983158850285","0.989407384494","0.995735852714","0.979036823854","0.989462896653","0.989462308851","0.000001188057","","","","",""
|
||||||
|
"HUNSPELL_CZECH_LUCENE_FILTER","CS_CZ","ALL_WORDS","PRIMARY_OUTPUT","5113","51676","2","51676","0","1","51676","10920","213552","11408","88283","1334865407","11408","1334876815","0.000855","88283","301835","29.248762","0.949288762447","0.707512382593","0.999991453893","0.999925335085","0.853751918243","0.888559718726","0.810759403563","0.745486280807","0.681745481942","0.819532521678","0.819499025505","0.000074664915","0.810722859062","0.995776551361","0.952852006662","0.973841506371","0.973841506371"
|
||||||
|
"HUNSPELL_CZECH_LUCENE_FILTER","CS_CZ","ALL_WORDS","ANY_CANDIDATE","5113","51676","2","48359","3317","5","55596","11359","224312","10102","77523","1334866713","10102","1334876815","0.000757","77523","301835","25.683900","0.956905304291","0.743160998559","0.999992432261","0.999934372078","0.871576715410","0.904855299474","0.836596431881","0.777913569166","0.719093919606","0.843288029954","0.843258147533","0.000065627922","","","","",""
|
||||||
|
"HUNSPELL_CZECH_LUCENE_FILTER","CS_CZ","ALL_WORDS","ALL_CANDIDATES","5113","51676","2","48359","3317","5","55596","11359","224312","13917","77523","1334862898","13917","1334876815","0.001043","77523","301835","25.683900","0.941581419558","0.743160998559","0.999989574319","0.999931514783","0.871575286439","0.893850652440","0.830686733424","0.775860578084","0.710405634802","0.836508570179","0.836476906392","0.000068485217","","","","",""
|
||||||
|
"HUNSPELL_CZECH_LUCENE_FILTER","CS_CZ","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","5038","50968","2","50968","0","1","50968","10816","210827","11239","87986","1298532976","11239","1298544215","0.000866","87986","298813","29.445171","0.949388920411","0.705548286052","0.999991344923","0.999923605087","0.852769815487","0.888008949714","0.809504702628","0.743753342581","0.679973036781","0.818437368155","0.818403143837","0.000076394913","0.809467327086","0.995812473772","0.952393750423","0.973619286126","0.973619286126"
|
||||||
|
"HUNSPELL_CZECH_LUCENE_FILTER","CS_CZ","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","5038","50968","2","47731","3237","5","54804","11240","221382","10028","77431","1298534187","10028","1298544215","0.000772","77431","298813","25.912862","0.956665658355","0.740871381098","0.999992277506","0.999932663918","0.870431829302","0.904003665310","0.835052421340","0.775874033233","0.716815448726","0.841882537861","0.841851913927","0.000067336082","","","","",""
|
||||||
|
"HUNSPELL_CZECH_LUCENE_FILTER","CS_CZ","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","5038","50968","2","47731","3237","5","54804","11240","221382","13601","77431","1298530614","13601","1298544215","0.001047","77431","298813","25.912862","0.942119217135","0.740871381098","0.999989525963","0.999929913009","0.870430453530","0.893573737936","0.829462940899","0.773935751817","0.708617411512","0.835457458856","0.835425115185","0.000070086991","","","","",""
|
||||||
|
"HUNSPELL_DUTCH_LUCENE_FILTER","NL_NL","ALL_WORDS","PRIMARY_OUTPUT","4992","26477","85","26477","0","1","26477","15909","18482","1333","46084","350436627","1333","350437960","0.000380","46084","64566","71.375027","0.932727731517","0.286249728960","0.999996196188","0.999864717095","0.643122962574","0.642512480358","0.438060700869","0.332315636923","0.280459491039","0.516713712165","0.516673857221","0.000135282905","0.438012080403","0.996931617211","0.889026094124","0.939891935467","0.939891935467"
|
||||||
|
"HUNSPELL_DUTCH_LUCENE_FILTER","NL_NL","ALL_WORDS","ANY_CANDIDATE","4992","26477","85","25223","1254","3","27763","16027","21374","1164","43192","350436796","1164","350437960","0.000332","43192","64566","66.895889","0.948353891206","0.331041105226","0.999996678442","0.999873450270","0.665518891834","0.690740573172","0.490769654666","0.380588457347","0.325178761600","0.560307166017","0.560267948638","0.000126549730","","","","",""
|
||||||
|
"HUNSPELL_DUTCH_LUCENE_FILTER","NL_NL","ALL_WORDS","ALL_CANDIDATES","4992","26477","85","25223","1254","3","27763","16027","21374","1738","43192","350436222","1738","350437960","0.000496","43192","64566","66.895889","0.924800969193","0.331041105226","0.999995040492","0.999871812621","0.665518072859","0.680639942935","0.487556741714","0.379812066416","0.322363658301","0.553305643343","0.553264631551","0.000128187379","","","","",""
|
||||||
|
"HUNSPELL_DUTCH_LUCENE_FILTER","NL_NL","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4796","25678","84","25678","0","1","25678","15258","18333","1310","44814","329602546","1310","329603856","0.000397","44814","63147","70.967742","0.933309575930","0.290322580645","0.999996025532","0.999860089122","0.645159303089","0.646808120294","0.442879574828","0.336717714000","0.284422172921","0.520538994337","0.520497519470","0.000139910878","0.442828931093","0.996884174988","0.889060613994","0.939890140871","0.939890140871"
|
||||||
|
"HUNSPELL_DUTCH_LUCENE_FILTER","NL_NL","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","4796","25678","84","24492","1186","3","26896","15323","21212","1141","41935","329602715","1141","329603856","0.000346","41935","63147","66.408539","0.948955397486","0.335914611937","0.999996538269","0.999869334815","0.667955575103","0.695206444720","0.496187134503","0.385755489360","0.329952712792","0.564595416287","0.564554662442","0.000130665185","","","","",""
|
||||||
|
"HUNSPELL_DUTCH_LUCENE_FILTER","NL_NL","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","4796","25678","84","24492","1186","3","26896","15323","21212","1712","41935","329602144","1712","329603856","0.000519","41935","63147","66.408539","0.925318443553","0.335914611937","0.999994805886","0.999867602764","0.667954708912","0.684951854459","0.492895400309","0.384956009176","0.327047903915","0.557519493726","0.557476858392","0.000132397236","","","","",""
|
||||||
|
"HUNSPELL_ENGLISH_LUCENE_FILTER","US_UK","ALL_WORDS","PRIMARY_OUTPUT","396939","607439","250964","607439","0","1","607439","557518","46002","1981986","267868","184488469785","1981986","184490451771","0.001074","267868","313870","85.343614","0.022683566175","0.146563864020","0.999989256972","0.999987805059","0.573276560496","0.027298226808","0.039286754363","0.070050933952","0.020036970960","0.057659267324","0.057655308782","0.000012194941","0.039283923590","0.993096189204","0.963677172906","0.978165531652","0.978165531652"
|
||||||
|
"HUNSPELL_ENGLISH_LUCENE_FILTER","US_UK","ALL_WORDS","ANY_CANDIDATE","396939","607439","250964","600602","6837","4","614296","557638","51229","1978852","262641","184488472919","1978852","184490451771","0.001073","262641","313870","83.678274","0.025234953679","0.163217255552","0.999989273960","0.999987850378","0.581603264756","0.030369825498","0.043711664621","0.077960810954","0.022344183028","0.064177721084","0.064173802046","0.000012149622","","","","",""
|
||||||
|
"HUNSPELL_ENGLISH_LUCENE_FILTER","US_UK","ALL_WORDS","ALL_CANDIDATES","396939","607439","250964","600602","6837","4","614296","557638","51229","2008917","262641","184488442854","2008917","184490451771","0.001089","262641","313870","83.678274","0.024866684206","0.163217255552","0.999989110997","0.999987687416","0.581603183275","0.029942881217","0.043158091605","0.077253888104","0.022054971033","0.063707707153","0.063703758400","0.000012312584","","","","",""
|
||||||
|
"HUNSPELL_ENGLISH_LUCENE_FILTER","US_UK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","374384","583910","228735","583910","0","1","583910","535362","45926","1978041","265965","170472862163","1978041","170474840204","0.001160","265965","311891","85.274984","0.022691081426","0.147250161114","0.999988396874","0.999986836756","0.573619278994","0.027311677226","0.039322595808","0.070190378755","0.020055617372","0.057803679777","0.057799415161","0.000013163244","0.039319549964","0.993066314983","0.962316521867","0.977449637185","0.977449637185"
|
||||||
|
"HUNSPELL_ENGLISH_LUCENE_FILTER","US_UK","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","374384","583910","228735","577124","6786","4","590716","535485","51150","1974950","260741","170472865254","1974950","170474840204","0.001158","260741","311891","83.600040","0.025245545630","0.163999602425","0.999988415006","0.999986885532","0.581994008716","0.030387494919","0.043755514884","0.078123472659","0.022367099418","0.064344847861","0.064340626006","0.000013114468","","","","",""
|
||||||
|
"HUNSPELL_ENGLISH_LUCENE_FILTER","US_UK","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","374384","583910","228735","577124","6786","4","590716","535485","51150","2004598","260741","170472835606","2004598","170474840204","0.001176","260741","311891","83.600040","0.024881454342","0.163999602425","0.999988241092","0.999986711618","0.581993921758","0.029965261387","0.043207600483","0.077422296168","0.022080830084","0.063879172034","0.063874918547","0.000013288382","","","","",""
|
||||||
|
"HUNSPELL_FRENCH_LUCENE_FILTER","FR_FR","ALL_WORDS","PRIMARY_OUTPUT","59240","425210","2301","425210","0","1","425210","154336","3422734","776728","2031881","90395328102","776728","90396104830","0.000859","2031881","5454615","37.250677","0.815041069547","0.627493232795","0.999991407506","0.999968931852","0.813742320150","0.769068574566","0.709075347131","0.657764674673","0.549277098051","0.715145268872","0.715130511338","0.000031068148","0.709060074832","0.978337247291","0.913705954219","0.944917713687","0.944917713687"
|
||||||
|
"HUNSPELL_FRENCH_LUCENE_FILTER","FR_FR","ALL_WORDS","ANY_CANDIDATE","59240","425210","2301","411699","13511","4","439015","154718","3610612","745831","1844003","90395358999","745831","90396104830","0.000825","1844003","5454615","33.806291","0.828798173189","0.661937093635","0.999991749302","0.999971351888","0.830964421468","0.789018996925","0.736029080656","0.689708764155","0.582314885091","0.740683639600","0.740669908387","0.000028648112","","","","",""
|
||||||
|
"HUNSPELL_FRENCH_LUCENE_FILTER","FR_FR","ALL_WORDS","ALL_CANDIDATES","59240","425210","2301","411699","13511","4","439015","154718","3610612","1043199","1844003","90395061631","1043199","90396104830","0.001154","1844003","5454615","33.806291","0.775839843947","0.661937093635","0.999988459691","0.999968062476","0.830962776663","0.750027659074","0.714376699201","0.681961135862","0.555665643861","0.716629033342","0.716613365354","0.000031937524","","","","",""
|
||||||
|
"HUNSPELL_FRENCH_LUCENE_FILTER","FR_FR","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","57698","421231","2133","421231","0","1","421231","153822","3412548","763305","2028011","88711363201","763305","88712126506","0.000860","2028011","5440559","37.275784","0.817209801207","0.627242163903","0.999991395708","0.999968537054","0.813616779806","0.770536594362","0.709734150326","0.657825640123","0.550068151075","0.715952822518","0.715937898033","0.000031462946","0.709718690125","0.979328164393","0.913161860024","0.945088340370","0.945088340370"
|
||||||
|
"HUNSPELL_FRENCH_LUCENE_FILTER","FR_FR","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","57698","421231","2133","407794","13437","4","434961","154205","3600083","733584","1840476","88711392922","733584","88712126506","0.000827","1840476","5440559","33.828803","0.830724418835","0.661711967465","0.999991730736","0.999970985904","0.830851849101","0.790350629656","0.736648201095","0.689779349655","0.583090317150","0.741417756470","0.741403865778","0.000029014096","","","","",""
|
||||||
|
"HUNSPELL_FRENCH_LUCENE_FILTER","FR_FR","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","57698","421231","2133","407794","13437","4","434961","154205","3600083","1027635","1840476","88711098871","1027635","88712126506","0.001158","1840476","5440559","33.828803","0.777939148410","0.661711967465","0.999988416071","0.999967671442","0.830850191768","0.751538185756","0.715133880405","0.682093458746","0.556582409247","0.717475884237","0.717460037184","0.000032328558","","","","",""
|
||||||
|
"HUNSPELL_GERMAN_LUCENE_FILTER","DE_DE","ALL_WORDS","PRIMARY_OUTPUT","54092","296974","1474","296974","0","1","296974","182774","391862","203883","992010","44095042096","203883","44095245979","0.000462","992010","1383872","71.683653","0.657768004767","0.283163471766","0.999995376304","0.999972880172","0.641579424035","0.520145203475","0.395896782054","0.319562150060","0.246802560848","0.431573715426","0.431562811676","0.000027119828","0.395885371175","0.980462638581","0.886872825924","0.931322397630","0.931322397630"
|
||||||
|
"HUNSPELL_GERMAN_LUCENE_FILTER","DE_DE","ALL_WORDS","ANY_CANDIDATE","54092","296974","1474","289083","7891","3","305052","183111","408175","158403","975697","44095087576","158403","44095245979","0.000359","975697","1383872","70.504859","0.720421548313","0.294951411691","0.999996407708","0.999974281481","0.647473909700","0.559115650060","0.418544438463","0.334456395588","0.264657729653","0.460965674088","0.460955788036","0.000025718519","","","","",""
|
||||||
|
"HUNSPELL_GERMAN_LUCENE_FILTER","DE_DE","ALL_WORDS","ALL_CANDIDATES","54092","296974","1474","289083","7891","3","305052","183111","408175","242551","975697","44095003428","242551","44095245979","0.000550","975697","1383872","70.504859","0.627260936247","0.294951411691","0.999994499384","0.999972373218","0.647472955538","0.511911128190","0.401234052132","0.329906951166","0.250964847398","0.430129630047","0.430118032816","0.000027626782","","","","",""
|
||||||
|
"HUNSPELL_GERMAN_LUCENE_FILTER","DE_DE","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","16007","150098","228","150098","0","1","150098","86983","278093","84679","595318","11263671663","84679","11263756342","0.000752","595318","873411","68.160122","0.766577905682","0.318398783620","0.999992482170","0.999939634323","0.659195632895","0.598178360154","0.449922058465","0.360558871242","0.290257700216","0.494041974653","0.494019111671","0.000060365677","0.449897024669","0.988041339480","0.865580709092","0.922765807515","0.922765807515"
|
||||||
|
"HUNSPELL_GERMAN_LUCENE_FILTER","DE_DE","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","16007","150098","228","145109","4989","3","155207","87393","288864","60996","584547","11263695346","60996","11263756342","0.000542","584547","873411","66.926911","0.825655976676","0.330730893016","0.999994584755","0.999942692923","0.665362738886","0.635466205220","0.472281285177","0.375782098835","0.309141519702","0.522560942370","0.522540242219","0.000057307077","","","","",""
|
||||||
|
"HUNSPELL_GERMAN_LUCENE_FILTER","DE_DE","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","16007","150098","228","145109","4989","3","155207","87393","288864","96545","584547","11263659797","96545","11263756342","0.000857","584547","873411","66.926911","0.749499881944","0.330730893016","0.999991428703","0.999939537116","0.665361160860","0.598050472724","0.458944090497","0.372338300095","0.297811447117","0.497878263505","0.497854575726","0.000060462884","","","","",""
|
||||||
|
"HUNSPELL_POLISH_LUCENE_FILTER","PL_PL","ALL_WORDS","PRIMARY_OUTPUT","9990","122341","1","122341","0","1","122341","18419","971262","52652","149705","7482425351","52652","7482478003","0.000704","149705","1120967","13.354987","0.948577712581","0.866450127435","0.999992963294","0.999972959935","0.933221545364","0.930929837176","0.905655838249","0.881717903868","0.827578626454","0.906584403102","0.906571161039","0.000027040065","0.905642343969","0.994545991966","0.970520439141","0.982386343372","0.982386343372"
|
||||||
|
"HUNSPELL_POLISH_LUCENE_FILTER","PL_PL","ALL_WORDS","ANY_CANDIDATE","9990","122341","1","110894","11447","6","135231","19068","1040224","42213","80743","7482435790","42213","7482478003","0.000564","80743","1120967","7.202977","0.961001887408","0.927970225707","0.999994358420","0.999983569937","0.963982292063","0.954208759768","0.944197251162","0.934393641743","0.894293230626","0.944341642819","0.944333470354","0.000016430063","","","","",""
|
||||||
|
"HUNSPELL_POLISH_LUCENE_FILTER","PL_PL","ALL_WORDS","ALL_CANDIDATES","9990","122341","1","110894","11447","6","135231","19068","1040224","82745","80743","7482395258","82745","7482478003","0.001106","80743","1120967","7.202977","0.926315864463","0.927970225707","0.999988941498","0.999978153827","0.963979583602","0.926646264647","0.927142307089","0.927638880888","0.864180136112","0.927142676087","0.927131751477","0.000021846173","","","","",""
|
||||||
|
"HUNSPELL_POLISH_LUCENE_FILTER","PL_PL","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","9846","120925","1","120925","0","1","120925","18149","965984","51950","148667","7310200749","51950","7310252699","0.000711","148667","1114651","13.337538","0.948965257080","0.866624620621","0.999992893543","0.999972560946","0.933308757082","0.931268723294","0.905927782480","0.881929423296","0.828032892137","0.906860880124","0.906847444801","0.000027439054","0.905914089179","0.994583905165","0.970514203019","0.982401644006","0.982401644006"
|
||||||
|
"HUNSPELL_POLISH_LUCENE_FILTER","PL_PL","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","9846","120925","1","109660","11265","6","133595","18789","1034283","41671","80368","7310211028","41671","7310252699","0.000570","80368","1114651","7.210149","0.961270649117","0.927898508143","0.999994299650","0.999983308321","0.963946403896","0.954405554191","0.944289819479","0.934386268967","0.894459328803","0.944437187555","0.944428885928","0.000016691679","","","","",""
|
||||||
|
"HUNSPELL_POLISH_LUCENE_FILTER","PL_PL","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","9846","120925","1","109660","11265","6","133595","18789","1034283","81865","80368","7310170834","81865","7310252699","0.001120","80368","1114651","7.210149","0.926653992123","0.927898508143","0.999988801345","0.999977810854","0.963943654744","0.926902628188","0.927275832560","0.927649337585","0.864412176686","0.927276041347","0.927264945147","0.000022189146","","","","",""
|
||||||
|
"HUNSPELL_SPANISH_LUCENE_FILTER","ES_ES","ALL_WORDS","PRIMARY_OUTPUT","65059","871332","3589","871332","0","1","871332","495840","9662476","536192","32310860","379566781918","536192","379567318110","0.000141","32310860","41973336","76.979490","0.947425291224","0.230205099733","0.999998587360","0.999913471422","0.615101843546","0.583708381625","0.370408466579","0.271277635967","0.227301418167","0.467014061518","0.466991649518","0.000086528578","0.370381248953","0.993314263125","0.790558492734","0.880413722434","0.880413722434"
|
||||||
|
"HUNSPELL_SPANISH_LUCENE_FILTER","ES_ES","ALL_WORDS","ANY_CANDIDATE","65059","871332","3589","853455","17877","5","890999","496361","10079118","416345","31894218","379566901765","416345","379567318110","0.000110","31894218","41973336","75.986855","0.960330954432","0.240131449166","0.999998903106","0.999914884689","0.620065176136","0.600267728541","0.384194728757","0.282504215637","0.237772914592","0.480214185303","0.480192080762","0.000085115311","","","","",""
|
||||||
|
"HUNSPELL_SPANISH_LUCENE_FILTER","ES_ES","ALL_WORDS","ALL_CANDIDATES","65059","871332","3589","853455","17877","5","890999","496361","10079118","888077","31894218","379566430033","888077","379567318110","0.000234","31894218","41973336","75.986855","0.919024235459","0.240131449166","0.999997660291","0.999913642011","0.620064554728","0.587073016700","0.380771322449","0.281759130783","0.235155989841","0.469772946730","0.469749183448","0.000086357989","","","","",""
|
||||||
|
"HUNSPELL_SPANISH_LUCENE_FILTER","ES_ES","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","64918","869371","3525","869371","0","1","869371","495045","9628515","531181","32234855","377860138584","531181","377860669765","0.000141","32234855","41863370","77.000144","0.947716841134","0.229998564377","0.999998594241","0.999913295008","0.614998579309","0.583531128169","0.370163304101","0.271052948234","0.227116805648","0.466876335765","0.466853897372","0.000086704992","0.370136051001","0.993362468962","0.790499503024","0.880396073652","0.880396073652"
|
||||||
|
"HUNSPELL_SPANISH_LUCENE_FILTER","ES_ES","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","64918","869371","3525","851564","17807","5","888962","495572","10041262","412198","31822108","377860257567","412198","377860669765","0.000109","31822108","41863370","76.014205","0.960568271175","0.239857947413","0.999998909127","0.999914702064","0.619928428270","0.599999808789","0.383863548307","0.282205460900","0.237519268813","0.479999931119","0.479977799265","0.000085297936","","","","",""
|
||||||
|
"HUNSPELL_SPANISH_LUCENE_FILTER","ES_ES","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","64918","869371","3525","851564","17807","5","888962","495572","10041262","878949","31822108","377859790816","878949","377860669765","0.000233","31822108","41863370","76.014205","0.919511720057","0.239857947413","0.999997673881","0.999913466955","0.619927810647","0.586904802235","0.380469146267","0.281467012980","0.234925531298","0.469629847641","0.469606065497","0.000086533045","","","","",""
|
||||||
|
"HUNSPELL_UKRAINIAN_LUCENE_FILTER","UK_UA","ALL_WORDS","PRIMARY_OUTPUT","1493","14245","4","14245","0","1","14245","3137","50416","794","14924","101386756","794","101387550","0.000783","14924","65340","22.840526","0.984495215778","0.771594735231","0.999992168664","0.999845070949","0.885793451947","0.933007624547","0.865139425139","0.806475349522","0.762331024889","0.871568313648","0.871498740996","0.000154929051","0.865063055969","0.998114340300","0.949803904722","0.973360047526","0.973360047526"
|
||||||
|
"HUNSPELL_UKRAINIAN_LUCENE_FILTER","UK_UA","ALL_WORDS","ANY_CANDIDATE","1493","14245","4","12923","1322","6","15740","3311","55875","326","9465","101387224","326","101387550","0.000322","9465","65340","14.485767","0.994199391470","0.855142332415","0.999996784615","0.999903492153","0.927569558515","0.962883947281","0.919442821764","0.879752236578","0.850896963421","0.922053136488","0.922008115880","0.000096507847","","","","",""
|
||||||
|
"HUNSPELL_UKRAINIAN_LUCENE_FILTER","UK_UA","ALL_WORDS","ALL_CANDIDATES","1493","14245","4","12923","1322","6","15740","3311","55875","1271","9465","101386279","1271","101387550","0.001254","9465","65340","14.485767","0.977758723270","0.855142332415","0.999987463944","0.999894177485","0.927564898180","0.950500809733","0.912349166435","0.877142031861","0.838825419225","0.914397547654","0.914347195556","0.000105822515","","","","",""
|
||||||
|
"HUNSPELL_UKRAINIAN_LUCENE_FILTER","UK_UA","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","1491","14236","4","14236","0","1","14236","3134","50404","794","14920","101258612","794","101259406","0.000784","14920","65324","22.839998","0.984491581702","0.771600024493","0.999992158753","0.999844914465","0.885796091623","0.933006560145","0.865141346698","0.806479484406","0.762334008893","0.871569692311","0.871500048785","0.000155085535","0.865064900251","0.998112856419","0.949788155938","0.973351072054","0.973351072054"
|
||||||
|
"HUNSPELL_UKRAINIAN_LUCENE_FILTER","UK_UA","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","1491","14236","4","12915","1321","6","15730","3308","55859","326","9465","101259080","326","101259406","0.000322","9465","65324","14.489315","0.994197739610","0.855106851999","0.999996780546","0.999903370085","0.927551816273","0.962873710629","0.919421606630","0.879721936116","0.850860624524","0.922033242016","0.921988165267","0.000096629915","","","","",""
|
||||||
|
"HUNSPELL_UKRAINIAN_LUCENE_FILTER","UK_UA","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","1491","14236","4","12915","1321","6","15730","3308","55859","1271","9465","101258135","1271","101259406","0.001255","9465","65324","14.489315","0.977752494311","0.855106851999","0.999987448080","0.999894043636","0.927547150039","0.950487333415","0.912326261290","0.877111165546","0.838786695698","0.914375665383","0.914325250217","0.000105956364","","","","",""
|
||||||
|
"ITALIAN_LUCENE_ITALIAN_LIGHT_STEM_FILTER","IT_IT","ALL_WORDS","PRIMARY_OUTPUT","10009","327551","0","327551","0","1","327551","244870","109684","10589","6034130","53638510622","10589","53638521211","0.000020","6034130","6143814","98.214725","0.911958627456","0.017852754006","0.999999802586","0.999887319289","0.508926278296","0.082781551919","0.035019947839","0.022207258650","0.017822037328","0.127596916262","0.127588341500","0.000112680711","0.035015703871","0.997481424185","0.737537113266","0.848036553212","0.848036553212"
|
||||||
|
"ITALIAN_LUCENE_ITALIAN_LIGHT_STEM_FILTER","IT_IT","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","10007","327469","0","327469","0","1","327469","244808","109658","10588","6032516","53611656484","10588","53611667072","0.000020","6032516","6142174","98.214671","0.911947174958","0.017853287777","0.999999802506","0.999887292971","0.508926545141","0.082783771729","0.035020966336","0.022207918023","0.017822564890","0.127598022525","0.127589445488","0.000112707029","0.035016721203","0.997481197125","0.737534145120","0.848034509070","0.848034509070"
|
||||||
|
"ITALIAN_RADIXOR","IT_IT","ALL_WORDS","PRIMARY_OUTPUT","10009","327551","0","327551","0","1","327551","10010","6100906","124172","42908","53638397039","124172","53638521211","0.000231","42908","6143814","0.698394","0.980052940702","0.993016064614","0.999997685022","0.999996885431","0.996506874818","0.982618418699","0.986491918597","0.990396078172","0.973343909830","0.986513210398","0.986511657877","0.000003114569","0.986490361200","0.995780270704","0.997112550353","0.996445965204","0.996445965204"
|
||||||
|
"ITALIAN_RADIXOR","IT_IT","ALL_WORDS","ANY_CANDIDATE","10009","327551","0","321297","6254","4","334175","10012","6143734","0","80","53638521211","0","53638521211","0.000000","80","6143814","0.001302","1.000000000000","0.999986978772","1.000000000000","0.999999998509","0.999993489386","0.999997395727","0.999993489344","0.999989582991","0.999986978772","0.999993489365","0.999993488619","0.000000001491","","","","",""
|
||||||
|
"ITALIAN_RADIXOR","IT_IT","ALL_WORDS","ALL_CANDIDATES","10009","327551","0","321297","6254","4","334175","10012","6143734","170950","80","53638350261","170950","53638521211","0.000319","80","6143814","0.001302","0.972928178195","0.999986978772","0.999996812925","0.999996811799","0.999991895849","0.978222150749","0.986272020913","0.994455476443","0.972915852437","0.986364795335","0.986363222748","0.000003188201","","","","",""
|
||||||
|
"ITALIAN_RADIXOR","IT_IT","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","10007","327469","0","327469","0","1","327469","10007","6099346","124171","42828","53611542901","124171","53611667072","0.000232","42828","6142174","0.697278","0.980048098206","0.993027224563","0.999997683881","0.999996885382","0.996512454222","0.982616709845","0.986494972258","0.990403969991","0.973349855458","0.986516316590","0.986514764059","0.000003114618","0.986493414837","0.995779575755","0.997114909855","0.996446795437","0.996446795437"
|
||||||
|
"ITALIAN_RADIXOR","IT_IT","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","10007","327469","0","321217","6252","4","334089","10007","6142174","0","0","53611667072","0","53611667072","0.000000","0","6142174","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"ITALIAN_RADIXOR","IT_IT","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","10007","327469","0","321217","6252","4","334089","10007","6142174","170949","0","53611496123","170949","53611667072","0.000319","0","6142174","0.000000","0.972921642743","1.000000000000","0.999996811347","0.999996811712","0.999998405674","0.978219357390","0.986274996092","0.994464412864","0.972921642743","0.986367904356","0.986366331762","0.000003188288","","","","",""
|
||||||
|
"NL_NL_RADIXOR","NL_NL","ALL_WORDS","PRIMARY_OUTPUT","4992","26477","85","26477","0","1","26477","5015","63102","1214","1464","350436746","1214","350437960","0.000346","1464","64566","2.267447","0.981124448038","0.977325527367","0.999996535763","0.999992359542","0.988661031565","0.980362303079","0.979221303208","0.978082956166","0.959288537549","0.979223145453","0.979219325206","0.000007640458","0.979217482290","0.997464133435","0.997003025118","0.997233525974","0.997233525974"
|
||||||
|
"NL_NL_RADIXOR","NL_NL","ALL_WORDS","ANY_CANDIDATE","4992","26477","85","25905","572","3","27061","5016","64566","0","0","350437960","0","350437960","0.000000","0","64566","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"NL_NL_RADIXOR","NL_NL","ALL_WORDS","ALL_CANDIDATES","4992","26477","85","25905","572","3","27061","5016","64566","2651","0","350435309","2651","350437960","0.000756","0","64566","0.000000","0.960560572474","1.000000000000","0.999992435180","0.999992436574","0.999996217590","0.968197604323","0.979883596519","0.991855131329","0.960560572474","0.980081921308","0.980078214229","0.000007563426","","","","",""
|
||||||
|
"NL_NL_RADIXOR","NL_NL","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4796","25678","84","25678","0","1","25678","4797","61763","1214","1384","329602642","1214","329603856","0.000368","1384","63147","2.191711","0.980723121139","0.978082885964","0.999996316791","0.999992119320","0.989039601378","0.980193934392","0.979401224192","0.978609795129","0.959633939808","0.979402113872","0.979398173122","0.000007880680","0.979397283106","0.997373193672","0.997139403762","0.997256285015","0.997256285015"
|
||||||
|
"NL_NL_RADIXOR","NL_NL","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","4796","25678","84","25129","549","3","26239","4797","63147","0","0","329603856","0","329603856","0.000000","0","63147","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"NL_NL_RADIXOR","NL_NL","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","4796","25678","84","25129","549","3","26239","4797","63147","2651","0","329601205","2651","329603856","0.000804","0","63147","0.000000","0.959710021581","1.000000000000","0.999991957012","0.999991958552","0.999995978506","0.967506182222","0.979440846873","0.991673628866","0.959710021581","0.979647906945","0.979643967288","0.000008041448","","","","",""
|
||||||
|
"NN_NO_RADIXOR","NN_NO","ALL_WORDS","PRIMARY_OUTPUT","4688","18250","23","18250","0","1","18250","4680","26716","6230","3936","166485243","6230","166491473","0.003742","3936","30652","12.840924","0.810902689249","0.871590760799","0.999962580666","0.999938951055","0.935776670732","0.822354650447","0.840152206044","0.858737158800","0.724364188493","0.840699287413","0.840668985911","0.000061048945","0.840121715471","0.983845117159","0.986801676045","0.985321178741","0.985321178741"
|
||||||
|
"NN_NO_RADIXOR","NN_NO","ALL_WORDS","ANY_CANDIDATE","4688","18250","23","15846","2404","5","21513","4693","30652","0","0","166491473","0","166491473","0.000000","0","30652","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"NN_NO_RADIXOR","NN_NO","ALL_WORDS","ALL_CANDIDATES","4688","18250","23","15846","2404","5","21513","4693","30652","13214","0","166478259","13214","166491473","0.007937","0","30652","0.000000","0.698764418912","1.000000000000","0.999920632572","0.999920647181","0.999960316286","0.743561877778","0.822673716418","0.920624241623","0.698764418912","0.835921299473","0.835888126353","0.000079352819","","","","",""
|
||||||
|
"NN_NO_RADIXOR","NN_NO","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4681","18219","23","18219","0","1","18219","4668","26671","6230","3924","165920046","6230","165926276","0.003755","3924","30595","12.825625","0.810644053372","0.871743748979","0.999962453204","0.999938815429","0.935853101091","0.822169063928","0.840084414766","0.858797921188","0.724263408011","0.840638974932","0.840608609146","0.000061184571","0.840053856992","0.983814668550","0.986841730061","0.985325874420","0.985325874420"
|
||||||
|
"NN_NO_RADIXOR","NN_NO","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","4681","18219","23","15820","2399","5","21477","4681","30595","0","0","165926276","0","165926276","0.000000","0","30595","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"NN_NO_RADIXOR","NN_NO","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","4681","18219","23","15820","2399","5","21477","4681","30595","13214","0","165913062","13214","165926276","0.007964","0","30595","0.000000","0.698372480541","1.000000000000","0.999920362222","0.999920376903","0.999960181111","0.743206805583","0.822402021397","0.920488118949","0.698372480541","0.835686831618","0.835653554835","0.000079623097","","","","",""
|
||||||
|
"NORWEGIAN_BOKMAL_LUCENE_NORWEGIAN_LIGHT_STEM_FILTER","NB_NO","ALL_WORDS","PRIMARY_OUTPUT","17929","75310","252","75310","0","1","75310","25999","99529","25171","42651","2835593044","25171","2835618215","0.000888","42651","142180","29.997890","0.798147554130","0.700021100014","0.999991123276","0.999976083311","0.850006111645","0.776381478361","0.745870803357","0.717667503101","0.594732030284","0.747475838282","0.747464055956","0.000023916689","0.745858895755","0.987774378220","0.965622291196","0.976572729149","0.976572729149"
|
||||||
|
"NORWEGIAN_BOKMAL_LUCENE_NORWEGIAN_LIGHT_STEM_FILTER","NB_NO","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","17914","75251","252","75251","0","1","75251","25985","99450","25118","42641","2831151666","25118","2831176784","0.000887","42641","142091","30.009642","0.798359129150","0.699903582915","0.999991128071","0.999976068044","0.849947355493","0.776512696705","0.745896444523","0.717602881668","0.594764635875","0.747512150366","0.747500361706","0.000023931956","0.745884529658","0.987789184092","0.965602788874","0.976569991255","0.976569991255"
|
||||||
|
"NORWEGIAN_BOKMAL_LUCENE_NORWEGIAN_MINIMAL_STEM_FILTER","NB_NO","ALL_WORDS","PRIMARY_OUTPUT","17929","75310","252","75310","0","1","75310","27457","94526","14772","47654","2835603443","14772","2835618215","0.000521","47654","142180","33.516669","0.864846566268","0.664833309889","0.999994790554","0.999977986151","0.832414050221","0.815762584315","0.751763573752","0.697075888841","0.602260563739","0.758273568838","0.758263230800","0.000022013849","0.751752754540","0.992088894987","0.962515968891","0.977078714611","0.977078714611"
|
||||||
|
"NORWEGIAN_BOKMAL_LUCENE_NORWEGIAN_MINIMAL_STEM_FILTER","NB_NO","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","17914","75251","252","75251","0","1","75251","27443","94447","14719","47644","2831162065","14719","2831176784","0.000520","47644","142091","33.530625","0.865168642251","0.664693752595","0.999994801102","0.999977973869","0.832344276848","0.815949754214","0.751795969864","0.696994967013","0.602302149098","0.758335144541","0.758324803793","0.000022026131","0.751785145440","0.992107474043","0.962494026490","0.977076419047","0.977076419047"
|
||||||
|
"NORWEGIAN_BOKMAL_RADIXOR","NB_NO","ALL_WORDS","PRIMARY_OUTPUT","17929","75310","252","75310","0","1","75310","17886","135010","11482","7170","2835606733","11482","2835618215","0.000405","7170","142180","5.042903","0.921620293258","0.949570966381","0.999995950795","0.999993422575","0.974783458588","0.927078011613","0.935386875069","0.943846020481","0.878616704195","0.935491246621","0.935487968734","0.000006577425","0.935383586923","0.993354053349","0.994615320153","0.993984286646","0.993984286646"
|
||||||
|
"NORWEGIAN_BOKMAL_RADIXOR","NB_NO","ALL_WORDS","ANY_CANDIDATE","17929","75310","252","71073","4237","9","79825","17962","142180","0","0","2835618215","0","2835618215","0.000000","0","142180","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"NORWEGIAN_BOKMAL_RADIXOR","NB_NO","ALL_WORDS","ALL_CANDIDATES","17929","75310","252","71073","4237","9","79825","17962","142180","20161","0","2835598054","20161","2835618215","0.000711","0","142180","0.000000","0.875810793330","1.000000000000","0.999992890087","0.999992890443","0.999996445043","0.898118108406","0.933794385280","0.972422273928","0.875810793330","0.935847633608","0.935844306704","0.000007109557","","","","",""
|
||||||
|
"NORWEGIAN_BOKMAL_RADIXOR","NB_NO","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","17914","75251","252","75251","0","1","75251","17838","134987","11482","7104","2831165302","11482","2831176784","0.000406","7104","142091","4.999613","0.921607985307","0.950003870759","0.999995944443","0.999993435568","0.974999907601","0.927150543912","0.935590518436","0.944185565020","0.878976122105","0.935698217036","0.935694946007","0.000006564432","0.935587236809","0.993348241255","0.994693619298","0.994020475044","0.994020475044"
|
||||||
|
"NORWEGIAN_BOKMAL_RADIXOR","NB_NO","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","17914","75251","252","71047","4204","9","79733","17914","142091","0","0","2831176784","0","2831176784","0.000000","0","142091","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"NORWEGIAN_BOKMAL_RADIXOR","NB_NO","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","17914","75251","252","71047","4204","9","79733","17914","142091","20161","0","2831156623","20161","2831176784","0.000712","0","142091","0.000000","0.875742671893","1.000000000000","0.999992878933","0.999992879290","0.999996439466","0.898060798964","0.933755663840","0.972405477022","0.875742671893","0.935811237319","0.935807905326","0.000007120710","","","","",""
|
||||||
|
"PERSIAN_LUCENE_PERSIAN_STEM_FILTER","FA_IR","ALL_WORDS","PRIMARY_OUTPUT","69","3701","0","3701","0","1","3701","3190","430","179","98018","6748223","179","6748402","0.002652","98018","98448","99.563221","0.706075533662","0.004367788071","0.999973475202","0.985658076342","0.502170631636","0.021311605408","0.008681870034","0.005451304637","0.004359860890","0.055533668104","0.054800599292","0.014341923658","0.008506575635","0.985685778881","0.520346904570","0.681125382676","0.681125382676"
|
||||||
|
"PERSIAN_LUCENE_PERSIAN_STEM_FILTER","FA_IR","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","69","3701","0","3701","0","1","3701","3190","430","179","98018","6748223","179","6748402","0.002652","98018","98448","99.563221","0.706075533662","0.004367788071","0.999973475202","0.985658076342","0.502170631636","0.021311605408","0.008681870034","0.005451304637","0.004359860890","0.055533668104","0.054800599292","0.014341923658","0.008506575635","0.985685778881","0.520346904570","0.681125382676","0.681125382676"
|
||||||
|
"PERSIAN_RADIXOR","FA_IR","ALL_WORDS","PRIMARY_OUTPUT","69","3701","0","3701","0","1","3701","69","93636","8621","4812","6739781","8621","6748402","0.127749","4812","98448","4.887860","0.915692813206","0.951121404193","0.998722512381","0.998038075904","0.974921958287","0.922565796215","0.933070924989","0.943818050233","0.874538848780","0.933239001706","0.932248664283","0.001961924096","0.932075735269","0.980342969277","0.984586447427","0.982460126226","0.982460126226"
|
||||||
|
"PERSIAN_RADIXOR","FA_IR","ALL_WORDS","ANY_CANDIDATE","69","3701","0","3387","314","2","4015","69","98448","0","0","6748402","0","6748402","0.000000","0","98448","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"PERSIAN_RADIXOR","FA_IR","ALL_WORDS","ALL_CANDIDATES","69","3701","0","3387","314","2","4015","69","98448","13433","0","6734969","13433","6748402","0.199055","0","98448","0.000000","0.879934930864","1.000000000000","0.998009454683","0.998038075904","0.999004727341","0.901584696651","0.936133391021","0.973435401930","0.879934930864","0.938048469358","0.937114390300","0.001961924096","","","","",""
|
||||||
|
"PERSIAN_RADIXOR","FA_IR","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","69","3701","0","3701","0","1","3701","69","93636","8621","4812","6739781","8621","6748402","0.127749","4812","98448","4.887860","0.915692813206","0.951121404193","0.998722512381","0.998038075904","0.974921958287","0.922565796215","0.933070924989","0.943818050233","0.874538848780","0.933239001706","0.932248664283","0.001961924096","0.932075735269","0.980342969277","0.984586447427","0.982460126226","0.982460126226"
|
||||||
|
"PERSIAN_RADIXOR","FA_IR","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","69","3701","0","3387","314","2","4015","69","98448","0","0","6748402","0","6748402","0.000000","0","98448","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"PERSIAN_RADIXOR","FA_IR","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","69","3701","0","3387","314","2","4015","69","98448","13433","0","6734969","13433","6748402","0.199055","0","98448","0.000000","0.879934930864","1.000000000000","0.998009454683","0.998038075904","0.999004727341","0.901584696651","0.936133391021","0.973435401930","0.879934930864","0.938048469358","0.937114390300","0.001961924096","","","","",""
|
||||||
|
"POLISH_LUCENE_MORFOLOGIK_FILTER","PL_PL","ALL_WORDS","PRIMARY_OUTPUT","9990","122341","1","122341","0","1","122341","15519","1004747","99228","116220","7482378775","99228","7482478003","0.001326","116220","1120967","10.367834","0.910117529835","0.896321657997","0.999986738618","0.999971210643","0.948154198307","0.907324485129","0.903166914014","0.899047271013","0.823431500703","0.903193253581","0.903178865015","0.000028789357","0.903152518035","0.990022261217","0.977053921984","0.983495343428","0.983495343428"
|
||||||
|
"POLISH_LUCENE_MORFOLOGIK_FILTER","PL_PL","ALL_WORDS","ANY_CANDIDATE","9990","122341","1","109468","12873","5","136636","16295","1093112","85532","27855","7482392471","85532","7482478003","0.001143","27855","1120967","2.484908","0.927431862377","0.975150918805","0.999988569028","0.999984848600","0.987569743916","0.936598359399","0.950692965028","0.965218263555","0.906019814355","0.950992130738","0.950984648194","0.000015151400","","","","",""
|
||||||
|
"POLISH_LUCENE_MORFOLOGIK_FILTER","PL_PL","ALL_WORDS","ALL_CANDIDATES","9990","122341","1","109468","12873","5","136636","16295","1093112","143096","27855","7482334907","143096","7482478003","0.001912","27855","1120967","2.484908","0.884246016852","0.975150918805","0.999980875854","0.999977156579","0.987565897330","0.901045352805","0.927476322293","0.955504786999","0.864760696263","0.928586730350","0.928575670129","0.000022843421","","","","",""
|
||||||
|
"POLISH_LUCENE_MORFOLOGIK_FILTER","PL_PL","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","9846","120925","1","120925","0","1","120925","15277","999138","99224","115513","7310153475","99224","7310252699","0.001357","115513","1114651","10.363154","0.909661841906","0.896368459724","0.999986426735","0.999970629707","0.948177443229","0.906971715650","0.902966227492","0.898995962905","0.823097930182","0.902990688822","0.902976009256","0.000029370293","0.902951540918","0.989889334726","0.977011745487","0.983408384376","0.983408384376"
|
||||||
|
"POLISH_LUCENE_MORFOLOGIK_FILTER","PL_PL","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","9846","120925","1","108162","12763","5","135105","16044","1087157","85532","27494","7310167167","85532","7310252699","0.001170","27494","1114651","2.466602","0.927063356099","0.975333983462","0.999988299720","0.999984541059","0.987661141591","0.936331423447","0.950586270515","0.965281863330","0.905826028197","0.950892420848","0.950884788442","0.000015458941","","","","",""
|
||||||
|
"POLISH_LUCENE_MORFOLOGIK_FILTER","PL_PL","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","9846","120925","1","108162","12763","5","135105","16044","1087157","143085","27494","7310109614","143085","7310252699","0.001957","27494","1114651","2.466602","0.883693614752","0.975333983462","0.999980426805","0.999976669344","0.987657205134","0.900617649988","0.927255102898","0.955516285728","0.864376148890","0.928383764096","0.928372472930","0.000023330656","","","","",""
|
||||||
|
"POLISH_LUCENE_STEMPEL_DIRECT","PL_PL","ALL_WORDS","PRIMARY_OUTPUT","9990","122341","1","122341","0","1","122341","31432","797573","66669","323394","7482411334","66669","7482478003","0.000891","323394","1120967","28.849556","0.922858412343","0.711504442147","0.999991089984","0.999947877619","0.855747766065","0.871105640425","0.803515398127","0.745658746735","0.671563509358","0.810319603524","0.810295555502","0.000052122381","0.803489769425","0.991766794523","0.931068514076","0.960459620808","0.960459620808"
|
||||||
|
"POLISH_LUCENE_STEMPEL_DIRECT","PL_PL","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","9846","120925","1","120925","0","1","120925","30830","794493","66274","320158","7310186425","66274","7310252699","0.000907","320158","1114651","28.722712","0.923005877316","0.712772876892","0.999990934103","0.999947146412","0.856381905497","0.871590591697","0.804379630033","0.746792242917","0.672771767894","0.811106376847","0.811081975971","0.000052853588","0.804353636298","0.991711086900","0.931317880193","0.960566151650","0.960566151650"
|
||||||
|
"POLISH_LUCENE_STEMPEL_FILTER","PL_PL","ALL_WORDS","PRIMARY_OUTPUT","9990","122341","1","122341","0","1","122341","31432","797573","66669","323394","7482411334","66669","7482478003","0.000891","323394","1120967","28.849556","0.922858412343","0.711504442147","0.999991089984","0.999947877619","0.855747766065","0.871105640425","0.803515398127","0.745658746735","0.671563509358","0.810319603524","0.810295555502","0.000052122381","0.803489769425","0.991766794523","0.931068514076","0.960459620808","0.960459620808"
|
||||||
|
"POLISH_LUCENE_STEMPEL_FILTER","PL_PL","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","9846","120925","1","120925","0","1","120925","30830","794493","66274","320158","7310186425","66274","7310252699","0.000907","320158","1114651","28.722712","0.923005877316","0.712772876892","0.999990934103","0.999947146412","0.856381905497","0.871590591697","0.804379630033","0.746792242917","0.672771767894","0.811106376847","0.811081975971","0.000052853588","0.804353636298","0.991711086900","0.931317880193","0.960566151650","0.960566151650"
|
||||||
|
"POLISH_RADIXOR","PL_PL","ALL_WORDS","PRIMARY_OUTPUT","9990","122341","1","122341","0","1","122341","10074","1099420","13669","21547","7482464334","13669","7482478003","0.000183","21547","1120967","1.922180","0.987719760055","0.980778203105","0.999998173199","0.999995294243","0.990388188152","0.986323599045","0.984236742499","0.982158698021","0.968962733423","0.984242862020","0.984240510632","0.000004705757","0.984234389298","0.996967243455","0.996469409869","0.996718264498","0.996718264498"
|
||||||
|
"POLISH_RADIXOR","PL_PL","ALL_WORDS","ANY_CANDIDATE","9990","122341","1","119475","2866","4","125778","10079","1120967","0","0","7482478003","0","7482478003","0.000000","0","1120967","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"POLISH_RADIXOR","PL_PL","ALL_WORDS","ALL_CANDIDATES","9990","122341","1","119475","2866","4","125778","10079","1120967","38073","0","7482439930","38073","7482478003","0.000509","0","1120967","0.000000","0.967151263114","1.000000000000","0.999994911712","0.999994912475","0.999997455856","0.973547222425","0.983301367057","0.993252946885","0.967151263114","0.983438489746","0.983435987734","0.000005087525","","","","",""
|
||||||
|
"POLISH_RADIXOR","PL_PL","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","9846","120925","1","120925","0","1","120925","9844","1093651","13669","21000","7310239030","13669","7310252699","0.000187","21000","1114651","1.883998","0.987655781527","0.981160022285","0.999998130160","0.999995258206","0.990579076223","0.986349757961","0.984397186102","0.982452329568","0.969273787578","0.984402543989","0.984400174373","0.000004741794","0.984394814870","0.996926141446","0.996646530259","0.996786316244","0.996786316244"
|
||||||
|
"POLISH_RADIXOR","PL_PL","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","9846","120925","1","118145","2780","4","124274","9847","1114651","0","0","7310252699","0","7310252699","0.000000","0","1114651","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"POLISH_RADIXOR","PL_PL","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","9846","120925","1","118145","2780","4","124274","9847","1114651","38073","0","7310214626","38073","7310252699","0.000521","0","1114651","0.000000","0.966971278467","1.000000000000","0.999994791835","0.999994792629","0.999997395918","0.973401318686","0.983208335630","0.993214975136","0.966971278467","0.983346977657","0.983344416937","0.000005207371","","","","",""
|
||||||
|
"PORTUGUESE_LUCENE_PORTUGUESE_LIGHT_STEM_FILTER","PT_PT","ALL_WORDS","PRIMARY_OUTPUT","4001","211489","0","211489","0","1","211489","112814","149830","2511","5339230","22358201245","2511","22358203756","0.000011","5339230","5489060","97.270389","0.983517240927","0.027296112631","0.999999887692","0.999761142266","0.513648000162","0.122843213263","0.053118010934","0.033885033146","0.027283631587","0.163848092400","0.163827867245","0.000238857734","0.053105458848","0.999226179883","0.720580123442","0.837329788424","0.837329788424"
|
||||||
|
"PORTUGUESE_LUCENE_PORTUGUESE_LIGHT_STEM_FILTER","PT_PT","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4001","211489","0","211489","0","1","211489","112814","149830","2511","5339230","22358201245","2511","22358203756","0.000011","5339230","5489060","97.270389","0.983517240927","0.027296112631","0.999999887692","0.999761142266","0.513648000162","0.122843213263","0.053118010934","0.033885033146","0.027283631587","0.163848092400","0.163827867245","0.000238857734","0.053105458848","0.999226179883","0.720580123442","0.837329788424","0.837329788424"
|
||||||
|
"PORTUGUESE_LUCENE_PORTUGUESE_MINIMAL_STEM_FILTER","PT_PT","ALL_WORDS","PRIMARY_OUTPUT","4001","211489","0","211489","0","1","211489","167745","43406","598","5445654","22358203158","598","22358203756","0.000003","5445654","5489060","99.209227","0.986410326334","0.007907729192","0.999999973254","0.999756469021","0.503953851223","0.038310165654","0.015689679353","0.009864890589","0.007906867787","0.088319113068","0.088308061848","0.000243530979","0.015685836581","0.999664059174","0.692382565086","0.818121623354","0.818121623354"
|
||||||
|
"PORTUGUESE_LUCENE_PORTUGUESE_MINIMAL_STEM_FILTER","PT_PT","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4001","211489","0","211489","0","1","211489","167745","43406","598","5445654","22358203158","598","22358203756","0.000003","5445654","5489060","99.209227","0.986410326334","0.007907729192","0.999999973254","0.999756469021","0.503953851223","0.038310165654","0.015689679353","0.009864890589","0.007906867787","0.088319113068","0.088308061848","0.000243530979","0.015685836581","0.999664059174","0.692382565086","0.818121623354","0.818121623354"
|
||||||
|
"PORTUGUESE_LUCENE_PORTUGUESE_STEM_FILTER","PT_PT","ALL_WORDS","PRIMARY_OUTPUT","4001","211489","0","211489","0","1","211489","27586","3803488","99075","1685572","22358104681","99075","22358203756","0.000443","1685572","5489060","30.707844","0.974612837768","0.692921556696","0.999995568741","0.999920198913","0.846458562719","0.901329863268","0.809974591186","0.735433886866","0.680636384053","0.821784792219","0.821750382444","0.000079801087","0.809935821347","0.996728545383","0.918475110538","0.956003146785","0.956003146785"
|
||||||
|
"PORTUGUESE_LUCENE_PORTUGUESE_STEM_FILTER","PT_PT","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4001","211489","0","211489","0","1","211489","27586","3803488","99075","1685572","22358104681","99075","22358203756","0.000443","1685572","5489060","30.707844","0.974612837768","0.692921556696","0.999995568741","0.999920198913","0.846458562719","0.901329863268","0.809974591186","0.735433886866","0.680636384053","0.821784792219","0.821750382444","0.000079801087","0.809935821347","0.996728545383","0.918475110538","0.956003146785","0.956003146785"
|
||||||
|
"PORTUGUESE_RADIXOR","PT_PT","ALL_WORDS","PRIMARY_OUTPUT","4001","211489","0","211489","0","1","211489","4001","5472616","20678","16444","22358183078","20678","22358203756","0.000092","16444","5489060","0.299578","0.996235774018","0.997004222945","0.999999075149","0.999998340077","0.998501649047","0.996389369023","0.996619850353","0.996850438335","0.993262474550","0.996619924417","0.996619094289","0.000001659923","0.996619020188","0.999299376330","0.999346803887","0.999323089546","0.999323089546"
|
||||||
|
"PORTUGUESE_RADIXOR","PT_PT","ALL_WORDS","ANY_CANDIDATE","4001","211489","0","210699","790","3","212297","4001","5489060","0","0","22358203756","0","22358203756","0.000000","0","5489060","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"PORTUGUESE_RADIXOR","PT_PT","ALL_WORDS","ALL_CANDIDATES","4001","211489","0","210699","790","3","212297","4001","5489060","38310","0","22358165446","38310","22358203756","0.000171","0","5489060","0.000000","0.993069036450","1.000000000000","0.999998286535","0.999998286956","0.999999143267","0.994447532369","0.996522466897","0.998606078314","0.993069036450","0.996528492543","0.996527638784","0.000001713044","","","","",""
|
||||||
|
"PORTUGUESE_RADIXOR","PT_PT","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4001","211489","0","211489","0","1","211489","4001","5472616","20678","16444","22358183078","20678","22358203756","0.000092","16444","5489060","0.299578","0.996235774018","0.997004222945","0.999999075149","0.999998340077","0.998501649047","0.996389369023","0.996619850353","0.996850438335","0.993262474550","0.996619924417","0.996619094289","0.000001659923","0.996619020188","0.999299376330","0.999346803887","0.999323089546","0.999323089546"
|
||||||
|
"PORTUGUESE_RADIXOR","PT_PT","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","4001","211489","0","210699","790","3","212297","4001","5489060","0","0","22358203756","0","22358203756","0.000000","0","5489060","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"PORTUGUESE_RADIXOR","PT_PT","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","4001","211489","0","210699","790","3","212297","4001","5489060","38310","0","22358165446","38310","22358203756","0.000171","0","5489060","0.000000","0.993069036450","1.000000000000","0.999998286535","0.999998286956","0.999999143267","0.994447532369","0.996522466897","0.998606078314","0.993069036450","0.996528492543","0.996527638784","0.000001713044","","","","",""
|
||||||
|
"RUSSIAN_LUCENE_RUSSIAN_LIGHT_STEM_FILTER","RU_RU","ALL_WORDS","PRIMARY_OUTPUT","37410","768882","10","768882","0","1","768882","232250","3081067","321183","10008438","295575969833","321183","295576291016","0.000109","10008438","13089505","76.461547","0.905596884415","0.235384531348","0.999998913367","0.999965054154","0.617691722357","0.577011147253","0.373649378129","0.276277984307","0.229747124085","0.461696326851","0.461686629842","0.000034945846","0.373637933830","0.994310930069","0.870888421754","0.928516167212","0.928516167212"
|
||||||
|
"RUSSIAN_LUCENE_RUSSIAN_LIGHT_STEM_FILTER","RU_RU","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","37297","768133","10","768133","0","1","768133","232143","3078888","318921","10008238","295000362731","318921","295000681652","0.000108","10008238","13087126","76.473918","0.906139220892","0.235260820443","0.999998918914","0.999964994315","0.617629869679","0.577038425373","0.373539598427","0.276151716079","0.229664120975","0.461713175622","0.461703471649","0.000035005685","0.373528142103","0.994349544377","0.870767393403","0.928464208705","0.928464208705"
|
||||||
|
"RUSSIAN_RADIXOR","RU_RU","ALL_WORDS","PRIMARY_OUTPUT","37410","768882","10","768882","0","1","768882","37561","12823203","155850","266302","295576135166","155850","295576291016","0.000053","266302","13089505","2.034470","0.987992190185","0.979655304001","0.999999472725","0.999998571830","0.989827388363","0.986313480705","0.983806085477","0.981311406466","0.968128298562","0.983814916245","0.983814202914","0.000001428170","0.983805371373","0.997699288696","0.997273959852","0.997486578934","0.997486578934"
|
||||||
|
"RUSSIAN_RADIXOR","RU_RU","ALL_WORDS","ANY_CANDIDATE","37410","768882","10","749720","19162","4","788492","37593","13089492","0","13","295576291016","0","295576291016","0.000000","13","13089505","0.000099","1.000000000000","0.999999006838","1.000000000000","0.999999999956","0.999999503419","0.999999801367","0.999999503419","0.999999205470","0.999999006838","0.999999503419","0.999999503397","0.000000000044","","","","",""
|
||||||
|
"RUSSIAN_RADIXOR","RU_RU","ALL_WORDS","ALL_CANDIDATES","37410","768882","10","749720","19162","4","788492","37593","13089492","434710","13","295575856306","434710","295576291016","0.000147","13","13089505","0.000099","0.967856883534","0.999999006838","0.999998529280","0.999998529301","0.999998768059","0.974118939969","0.983665447282","0.993400920813","0.967855953192","0.983796687479","0.983795964011","0.000001470699","","","","",""
|
||||||
|
"RUSSIAN_RADIXOR","RU_RU","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","37297","768133","10","768133","0","1","768133","37282","12821513","155850","265613","295000525802","155850","295000681652","0.000053","265613","13087126","2.029575","0.987990626447","0.979704252867","0.999999471696","0.999998571379","0.989851862281","0.986322156837","0.983829991833","0.981350389119","0.968174600634","0.983838715706","0.983838002141","0.000001428621","0.983829277503","0.997696524283","0.997321167437","0.997508810549","0.997508810549"
|
||||||
|
"RUSSIAN_RADIXOR","RU_RU","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","37297","768133","10","749142","18991","4","787549","37306","13087126","0","0","295000681652","0","295000681652","0.000000","0","13087126","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"RUSSIAN_RADIXOR","RU_RU","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","37297","768133","10","749142","18991","4","787549","37306","13087126","434710","0","295000246942","434710","295000681652","0.000147","0","13087126","0.000000","0.967851259252","1.000000000000","0.999998526410","0.999998526476","0.999999263205","0.974114570610","0.983663023007","0.993400519870","0.967851259252","0.983794317554","0.983793592699","0.000001473524","","","","",""
|
||||||
|
"SNOWBALL_DANISH_DIRECT","DA_DK","ALL_WORDS","PRIMARY_OUTPUT","4179","28079","32","28079","0","1","28079","5553","78732","6341","11163","394104845","6341","394111186","0.001609","11163","89895","12.417821","0.925464013259","0.875821792091","0.999983910632","0.999955596266","0.937902851361","0.915090414169","0.899958849618","0.885319563795","0.818113803566","0.900300811178","0.900278764621","0.000044403734","0.899936659693","0.994195946109","0.978579164615","0.986325742978","0.986325742978"
|
||||||
|
"SNOWBALL_DANISH_DIRECT","DA_DK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4173","28033","32","28033","0","1","28033","5539","78627","6341","11113","392814447","6341","392820788","0.001614","11113","89740","12.383552","0.925371904717","0.876164475150","0.999983857779","0.999955577673","0.938074166465","0.915093153823","0.900096160451","0.885582797210","0.818340774971","0.900432112497","0.900410054086","0.000044422327","0.900073960926","0.994185354918","0.978644331817","0.986353630937","0.986353630937"
|
||||||
|
"SNOWBALL_DANISH_LUCENE_FILTER","DA_DK","ALL_WORDS","PRIMARY_OUTPUT","4179","28079","32","28079","0","1","28079","5546","78744","6507","11151","394104679","6507","394111186","0.001651","11151","89895","12.404472","0.923672449590","0.875955281161","0.999983489431","0.999955205602","0.937969385296","0.913717599716","0.899181254496","0.885100184115","0.816829526358","0.899497504322","0.899475250557","0.000044794398","0.899158868074","0.994052746860","0.978603476314","0.986267614487","0.986267614487"
|
||||||
|
"SNOWBALL_DANISH_LUCENE_FILTER","DA_DK","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4173","28033","32","28033","0","1","28033","5539","78627","6341","11113","392814447","6341","392820788","0.001614","11113","89740","12.383552","0.925371904717","0.876164475150","0.999983857779","0.999955577673","0.938074166465","0.915093153823","0.900096160451","0.885582797210","0.818340774971","0.900432112497","0.900410054086","0.000044422327","0.900073960926","0.994185354918","0.978644331817","0.986353630937","0.986353630937"
|
||||||
|
"SNOWBALL_DUTCH_DIRECT","NL_NL","ALL_WORDS","PRIMARY_OUTPUT","4992","26477","85","26477","0","1","26477","12051","29325","4382","35241","350433578","4382","350437960","0.001250","35241","64566","54.581359","0.869997329931","0.454186413902","0.999987495647","0.999886953739","0.727086954774","0.735353119953","0.596806854375","0.502190286022","0.425320531415","0.628602392126","0.628557411604","0.000113046261","0.596755898222","0.992814719235","0.917346281080","0.953589661124","0.953589661124"
|
||||||
|
"SNOWBALL_DUTCH_DIRECT","NL_NL","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4796","25678","84","25678","0","1","25678","11466","29111","4382","34036","329599474","4382","329603856","0.001329","34036","63147","53.899631","0.869166691547","0.461003689803","0.999986705253","0.999883464224","0.730495197528","0.738411822300","0.602462748344","0.508789468717","0.431088865524","0.633000040962","0.632953313739","0.000116535776","0.602409959779","0.992557044900","0.918058993802","0.953855618789","0.953855618789"
|
||||||
|
"SNOWBALL_DUTCH_LUCENE_FILTER","NL_NL","ALL_WORDS","PRIMARY_OUTPUT","4992","26477","85","26477","0","1","26477","14573","15302","1588","49264","350436372","1588","350437960","0.000453","49264","64566","76.300220","0.905979869745","0.236997800700","0.999995468527","0.999854916880","0.618496634614","0.579068464950","0.375712040856","0.278062466837","0.231308764398","0.463373754768","0.463333378452","0.000145083120","0.375664346452","0.995827666179","0.888410302915","0.939057139355","0.939057139355"
|
||||||
|
"SNOWBALL_DUTCH_LUCENE_FILTER","NL_NL","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4796","25678","84","25678","0","1","25678","14116","14972","1544","48175","329602312","1544","329603856","0.000468","48175","63147","76.290243","0.906514894648","0.237097565997","0.999995315589","0.999849184178","0.618546440793","0.579362438183","0.375883408860","0.278182412747","0.231438685443","0.463608105042","0.463566154833","0.000150815822","0.375833834664","0.995816517119","0.887491664267","0.938538755173","0.938538755173"
|
||||||
|
"SNOWBALL_FINNISH_DIRECT","FI_FI","ALL_WORDS","PRIMARY_OUTPUT","57027","1811717","292","1811717","0","1","1811717","381483","15114332","1544812","16409363","1641125269679","1544812","1641126814491","0.000094","16409363","31523695","52.054060","0.907269425128","0.479459403474","0.999999058688","0.999989060059","0.739729231081","0.769880311353","0.627374073993","0.529384116965","0.457061215373","0.659544431681","0.659540149918","0.000010939941","0.627369124557","0.991871857177","0.904138579582","0.945975396220","0.945975396220"
|
||||||
|
"SNOWBALL_FINNISH_DIRECT","FI_FI","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","54762","1757055","274","1757055","0","1","1757055","372232","14692070","1513705","16121763","1543587930447","1513705","1543589444152","0.000098","16121763","30813833","52.319888","0.906594717007","0.476801117213","0.999999019360","0.999988575255","0.738400068286","0.768116957494","0.624933751043","0.526744348874","0.454475376380","0.657468914800","0.657464451566","0.000011424745","0.624928589970","0.991731678095","0.902933406396","0.945251664438","0.945251664438"
|
||||||
|
"SNOWBALL_FINNISH_LUCENE_FILTER","FI_FI","ALL_WORDS","PRIMARY_OUTPUT","57027","1811717","292","1811717","0","1","1811717","377778","15153638","1922153","16370057","1641124892338","1922153","1641126814491","0.000117","16370057","31523695","51.929372","0.887434028678","0.480706275073","0.999998828760","0.999988854086","0.740352551917","0.758996033322","0.623613097472","0.529216231176","0.453079796332","0.653142485450","0.653138019077","0.000011145914","0.623608016975","0.990717710840","0.904385055188","0.945584912497","0.945584912497"
|
||||||
|
"SNOWBALL_FINNISH_LUCENE_FILTER","FI_FI","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","54762","1757055","274","1757055","0","1","1757055","372232","14692070","1513705","16121763","1543587930447","1513705","1543589444152","0.000098","16121763","30813833","52.319888","0.906594717007","0.476801117213","0.999999019360","0.999988575255","0.738400068286","0.768116957494","0.624933751043","0.526744348874","0.454475376380","0.657468914800","0.657464451566","0.000011424745","0.624928589970","0.991731678095","0.902933406396","0.945251664438","0.945251664438"
|
||||||
|
"SNOWBALL_FRENCH_DIRECT","FR_FR","ALL_WORDS","PRIMARY_OUTPUT","59240","425210","2301","425210","0","1","425210","85627","3766640","1654723","1687975","90394450107","1654723","90396104830","0.001831","1687975","5454615","30.945814","0.694777309691","0.690541862258","0.999981694753","0.999963023890","0.845261778506","0.693926068790","0.692653111288","0.691384815533","0.529815856272","0.692656348624","0.692637859933","0.000036976110","0.692634622294","0.959459328254","0.944947915186","0.952148333884","0.952148333884"
|
||||||
|
"SNOWBALL_FRENCH_DIRECT","FR_FR","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","57698","421231","2133","421231","0","1","421231","84526","3758589","1646111","1681970","88710480395","1646111","88712126506","0.001856","1681970","5440559","30.915389","0.695429718578","0.690846106071","0.999981444352","0.999962486787","0.845413775212","0.694508136723","0.693130334647","0.691757988461","0.530374491828","0.693134123475","0.693115366288","0.000037513213","0.693111577099","0.959520798119","0.944537159644","0.951970023370","0.951970023370"
|
||||||
|
"SNOWBALL_FRENCH_LUCENE_FILTER","FR_FR","ALL_WORDS","PRIMARY_OUTPUT","59240","425210","2301","425210","0","1","425210","85202","3763777","1661388","1690838","90394443442","1661388","90396104830","0.001838","1690838","5454615","30.998301","0.693762678186","0.690016985617","0.999981621022","0.999962918494","0.844999303320","0.693010289898","0.691884762376","0.690762884895","0.528917286853","0.691887297134","0.691868755638","0.000037081506","0.691866220643","0.958697792387","0.944714715363","0.951654891797","0.951654891797"
|
||||||
|
"SNOWBALL_FRENCH_LUCENE_FILTER","FR_FR","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","57698","421231","2133","421231","0","1","421231","84810","3755856","1641925","1684703","88710484581","1641925","88712126506","0.001851","1684703","5440559","30.965623","0.695814817237","0.690343767984","0.999981491538","0.999962503165","0.845162629761","0.694713680979","0.693068495729","0.691431084156","0.530302080457","0.693073894149","0.693055145391","0.000037496835","0.693049746458","0.959566165512","0.944384738154","0.951914926182","0.951914926182"
|
||||||
|
"SNOWBALL_GERMAN_DIRECT","DE_DE","ALL_WORDS","PRIMARY_OUTPUT","54092","296974","1474","296974","0","1","296974","81649","771138","190680","612734","44095055299","190680","44095245979","0.000432","612734","1383872","44.276783","0.801750435114","0.557232171762","0.999995675724","0.999981780603","0.778613923743","0.737064397386","0.657493530688","0.593429030432","0.489750735447","0.668401927114","0.668393541401","0.000018219397","0.657484715679","0.983724573695","0.949324273697","0.966218331938","0.966218331938"
|
||||||
|
"SNOWBALL_GERMAN_DIRECT","DE_DE","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","16007","150098","228","150098","0","1","150098","37843","516936","87697","356475","11263668645","87697","11263756342","0.000779","356475","873411","40.814118","0.854958297017","0.591858815609","0.999992214231","0.999960569321","0.795925514920","0.785153327381","0.699486618802","0.630674793334","0.537854226580","0.711347035607","0.711329191110","0.000039430679","0.699467554207","0.988417636496","0.932451723900","0.959619376607","0.959619376607"
|
||||||
|
"SNOWBALL_GERMAN_LUCENE_FILTER","DE_DE","ALL_WORDS","PRIMARY_OUTPUT","54092","296974","1474","296974","0","1","296974","86669","751056","295701","632816","44094950278","295701","44095245979","0.000671","632816","1383872","45.727929","0.717507501741","0.542720714054","0.999993294039","0.999978943584","0.771357004047","0.674088567377","0.617993120299","0.570516594262","0.447170798768","0.624024185176","0.624014089276","0.000021056416","0.617982794334","0.975844522648","0.942925157079","0.959102449371","0.959102449371"
|
||||||
|
"SNOWBALL_GERMAN_LUCENE_FILTER","DE_DE","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","16007","150098","228","150098","0","1","150098","46077","481501","77653","391910","11263678689","77653","11263756342","0.000689","391910","873411","44.871200","0.861124126806","0.551287996144","0.999993105941","0.999958315274","0.775640551042","0.774110642769","0.672222202832","0.594035281304","0.506276128631","0.689004640259","0.688986412764","0.000041684726","0.672202362239","0.989021274644","0.919542244735","0.953017108147","0.953017108147"
|
||||||
|
"SNOWBALL_HUNGARIAN_DIRECT","HU_HU","ALL_WORDS","PRIMARY_OUTPUT","19406","916344","1","916344","0","1","916344","116105","14287912","1506056","7874191","419819036837","1506056","419820542893","0.000359","7874191","22162103","35.529981","0.904643595580","0.644700189328","0.999996412620","0.999977657711","0.822348300974","0.837136808086","0.752865700984","0.684009307333","0.603676525918","0.763690969794","0.763680928295","0.000022342289","0.752854843818","0.991947513126","0.924304257644","0.956931989551","0.956931989551"
|
||||||
|
"SNOWBALL_HUNGARIAN_DIRECT","HU_HU","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","18360","878513","1","878513","0","1","878513","111379","13776526","1496670","7634885","385869198247","1496670","385870694917","0.000388","7634885","21411411","35.658019","0.902006757459","0.643419810119","0.999996121317","0.999976336507","0.821707965718","0.834898516372","0.751079383241","0.682554714263","0.601382804609","0.761819543337","0.761808891723","0.000023663493","0.751067882221","0.991609896137","0.923288102987","0.956230170303","0.956230170303"
|
||||||
|
"SNOWBALL_HUNGARIAN_LUCENE_FILTER","HU_HU","ALL_WORDS","PRIMARY_OUTPUT","19406","916344","1","916344","0","1","916344","114867","14299358","1792049","7862745","419818750844","1792049","419820542893","0.000427","7862745","22162103","35.478334","0.888633169244","0.645216656560","0.999995731393","0.999977003783","0.822606193976","0.826287586346","0.747610245439","0.682613266689","0.596946950992","0.757205997314","0.757195513225","0.000022996217","0.747599036407","0.990687085622","0.924490230693","0.956444632598","0.956444632598"
|
||||||
|
"SNOWBALL_HUNGARIAN_LUCENE_FILTER","HU_HU","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","18360","878513","1","878513","0","1","878513","111379","13776526","1496670","7634885","385869198247","1496670","385870694917","0.000388","7634885","21411411","35.658019","0.902006757459","0.643419810119","0.999996121317","0.999976336507","0.821707965718","0.834898516372","0.751079383241","0.682554714263","0.601382804609","0.761819543337","0.761808891723","0.000023663493","0.751067882221","0.991609896137","0.923288102987","0.956230170303","0.956230170303"
|
||||||
|
"SNOWBALL_ITALIAN_DIRECT","IT_IT","ALL_WORDS","PRIMARY_OUTPUT","10009","327551","0","327551","0","1","327551","46828","4499650","504775","1644164","53638016436","504775","53638521211","0.000941","1644164","6143814","26.761292","0.899134266174","0.732387080729","0.999990589319","0.999959941236","0.866188835024","0.859975076366","0.807239600802","0.760598128154","0.676782697802","0.811488952720","0.811469907061","0.000040058764","0.807219778599","0.987993752409","0.933407841859","0.959925420024","0.959925420024"
|
||||||
|
"SNOWBALL_ITALIAN_DIRECT","IT_IT","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","10007","327469","0","327469","0","1","327469","46814","4498652","504774","1643522","53611162298","504774","53611667072","0.000942","1643522","6142174","26.757985","0.899114326863","0.732420149608","0.999990584624","0.999959933163","0.866205367116","0.859969602244","0.807251650876","0.760623806435","0.676799637969","0.811498274672","0.811479224597","0.000040066837","0.807231824542","0.987990915845","0.933413003102","0.959926810498","0.959926810498"
|
||||||
|
"SNOWBALL_ITALIAN_LUCENE_FILTER","IT_IT","ALL_WORDS","PRIMARY_OUTPUT","10009","327551","0","327551","0","1","327551","46828","4499650","504775","1644164","53638016436","504775","53638521211","0.000941","1644164","6143814","26.761292","0.899134266174","0.732387080729","0.999990589319","0.999959941236","0.866188835024","0.859975076366","0.807239600802","0.760598128154","0.676782697802","0.811488952720","0.811469907061","0.000040058764","0.807219778599","0.987993752409","0.933407841859","0.959925420024","0.959925420024"
|
||||||
|
"SNOWBALL_ITALIAN_LUCENE_FILTER","IT_IT","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","10007","327469","0","327469","0","1","327469","46814","4498652","504774","1643522","53611162298","504774","53611667072","0.000942","1643522","6142174","26.757985","0.899114326863","0.732420149608","0.999990584624","0.999959933163","0.866205367116","0.859969602244","0.807251650876","0.760623806435","0.676799637969","0.811498274672","0.811479224597","0.000040066837","0.807231824542","0.987990915845","0.933413003102","0.959926810498","0.959926810498"
|
||||||
|
"SNOWBALL_NORWEGIAN_BOKMAL_DIRECT","NB_NO","ALL_WORDS","PRIMARY_OUTPUT","17929","75310","252","75310","0","1","75310","24394","106626","23997","35554","2835594218","23997","2835618215","0.000846","35554","142180","25.006330","0.816288096277","0.749936699958","0.999991537295","0.999978999989","0.874964118626","0.802094867845","0.781706946038","0.762329786671","0.641641141674","0.782409356499","0.782398932962","0.000021000011","0.781696464373","0.988328173631","0.971119668400","0.979648355684","0.979648355684"
|
||||||
|
"SNOWBALL_NORWEGIAN_BOKMAL_DIRECT","NB_NO","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","17914","75251","252","75251","0","1","75251","24367","106567","23997","35524","2831152787","23997","2831176784","0.000848","35524","142091","25.000880","0.816205079501","0.749991202821","0.999991524019","0.999978977642","0.874991363420","0.802043209347","0.781698483431","0.762360357576","0.641629738452","0.782397999310","0.782387564360","0.000021022358","0.781687990535","0.988317966243","0.971127035574","0.979647089734","0.979647089734"
|
||||||
|
"SNOWBALL_NORWEGIAN_BOKMAL_LUCENE_FILTER","NB_NO","ALL_WORDS","PRIMARY_OUTPUT","17929","75310","252","75310","0","1","75310","24396","106589","24046","35591","2835594169","24046","2835618215","0.000848","35591","142180","25.032353","0.815929880966","0.749676466451","0.999991520015","0.999978969662","0.874833993233","0.801758635215","0.781401315910","0.762052176648","0.641229410562","0.782101930719","0.782091491842","0.000021030338","0.781390819068","0.988295184140","0.971086244692","0.979615142710","0.979615142710"
|
||||||
|
"SNOWBALL_NORWEGIAN_BOKMAL_LUCENE_FILTER","NB_NO","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","17914","75251","252","75251","0","1","75251","24381","106512","23993","35579","2831152791","23993","2831176784","0.000847","35579","142091","25.039587","0.816152637830","0.749604126933","0.999991525432","0.999978959629","0.874797826182","0.801914137847","0.781464144742","0.762031224736","0.641314033862","0.782170943927","0.782160500766","0.000021040371","0.781453643056","0.988310445474","0.971073741466","0.979616277822","0.979616277822"
|
||||||
|
"SNOWBALL_NORWEGIAN_NYNORSK_DIRECT","NN_NO","ALL_WORDS","PRIMARY_OUTPUT","4688","18250","23","18250","0","1","18250","6138","22004","8274","8648","166483199","8274","166491473","0.004970","8648","30652","28.213493","0.726732280864","0.717865065901","0.999950303761","0.999898379870","0.858907684831","0.724941356316","0.722271459051","0.719621155632","0.565277706417","0.722285066089","0.722234252664","0.000101620130","0.722220641604","0.980542486408","0.964998187466","0.972708239744","0.972708239744"
|
||||||
|
"SNOWBALL_NORWEGIAN_NYNORSK_DIRECT","NN_NO","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4681","18219","23","18219","0","1","18219","6120","21971","8274","8624","165918002","8274","165926276","0.004987","8624","30595","28.187612","0.726434121342","0.718123876450","0.999950134480","0.999898178365","0.859037005465","0.724756721095","0.722255095332","0.719770679771","0.565257660346","0.722267047015","0.722216132089","0.000101821635","0.722204176866","0.980505813025","0.965064509051","0.972723884952","0.972723884952"
|
||||||
|
"SNOWBALL_NORWEGIAN_NYNORSK_LUCENE_FILTER","NN_NO","ALL_WORDS","PRIMARY_OUTPUT","4688","18250","23","18250","0","1","18250","6144","21978","8295","8674","166483178","8295","166491473","0.004982","8674","30652","28.298317","0.725993459518","0.717016834138","0.999950177629","0.999898097625","0.858483505883","0.724180198229","0.721477226098","0.718794356395","0.564305338023","0.721491186328","0.721440231913","0.000101902375","0.721426267560","0.980461058483","0.964862123312","0.972599049418","0.972599049418"
|
||||||
|
"SNOWBALL_NORWEGIAN_NYNORSK_LUCENE_FILTER","NN_NO","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4681","18219","23","18219","0","1","18219","6130","21948","8274","8647","165918002","8274","165926276","0.004987","8647","30595","28.262788","0.726225928132","0.717372119627","0.999950134480","0.999898039774","0.858661127054","0.724437725685","0.721771872996","0.719125568472","0.564665929147","0.721785448310","0.721734464790","0.000101960226","0.721720885459","0.980505813025","0.964944952705","0.972663150405","0.972663150405"
|
||||||
|
"SNOWBALL_PORTUGUESE_DIRECT","PT_PT","ALL_WORDS","PRIMARY_OUTPUT","4001","211489","0","211489","0","1","211489","11315","4817239","167230","671821","22358036526","167230","22358203756","0.000748","671821","5489060","12.239272","0.966449786326","0.877607277020","0.999992520419","0.999962481554","0.938799898719","0.947270839082","0.919888415834","0.894044585092","0.851660540743","0.920957852105","0.920939611009","0.000037518446","0.919869695779","0.996663145176","0.967923515462","0.982083116554","0.982083116554"
|
||||||
|
"SNOWBALL_PORTUGUESE_DIRECT","PT_PT","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4001","211489","0","211489","0","1","211489","11315","4817239","167230","671821","22358036526","167230","22358203756","0.000748","671821","5489060","12.239272","0.966449786326","0.877607277020","0.999992520419","0.999962481554","0.938799898719","0.947270839082","0.919888415834","0.894044585092","0.851660540743","0.920957852105","0.920939611009","0.000037518446","0.919869695779","0.996663145176","0.967923515462","0.982083116554","0.982083116554"
|
||||||
|
"SNOWBALL_PORTUGUESE_LUCENE_FILTER","PT_PT","ALL_WORDS","PRIMARY_OUTPUT","4001","211489","0","211489","0","1","211489","11315","4817239","167230","671821","22358036526","167230","22358203756","0.000748","671821","5489060","12.239272","0.966449786326","0.877607277020","0.999992520419","0.999962481554","0.938799898719","0.947270839082","0.919888415834","0.894044585092","0.851660540743","0.920957852105","0.920939611009","0.000037518446","0.919869695779","0.996663145176","0.967923515462","0.982083116554","0.982083116554"
|
||||||
|
"SNOWBALL_PORTUGUESE_LUCENE_FILTER","PT_PT","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","4001","211489","0","211489","0","1","211489","11315","4817239","167230","671821","22358036526","167230","22358203756","0.000748","671821","5489060","12.239272","0.966449786326","0.877607277020","0.999992520419","0.999962481554","0.938799898719","0.947270839082","0.919888415834","0.894044585092","0.851660540743","0.920957852105","0.920939611009","0.000037518446","0.919869695779","0.996663145176","0.967923515462","0.982083116554","0.982083116554"
|
||||||
|
"SNOWBALL_RUSSIAN_DIRECT","RU_RU","ALL_WORDS","PRIMARY_OUTPUT","37410","768882","10","768882","0","1","768882","64358","8766656","3782908","4322849","295572508108","3782908","295576291016","0.001280","4322849","13089505","33.025305","0.698562595481","0.669746946122","0.999987201585","0.999972577645","0.834867073854","0.692602792505","0.683851352013","0.675318311031","0.519585195076","0.684003044583","0.683989349009","0.000027422355","0.683837646322","0.974179960240","0.953661001039","0.963811283954","0.963811283954"
|
||||||
|
"SNOWBALL_RUSSIAN_DIRECT","RU_RU","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","37297","768133","10","768133","0","1","768133","64159","8764719","3782908","4322407","294996898744","3782908","295000681652","0.001282","4322407","13087126","33.027931","0.698516062041","0.669720685810","0.999987176613","0.999972525638","0.834853931211","0.692560581516","0.683815365804","0.675288254087","0.519543647630","0.683966853085","0.683953131513","0.000027474362","0.683801634112","0.974148936252","0.953633561396","0.963782086975","0.963782086975"
|
||||||
|
"SNOWBALL_RUSSIAN_LUCENE_FILTER","RU_RU","ALL_WORDS","PRIMARY_OUTPUT","37410","768882","10","768882","0","1","768882","64266","8766889","3785790","4322616","295572505226","3785790","295576291016","0.001281","4322616","13089505","33.023525","0.698407806015","0.669764746642","0.999987191835","0.999972568683","0.834875969239","0.692484865100","0.683786451263","0.675303850911","0.519510266339","0.683936347366","0.683922647122","0.000027431317","0.683772741022","0.974131099393","0.953673855106","0.963793934411","0.963793934411"
|
||||||
|
"SNOWBALL_RUSSIAN_LUCENE_FILTER","RU_RU","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","37297","768133","10","768133","0","1","768133","64159","8764719","3782908","4322407","294996898744","3782908","295000681652","0.001282","4322407","13087126","33.027931","0.698516062041","0.669720685810","0.999987176613","0.999972525638","0.834853931211","0.692560581516","0.683815365804","0.675288254087","0.519543647630","0.683966853085","0.683953131513","0.000027474362","0.683801634112","0.974148936252","0.953633561396","0.963782086975","0.963782086975"
|
||||||
|
"SNOWBALL_SPANISH_DIRECT","ES_ES","ALL_WORDS","PRIMARY_OUTPUT","65059","871332","3589","871332","0","1","871332","195021","12811687","2228819","29161649","379565089291","2228819","379567318110","0.000587","29161649","41973336","69.476605","0.851812232913","0.305233946618","0.999994128001","0.999917308483","0.652614037309","0.627191552465","0.449423738186","0.350172671706","0.289843040458","0.509903921959","0.509876023351","0.000082691517","0.449391616998","0.981405614580","0.852462513401","0.912400934512","0.912400934512"
|
||||||
|
"SNOWBALL_SPANISH_DIRECT","ES_ES","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","64918","869371","3525","869371","0","1","869371","194444","12787018","2201196","29076352","377858468569","2201196","377860669765","0.000583","29076352","41863370","69.455354","0.853138205793","0.305446455935","0.999994174583","0.999917233823","0.652720315259","0.627945981812","0.449838583213","0.350441220963","0.290188220622","0.510478247707","0.510450359519","0.000082766177","0.449806446076","0.981468761133","0.852555702466","0.912481600659","0.912481600659"
|
||||||
|
"SNOWBALL_SPANISH_LUCENE_FILTER","ES_ES","ALL_WORDS","PRIMARY_OUTPUT","65059","871332","3589","871332","0","1","871332","194971","12811693","2230481","29161643","379565087629","2230481","379567318110","0.000588","29161643","41973336","69.476591","0.851718175843","0.305234089566","0.999994123622","0.999917304121","0.652614106594","0.627150877515","0.449410800675","0.350169642836","0.289832278511","0.509875888791","0.509847985527","0.000082695879","0.449378676109","0.981386049614","0.852460744495","0.912391466047","0.912391466047"
|
||||||
|
"SNOWBALL_SPANISH_LUCENE_FILTER","ES_ES","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","64918","869371","3525","869371","0","1","869371","194444","12787018","2201196","29076352","377858468569","2201196","377860669765","0.000583","29076352","41863370","69.455354","0.853138205793","0.305446455935","0.999994174583","0.999917233823","0.652720315259","0.627945981812","0.449838583213","0.350441220963","0.290188220622","0.510478247707","0.510450359519","0.000082766177","0.449806446076","0.981468761133","0.852555702466","0.912481600659","0.912481600659"
|
||||||
|
"SNOWBALL_SWEDISH_DIRECT","SV_SE","ALL_WORDS","PRIMARY_OUTPUT","12371","98108","68","98108","0","1","98108","25915","237017","67105","148325","4812088331","67105","4812155436","0.001394","148325","385342","38.491781","0.779348419384","0.615082186733","0.999986055105","0.999955235704","0.807534120919","0.739831942216","0.687539886056","0.642151948805","0.523855832838","0.692360693585","0.692339154006","0.000044764296","0.687517812951","0.984860422704","0.942685282143","0.963311451566","0.963311451566"
|
||||||
|
"SNOWBALL_SWEDISH_DIRECT","SV_SE","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","12342","97881","68","97881","0","1","97881","25840","236588","67105","147975","4789844472","67105","4789911577","0.001401","147975","384563","38.478741","0.779036724587","0.615212591955","0.999985990347","0.999955100897","0.807599291151","0.739644914918","0.687500000000","0.642223301999","0.523809523810","0.692295603454","0.692273994517","0.000044899103","0.687477858823","0.984821307273","0.942694565973","0.963297587020","0.963297587020"
|
||||||
|
"SNOWBALL_SWEDISH_LUCENE_FILTER","SV_SE","ALL_WORDS","PRIMARY_OUTPUT","12371","98108","68","98108","0","1","98108","26781","230676","64262","154666","4812091174","64262","4812155436","0.001335","154666","385342","40.137333","0.782116919488","0.598626674487","0.999986645901","0.999954508853","0.799306660194","0.736939762085","0.678179573117","0.628097931391","0.513064830384","0.684248529829","0.684226838572","0.000045491147","0.678157227687","0.985207247898","0.939659207875","0.961894327137","0.961894327137"
|
||||||
|
"SNOWBALL_SWEDISH_LUCENE_FILTER","SV_SE","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","12342","97881","68","97881","0","1","97881","26706","230247","64262","154316","4789847315","64262","4789911577","0.001342","154316","384563","40.127625","0.781799537535","0.598723746174","0.999986583886","0.999954370671","0.799355165030","0.736743719918","0.678122496584","0.628142458291","0.512999498691","0.684165146635","0.684143384784","0.000045629329","0.678100081584","0.985169028543","0.939660518299","0.961876797342","0.961876797342"
|
||||||
|
"SNOWBALL_YIDDISH_DIRECT","YI","ALL_WORDS","PRIMARY_OUTPUT","802","3578","0","3578","0","1","3578","1087","4962","1151","1382","6391758","1151","6392909","0.018004","1382","6344","21.784363","0.811712743334","0.782156368222","0.999819956768","0.999604172550","0.890988162495","0.805624107027","0.796660512162","0.787894185271","0.662041360907","0.796797522188","0.796599716782","0.000395827450","0.796462473806","0.982918530193","0.962014249878","0.972354049666","0.972354049666"
|
||||||
|
"SNOWBALL_YIDDISH_DIRECT","YI","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","802","3578","0","3578","0","1","3578","1087","4962","1151","1382","6391758","1151","6392909","0.018004","1382","6344","21.784363","0.811712743334","0.782156368222","0.999819956768","0.999604172550","0.890988162495","0.805624107027","0.796660512162","0.787894185271","0.662041360907","0.796797522188","0.796599716782","0.000395827450","0.796462473806","0.982918530193","0.962014249878","0.972354049666","0.972354049666"
|
||||||
|
"SNOWBALL_YIDDISH_LUCENE_FILTER","YI","ALL_WORDS","PRIMARY_OUTPUT","802","3578","0","3578","0","1","3578","1087","4962","1151","1382","6391758","1151","6392909","0.018004","1382","6344","21.784363","0.811712743334","0.782156368222","0.999819956768","0.999604172550","0.890988162495","0.805624107027","0.796660512162","0.787894185271","0.662041360907","0.796797522188","0.796599716782","0.000395827450","0.796462473806","0.982918530193","0.962014249878","0.972354049666","0.972354049666"
|
||||||
|
"SNOWBALL_YIDDISH_LUCENE_FILTER","YI","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","802","3578","0","3578","0","1","3578","1087","4962","1151","1382","6391758","1151","6392909","0.018004","1382","6344","21.784363","0.811712743334","0.782156368222","0.999819956768","0.999604172550","0.890988162495","0.805624107027","0.796660512162","0.787894185271","0.662041360907","0.796797522188","0.796599716782","0.000395827450","0.796462473806","0.982918530193","0.962014249878","0.972354049666","0.972354049666"
|
||||||
|
"SPANISH_LUCENE_SPANISH_LIGHT_STEM_FILTER","ES_ES","ALL_WORDS","PRIMARY_OUTPUT","65059","871332","3589","871332","0","1","871332","405552","1244317","147956","40729019","379567170154","147956","379567318110","0.000039","40729019","41973336","97.035458","0.893730611741","0.029645415842","0.999999610198","0.999892318297","0.514822513020","0.130863846499","0.057387272020","0.036752000024","0.029541282827","0.162772895888","0.162762055080","0.000107681703","0.057380579619","0.993823553768","0.756690454887","0.859195405761","0.859195405761"
|
||||||
|
"SPANISH_LUCENE_SPANISH_LIGHT_STEM_FILTER","ES_ES","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","64918","869371","3525","869371","0","1","869371","404617","1241848","146613","40621522","377860523152","146613","377860669765","0.000039","40621522","41863370","97.033569","0.894406108634","0.029664310351","0.999999611992","0.999892119974","0.514831961171","0.130949068412","0.057424066047","0.036775459718","0.029560783207","0.162886280533","0.162875426941","0.000107880026","0.057417362063","0.993866319748","0.756725425267","0.859233931168","0.859233931168"
|
||||||
|
"SPANISH_LUCENE_SPANISH_MINIMAL_STEM_FILTER","ES_ES","ALL_WORDS","PRIMARY_OUTPUT","65059","871332","3589","871332","0","1","871332","718633","148463","47859","41824873","379567270251","47859","379567318110","0.000013","41824873","41973336","99.646292","0.756221921130","0.003537078873","0.999999873912","0.999889695187","0.501768476392","0.017360591398","0.007041223811","0.004416184633","0.003533050405","0.051718628951","0.051713939443","0.000110304813","0.007040201537","0.995635307295","0.710609719902","0.829315972283","0.829315972283"
|
||||||
|
"SPANISH_LUCENE_SPANISH_MINIMAL_STEM_FILTER","ES_ES","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","64918","869371","3525","869371","0","1","869371","717093","148226","47148","41715144","377860622617","47148","377860669765","0.000012","41715144","41863370","99.645929","0.758678227400","0.003540708739","0.999999875224","0.999889489251","0.501770291981","0.017379114288","0.007048522419","0.004420728101","0.003536725554","0.051829129163","0.051824445330","0.000110510749","0.007047500484","0.995675746140","0.710626433954","0.829341382913","0.829341382913"
|
||||||
|
"SPANISH_LUCENE_SPANISH_PLURAL_STEM_FILTER","ES_ES","ALL_WORDS","PRIMARY_OUTPUT","65059","871332","3589","871332","0","1","871332","578805","325245","58578","41648091","379567259532","58578","379567318110","0.000015","41648091","41973336","99.225115","0.847382777999","0.007748847983","0.999999845672","0.999890132644","0.503874346827","0.037377069210","0.015357262275","0.009663967067","0.007738048760","0.081032341260","0.081026288458","0.000109867356","0.015355289170","0.995442321769","0.723731297627","0.838115191065","0.838115191065"
|
||||||
|
"SPANISH_LUCENE_SPANISH_PLURAL_STEM_FILTER","ES_ES","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","64918","869371","3525","869371","0","1","869371","577533","324656","57716","41538714","377860612049","57716","377860669765","0.000015","41538714","41863370","99.224487","0.849057985417","0.007755132948","0.999999847256","0.999889928152","0.503877490102","0.037408921072","0.015369880354","0.009671831022","0.007744455857","0.081145286724","0.081139234944","0.000110071848","0.015367905834","0.995484117647","0.723752971494","0.838144538380","0.838144538380"
|
||||||
|
"SPANISH_RADIXOR","ES_ES","ALL_WORDS","PRIMARY_OUTPUT","65059","871332","3589","871332","0","1","871332","64995","41074684","288483","898652","379567029627","288483","379567318110","0.000076","898652","41973336","2.141007","0.993025606574","0.978589931475","0.999999239969","0.999996872745","0.989294585722","0.990104500109","0.985754921826","0.981443392220","0.971909988067","0.985781345071","0.985779787115","0.000003127255","0.985753358111","0.995417814373","0.993266303762","0.994340895233","0.994340895233"
|
||||||
|
"SPANISH_RADIXOR","ES_ES","ALL_WORDS","ANY_CANDIDATE","65059","871332","3589","828695","42637","21","916797","65118","41972710","2","626","379567318108","2","379567318110","0.000000","626","41973336","0.001491","0.999999952350","0.999985085770","0.999999999995","0.999999998346","0.999992542882","0.999996978999","0.999992519005","0.999988059050","0.999985038121","0.999992519032","0.999992518205","0.000000001654","","","","",""
|
||||||
|
"SPANISH_RADIXOR","ES_ES","ALL_WORDS","ALL_CANDIDATES","65059","871332","3589","828695","42637","21","916797","65118","41972710","1349800","626","379565968310","1349800","379567318110","0.000356","626","41973336","0.001491","0.968842987168","0.999985085770","0.999996443846","0.999996442590","0.999990764808","0.974915259157","0.984167740127","0.993597526064","0.968828987818","0.984290880594","0.984289129583","0.000003557410","","","","",""
|
||||||
|
"SPANISH_RADIXOR","ES_ES","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","64918","869371","3525","869371","0","1","869371","64814","40978337","276044","885033","377860393721","276044","377860669765","0.000073","885033","41863370","2.114099","0.993308734895","0.978859012067","0.999999269456","0.999996927576","0.989429140761","0.990384762162","0.986030938205","0.981715226337","0.972446769193","0.986057405488","0.986055874970","0.000003072424","0.986029401906","0.995463637710","0.993323040564","0.994392187139","0.994392187139"
|
||||||
|
"SPANISH_RADIXOR","ES_ES","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","64918","869371","3525","826968","42403","21","914127","64933","41863370","0","0","377860669765","0","377860669765","0.000000","0","41863370","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"SPANISH_RADIXOR","ES_ES","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","64918","869371","3525","826968","42403","21","914127","64933","41863370","1255381","0","377859414384","1255381","377860669765","0.000332","0","41863370","0.000000","0.970885497124","1.000000000000","0.999996677662","0.999996678030","0.999998338831","0.976571978660","0.985227704543","0.994038240493","0.970885497124","0.985335220686","0.985333583876","0.000003321970","","","","",""
|
||||||
|
"SWEDISH_LUCENE_SWEDISH_LIGHT_STEM_FILTER","SV_SE","ALL_WORDS","PRIMARY_OUTPUT","12371","98108","68","98108","0","1","98108","22392","218635","45941","166707","4812109495","45941","4812155436","0.000955","166707","385342","43.262089","0.826359911708","0.567379107390","0.999990453135","0.999955813777","0.783684780262","0.757232036109","0.672807954234","0.605320541501","0.506940918144","0.684733049508","0.684712936280","0.000044186223","0.672786622564","0.986795482859","0.942302523776","0.964035907715","0.964035907715"
|
||||||
|
"SWEDISH_LUCENE_SWEDISH_LIGHT_STEM_FILTER","SV_SE","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","12342","97881","68","97881","0","1","97881","22338","218126","45941","166437","4789865636","45941","4789911577","0.000959","166437","384563","43.279515","0.826025213298","0.567204853301","0.999990408800","0.999955664954","0.783597631051","0.756945124029","0.672574503184","0.605125951621","0.506675896159","0.684489232882","0.684469049184","0.000044335046","0.672553099274","0.986761366955","0.942264620928","0.963999791833","0.963999791833"
|
||||||
|
"SWEDISH_LUCENE_SWEDISH_MINIMAL_STEM_FILTER","SV_SE","ALL_WORDS","PRIMARY_OUTPUT","12371","98108","68","98108","0","1","98108","23360","228181","40227","157161","4812115209","40227","4812155436","0.000836","157161","385342","40.784809","0.850127417961","0.592151906618","0.999991640544","0.999958984659","0.796071773581","0.781991317186","0.698068068834","0.630412272016","0.536178622033","0.709510092538","0.709491456160","0.000041015341","0.698048215965","0.988492665376","0.944581755622","0.966038479572","0.966038479572"
|
||||||
|
"SWEDISH_LUCENE_SWEDISH_MINIMAL_STEM_FILTER","SV_SE","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","12342","97881","68","97881","0","1","97881","23312","227624","40227","156939","4789871350","40227","4789911577","0.000840","156939","384563","40.809698","0.849815755775","0.591903017191","0.999991601724","0.999958840540","0.795947309457","0.781693541131","0.697790053555","0.630152322431","0.535850655618","0.709230928471","0.709212225121","0.000041159460","0.697770131116","0.988462934404","0.944527865197","0.965996098328","0.965996098328"
|
||||||
|
"SWEDISH_RADIXOR","SV_SE","ALL_WORDS","PRIMARY_OUTPUT","12371","98108","68","98108","0","1","98108","12330","365796","24473","19546","4812130963","24473","4812155436","0.000509","19546","385342","5.072377","0.937291970410","0.949276227351","0.999994914337","0.999990853272","0.974635570844","0.939664553041","0.943246034417","0.946854921499","0.892588119029","0.943265066457","0.943260495884","0.000009146728","0.943241460869","0.992630770222","0.993394969179","0.993012722673","0.993012722673"
|
||||||
|
"SWEDISH_RADIXOR","SV_SE","ALL_WORDS","ANY_CANDIDATE","12371","98108","68","92341","5767","5","104148","12371","385342","0","0","4812155436","0","4812155436","0.000000","0","385342","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"SWEDISH_RADIXOR","SV_SE","ALL_WORDS","ALL_CANDIDATES","12371","98108","68","92341","5767","5","104148","12371","385342","47848","0","4812107588","47848","4812155436","0.000994","0","385342","0.000000","0.889545003347","1.000000000000","0.999990056847","0.999990057643","0.999995028423","0.909639856815","0.941544130223","0.975767741439","0.889545003347","0.943156934634","0.943152245645","0.000009942357","","","","",""
|
||||||
|
"SWEDISH_RADIXOR","SV_SE","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","12342","97881","68","97881","0","1","97881","12301","365017","24473","19546","4789887104","24473","4789911577","0.000511","19546","384563","5.082652","0.937166551131","0.949173477428","0.999994890720","0.999990810798","0.974584184074","0.939543572972","0.943131801052","0.946747541943","0.892383555482","0.943150907472","0.943146315681","0.000009189202","0.943127206266","0.992611730682","0.993377890892","0.992994663001","0.992994663001"
|
||||||
|
"SWEDISH_RADIXOR","SV_SE","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","12342","97881","68","92114","5767","5","103921","12342","384563","0","0","4789911577","0","4789911577","0.000000","0","384563","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"SWEDISH_RADIXOR","SV_SE","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","12342","97881","68","92114","5767","5","103921","12342","384563","47848","0","4789863729","47848","4789911577","0.000999","0","384563","0.000000","0.889346015712","1.000000000000","0.999990010672","0.999990011473","0.999995005336","0.909473386475","0.941432652692","0.975719846569","0.889346015712","0.943051438529","0.943046728292","0.000009988527","","","","",""
|
||||||
|
"UKRAINIAN_LUCENE_MORFOLOGIK_FILTER","UK_UA","ALL_WORDS","PRIMARY_OUTPUT","1493","14245","4","14245","0","1","14245","2358","56032","828","9308","101386722","828","101387550","0.000817","9308","65340","14.245485","0.985437917693","0.857545148454","0.999991833317","0.999900091560","0.928768490886","0.956895962839","0.917054009820","0.880397209478","0.846814169992","0.919270093835","0.919222898475","0.000099908440","0.917004266345","0.997989675681","0.970999849348","0.984309781661","0.984309781661"
|
||||||
|
"UKRAINIAN_LUCENE_MORFOLOGIK_FILTER","UK_UA","ALL_WORDS","ANY_CANDIDATE","1493","14245","4","12038","2207","6","16937","2912","60394","122","4946","101387428","122","101387550","0.000120","4946","65340","7.569636","0.997984004230","0.924303642485","0.999998796696","0.999950045780","0.962151219591","0.982322936592","0.959731756929","0.938156308641","0.922581039382","0.960437530635","0.960413432420","0.000049954220","","","","",""
|
||||||
|
"UKRAINIAN_LUCENE_MORFOLOGIK_FILTER","UK_UA","ALL_WORDS","ALL_CANDIDATES","1493","14245","4","12038","2207","6","16937","2912","60394","1368","4946","101386182","1368","101387550","0.001349","4946","65340","7.569636","0.977850458211","0.924303642485","0.999986507219","0.999937764217","0.962145074852","0.966650447520","0.950323362339","0.934538657225","0.905348683816","0.950700131656","0.950669478973","0.000062235783","","","","",""
|
||||||
|
"UKRAINIAN_LUCENE_MORFOLOGIK_FILTER","UK_UA","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","1491","14236","4","14236","0","1","14236","2356","56016","828","9308","101258578","828","101259406","0.000818","9308","65324","14.248974","0.985433818873","0.857510256567","0.999991822982","0.999899965191","0.928751039775","0.956884181756","0.917032283413","0.880367133966","0.846777119361","0.919249480202","0.919202225823","0.000100034809","0.916982477117","0.997988093697","0.970977645923","0.984297603943","0.984297603943"
|
||||||
|
"UKRAINIAN_LUCENE_MORFOLOGIK_FILTER","UK_UA","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","1491","14236","4","12029","2207","6","16928","2910","60378","122","4946","101259284","122","101259406","0.000120","4946","65324","7.571490","0.997983471074","0.924285101953","0.999998795174","0.999949982596","0.962141948563","0.982318335047","0.959721515768","0.938140934008","0.922562112276","0.960427641371","0.960403512884","0.000050017404","","","","",""
|
||||||
|
"UKRAINIAN_LUCENE_MORFOLOGIK_FILTER","UK_UA","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","1491","14236","4","12029","2207","6","16928","2910","60378","1368","4946","101258038","1368","101259406","0.001351","4946","65324","7.571490","0.977844718686","0.924285101953","0.999986490144","0.999937685499","0.962135796049","0.966641904786","0.950310852286","0.934522445998","0.905325976129","0.950687806541","0.950657115187","0.000062314501","","","","",""
|
||||||
|
"UKRAINIAN_MORFOLOGIK_DIRECT","UK_UA","ALL_WORDS","PRIMARY_OUTPUT","1493","14245","4","14245","0","1","14245","2365","56016","828","9324","101386722","828","101387550","0.000817","9324","65340","14.269972","0.985433818873","0.857300275482","0.999991833317","0.999899933851","0.928646054399","0.956831877998","0.916912197996","0.880190066750","0.846572361262","0.919136923635","0.919089660097","0.000100066149","0.916862376978","0.997989675681","0.970875945953","0.984246115917","0.984246115917"
|
||||||
|
"UKRAINIAN_MORFOLOGIK_DIRECT","UK_UA","ALL_WORDS","ANY_CANDIDATE","1493","14245","4","12038","2207","6","16937","2919","60378","122","4962","101387428","122","101387550","0.000120","4962","65340","7.594123","0.997983471074","0.924058769513","0.999998796696","0.999949888071","0.962028783105","0.982267195939","0.959599491418","0.937954390108","0.922336622774","0.960310042786","0.960285871670","0.000050111929","","","","",""
|
||||||
|
"UKRAINIAN_MORFOLOGIK_DIRECT","UK_UA","ALL_WORDS","ALL_CANDIDATES","1493","14245","4","12038","2207","6","16937","2919","60378","1368","4962","101386182","1368","101387550","0.001349","4962","65340","7.594123","0.977844718686","0.924058769513","0.999986507219","0.999937606509","0.962022638366","0.966592384831","0.950191209102","0.934337338211","0.905108832524","0.950571400540","0.950540673331","0.000062393491","","","","",""
|
||||||
|
"UKRAINIAN_MORFOLOGIK_DIRECT","UK_UA","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","1491","14236","4","14236","0","1","14236","2356","56016","828","9308","101258578","828","101259406","0.000818","9308","65324","14.248974","0.985433818873","0.857510256567","0.999991822982","0.999899965191","0.928751039775","0.956884181756","0.917032283413","0.880367133966","0.846777119361","0.919249480202","0.919202225823","0.000100034809","0.916982477117","0.997988093697","0.970977645923","0.984297603943","0.984297603943"
|
||||||
|
"UKRAINIAN_MORFOLOGIK_DIRECT","UK_UA","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","1491","14236","4","12029","2207","6","16928","2910","60378","122","4946","101259284","122","101259406","0.000120","4946","65324","7.571490","0.997983471074","0.924285101953","0.999998795174","0.999949982596","0.962141948563","0.982318335047","0.959721515768","0.938140934008","0.922562112276","0.960427641371","0.960403512884","0.000050017404","","","","",""
|
||||||
|
"UKRAINIAN_MORFOLOGIK_DIRECT","UK_UA","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","1491","14236","4","12029","2207","6","16928","2910","60378","1368","4946","101258038","1368","101259406","0.001351","4946","65324","7.571490","0.977844718686","0.924285101953","0.999986490144","0.999937685499","0.962135796049","0.966641904786","0.950310852286","0.934522445998","0.905325976129","0.950687806541","0.950657115187","0.000062314501","","","","",""
|
||||||
|
"UKRAINIAN_RADIXOR","UK_UA","ALL_WORDS","PRIMARY_OUTPUT","1493","14245","4","14245","0","1","14245","1493","64732","880","608","101386670","880","101387550","0.000868","608","65340","0.930517","0.986587819301","0.990694827058","0.999991320433","0.999985333094","0.995343073746","0.987406494442","0.988637057853","0.989870692292","0.977529447297","0.988639190514","0.988631855097","0.000014666906","0.988629719696","0.997993591453","0.998265712624","0.998129633491","0.998129633491"
|
||||||
|
"UKRAINIAN_RADIXOR","UK_UA","ALL_WORDS","ANY_CANDIDATE","1493","14245","4","14055","190","2","14435","1493","65340","0","0","101387550","0","101387550","0.000000","0","65340","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"UKRAINIAN_RADIXOR","UK_UA","ALL_WORDS","ALL_CANDIDATES","1493","14245","4","14055","190","2","14435","1493","65340","1490","0","101386060","1490","101387550","0.001470","0","65340","0.000000","0.977704623672","1.000000000000","0.999985303916","0.999985313380","0.999992651958","0.982083809295","0.988726639933","0.995459946982","0.977704623672","0.988789473888","0.988782208195","0.000014686620","","","","",""
|
||||||
|
"UKRAINIAN_RADIXOR","UK_UA","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","1491","14236","4","14236","0","1","14236","1491","64716","880","608","101258526","880","101259406","0.000869","608","65324","0.930745","0.986584547838","0.990692547915","0.999991309449","0.999985314543","0.995341928682","0.987403420118","0.988634280477","0.989868213355","0.977524016676","0.988636414174","0.988629069474","0.000014685457","0.988626933033","0.997992012551","0.998264347489","0.998128161444","0.998128161444"
|
||||||
|
"UKRAINIAN_RADIXOR","UK_UA","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","1491","14236","4","14046","190","2","14426","1491","65324","0","0","101259406","0","101259406","0.000000","0","65324","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"UKRAINIAN_RADIXOR","UK_UA","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","1491","14236","4","14046","190","2","14426","1491","65324","1490","0","101257916","1490","101259406","0.001471","0","65324","0.000000","0.977699284581","1.000000000000","0.999985285318","0.999985294804","0.999992642659","0.982079499669","0.988723909852","0.995458840023","0.977699284581","0.988786774073","0.988779499204","0.000014705196","","","","",""
|
||||||
|
"YI_RADIXOR","YI","ALL_WORDS","PRIMARY_OUTPUT","802","3578","0","3578","0","1","3578","802","6195","195","149","6392714","195","6392909","0.003050","149","6344","2.348676","0.969483568075","0.976513240858","0.999969497454","0.999946243726","0.988241369156","0.970881394183","0.972985707555","0.975099162627","0.947392567671","0.972992055990","0.972965163911","0.000053756274","0.972958803000","0.995691103897","0.996142223728","0.995916612726","0.995916612726"
|
||||||
|
"YI_RADIXOR","YI","ALL_WORDS","ANY_CANDIDATE","802","3578","0","3489","89","3","3676","802","6344","0","0","6392909","0","6392909","0.000000","0","6344","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"YI_RADIXOR","YI","ALL_WORDS","ALL_CANDIDATES","802","3578","0","3489","89","3","3676","802","6344","389","0","6392520","389","6392909","0.006085","0","6344","0.000000","0.942224862617","1.000000000000","0.999939151332","0.999939211655","0.999969575666","0.953239572064","0.970253116158","0.987885016662","0.942224862617","0.970682678643","0.970653145819","0.000060788345","","","","",""
|
||||||
|
"YI_RADIXOR","YI","LOWERCASE_GROUPS_ONLY","PRIMARY_OUTPUT","802","3578","0","3578","0","1","3578","802","6195","195","149","6392714","195","6392909","0.003050","149","6344","2.348676","0.969483568075","0.976513240858","0.999969497454","0.999946243726","0.988241369156","0.970881394183","0.972985707555","0.975099162627","0.947392567671","0.972992055990","0.972965163911","0.000053756274","0.972958803000","0.995691103897","0.996142223728","0.995916612726","0.995916612726"
|
||||||
|
"YI_RADIXOR","YI","LOWERCASE_GROUPS_ONLY","ANY_CANDIDATE","802","3578","0","3489","89","3","3676","802","6344","0","0","6392909","0","6392909","0.000000","0","6344","0.000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","1.000000000000","0.000000000000","","","","",""
|
||||||
|
"YI_RADIXOR","YI","LOWERCASE_GROUPS_ONLY","ALL_CANDIDATES","802","3578","0","3489","89","3","3676","802","6344","389","0","6392520","389","6392909","0.006085","0","6344","0.000000","0.942224862617","1.000000000000","0.999939151332","0.999939211655","0.999969575666","0.953239572064","0.970253116158","0.987885016662","0.942224862617","0.970682678643","0.970653145819","0.000060788345","","","","",""
|
||||||
|
1
docs/benchmarks/data/stemming-quality.sha256
Normal file
1
docs/benchmarks/data/stemming-quality.sha256
Normal file
@@ -0,0 +1 @@
|
|||||||
|
5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28 stemming-quality.csv
|
||||||
@@ -6,6 +6,8 @@ two layers:
|
|||||||
- **benchmark reference pages**, which explain methodology, corpora, environment, candidate
|
- **benchmark reference pages**, which explain methodology, corpora, environment, candidate
|
||||||
selection, and the English dictionary coverage experiment;
|
selection, and the English dictionary coverage experiment;
|
||||||
- **language result pages**, which contain the actual same-language accuracy and throughput tables.
|
- **language result pages**, which contain the actual same-language accuracy and throughput tables.
|
||||||
|
- **pairwise quality pages and generated sections**, which publish over-stemming, under-stemming,
|
||||||
|
candidate-policy, classification, and partition measurements from one checked result snapshot.
|
||||||
|
|
||||||
This structure keeps methodology separate from per-language result pages, while preserving all
|
This structure keeps methodology separate from per-language result pages, while preserving all
|
||||||
measured data and the command-class analysis for each Radixor language resource.
|
measured data and the command-class analysis for each Radixor language resource.
|
||||||
@@ -26,6 +28,9 @@ the preferred result measured by the accuracy pass.
|
|||||||
| Page | Purpose |
|
| Page | Purpose |
|
||||||
| --- | --- |
|
| --- | --- |
|
||||||
| [Methodology](reference/methodology.md) | Workload design, normalization, speed metrics, quality metrics, and interpretation rules. |
|
| [Methodology](reference/methodology.md) | Workload design, normalization, speed metrics, quality metrics, and interpretation rules. |
|
||||||
|
| [Linguistic quality methodology](reference/linguistic-quality.md) | Gold-standard groups, output policies, pairwise formulas, ranking rules, aggregation, and limitations. |
|
||||||
|
| [Tested stemmers](reference/tested-stemmers.md) | Versions, upstream attribution, evaluated coverage, adapters, preprocessing, and output capability. |
|
||||||
|
| [Reproducibility and raw data](reference/reproducibility.md) | Commands, versioned CSV snapshot, checksum, generated artifacts, and unavailable provenance. |
|
||||||
| [Corpora](reference/corpora.md) | Dictionary row counts, complete quality tokens, already-root tokens, changed speed tokens, and timing token counts. |
|
| [Corpora](reference/corpora.md) | Dictionary row counts, complete quality tokens, already-root tokens, changed speed tokens, and timing token counts. |
|
||||||
| [Environment and reports](reference/environment.md) | Hardware, JVM, JMH settings, report files, and badge/report policy. |
|
| [Environment and reports](reference/environment.md) | Hardware, JVM, JMH settings, report files, and badge/report policy. |
|
||||||
| [English dictionary coverage](reference/english-coverage.md) | Quality/speed operating curve for contracted Radixor tries built from 100% down to 10% of English dictionary rows. |
|
| [English dictionary coverage](reference/english-coverage.md) | Quality/speed operating curve for contracted Radixor tries built from 100% down to 10% of English dictionary rows. |
|
||||||
@@ -47,9 +52,276 @@ Open [Language Benchmark Pages](languages/index.md) for the complete language li
|
|||||||
|
|
||||||
The English dictionary coverage benchmark shows the current contracted-trie operating curve. With
|
The English dictionary coverage benchmark shows the current contracted-trie operating curve. With
|
||||||
the full English dictionary, Radixor reaches `97.478%` all-token exactness and `97.197%`
|
the full English dictionary, Radixor reaches `97.478%` all-token exactness and `97.197%`
|
||||||
changed-token exactness at `109.8 ns/token`. Even with a deterministic 10% dictionary slice, it
|
changed-token exactness at `135.8 ns/token`. Even with a deterministic 10% dictionary slice, it
|
||||||
keeps `92.868%` all-token exactness and `76.516%` changed-token exactness at `90.9 ns/token`.
|
keeps `92.868%` all-token exactness and `76.516%` changed-token exactness at `86.0 ns/token`.
|
||||||
|
|
||||||
Those figures should not be reduced to a single speed badge. The professional interpretation is a
|
Those figures should not be reduced to a single speed badge. The professional interpretation is a
|
||||||
quality/speed envelope: the amount and quality of dictionary knowledge affect stemming precision,
|
quality/speed envelope: the amount and quality of dictionary knowledge affect stemming precision,
|
||||||
while contracted tries reduce lookup cost in uniform regions of the compiled graph.
|
while contracted tries reduce lookup cost in uniform regions of the compiled graph.
|
||||||
|
|
||||||
|
## Quality versus performance
|
||||||
|
|
||||||
|
Each language page keeps exact-root accuracy, JMH latency, and pairwise linguistic-quality results in separate tables. No undocumented scalar combines them. The current repository checkout does not contain the dated machine-readable JMH CSV files named by the performance provenance page, so this revision preserves the existing performance tables but does not regenerate a cross-language Pareto frontier from rounded Markdown values. A defensible Pareto analysis requires the original unrounded JMH snapshot on the same hardware and JVM. Readers can still inspect the quality and speed dimensions side by side on every language page.
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY-OVERVIEW:START -->
|
||||||
|
|
||||||
|
## Pairwise Quality Findings
|
||||||
|
|
||||||
|
The validated snapshot is a broad multilingual comparison covering the complete 20-language Radixor dictionary universe; 19 languages have existing benchmark pages. The direct ranking below uses only deterministic `PRIMARY_OUTPUT` rows over identical per-language inputs. Candidate-aware rows are intentionally excluded from this claim.
|
||||||
|
|
||||||
|
!!! success "Evidence-based primary-output result"
|
||||||
|
Radixor achieved the highest balanced accuracy among the evaluated deterministic stemmers for every documented language in both `ALL_WORDS` and `LOWERCASE_GROUPS_ONLY`: **38 wins in 38 language-mode comparisons, with no exact first-place ties**. This statement is limited to the evaluated implementations, versions, dictionaries, adapters, and balanced-accuracy metric; it is not a universal claim about every stemming use case.
|
||||||
|
|
||||||
|
### Per-language winner matrix
|
||||||
|
|
||||||
|
| Language | Dictionary mode | Winner | Balanced accuracy | Runner-up | Difference | Exact tie | Deterministic stemmers |
|
||||||
|
|---|---|---|---:|---|---:|---|---:|
|
||||||
|
|Czech (`CS_CZ`)|ALL_WORDS|Radixor|0.996565|HUNSPELL CZECH LUCENE FILTER|0.142812638|no|3|
|
||||||
|
|Czech (`CS_CZ`)|LOWERCASE_GROUPS_ONLY|Radixor|0.997139|HUNSPELL CZECH LUCENE FILTER|0.144369049|no|3|
|
||||||
|
|Danish (`DA_DK`)|ALL_WORDS|Radixor|0.996066|SNOWBALL DANISH LUCENE FILTER|0.058096771|no|3|
|
||||||
|
|Danish (`DA_DK`)|LOWERCASE_GROUPS_ONLY|Radixor|0.996305|SNOWBALL DANISH DIRECT|0.058230346|no|3|
|
||||||
|
|Dutch (`NL_NL`)|ALL_WORDS|Radixor|0.988661|SNOWBALL DUTCH DIRECT|0.261574077|no|4|
|
||||||
|
|Dutch (`NL_NL`)|LOWERCASE_GROUPS_ONLY|Radixor|0.989040|SNOWBALL DUTCH DIRECT|0.258544404|no|4|
|
||||||
|
|English (`US_UK`)|ALL_WORDS|Radixor|0.965159|ENGLISH LUCENE PORTER COPIED|0.010532535|no|11|
|
||||||
|
|English (`US_UK`)|LOWERCASE_GROUPS_ONLY|Radixor|0.965820|ENGLISH LUCENE PORTER COPIED|0.010920064|no|11|
|
||||||
|
|Finnish (`FI_FI`)|ALL_WORDS|Radixor|0.984594|SNOWBALL FINNISH LUCENE FILTER|0.244241861|no|4|
|
||||||
|
|Finnish (`FI_FI`)|LOWERCASE_GROUPS_ONLY|Radixor|0.988068|SNOWBALL FINNISH DIRECT|0.249668284|no|4|
|
||||||
|
|French (`FR_FR`)|ALL_WORDS|Radixor|0.956992|SNOWBALL FRENCH DIRECT|0.111730673|no|6|
|
||||||
|
|French (`FR_FR`)|LOWERCASE_GROUPS_ONLY|Radixor|0.957224|SNOWBALL FRENCH DIRECT|0.111809799|no|6|
|
||||||
|
|German (`DE_DE`)|ALL_WORDS|Radixor|0.907901|GERMAN CISTEM|0.027131083|no|8|
|
||||||
|
|German (`DE_DE`)|LOWERCASE_GROUPS_ONLY|Radixor|0.966157|GERMAN CISTEM|0.050868631|no|8|
|
||||||
|
|Hungarian (`HU_HU`)|ALL_WORDS|Radixor|0.995491|SNOWBALL HUNGARIAN LUCENE FILTER|0.172884951|no|4|
|
||||||
|
|Hungarian (`HU_HU`)|LOWERCASE_GROUPS_ONLY|Radixor|0.996163|SNOWBALL HUNGARIAN DIRECT|0.174455479|no|4|
|
||||||
|
|Italian (`IT_IT`)|ALL_WORDS|Radixor|0.996507|SNOWBALL ITALIAN DIRECT|0.130318040|no|4|
|
||||||
|
|Italian (`IT_IT`)|LOWERCASE_GROUPS_ONLY|Radixor|0.996512|SNOWBALL ITALIAN DIRECT|0.130307087|no|4|
|
||||||
|
|Norwegian Bokmal (`NB_NO`)|ALL_WORDS|Radixor|0.974783|SNOWBALL NORWEGIAN BOKMAL DIRECT|0.099819340|no|5|
|
||||||
|
|Norwegian Bokmal (`NB_NO`)|LOWERCASE_GROUPS_ONLY|Radixor|0.975000|SNOWBALL NORWEGIAN BOKMAL DIRECT|0.100008544|no|5|
|
||||||
|
|Norwegian Nynorsk (`NN_NO`)|ALL_WORDS|Radixor|0.935777|SNOWBALL NORWEGIAN NYNORSK DIRECT|0.076868986|no|3|
|
||||||
|
|Norwegian Nynorsk (`NN_NO`)|LOWERCASE_GROUPS_ONLY|Radixor|0.935853|SNOWBALL NORWEGIAN NYNORSK DIRECT|0.076816096|no|3|
|
||||||
|
|Persian (`FA_IR`)|ALL_WORDS|Radixor|0.974922|PERSIAN LUCENE PERSIAN STEM FILTER|0.472751327|no|2|
|
||||||
|
|Persian (`FA_IR`)|LOWERCASE_GROUPS_ONLY|Radixor|0.974922|PERSIAN LUCENE PERSIAN STEM FILTER|0.472751327|no|2|
|
||||||
|
|Polish (`PL_PL`)|ALL_WORDS|Radixor|0.990388|POLISH LUCENE MORFOLOGIK FILTER|0.042233990|no|5|
|
||||||
|
|Polish (`PL_PL`)|LOWERCASE_GROUPS_ONLY|Radixor|0.990579|POLISH LUCENE MORFOLOGIK FILTER|0.042401633|no|5|
|
||||||
|
|Portuguese (`PT_PT`)|ALL_WORDS|Radixor|0.998502|SNOWBALL PORTUGUESE DIRECT|0.059701750|no|6|
|
||||||
|
|Portuguese (`PT_PT`)|LOWERCASE_GROUPS_ONLY|Radixor|0.998502|SNOWBALL PORTUGUESE DIRECT|0.059701750|no|6|
|
||||||
|
|Russian (`RU_RU`)|ALL_WORDS|Radixor|0.989827|SNOWBALL RUSSIAN LUCENE FILTER|0.154951419|no|4|
|
||||||
|
|Russian (`RU_RU`)|LOWERCASE_GROUPS_ONLY|Radixor|0.989852|SNOWBALL RUSSIAN DIRECT|0.154997931|no|4|
|
||||||
|
|Spanish (`ES_ES`)|ALL_WORDS|Radixor|0.989295|SNOWBALL SPANISH LUCENE FILTER|0.336680479|no|7|
|
||||||
|
|Spanish (`ES_ES`)|LOWERCASE_GROUPS_ONLY|Radixor|0.989429|SNOWBALL SPANISH DIRECT|0.336708826|no|7|
|
||||||
|
|Swedish (`SV_SE`)|ALL_WORDS|Radixor|0.974636|SNOWBALL SWEDISH DIRECT|0.167101450|no|5|
|
||||||
|
|Swedish (`SV_SE`)|LOWERCASE_GROUPS_ONLY|Radixor|0.974584|SNOWBALL SWEDISH DIRECT|0.166984893|no|5|
|
||||||
|
|Ukrainian (`UK_UA`)|ALL_WORDS|Radixor|0.995343|UKRAINIAN LUCENE MORFOLOGIK FILTER|0.066574583|no|4|
|
||||||
|
|Ukrainian (`UK_UA`)|LOWERCASE_GROUPS_ONLY|Radixor|0.995342|UKRAINIAN LUCENE MORFOLOGIK FILTER|0.066590889|no|4|
|
||||||
|
|Yiddish (`YI`)|ALL_WORDS|Radixor|0.988241|SNOWBALL YIDDISH DIRECT|0.097253207|no|3|
|
||||||
|
|Yiddish (`YI`)|LOWERCASE_GROUPS_ONLY|Radixor|0.988241|SNOWBALL YIDDISH DIRECT|0.097253207|no|3|
|
||||||
|
|
||||||
|
### Secondary-metric trade-offs
|
||||||
|
|
||||||
|
Balanced-accuracy leadership does not imply leadership on every error trade-off. The table below lists all **15** deterministic primary-output language-mode-metric cases where a non-Radixor adapter has the best displayed value. Equal values are resolved by the authoritative row ordering and should be read as ties when the unrounded values are equal. Throughput leadership remains in the separate performance tables.
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Non-Radixor secondary-metric leaders</summary>
|
||||||
|
|
||||||
|
| Language | Dictionary mode | Metric | Leader | Value |
|
||||||
|
|---|---|---|---|---:|
|
||||||
|
|English|ALL_WORDS|Over-stemming percentage|ENGLISH LUCENE POSSESSIVE FILTER|0.000604|
|
||||||
|
|English|LOWERCASE_GROUPS_ONLY|Over-stemming percentage|ENGLISH LUCENE POSSESSIVE FILTER|0.000653|
|
||||||
|
|French|ALL_WORDS|Over-stemming percentage|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|0.000177|
|
||||||
|
|French|LOWERCASE_GROUPS_ONLY|Over-stemming percentage|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|0.000166|
|
||||||
|
|German|LOWERCASE_GROUPS_ONLY|Over-stemming percentage|GERMAN LUCENE GERMAN MINIMAL STEM FILTER|0.000188|
|
||||||
|
|Italian|ALL_WORDS|Over-stemming percentage|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|0.000020|
|
||||||
|
|Italian|LOWERCASE_GROUPS_ONLY|Over-stemming percentage|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|0.000020|
|
||||||
|
|Persian|ALL_WORDS|Over-stemming percentage|PERSIAN LUCENE PERSIAN STEM FILTER|0.002652|
|
||||||
|
|Persian|LOWERCASE_GROUPS_ONLY|Over-stemming percentage|PERSIAN LUCENE PERSIAN STEM FILTER|0.002652|
|
||||||
|
|Portuguese|ALL_WORDS|Over-stemming percentage|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|0.000003|
|
||||||
|
|Portuguese|LOWERCASE_GROUPS_ONLY|Over-stemming percentage|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|0.000003|
|
||||||
|
|Spanish|ALL_WORDS|Over-stemming percentage|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|0.000013|
|
||||||
|
|Spanish|LOWERCASE_GROUPS_ONLY|Over-stemming percentage|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|0.000012|
|
||||||
|
|Ukrainian|ALL_WORDS|Over-stemming percentage|HUNSPELL UKRAINIAN LUCENE FILTER|0.000783|
|
||||||
|
|Ukrainian|LOWERCASE_GROUPS_ONLY|Over-stemming percentage|HUNSPELL UKRAINIAN LUCENE FILTER|0.000784|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
### Win, tie, and placement summary
|
||||||
|
|
||||||
|
Counts use `PRIMARY_OUTPUT` only and retain each adapter configuration as a separate stemmer except that language-specific Radixor identifiers are combined as Radixor. Coverage is displayed explicitly; unsupported languages are absent, not losses.
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>ALL_WORDS placements</summary>
|
||||||
|
|
||||||
|
| Stemmer | Evaluated languages | Wins | Exact first-place ties | Top-three placements | Average rank | Median rank |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|19|19|0|19|1.000|1.000|
|
||||||
|
|CZECH LUCENE CZECH STEM FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|ENGLISH LUCENE KSTEM FILTER|1|0|0|0|8.000|8.000|
|
||||||
|
|ENGLISH LUCENE MINIMAL FILTER|1|0|0|0|9.000|9.000|
|
||||||
|
|ENGLISH LUCENE PORTER COPIED|1|0|0|1|2.000|2.000|
|
||||||
|
|ENGLISH LUCENE PORTER FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|ENGLISH LUCENE POSSESSIVE FILTER|1|0|0|0|11.000|11.000|
|
||||||
|
|ENGLISH OPENNLP PORTER|1|0|0|0|4.000|4.000|
|
||||||
|
|ENGLISH PAICE HUSK LANCASTER|1|0|0|0|7.000|7.000|
|
||||||
|
|ENGLISH SNOWBALL ORIGINAL PORTER|1|0|0|0|6.000|6.000|
|
||||||
|
|ENGLISH SNOWBALL PORTER2|1|0|0|0|5.000|5.000|
|
||||||
|
|FINNISH LUCENE FINNISH LIGHT STEM FILTER|1|0|0|0|4.000|4.000|
|
||||||
|
|FRENCH LUCENE FRENCH LIGHT STEM FILTER|1|0|0|0|5.000|5.000|
|
||||||
|
|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|1|0|0|0|6.000|6.000|
|
||||||
|
|GERMAN CISTEM|1|0|0|1|2.000|2.000|
|
||||||
|
|GERMAN LUCENE GERMAN LIGHT STEM FILTER|1|0|0|0|5.000|5.000|
|
||||||
|
|GERMAN LUCENE GERMAN MINIMAL STEM FILTER|1|0|0|0|8.000|8.000|
|
||||||
|
|GERMAN LUCENE GERMAN STEM FILTER|1|0|0|0|6.000|6.000|
|
||||||
|
|HUNGARIAN LUCENE HUNGARIAN LIGHT STEM FILTER|1|0|0|0|4.000|4.000|
|
||||||
|
|HUNSPELL CZECH LUCENE FILTER|1|0|0|1|2.000|2.000|
|
||||||
|
|HUNSPELL DUTCH LUCENE FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|HUNSPELL ENGLISH LUCENE FILTER|1|0|0|0|10.000|10.000|
|
||||||
|
|HUNSPELL FRENCH LUCENE FILTER|1|0|0|0|4.000|4.000|
|
||||||
|
|HUNSPELL GERMAN LUCENE FILTER|1|0|0|0|7.000|7.000|
|
||||||
|
|HUNSPELL POLISH LUCENE FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|HUNSPELL SPANISH LUCENE FILTER|1|0|0|0|4.000|4.000|
|
||||||
|
|HUNSPELL UKRAINIAN LUCENE FILTER|1|0|0|0|4.000|4.000|
|
||||||
|
|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|1|0|0|0|4.000|4.000|
|
||||||
|
|NORWEGIAN BOKMAL LUCENE NORWEGIAN LIGHT STEM FILTER|1|0|0|0|4.000|4.000|
|
||||||
|
|NORWEGIAN BOKMAL LUCENE NORWEGIAN MINIMAL STEM FILTER|1|0|0|0|5.000|5.000|
|
||||||
|
|PERSIAN LUCENE PERSIAN STEM FILTER|1|0|0|1|2.000|2.000|
|
||||||
|
|POLISH LUCENE MORFOLOGIK FILTER|1|0|0|1|2.000|2.000|
|
||||||
|
|POLISH LUCENE STEMPEL DIRECT|1|0|0|0|4.000|4.000|
|
||||||
|
|POLISH LUCENE STEMPEL FILTER|1|0|0|0|5.000|5.000|
|
||||||
|
|PORTUGUESE LUCENE PORTUGUESE LIGHT STEM FILTER|1|0|0|0|5.000|5.000|
|
||||||
|
|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|1|0|0|0|6.000|6.000|
|
||||||
|
|PORTUGUESE LUCENE PORTUGUESE STEM FILTER|1|0|0|0|4.000|4.000|
|
||||||
|
|RUSSIAN LUCENE RUSSIAN LIGHT STEM FILTER|1|0|0|0|4.000|4.000|
|
||||||
|
|SNOWBALL DANISH DIRECT|1|0|0|1|3.000|3.000|
|
||||||
|
|SNOWBALL DANISH LUCENE FILTER|1|0|0|1|2.000|2.000|
|
||||||
|
|SNOWBALL DUTCH DIRECT|1|0|0|1|2.000|2.000|
|
||||||
|
|SNOWBALL DUTCH LUCENE FILTER|1|0|0|0|4.000|4.000|
|
||||||
|
|SNOWBALL FINNISH DIRECT|1|0|0|1|3.000|3.000|
|
||||||
|
|SNOWBALL FINNISH LUCENE FILTER|1|0|0|1|2.000|2.000|
|
||||||
|
|SNOWBALL FRENCH DIRECT|1|0|0|1|2.000|2.000|
|
||||||
|
|SNOWBALL FRENCH LUCENE FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|SNOWBALL GERMAN DIRECT|1|0|0|1|3.000|3.000|
|
||||||
|
|SNOWBALL GERMAN LUCENE FILTER|1|0|0|0|4.000|4.000|
|
||||||
|
|SNOWBALL HUNGARIAN DIRECT|1|0|0|1|3.000|3.000|
|
||||||
|
|SNOWBALL HUNGARIAN LUCENE FILTER|1|0|0|1|2.000|2.000|
|
||||||
|
|SNOWBALL ITALIAN DIRECT|1|0|0|1|2.000|2.000|
|
||||||
|
|SNOWBALL ITALIAN LUCENE FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|SNOWBALL NORWEGIAN BOKMAL DIRECT|1|0|0|1|2.000|2.000|
|
||||||
|
|SNOWBALL NORWEGIAN BOKMAL LUCENE FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|SNOWBALL NORWEGIAN NYNORSK DIRECT|1|0|0|1|2.000|2.000|
|
||||||
|
|SNOWBALL NORWEGIAN NYNORSK LUCENE FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|SNOWBALL PORTUGUESE DIRECT|1|0|0|1|2.000|2.000|
|
||||||
|
|SNOWBALL PORTUGUESE LUCENE FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|SNOWBALL RUSSIAN DIRECT|1|0|0|1|3.000|3.000|
|
||||||
|
|SNOWBALL RUSSIAN LUCENE FILTER|1|0|0|1|2.000|2.000|
|
||||||
|
|SNOWBALL SPANISH DIRECT|1|0|0|1|3.000|3.000|
|
||||||
|
|SNOWBALL SPANISH LUCENE FILTER|1|0|0|1|2.000|2.000|
|
||||||
|
|SNOWBALL SWEDISH DIRECT|1|0|0|1|2.000|2.000|
|
||||||
|
|SNOWBALL SWEDISH LUCENE FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|SNOWBALL YIDDISH DIRECT|1|0|0|1|2.000|2.000|
|
||||||
|
|SNOWBALL YIDDISH LUCENE FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|SPANISH LUCENE SPANISH LIGHT STEM FILTER|1|0|0|0|5.000|5.000|
|
||||||
|
|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|1|0|0|0|7.000|7.000|
|
||||||
|
|SPANISH LUCENE SPANISH PLURAL STEM FILTER|1|0|0|0|6.000|6.000|
|
||||||
|
|SWEDISH LUCENE SWEDISH LIGHT STEM FILTER|1|0|0|0|5.000|5.000|
|
||||||
|
|SWEDISH LUCENE SWEDISH MINIMAL STEM FILTER|1|0|0|0|4.000|4.000|
|
||||||
|
|UKRAINIAN LUCENE MORFOLOGIK FILTER|1|0|0|1|2.000|2.000|
|
||||||
|
|UKRAINIAN MORFOLOGIK DIRECT|1|0|0|1|3.000|3.000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>LOWERCASE_GROUPS_ONLY placements</summary>
|
||||||
|
|
||||||
|
| Stemmer | Evaluated languages | Wins | Exact first-place ties | Top-three placements | Average rank | Median rank |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|19|19|0|19|1.000|1.000|
|
||||||
|
|CZECH LUCENE CZECH STEM FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|ENGLISH LUCENE KSTEM FILTER|1|0|0|0|8.000|8.000|
|
||||||
|
|ENGLISH LUCENE MINIMAL FILTER|1|0|0|0|9.000|9.000|
|
||||||
|
|ENGLISH LUCENE PORTER COPIED|1|0|0|1|2.000|2.000|
|
||||||
|
|ENGLISH LUCENE PORTER FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|ENGLISH LUCENE POSSESSIVE FILTER|1|0|0|0|11.000|11.000|
|
||||||
|
|ENGLISH OPENNLP PORTER|1|0|0|0|4.000|4.000|
|
||||||
|
|ENGLISH PAICE HUSK LANCASTER|1|0|0|0|7.000|7.000|
|
||||||
|
|ENGLISH SNOWBALL ORIGINAL PORTER|1|0|0|0|6.000|6.000|
|
||||||
|
|ENGLISH SNOWBALL PORTER2|1|0|0|0|5.000|5.000|
|
||||||
|
|FINNISH LUCENE FINNISH LIGHT STEM FILTER|1|0|0|0|4.000|4.000|
|
||||||
|
|FRENCH LUCENE FRENCH LIGHT STEM FILTER|1|0|0|0|5.000|5.000|
|
||||||
|
|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|1|0|0|0|6.000|6.000|
|
||||||
|
|GERMAN CISTEM|1|0|0|1|2.000|2.000|
|
||||||
|
|GERMAN LUCENE GERMAN LIGHT STEM FILTER|1|0|0|0|5.000|5.000|
|
||||||
|
|GERMAN LUCENE GERMAN MINIMAL STEM FILTER|1|0|0|0|8.000|8.000|
|
||||||
|
|GERMAN LUCENE GERMAN STEM FILTER|1|0|0|0|6.000|6.000|
|
||||||
|
|HUNGARIAN LUCENE HUNGARIAN LIGHT STEM FILTER|1|0|0|0|4.000|4.000|
|
||||||
|
|HUNSPELL CZECH LUCENE FILTER|1|0|0|1|2.000|2.000|
|
||||||
|
|HUNSPELL DUTCH LUCENE FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|HUNSPELL ENGLISH LUCENE FILTER|1|0|0|0|10.000|10.000|
|
||||||
|
|HUNSPELL FRENCH LUCENE FILTER|1|0|0|0|4.000|4.000|
|
||||||
|
|HUNSPELL GERMAN LUCENE FILTER|1|0|0|0|7.000|7.000|
|
||||||
|
|HUNSPELL POLISH LUCENE FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|HUNSPELL SPANISH LUCENE FILTER|1|0|0|0|4.000|4.000|
|
||||||
|
|HUNSPELL UKRAINIAN LUCENE FILTER|1|0|0|0|4.000|4.000|
|
||||||
|
|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|1|0|0|0|4.000|4.000|
|
||||||
|
|NORWEGIAN BOKMAL LUCENE NORWEGIAN LIGHT STEM FILTER|1|0|0|0|4.000|4.000|
|
||||||
|
|NORWEGIAN BOKMAL LUCENE NORWEGIAN MINIMAL STEM FILTER|1|0|0|0|5.000|5.000|
|
||||||
|
|PERSIAN LUCENE PERSIAN STEM FILTER|1|0|0|1|2.000|2.000|
|
||||||
|
|POLISH LUCENE MORFOLOGIK FILTER|1|0|0|1|2.000|2.000|
|
||||||
|
|POLISH LUCENE STEMPEL DIRECT|1|0|0|0|4.000|4.000|
|
||||||
|
|POLISH LUCENE STEMPEL FILTER|1|0|0|0|5.000|5.000|
|
||||||
|
|PORTUGUESE LUCENE PORTUGUESE LIGHT STEM FILTER|1|0|0|0|5.000|5.000|
|
||||||
|
|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|1|0|0|0|6.000|6.000|
|
||||||
|
|PORTUGUESE LUCENE PORTUGUESE STEM FILTER|1|0|0|0|4.000|4.000|
|
||||||
|
|RUSSIAN LUCENE RUSSIAN LIGHT STEM FILTER|1|0|0|0|4.000|4.000|
|
||||||
|
|SNOWBALL DANISH DIRECT|1|0|0|1|2.000|2.000|
|
||||||
|
|SNOWBALL DANISH LUCENE FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|SNOWBALL DUTCH DIRECT|1|0|0|1|2.000|2.000|
|
||||||
|
|SNOWBALL DUTCH LUCENE FILTER|1|0|0|0|4.000|4.000|
|
||||||
|
|SNOWBALL FINNISH DIRECT|1|0|0|1|2.000|2.000|
|
||||||
|
|SNOWBALL FINNISH LUCENE FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|SNOWBALL FRENCH DIRECT|1|0|0|1|2.000|2.000|
|
||||||
|
|SNOWBALL FRENCH LUCENE FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|SNOWBALL GERMAN DIRECT|1|0|0|1|3.000|3.000|
|
||||||
|
|SNOWBALL GERMAN LUCENE FILTER|1|0|0|0|4.000|4.000|
|
||||||
|
|SNOWBALL HUNGARIAN DIRECT|1|0|0|1|2.000|2.000|
|
||||||
|
|SNOWBALL HUNGARIAN LUCENE FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|SNOWBALL ITALIAN DIRECT|1|0|0|1|2.000|2.000|
|
||||||
|
|SNOWBALL ITALIAN LUCENE FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|SNOWBALL NORWEGIAN BOKMAL DIRECT|1|0|0|1|2.000|2.000|
|
||||||
|
|SNOWBALL NORWEGIAN BOKMAL LUCENE FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|SNOWBALL NORWEGIAN NYNORSK DIRECT|1|0|0|1|2.000|2.000|
|
||||||
|
|SNOWBALL NORWEGIAN NYNORSK LUCENE FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|SNOWBALL PORTUGUESE DIRECT|1|0|0|1|2.000|2.000|
|
||||||
|
|SNOWBALL PORTUGUESE LUCENE FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|SNOWBALL RUSSIAN DIRECT|1|0|0|1|2.000|2.000|
|
||||||
|
|SNOWBALL RUSSIAN LUCENE FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|SNOWBALL SPANISH DIRECT|1|0|0|1|2.000|2.000|
|
||||||
|
|SNOWBALL SPANISH LUCENE FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|SNOWBALL SWEDISH DIRECT|1|0|0|1|2.000|2.000|
|
||||||
|
|SNOWBALL SWEDISH LUCENE FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|SNOWBALL YIDDISH DIRECT|1|0|0|1|2.000|2.000|
|
||||||
|
|SNOWBALL YIDDISH LUCENE FILTER|1|0|0|1|3.000|3.000|
|
||||||
|
|SPANISH LUCENE SPANISH LIGHT STEM FILTER|1|0|0|0|5.000|5.000|
|
||||||
|
|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|1|0|0|0|7.000|7.000|
|
||||||
|
|SPANISH LUCENE SPANISH PLURAL STEM FILTER|1|0|0|0|6.000|6.000|
|
||||||
|
|SWEDISH LUCENE SWEDISH LIGHT STEM FILTER|1|0|0|0|5.000|5.000|
|
||||||
|
|SWEDISH LUCENE SWEDISH MINIMAL STEM FILTER|1|0|0|0|4.000|4.000|
|
||||||
|
|UKRAINIAN LUCENE MORFOLOGIK FILTER|1|0|0|1|2.000|2.000|
|
||||||
|
|UKRAINIAN MORFOLOGIK DIRECT|1|0|0|1|3.000|3.000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
|
||||||
|
### Radixor full-coverage aggregates
|
||||||
|
|
||||||
|
These aggregates cover all 19 documented languages. Macro balanced accuracy gives each language equal weight. Micro metrics first sum raw pair counts across languages. Unsupported third-party languages are never inserted as zero results, so this full-coverage table is not presented as a cross-stemmer common-language ranking.
|
||||||
|
|
||||||
|
| Dictionary mode | Languages | Macro balanced accuracy | Micro balanced accuracy | Micro precision | Micro recall | Micro F1 |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|ALL_WORDS|19|0.978929|0.987664|0.975113|0.975328|0.975221|
|
||||||
|
|LOWERCASE_GROUPS_ONLY|19|0.982354|0.989366|0.975322|0.978734|0.977025|
|
||||||
|
|
||||||
|
### Reproducible data
|
||||||
|
|
||||||
|
- [Machine-readable quality snapshot](data/stemming-quality.csv)
|
||||||
|
- SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
|
||||||
|
- [Linguistic quality methodology](reference/linguistic-quality.md)
|
||||||
|
- [Tested stemmer inventory](reference/tested-stemmers.md)
|
||||||
|
- [Reproducibility and raw data](reference/reproducibility.md)
|
||||||
|
- Pearson and Spearman correlation files are generated under `build/reports/stemming-quality/`; they are separated by dictionary mode and output policy. Correlation does not establish metric equivalence.
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY-OVERVIEW:END -->
|
||||||
|
|||||||
@@ -26,11 +26,12 @@ Radixor stores the preferred transformation for each normalized dictionary word
|
|||||||
|
|
||||||
## Accuracy
|
## Accuracy
|
||||||
|
|
||||||
Accuracy is computed from one deterministic JMH measurement iteration without warmup. The benchmark may execute the full dictionary pass more than once inside that single timed iteration; percentages divide matching counters by evaluated counters from the same iteration.
|
Accuracy is computed from JMH auxiliary counters in the current report. The counters are deterministic for a fixed corpus and stemmer; percentages divide matching counters by evaluated counters from the same report and are not timing metrics.
|
||||||
|
|
||||||
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
||||||
| --- | ---: | ---: | ---: | --- |
|
| --- | ---: | ---: | ---: | --- |
|
||||||
| Radixor | 99.465% | 99.439% | 99.582% | Full Radixor dictionary patch-command stemmer. |
|
| Radixor | 99.465% | 99.439% | 99.582% | Full Radixor dictionary patch-command stemmer. |
|
||||||
|
| Lucene HunspellStemFilter | 84.850% | 82.269% | 96.806% | Benchmark-only Czech Hunspell dictionary compared via Lucene HunspellStemFilter. |
|
||||||
| Lucene CzechStemFilter | 16.784% | 15.538% | 22.559% | Lucene Czech suffix stemmer implemented as a TokenFilter. |
|
| Lucene CzechStemFilter | 16.784% | 15.538% | 22.559% | Lucene Czech suffix stemmer implemented as a TokenFilter. |
|
||||||
|
|
||||||
## Speed
|
## Speed
|
||||||
@@ -39,8 +40,9 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
|
|
||||||
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
||||||
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
||||||
| Radixor | `czechRadixor` | 3.117 | 0.454 | 66.9 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
| Radixor | `czechRadixor` | 3.332 | 0.240 | 71.6 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
||||||
| Lucene CzechStemFilter | `czechLuceneCzechStemFilter` | 2.921 | 0.202 | 62.7 | 0.937 | Czech suffix stemmer implemented as a Lucene TokenFilter. |
|
| Lucene HunspellStemFilter | `luceneHunspellStemFilter` | 346.819 | 3.622 | 7448.4 | 104.091 | Benchmark-only Czech Hunspell dictionary compared via Lucene HunspellStemFilter. |
|
||||||
|
| Lucene CzechStemFilter | `czechLuceneCzechStemFilter` | 3.163 | 0.253 | 67.9 | 0.949 | Czech suffix stemmer implemented as a Lucene TokenFilter. |
|
||||||
|
|
||||||
## Interpretation Notes
|
## Interpretation Notes
|
||||||
|
|
||||||
@@ -49,3 +51,368 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
||||||
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
||||||
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:START -->
|
||||||
|
|
||||||
|
## Stemming Quality
|
||||||
|
|
||||||
|
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `CS_CZ` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
|
||||||
|
|
||||||
|
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
|
||||||
|
|
||||||
|
### Evaluation Scope and Key Findings
|
||||||
|
|
||||||
|
The dictionary resource is `src/main/resources/cs_cz/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
|
||||||
|
|
||||||
|
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.996565** among 3 deterministic stemmers. The runner-up is `HUNSPELL CZECH LUCENE FILTER` at 0.853752, a difference of 0.142813. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.997139** among 3 deterministic stemmers. The runner-up is `HUNSPELL CZECH LUCENE FILTER` at 0.852770, a difference of 0.144369. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
### `ALL_WORDS`
|
||||||
|
|
||||||
|
This mode contains **7 result rows**, **3 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.996565|3867 / 1334876815 (0.000290%)|2073 / 301835 (0.686799%)|0.988432|0.990189|0.990191|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|PRIMARY_OUTPUT|0.853752|11408 / 1334876815 (0.000855%)|88283 / 301835 (29.248762%)|0.888560|0.810759|0.819499|
|
||||||
|
|3|CZECH LUCENE CZECH STEM FILTER|PRIMARY_OUTPUT|0.793614|14480 / 1334876815 (0.001085%)|124586 / 301835 (41.276194%)|0.829234|0.718241|0.736765|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.987264|0.993132|0.999997|0.996565|0.999996|0.000004|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|PRIMARY_OUTPUT|0.949289|0.707512|0.999991|0.853752|0.999925|0.000075|
|
||||||
|
|3|CZECH LUCENE CZECH STEM FILTER|PRIMARY_OUTPUT|0.924477|0.587238|0.999989|0.793614|0.999896|0.000104|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.988432|0.990189|0.991953|0.980569|0.990194|0.990191|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|PRIMARY_OUTPUT|0.888560|0.810759|0.745486|0.681745|0.819533|0.819499|
|
||||||
|
|3|CZECH LUCENE CZECH STEM FILTER|PRIMARY_OUTPUT|0.829234|0.718241|0.633453|0.560356|0.736809|0.736765|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.990187|0.998733|0.998686|0.998709|0.998709|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|PRIMARY_OUTPUT|0.810723|0.995777|0.952852|0.973842|0.973842|
|
||||||
|
|3|CZECH LUCENE CZECH STEM FILTER|PRIMARY_OUTPUT|0.718192|0.993801|0.944977|0.968774|0.968774|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|299762|3867|2073|1334872948|3867 / 1334876815|2073 / 301835|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|PRIMARY_OUTPUT|213552|11408|88283|1334865407|11408 / 1334876815|88283 / 301835|
|
||||||
|
|3|CZECH LUCENE CZECH STEM FILTER|PRIMARY_OUTPUT|177249|14480|124586|1334862335|14480 / 1334876815|124586 / 301835|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 1334876815 (0.000000%)|0 / 301835 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|ANY_CANDIDATE|0.871577|10102 / 1334876815 (0.000757%)|77523 / 301835 (25.683900%)|0.904855|0.836596|0.843258|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|ANY_CANDIDATE|0.956905|0.743161|0.999992|0.871577|0.999934|0.000066|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|ANY_CANDIDATE|0.904855|0.836596|0.777914|0.719094|0.843288|0.843258|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|301835|0|0|1334876815|0 / 1334876815|0 / 301835|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|ANY_CANDIDATE|224312|10102|77523|1334866713|10102 / 1334876815|77523 / 301835|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999998|5850 / 1334876815 (0.000438%)|0 / 301835 (0.000000%)|0.984732|0.990402|0.990446|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|ALL_CANDIDATES|0.871575|13917 / 1334876815 (0.001043%)|77523 / 301835 (25.683900%)|0.893851|0.830687|0.836477|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.980987|1.000000|0.999996|0.999998|0.999996|0.000004|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|ALL_CANDIDATES|0.941581|0.743161|0.999990|0.871575|0.999932|0.000068|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.984732|0.990402|0.996139|0.980987|0.990448|0.990446|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|ALL_CANDIDATES|0.893851|0.830687|0.775861|0.710406|0.836509|0.836477|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|301835|5850|0|1334870965|5850 / 1334876815|0 / 301835|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|ALL_CANDIDATES|224312|13917|77523|1334862898|13917 / 1334876815|77523 / 301835|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|2073|3867|1983|596|1.153340%|4|52319|
|
||||||
|
|HUNSPELL CZECH LUCENE FILTER|10760|1306|2509|3317|6.418840%|5|55596|
|
||||||
|
|
||||||
|
### `LOWERCASE_GROUPS_ONLY`
|
||||||
|
|
||||||
|
This mode contains **7 result rows**, **3 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.997139|3863 / 1298544215 (0.000297%)|1709 / 298813 (0.571930%)|0.988580|0.990710|0.990714|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|PRIMARY_OUTPUT|0.852770|11239 / 1298544215 (0.000866%)|87986 / 298813 (29.445171%)|0.888009|0.809505|0.818403|
|
||||||
|
|3|CZECH LUCENE CZECH STEM FILTER|PRIMARY_OUTPUT|0.791794|13950 / 1298544215 (0.001074%)|124426 / 298813 (41.640089%)|0.828709|0.715948|0.735055|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.987165|0.994281|0.999997|0.997139|0.999996|0.000004|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|PRIMARY_OUTPUT|0.949389|0.705548|0.999991|0.852770|0.999924|0.000076|
|
||||||
|
|3|CZECH LUCENE CZECH STEM FILTER|PRIMARY_OUTPUT|0.925931|0.583599|0.999989|0.791794|0.999893|0.000107|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.988580|0.990710|0.992849|0.981591|0.990716|0.990714|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|PRIMARY_OUTPUT|0.888009|0.809505|0.743753|0.679973|0.818437|0.818403|
|
||||||
|
|3|CZECH LUCENE CZECH STEM FILTER|PRIMARY_OUTPUT|0.828709|0.715948|0.630198|0.557569|0.735100|0.735055|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.990708|0.998726|0.999030|0.998878|0.998878|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|PRIMARY_OUTPUT|0.809467|0.995812|0.952394|0.973619|0.973619|
|
||||||
|
|3|CZECH LUCENE CZECH STEM FILTER|PRIMARY_OUTPUT|0.715897|0.993897|0.944297|0.968463|0.968463|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|297104|3863|1709|1298540352|3863 / 1298544215|1709 / 298813|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|PRIMARY_OUTPUT|210827|11239|87986|1298532976|11239 / 1298544215|87986 / 298813|
|
||||||
|
|3|CZECH LUCENE CZECH STEM FILTER|PRIMARY_OUTPUT|174387|13950|124426|1298530265|13950 / 1298544215|124426 / 298813|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 1298544215 (0.000000%)|0 / 298813 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|ANY_CANDIDATE|0.870432|10028 / 1298544215 (0.000772%)|77431 / 298813 (25.912862%)|0.904004|0.835052|0.841852|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|ANY_CANDIDATE|0.956666|0.740871|0.999992|0.870432|0.999933|0.000067|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|ANY_CANDIDATE|0.904004|0.835052|0.775874|0.716815|0.841883|0.841852|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|298813|0|0|1298544215|0 / 1298544215|0 / 298813|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|ANY_CANDIDATE|221382|10028|77431|1298534187|10028 / 1298544215|77431 / 298813|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999998|5782 / 1298544215 (0.000445%)|0 / 298813 (0.000000%)|0.984756|0.990418|0.990461|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|ALL_CANDIDATES|0.870430|13601 / 1298544215 (0.001047%)|77431 / 298813 (25.912862%)|0.893574|0.829463|0.835425|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.981017|1.000000|0.999996|0.999998|0.999996|0.000004|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|ALL_CANDIDATES|0.942119|0.740871|0.999990|0.870430|0.999930|0.000070|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.984756|0.990418|0.996145|0.981017|0.990463|0.990461|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|ALL_CANDIDATES|0.893574|0.829463|0.773936|0.708617|0.835457|0.835425|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|298813|5782|0|1298538433|5782 / 1298544215|0 / 298813|
|
||||||
|
|2|HUNSPELL CZECH LUCENE FILTER|ALL_CANDIDATES|221382|13601|77431|1298530614|13601 / 1298544215|77431 / 298813|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|1709|3863|1919|540|1.059488%|4|51543|
|
||||||
|
|HUNSPELL CZECH LUCENE FILTER|10555|1211|2362|3237|6.351044%|5|54804|
|
||||||
|
|
||||||
|
### Output Policies and Metric Definitions
|
||||||
|
|
||||||
|
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
|
||||||
|
|
||||||
|
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
|
||||||
|
|
||||||
|
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
|
||||||
|
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
|
||||||
|
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
|
||||||
|
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
|
||||||
|
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
|
||||||
|
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
|
||||||
|
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
|
||||||
|
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
|
||||||
|
- Jaccard index: `TP / (TP + FP + FN)`.
|
||||||
|
- Fowlkes–Mallows index: `sqrt(precision * recall)`.
|
||||||
|
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
|
||||||
|
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
|
||||||
|
|
||||||
|
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
|
||||||
|
|
||||||
|
### Provenance
|
||||||
|
|
||||||
|
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
|
||||||
|
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
|
||||||
|
- Evaluation command: `./gradlew stemmingQuality`
|
||||||
|
- Dictionary language: `CS_CZ`
|
||||||
|
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
|
||||||
|
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
|
||||||
|
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:END -->
|
||||||
|
|||||||
@@ -26,11 +26,11 @@ Radixor stores the preferred transformation for each normalized dictionary word
|
|||||||
|
|
||||||
## Accuracy
|
## Accuracy
|
||||||
|
|
||||||
Accuracy is computed from one deterministic JMH measurement iteration without warmup. The benchmark may execute the full dictionary pass more than once inside that single timed iteration; percentages divide matching counters by evaluated counters from the same iteration.
|
Accuracy is computed from JMH auxiliary counters in the current report. The counters are deterministic for a fixed corpus and stemmer; percentages divide matching counters by evaluated counters from the same report and are not timing metrics.
|
||||||
|
|
||||||
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
||||||
| --- | ---: | ---: | ---: | --- |
|
| --- | ---: | ---: | ---: | --- |
|
||||||
| Radixor | 99.371% | 99.527% | 98.923% | Radixor baseline in the Snowball-language comparison family. |
|
| Radixor | 99.371% | 99.527% | 98.923% | Full Radixor dictionary patch-command stemmer. |
|
||||||
| Lucene SnowballFilter | 55.509% | 54.159% | 59.371% | Lucene TokenFilter integration path around the Snowball algorithm. |
|
| Lucene SnowballFilter | 55.509% | 54.159% | 59.371% | Lucene TokenFilter integration path around the Snowball algorithm. |
|
||||||
| Official Snowball direct | 55.509% | 54.159% | 59.371% | Official Snowball generated Java stemmer; rule-based suffix algorithm. |
|
| Official Snowball direct | 55.509% | 54.159% | 59.371% | Official Snowball generated Java stemmer; rule-based suffix algorithm. |
|
||||||
|
|
||||||
@@ -40,9 +40,9 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
|
|
||||||
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
||||||
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
||||||
| Radixor | `radixor[DANISH]` | 1.065 | 0.019 | 44.6 | 1.000 | Radixor baseline for the Snowball-language comparison family. |
|
| Radixor | `radixor[DANISH]` | 1.143 | 0.017 | 47.8 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
||||||
| Official Snowball direct | `snowballDirect[DANISH]` | 2.028 | 0.011 | 84.9 | 1.904 | Official Snowball generated Java stemmer; direct API. |
|
| Official Snowball direct | `snowballDirect[DANISH]` | 2.168 | 0.058 | 90.7 | 1.896 | Official Snowball generated Java stemmer; direct API. |
|
||||||
| Lucene SnowballFilter | `luceneSnowballFilter[DANISH]` | 2.692 | 0.028 | 112.6 | 2.527 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
| Lucene SnowballFilter | `luceneSnowballFilter[DANISH]` | 2.975 | 0.143 | 124.5 | 2.602 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
||||||
|
|
||||||
## Interpretation Notes
|
## Interpretation Notes
|
||||||
|
|
||||||
@@ -51,3 +51,346 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
||||||
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
||||||
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:START -->
|
||||||
|
|
||||||
|
## Stemming Quality
|
||||||
|
|
||||||
|
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `DA_DK` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
|
||||||
|
|
||||||
|
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
|
||||||
|
|
||||||
|
### Evaluation Scope and Key Findings
|
||||||
|
|
||||||
|
The dictionary resource is `src/main/resources/da_dk/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
|
||||||
|
|
||||||
|
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.996066** among 3 deterministic stemmers. The runner-up is `SNOWBALL DANISH LUCENE FILTER` at 0.937969, a difference of 0.058097. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.996305** among 3 deterministic stemmers. The runner-up is `SNOWBALL DANISH DIRECT` at 0.938074, a difference of 0.058230. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
### `ALL_WORDS`
|
||||||
|
|
||||||
|
This mode contains **5 result rows**, **3 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.996066|1165 / 394111186 (0.000296%)|707 / 89895 (0.786473%)|0.988108|0.989614|0.989615|
|
||||||
|
|2|SNOWBALL DANISH LUCENE FILTER|PRIMARY_OUTPUT|0.937969|6507 / 394111186 (0.001651%)|11151 / 89895 (12.404472%)|0.913718|0.899181|0.899475|
|
||||||
|
|3|SNOWBALL DANISH DIRECT|PRIMARY_OUTPUT|0.937903|6341 / 394111186 (0.001609%)|11163 / 89895 (12.417821%)|0.915090|0.899959|0.900279|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.987106|0.992135|0.999997|0.996066|0.999995|0.000005|
|
||||||
|
|2|SNOWBALL DANISH LUCENE FILTER|PRIMARY_OUTPUT|0.923672|0.875955|0.999983|0.937969|0.999955|0.000045|
|
||||||
|
|3|SNOWBALL DANISH DIRECT|PRIMARY_OUTPUT|0.925464|0.875822|0.999984|0.937903|0.999956|0.000044|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.988108|0.989614|0.991125|0.979442|0.989618|0.989615|
|
||||||
|
|2|SNOWBALL DANISH LUCENE FILTER|PRIMARY_OUTPUT|0.913718|0.899181|0.885100|0.816830|0.899498|0.899475|
|
||||||
|
|3|SNOWBALL DANISH DIRECT|PRIMARY_OUTPUT|0.915090|0.899959|0.885320|0.818114|0.900301|0.900279|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.989612|0.998466|0.998719|0.998592|0.998592|
|
||||||
|
|2|SNOWBALL DANISH LUCENE FILTER|PRIMARY_OUTPUT|0.899159|0.994053|0.978603|0.986268|0.986268|
|
||||||
|
|3|SNOWBALL DANISH DIRECT|PRIMARY_OUTPUT|0.899937|0.994196|0.978579|0.986326|0.986326|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|89188|1165|707|394110021|1165 / 394111186|707 / 89895|
|
||||||
|
|2|SNOWBALL DANISH LUCENE FILTER|PRIMARY_OUTPUT|78744|6507|11151|394104679|6507 / 394111186|11151 / 89895|
|
||||||
|
|3|SNOWBALL DANISH DIRECT|PRIMARY_OUTPUT|78732|6341|11163|394104845|6341 / 394111186|11163 / 89895|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 394111186 (0.000000%)|0 / 89895 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|89895|0|0|394111186|0 / 394111186|0 / 89895|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999998|1849 / 394111186 (0.000469%)|0 / 89895 (0.000000%)|0.983812|0.989820|0.989869|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.979846|1.000000|0.999995|0.999998|0.999995|0.000005|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.983812|0.989820|0.995903|0.979846|0.989872|0.989869|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|89895|1849|0|394109337|1849 / 394111186|0 / 89895|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|707|1165|684|323|1.150326%|3|28405|
|
||||||
|
|
||||||
|
### `LOWERCASE_GROUPS_ONLY`
|
||||||
|
|
||||||
|
This mode contains **5 result rows**, **3 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.996305|1165 / 392820788 (0.000297%)|663 / 89740 (0.738801%)|0.988190|0.989843|0.989845|
|
||||||
|
|2|SNOWBALL DANISH DIRECT|PRIMARY_OUTPUT|0.938074|6341 / 392820788 (0.001614%)|11113 / 89740 (12.383552%)|0.915093|0.900096|0.900410|
|
||||||
|
|3|SNOWBALL DANISH LUCENE FILTER|PRIMARY_OUTPUT|0.938074|6341 / 392820788 (0.001614%)|11113 / 89740 (12.383552%)|0.915093|0.900096|0.900410|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.987090|0.992612|0.999997|0.996305|0.999995|0.000005|
|
||||||
|
|2|SNOWBALL DANISH DIRECT|PRIMARY_OUTPUT|0.925372|0.876164|0.999984|0.938074|0.999956|0.000044|
|
||||||
|
|3|SNOWBALL DANISH LUCENE FILTER|PRIMARY_OUTPUT|0.925372|0.876164|0.999984|0.938074|0.999956|0.000044|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.988190|0.989843|0.991503|0.979891|0.989847|0.989845|
|
||||||
|
|2|SNOWBALL DANISH DIRECT|PRIMARY_OUTPUT|0.915093|0.900096|0.885583|0.818341|0.900432|0.900410|
|
||||||
|
|3|SNOWBALL DANISH LUCENE FILTER|PRIMARY_OUTPUT|0.915093|0.900096|0.885583|0.818341|0.900432|0.900410|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.989841|0.998463|0.998812|0.998637|0.998637|
|
||||||
|
|2|SNOWBALL DANISH DIRECT|PRIMARY_OUTPUT|0.900074|0.994185|0.978644|0.986354|0.986354|
|
||||||
|
|3|SNOWBALL DANISH LUCENE FILTER|PRIMARY_OUTPUT|0.900074|0.994185|0.978644|0.986354|0.986354|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|89077|1165|663|392819623|1165 / 392820788|663 / 89740|
|
||||||
|
|2|SNOWBALL DANISH DIRECT|PRIMARY_OUTPUT|78627|6341|11113|392814447|6341 / 392820788|11113 / 89740|
|
||||||
|
|3|SNOWBALL DANISH LUCENE FILTER|PRIMARY_OUTPUT|78627|6341|11113|392814447|6341 / 392820788|11113 / 89740|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 392820788 (0.000000%)|0 / 89740 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|89740|0|0|392820788|0 / 392820788|0 / 89740|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999998|1849 / 392820788 (0.000471%)|0 / 89740 (0.000000%)|0.983784|0.989803|0.989852|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.979812|1.000000|0.999995|0.999998|0.999995|0.000005|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.983784|0.989803|0.995896|0.979812|0.989855|0.989852|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|89740|1849|0|392818939|1849 / 392820788|0 / 89740|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|663|1165|684|315|1.123676%|3|28351|
|
||||||
|
|
||||||
|
### Output Policies and Metric Definitions
|
||||||
|
|
||||||
|
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
|
||||||
|
|
||||||
|
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
|
||||||
|
|
||||||
|
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
|
||||||
|
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
|
||||||
|
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
|
||||||
|
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
|
||||||
|
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
|
||||||
|
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
|
||||||
|
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
|
||||||
|
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
|
||||||
|
- Jaccard index: `TP / (TP + FP + FN)`.
|
||||||
|
- Fowlkes–Mallows index: `sqrt(precision * recall)`.
|
||||||
|
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
|
||||||
|
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
|
||||||
|
|
||||||
|
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
|
||||||
|
|
||||||
|
### Provenance
|
||||||
|
|
||||||
|
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
|
||||||
|
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
|
||||||
|
- Evaluation command: `./gradlew stemmingQuality`
|
||||||
|
- Dictionary language: `DA_DK`
|
||||||
|
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
|
||||||
|
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
|
||||||
|
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:END -->
|
||||||
|
|||||||
@@ -26,11 +26,12 @@ Radixor stores the preferred transformation for each normalized dictionary word
|
|||||||
|
|
||||||
## Accuracy
|
## Accuracy
|
||||||
|
|
||||||
Accuracy is computed from one deterministic JMH measurement iteration without warmup. The benchmark may execute the full dictionary pass more than once inside that single timed iteration; percentages divide matching counters by evaluated counters from the same iteration.
|
Accuracy is computed from JMH auxiliary counters in the current report. The counters are deterministic for a fixed corpus and stemmer; percentages divide matching counters by evaluated counters from the same report and are not timing metrics.
|
||||||
|
|
||||||
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
||||||
| --- | ---: | ---: | ---: | --- |
|
| --- | ---: | ---: | ---: | --- |
|
||||||
| Radixor | 99.120% | 98.711% | 100.000% | Radixor baseline in the Snowball-language comparison family. |
|
| Radixor | 99.120% | 98.711% | 100.000% | Full Radixor dictionary patch-command stemmer. |
|
||||||
|
| Lucene HunspellStemFilter | 46.590% | 22.718% | 97.976% | Benchmark-only Dutch Hunspell dictionary compared via Lucene HunspellStemFilter. |
|
||||||
| Official Snowball direct | 15.954% | 8.992% | 30.939% | Official Snowball generated Java stemmer; rule-based suffix algorithm. |
|
| Official Snowball direct | 15.954% | 8.992% | 30.939% | Official Snowball generated Java stemmer; rule-based suffix algorithm. |
|
||||||
| Lucene SnowballFilter | 12.620% | 5.441% | 28.073% | Lucene TokenFilter integration path around the Snowball algorithm. |
|
| Lucene SnowballFilter | 12.620% | 5.441% | 28.073% | Lucene TokenFilter integration path around the Snowball algorithm. |
|
||||||
|
|
||||||
@@ -40,9 +41,10 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
|
|
||||||
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
||||||
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
||||||
| Radixor | `radixor[DUTCH]` | 1.262 | 0.039 | 58.7 | 1.000 | Radixor baseline for the Snowball-language comparison family. |
|
| Radixor | `radixor[DUTCH]` | 1.331 | 0.114 | 61.9 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
||||||
| Official Snowball direct | `snowballDirect[DUTCH]` | 3.968 | 0.258 | 184.7 | 3.145 | Official Snowball generated Java stemmer; direct API. |
|
| Lucene HunspellStemFilter | `luceneHunspellStemFilter` | 22.760 | 1.387 | 1059.3 | 17.105 | Benchmark-only Dutch Hunspell dictionary compared via Lucene HunspellStemFilter. |
|
||||||
| Lucene SnowballFilter | `luceneSnowballFilter[DUTCH]` | 6.866 | 0.337 | 319.6 | 5.441 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
| Official Snowball direct | `snowballDirect[DUTCH]` | 4.146 | 0.291 | 193.0 | 3.116 | Official Snowball generated Java stemmer; direct API. |
|
||||||
|
| Lucene SnowballFilter | `luceneSnowballFilter[DUTCH]` | 7.375 | 0.595 | 343.3 | 5.543 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
||||||
|
|
||||||
## Interpretation Notes
|
## Interpretation Notes
|
||||||
|
|
||||||
@@ -51,3 +53,378 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
||||||
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
||||||
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:START -->
|
||||||
|
|
||||||
|
## Stemming Quality
|
||||||
|
|
||||||
|
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `NL_NL` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
|
||||||
|
|
||||||
|
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
|
||||||
|
|
||||||
|
### Evaluation Scope and Key Findings
|
||||||
|
|
||||||
|
The dictionary resource is `src/main/resources/nl_nl/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
|
||||||
|
|
||||||
|
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.988661** among 4 deterministic stemmers. The runner-up is `SNOWBALL DUTCH DIRECT` at 0.727087, a difference of 0.261574. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.989040** among 4 deterministic stemmers. The runner-up is `SNOWBALL DUTCH DIRECT` at 0.730495, a difference of 0.258544. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
### `ALL_WORDS`
|
||||||
|
|
||||||
|
This mode contains **8 result rows**, **4 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.988661|1214 / 350437960 (0.000346%)|1464 / 64566 (2.267447%)|0.980362|0.979221|0.979219|
|
||||||
|
|2|SNOWBALL DUTCH DIRECT|PRIMARY_OUTPUT|0.727087|4382 / 350437960 (0.001250%)|35241 / 64566 (54.581359%)|0.735353|0.596807|0.628557|
|
||||||
|
|3|HUNSPELL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.643123|1333 / 350437960 (0.000380%)|46084 / 64566 (71.375027%)|0.642512|0.438061|0.516674|
|
||||||
|
|4|SNOWBALL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.618497|1588 / 350437960 (0.000453%)|49264 / 64566 (76.300220%)|0.579068|0.375712|0.463333|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.981124|0.977326|0.999997|0.988661|0.999992|0.000008|
|
||||||
|
|2|SNOWBALL DUTCH DIRECT|PRIMARY_OUTPUT|0.869997|0.454186|0.999987|0.727087|0.999887|0.000113|
|
||||||
|
|3|HUNSPELL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.932728|0.286250|0.999996|0.643123|0.999865|0.000135|
|
||||||
|
|4|SNOWBALL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.905980|0.236998|0.999995|0.618497|0.999855|0.000145|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.980362|0.979221|0.978083|0.959289|0.979223|0.979219|
|
||||||
|
|2|SNOWBALL DUTCH DIRECT|PRIMARY_OUTPUT|0.735353|0.596807|0.502190|0.425321|0.628602|0.628557|
|
||||||
|
|3|HUNSPELL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.642512|0.438061|0.332316|0.280459|0.516714|0.516674|
|
||||||
|
|4|SNOWBALL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.579068|0.375712|0.278062|0.231309|0.463374|0.463333|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.979217|0.997464|0.997003|0.997234|0.997234|
|
||||||
|
|2|SNOWBALL DUTCH DIRECT|PRIMARY_OUTPUT|0.596756|0.992815|0.917346|0.953590|0.953590|
|
||||||
|
|3|HUNSPELL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.438012|0.996932|0.889026|0.939892|0.939892|
|
||||||
|
|4|SNOWBALL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.375664|0.995828|0.888410|0.939057|0.939057|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|63102|1214|1464|350436746|1214 / 350437960|1464 / 64566|
|
||||||
|
|2|SNOWBALL DUTCH DIRECT|PRIMARY_OUTPUT|29325|4382|35241|350433578|4382 / 350437960|35241 / 64566|
|
||||||
|
|3|HUNSPELL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|18482|1333|46084|350436627|1333 / 350437960|46084 / 64566|
|
||||||
|
|4|SNOWBALL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|15302|1588|49264|350436372|1588 / 350437960|49264 / 64566|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 350437960 (0.000000%)|0 / 64566 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|2|HUNSPELL DUTCH LUCENE FILTER|ANY_CANDIDATE|0.665519|1164 / 350437960 (0.000332%)|43192 / 64566 (66.895889%)|0.690741|0.490770|0.560268|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|2|HUNSPELL DUTCH LUCENE FILTER|ANY_CANDIDATE|0.948354|0.331041|0.999997|0.665519|0.999873|0.000127|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|2|HUNSPELL DUTCH LUCENE FILTER|ANY_CANDIDATE|0.690741|0.490770|0.380588|0.325179|0.560307|0.560268|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|HUNSPELL DUTCH LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|64566|0|0|350437960|0 / 350437960|0 / 64566|
|
||||||
|
|2|HUNSPELL DUTCH LUCENE FILTER|ANY_CANDIDATE|21374|1164|43192|350436796|1164 / 350437960|43192 / 64566|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999996|2651 / 350437960 (0.000756%)|0 / 64566 (0.000000%)|0.968198|0.979884|0.980078|
|
||||||
|
|2|HUNSPELL DUTCH LUCENE FILTER|ALL_CANDIDATES|0.665518|1738 / 350437960 (0.000496%)|43192 / 64566 (66.895889%)|0.680640|0.487557|0.553265|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.960561|1.000000|0.999992|0.999996|0.999992|0.000008|
|
||||||
|
|2|HUNSPELL DUTCH LUCENE FILTER|ALL_CANDIDATES|0.924801|0.331041|0.999995|0.665518|0.999872|0.000128|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.968198|0.979884|0.991855|0.960561|0.980082|0.980078|
|
||||||
|
|2|HUNSPELL DUTCH LUCENE FILTER|ALL_CANDIDATES|0.680640|0.487557|0.379812|0.322364|0.553306|0.553265|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|HUNSPELL DUTCH LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|64566|2651|0|350435309|2651 / 350437960|0 / 64566|
|
||||||
|
|2|HUNSPELL DUTCH LUCENE FILTER|ALL_CANDIDATES|21374|1738|43192|350436222|1738 / 350437960|43192 / 64566|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|HUNSPELL DUTCH LUCENE FILTER|2892|169|405|1254|4.736186%|3|27763|
|
||||||
|
|Radixor|1464|1214|1437|572|2.160366%|3|27061|
|
||||||
|
|
||||||
|
### `LOWERCASE_GROUPS_ONLY`
|
||||||
|
|
||||||
|
This mode contains **8 result rows**, **4 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.989040|1214 / 329603856 (0.000368%)|1384 / 63147 (2.191711%)|0.980194|0.979401|0.979398|
|
||||||
|
|2|SNOWBALL DUTCH DIRECT|PRIMARY_OUTPUT|0.730495|4382 / 329603856 (0.001329%)|34036 / 63147 (53.899631%)|0.738412|0.602463|0.632953|
|
||||||
|
|3|HUNSPELL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.645159|1310 / 329603856 (0.000397%)|44814 / 63147 (70.967742%)|0.646808|0.442880|0.520498|
|
||||||
|
|4|SNOWBALL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.618546|1544 / 329603856 (0.000468%)|48175 / 63147 (76.290243%)|0.579362|0.375883|0.463566|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.980723|0.978083|0.999996|0.989040|0.999992|0.000008|
|
||||||
|
|2|SNOWBALL DUTCH DIRECT|PRIMARY_OUTPUT|0.869167|0.461004|0.999987|0.730495|0.999883|0.000117|
|
||||||
|
|3|HUNSPELL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.933310|0.290323|0.999996|0.645159|0.999860|0.000140|
|
||||||
|
|4|SNOWBALL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.906515|0.237098|0.999995|0.618546|0.999849|0.000151|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.980194|0.979401|0.978610|0.959634|0.979402|0.979398|
|
||||||
|
|2|SNOWBALL DUTCH DIRECT|PRIMARY_OUTPUT|0.738412|0.602463|0.508789|0.431089|0.633000|0.632953|
|
||||||
|
|3|HUNSPELL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.646808|0.442880|0.336718|0.284422|0.520539|0.520498|
|
||||||
|
|4|SNOWBALL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.579362|0.375883|0.278182|0.231439|0.463608|0.463566|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.979397|0.997373|0.997139|0.997256|0.997256|
|
||||||
|
|2|SNOWBALL DUTCH DIRECT|PRIMARY_OUTPUT|0.602410|0.992557|0.918059|0.953856|0.953856|
|
||||||
|
|3|HUNSPELL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.442829|0.996884|0.889061|0.939890|0.939890|
|
||||||
|
|4|SNOWBALL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|0.375834|0.995817|0.887492|0.938539|0.938539|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|61763|1214|1384|329602642|1214 / 329603856|1384 / 63147|
|
||||||
|
|2|SNOWBALL DUTCH DIRECT|PRIMARY_OUTPUT|29111|4382|34036|329599474|4382 / 329603856|34036 / 63147|
|
||||||
|
|3|HUNSPELL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|18333|1310|44814|329602546|1310 / 329603856|44814 / 63147|
|
||||||
|
|4|SNOWBALL DUTCH LUCENE FILTER|PRIMARY_OUTPUT|14972|1544|48175|329602312|1544 / 329603856|48175 / 63147|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 329603856 (0.000000%)|0 / 63147 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|2|HUNSPELL DUTCH LUCENE FILTER|ANY_CANDIDATE|0.667956|1141 / 329603856 (0.000346%)|41935 / 63147 (66.408539%)|0.695206|0.496187|0.564555|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|2|HUNSPELL DUTCH LUCENE FILTER|ANY_CANDIDATE|0.948955|0.335915|0.999997|0.667956|0.999869|0.000131|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|2|HUNSPELL DUTCH LUCENE FILTER|ANY_CANDIDATE|0.695206|0.496187|0.385755|0.329953|0.564595|0.564555|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|HUNSPELL DUTCH LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|63147|0|0|329603856|0 / 329603856|0 / 63147|
|
||||||
|
|2|HUNSPELL DUTCH LUCENE FILTER|ANY_CANDIDATE|21212|1141|41935|329602715|1141 / 329603856|41935 / 63147|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999996|2651 / 329603856 (0.000804%)|0 / 63147 (0.000000%)|0.967506|0.979441|0.979644|
|
||||||
|
|2|HUNSPELL DUTCH LUCENE FILTER|ALL_CANDIDATES|0.667955|1712 / 329603856 (0.000519%)|41935 / 63147 (66.408539%)|0.684952|0.492895|0.557477|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.959710|1.000000|0.999992|0.999996|0.999992|0.000008|
|
||||||
|
|2|HUNSPELL DUTCH LUCENE FILTER|ALL_CANDIDATES|0.925318|0.335915|0.999995|0.667955|0.999868|0.000132|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.967506|0.979441|0.991674|0.959710|0.979648|0.979644|
|
||||||
|
|2|HUNSPELL DUTCH LUCENE FILTER|ALL_CANDIDATES|0.684952|0.492895|0.384956|0.327048|0.557519|0.557477|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|HUNSPELL DUTCH LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|63147|2651|0|329601205|2651 / 329603856|0 / 63147|
|
||||||
|
|2|HUNSPELL DUTCH LUCENE FILTER|ALL_CANDIDATES|21212|1712|41935|329602144|1712 / 329603856|41935 / 63147|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|HUNSPELL DUTCH LUCENE FILTER|2879|169|402|1186|4.618740%|3|26896|
|
||||||
|
|Radixor|1384|1214|1437|549|2.138017%|3|26239|
|
||||||
|
|
||||||
|
### Output Policies and Metric Definitions
|
||||||
|
|
||||||
|
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
|
||||||
|
|
||||||
|
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
|
||||||
|
|
||||||
|
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
|
||||||
|
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
|
||||||
|
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
|
||||||
|
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
|
||||||
|
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
|
||||||
|
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
|
||||||
|
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
|
||||||
|
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
|
||||||
|
- Jaccard index: `TP / (TP + FP + FN)`.
|
||||||
|
- Fowlkes–Mallows index: `sqrt(precision * recall)`.
|
||||||
|
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
|
||||||
|
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
|
||||||
|
|
||||||
|
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
|
||||||
|
|
||||||
|
### Provenance
|
||||||
|
|
||||||
|
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
|
||||||
|
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
|
||||||
|
- Evaluation command: `./gradlew stemmingQuality`
|
||||||
|
- Dictionary language: `NL_NL`
|
||||||
|
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
|
||||||
|
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
|
||||||
|
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:END -->
|
||||||
|
|||||||
@@ -26,13 +26,14 @@ Radixor stores the preferred transformation for each normalized dictionary word
|
|||||||
|
|
||||||
## Accuracy
|
## Accuracy
|
||||||
|
|
||||||
Accuracy is computed from one deterministic JMH measurement iteration without warmup. The benchmark may execute the full dictionary pass more than once inside that single timed iteration; percentages divide matching counters by evaluated counters from the same iteration.
|
Accuracy is computed from JMH auxiliary counters in the current report. The counters are deterministic for a fixed corpus and stemmer; percentages divide matching counters by evaluated counters from the same report and are not timing metrics.
|
||||||
|
|
||||||
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
||||||
| --- | ---: | ---: | ---: | --- |
|
| --- | ---: | ---: | ---: | --- |
|
||||||
| Radixor | 97.478% | 97.197% | 97.552% | Full Radixor dictionary patch-command stemmer. |
|
| Radixor | 97.478% | 97.197% | 97.552% | Full Radixor dictionary patch-command stemmer. |
|
||||||
| Lucene EnglishMinimalStemFilter | 90.981% | 65.189% | 97.820% | Minimal English plural reduction, not a full stemmer. |
|
| Lucene EnglishMinimalStemFilter | 90.981% | 65.189% | 97.820% | Minimal English plural reduction, not a full stemmer. |
|
||||||
| Lucene KStemFilter | 80.076% | 76.608% | 80.996% | Krovetz-style English stemming TokenFilter; broader than minimal suffix reducers. |
|
| Lucene KStemFilter | 80.076% | 76.608% | 80.996% | Krovetz-style English stemming TokenFilter; broader than minimal suffix reducers. |
|
||||||
|
| Lucene HunspellStemFilter | 80.243% | 12.750% | 98.139% | Benchmark-only English Hunspell dictionary compared via Lucene HunspellStemFilter. |
|
||||||
| Lucene EnglishPossessiveFilter | 79.032% | 0.003% | 99.987% | Possessive-ending remover only, not a full stemmer. |
|
| Lucene EnglishPossessiveFilter | 79.032% | 0.003% | 99.987% | Possessive-ending remover only, not a full stemmer. |
|
||||||
| Snowball English / Porter2 | 40.342% | 46.296% | 38.763% | Porter2 rule-based suffix stemmer, distinct from original Porter. |
|
| Snowball English / Porter2 | 40.342% | 46.296% | 38.763% | Porter2 rule-based suffix stemmer, distinct from original Porter. |
|
||||||
| Lucene PorterStemFilter | 39.538% | 46.201% | 37.772% | Lucene TokenFilter path for Porter suffix rules; not dictionary-root equivalent. |
|
| Lucene PorterStemFilter | 39.538% | 46.201% | 37.772% | Lucene TokenFilter path for Porter suffix rules; not dictionary-root equivalent. |
|
||||||
@@ -47,16 +48,17 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
|
|
||||||
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
||||||
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
||||||
| Radixor | `radixorUsUkProfiPreferredStem` | 16.621 | 8.532 | 79.0 | 1.000 | Full dictionary patch-command stemmer using compiled patch commands. |
|
| Radixor | `radixorUsUkProfiPreferredStem` | 21.987 | 8.707 | 104.5 | 1.000 | Full dictionary patch-command stemmer using compiled patch commands. |
|
||||||
| Lucene EnglishPossessiveFilter | `luceneEnglishPossessiveFilter` | 23.845 | 0.833 | 113.3 | 1.435 | Possessive-ending remover only; not a full stemmer. |
|
| Lucene EnglishPossessiveFilter | `luceneEnglishPossessiveFilter` | 24.539 | 1.515 | 116.6 | 1.116 | Possessive-ending remover only; not a full stemmer. |
|
||||||
| Lucene EnglishMinimalStemFilter | `luceneEnglishMinimalStemFilter` | 17.091 | 0.198 | 81.2 | 1.028 | Narrow plural reduction filter; not a full stemmer. |
|
| Lucene EnglishMinimalStemFilter | `luceneEnglishMinimalStemFilter` | 22.702 | 1.195 | 107.8 | 1.032 | Narrow plural reduction filter; not a full stemmer. |
|
||||||
| Lucene PorterStemmer direct copy | `lucenePorterStemmerCopied` | 18.598 | 10.954 | 88.4 | 1.119 | Benchmark-only generated copy of Lucene package-private Porter implementation. |
|
| Lucene PorterStemmer direct copy | `lucenePorterStemmerCopied` | 24.696 | 13.235 | 117.3 | 1.123 | Benchmark-only generated copy of Lucene package-private Porter implementation. |
|
||||||
| OpenNLP PorterStemmer | `opennlpPorterStemmer` | 18.213 | 10.674 | 86.5 | 1.096 | Apache OpenNLP Porter implementation. |
|
| OpenNLP PorterStemmer | `opennlpPorterStemmer` | 23.121 | 12.528 | 109.8 | 1.052 | Apache OpenNLP Porter implementation. |
|
||||||
| Snowball original Porter | `snowballOriginalPorter` | 32.921 | 11.520 | 156.4 | 1.981 | Classic Porter suffix-rule stemmer; historical English baseline, not a dictionary-equivalent stemmer. |
|
| Snowball original Porter | `snowballOriginalPorter` | 38.904 | 10.353 | 184.8 | 1.769 | Classic Porter suffix-rule stemmer; historical English baseline, not a dictionary-equivalent stemmer. |
|
||||||
| Lucene PorterStemFilter | `lucenePorterStemFilter` | 42.874 | 1.321 | 203.7 | 2.579 | Lucene TokenFilter integration path for Porter; includes TokenStream overhead. |
|
| Lucene PorterStemFilter | `lucenePorterStemFilter` | 37.021 | 1.196 | 175.9 | 1.684 | Lucene TokenFilter integration path for Porter; includes TokenStream overhead. |
|
||||||
| Lucene KStemFilter | `luceneKStemFilter` | 50.483 | 3.624 | 239.8 | 3.037 | Krovetz-style English TokenFilter; broader than minimal suffix filters. |
|
| Lucene KStemFilter | `luceneKStemFilter` | 51.640 | 2.591 | 245.3 | 2.349 | Krovetz-style English TokenFilter; broader than minimal suffix filters. |
|
||||||
| Snowball English / Porter2 | `snowballEnglishPorter2` | 47.844 | 1.887 | 227.3 | 2.878 | Porter2 suffix-rule stemmer, distinct from original Porter. |
|
| Lucene HunspellStemFilter | `luceneHunspellStemFilter` | 79.785 | 1.347 | 379.0 | 3.629 | Benchmark-only English Hunspell comparison using the benchmark Hunspell corpus. |
|
||||||
| Paice/Husk Lancaster | `paiceHuskLancaster` | 135.050 | 11.088 | 641.6 | 8.125 | Aggressive rule-based English stemmer. |
|
| Snowball English / Porter2 | `snowballEnglishPorter2` | 52.437 | 0.773 | 249.1 | 2.385 | Porter2 suffix-rule stemmer, distinct from original Porter. |
|
||||||
|
| Paice/Husk Lancaster | `paiceHuskLancaster` | 141.556 | 12.324 | 672.5 | 6.438 | Aggressive rule-based English stemmer. |
|
||||||
|
|
||||||
## Interpretation Notes
|
## Interpretation Notes
|
||||||
|
|
||||||
@@ -65,3 +67,448 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
||||||
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
||||||
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:START -->
|
||||||
|
|
||||||
|
## Stemming Quality
|
||||||
|
|
||||||
|
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `US_UK` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
|
||||||
|
|
||||||
|
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
|
||||||
|
|
||||||
|
### Evaluation Scope and Key Findings
|
||||||
|
|
||||||
|
The dictionary resource is `src/main/resources/us_uk/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
|
||||||
|
|
||||||
|
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.965159** among 11 deterministic stemmers. The runner-up is `ENGLISH LUCENE PORTER COPIED` at 0.954627, a difference of 0.010533. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.965820** among 11 deterministic stemmers. The runner-up is `ENGLISH LUCENE PORTER COPIED` at 0.954900, a difference of 0.010920. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
### `ALL_WORDS`
|
||||||
|
|
||||||
|
This mode contains **15 result rows**, **11 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.965159|1149886 / 184490451771 (0.000623%)|21869 / 313870 (6.967534%)|0.240076|0.332621|0.434052|
|
||||||
|
|2|ENGLISH LUCENE PORTER COPIED|PRIMARY_OUTPUT|0.954627|1557406 / 184490451771 (0.000844%)|28480 / 313870 (9.073820%)|0.185679|0.264659|0.375252|
|
||||||
|
|3|ENGLISH LUCENE PORTER FILTER|PRIMARY_OUTPUT|0.954627|1557406 / 184490451771 (0.000844%)|28480 / 313870 (9.073820%)|0.185679|0.264659|0.375252|
|
||||||
|
|4|ENGLISH OPENNLP PORTER|PRIMARY_OUTPUT|0.954627|1557406 / 184490451771 (0.000844%)|28480 / 313870 (9.073820%)|0.185679|0.264659|0.375252|
|
||||||
|
|5|ENGLISH SNOWBALL PORTER2|PRIMARY_OUTPUT|0.954537|1566711 / 184490451771 (0.000849%)|28536 / 313870 (9.091662%)|0.184753|0.263477|0.374240|
|
||||||
|
|6|ENGLISH SNOWBALL ORIGINAL PORTER|PRIMARY_OUTPUT|0.954490|1555293 / 184490451771 (0.000843%)|28566 / 313870 (9.101220%)|0.185835|0.264849|0.375363|
|
||||||
|
|7|ENGLISH PAICE HUSK LANCASTER|PRIMARY_OUTPUT|0.952394|3062661 / 184490451771 (0.001660%)|29879 / 313870 (9.519546%)|0.103643|0.155164|0.277089|
|
||||||
|
|8|ENGLISH LUCENE KSTEM FILTER|PRIMARY_OUTPUT|0.878441|1368501 / 184490451771 (0.000742%)|76305 / 313870 (24.311020%)|0.176284|0.247472|0.334598|
|
||||||
|
|9|ENGLISH LUCENE MINIMAL FILTER|PRIMARY_OUTPUT|0.718599|1122264 / 184490451771 (0.000608%)|176645 / 313870 (56.279670%)|0.128204|0.174436|0.218251|
|
||||||
|
|10|HUNSPELL ENGLISH LUCENE FILTER|PRIMARY_OUTPUT|0.573277|1981986 / 184490451771 (0.001074%)|267868 / 313870 (85.343614%)|0.027298|0.039287|0.057655|
|
||||||
|
|11|ENGLISH LUCENE POSSESSIVE FILTER|PRIMARY_OUTPUT|0.500008|1115154 / 184490451771 (0.000604%)|313863 / 313870 (99.997770%)|0.000007|0.000010|0.000009|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.202513|0.930325|0.999994|0.965159|0.999994|0.000006|
|
||||||
|
|2|ENGLISH LUCENE PORTER COPIED|PRIMARY_OUTPUT|0.154868|0.909262|0.999992|0.954627|0.999991|0.000009|
|
||||||
|
|3|ENGLISH LUCENE PORTER FILTER|PRIMARY_OUTPUT|0.154868|0.909262|0.999992|0.954627|0.999991|0.000009|
|
||||||
|
|4|ENGLISH OPENNLP PORTER|PRIMARY_OUTPUT|0.154868|0.909262|0.999992|0.954627|0.999991|0.000009|
|
||||||
|
|5|ENGLISH SNOWBALL PORTER2|PRIMARY_OUTPUT|0.154064|0.909083|0.999992|0.954537|0.999991|0.000009|
|
||||||
|
|6|ENGLISH SNOWBALL ORIGINAL PORTER|PRIMARY_OUTPUT|0.155006|0.908988|0.999992|0.954490|0.999991|0.000009|
|
||||||
|
|7|ENGLISH PAICE HUSK LANCASTER|PRIMARY_OUTPUT|0.084858|0.904805|0.999983|0.952394|0.999983|0.000017|
|
||||||
|
|8|ENGLISH LUCENE KSTEM FILTER|PRIMARY_OUTPUT|0.147917|0.756890|0.999993|0.878441|0.999992|0.000008|
|
||||||
|
|9|ENGLISH LUCENE MINIMAL FILTER|PRIMARY_OUTPUT|0.108953|0.437203|0.999994|0.718599|0.999993|0.000007|
|
||||||
|
|10|HUNSPELL ENGLISH LUCENE FILTER|PRIMARY_OUTPUT|0.022684|0.146564|0.999989|0.573277|0.999988|0.000012|
|
||||||
|
|11|ENGLISH LUCENE POSSESSIVE FILTER|PRIMARY_OUTPUT|0.000006|0.000022|0.999994|0.500008|0.999992|0.000008|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.240076|0.332621|0.541270|0.199487|0.434054|0.434052|
|
||||||
|
|2|ENGLISH LUCENE PORTER COPIED|PRIMARY_OUTPUT|0.185679|0.264659|0.460563|0.152511|0.375254|0.375252|
|
||||||
|
|3|ENGLISH LUCENE PORTER FILTER|PRIMARY_OUTPUT|0.185679|0.264659|0.460563|0.152511|0.375254|0.375252|
|
||||||
|
|4|ENGLISH OPENNLP PORTER|PRIMARY_OUTPUT|0.185679|0.264659|0.460563|0.152511|0.375254|0.375252|
|
||||||
|
|5|ENGLISH SNOWBALL PORTER2|PRIMARY_OUTPUT|0.184753|0.263477|0.459102|0.151727|0.374242|0.374240|
|
||||||
|
|6|ENGLISH SNOWBALL ORIGINAL PORTER|PRIMARY_OUTPUT|0.185835|0.264849|0.460751|0.152637|0.375365|0.375363|
|
||||||
|
|7|ENGLISH PAICE HUSK LANCASTER|PRIMARY_OUTPUT|0.103643|0.155164|0.308543|0.084107|0.277092|0.277089|
|
||||||
|
|8|ENGLISH LUCENE KSTEM FILTER|PRIMARY_OUTPUT|0.176284|0.247472|0.415099|0.141208|0.334600|0.334598|
|
||||||
|
|9|ENGLISH LUCENE MINIMAL FILTER|PRIMARY_OUTPUT|0.128204|0.174436|0.272816|0.095552|0.218253|0.218251|
|
||||||
|
|10|HUNSPELL ENGLISH LUCENE FILTER|PRIMARY_OUTPUT|0.027298|0.039287|0.070051|0.020037|0.057659|0.057655|
|
||||||
|
|11|ENGLISH LUCENE POSSESSIVE FILTER|PRIMARY_OUTPUT|0.000007|0.000010|0.000015|0.000005|0.000012|0.000009|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.332619|0.994215|0.997770|0.995989|0.995989|
|
||||||
|
|2|ENGLISH LUCENE PORTER COPIED|PRIMARY_OUTPUT|0.264656|0.969648|0.997199|0.983231|0.983231|
|
||||||
|
|3|ENGLISH LUCENE PORTER FILTER|PRIMARY_OUTPUT|0.264656|0.969648|0.997199|0.983231|0.983231|
|
||||||
|
|4|ENGLISH OPENNLP PORTER|PRIMARY_OUTPUT|0.264656|0.969648|0.997199|0.983231|0.983231|
|
||||||
|
|5|ENGLISH SNOWBALL PORTER2|PRIMARY_OUTPUT|0.263474|0.969037|0.997182|0.982908|0.982908|
|
||||||
|
|6|ENGLISH SNOWBALL ORIGINAL PORTER|PRIMARY_OUTPUT|0.264847|0.969891|0.997193|0.983353|0.983353|
|
||||||
|
|7|ENGLISH PAICE HUSK LANCASTER|PRIMARY_OUTPUT|0.155162|0.937768|0.996600|0.966289|0.966289|
|
||||||
|
|8|ENGLISH LUCENE KSTEM FILTER|PRIMARY_OUTPUT|0.247470|0.980687|0.992108|0.986364|0.986364|
|
||||||
|
|9|ENGLISH LUCENE MINIMAL FILTER|PRIMARY_OUTPUT|0.174433|0.995202|0.981174|0.988138|0.988138|
|
||||||
|
|10|HUNSPELL ENGLISH LUCENE FILTER|PRIMARY_OUTPUT|0.039284|0.993096|0.963677|0.978166|0.978166|
|
||||||
|
|11|ENGLISH LUCENE POSSESSIVE FILTER|PRIMARY_OUTPUT|0.000007|0.995789|0.958019|0.976539|0.976539|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|292001|1149886|21869|184489301885|1149886 / 184490451771|21869 / 313870|
|
||||||
|
|2|ENGLISH LUCENE PORTER COPIED|PRIMARY_OUTPUT|285390|1557406|28480|184488894365|1557406 / 184490451771|28480 / 313870|
|
||||||
|
|3|ENGLISH LUCENE PORTER FILTER|PRIMARY_OUTPUT|285390|1557406|28480|184488894365|1557406 / 184490451771|28480 / 313870|
|
||||||
|
|4|ENGLISH OPENNLP PORTER|PRIMARY_OUTPUT|285390|1557406|28480|184488894365|1557406 / 184490451771|28480 / 313870|
|
||||||
|
|5|ENGLISH SNOWBALL PORTER2|PRIMARY_OUTPUT|285334|1566711|28536|184488885060|1566711 / 184490451771|28536 / 313870|
|
||||||
|
|6|ENGLISH SNOWBALL ORIGINAL PORTER|PRIMARY_OUTPUT|285304|1555293|28566|184488896478|1555293 / 184490451771|28566 / 313870|
|
||||||
|
|7|ENGLISH PAICE HUSK LANCASTER|PRIMARY_OUTPUT|283991|3062661|29879|184487389110|3062661 / 184490451771|29879 / 313870|
|
||||||
|
|8|ENGLISH LUCENE KSTEM FILTER|PRIMARY_OUTPUT|237565|1368501|76305|184489083270|1368501 / 184490451771|76305 / 313870|
|
||||||
|
|9|ENGLISH LUCENE MINIMAL FILTER|PRIMARY_OUTPUT|137225|1122264|176645|184489329507|1122264 / 184490451771|176645 / 313870|
|
||||||
|
|10|HUNSPELL ENGLISH LUCENE FILTER|PRIMARY_OUTPUT|46002|1981986|267868|184488469785|1981986 / 184490451771|267868 / 313870|
|
||||||
|
|11|ENGLISH LUCENE POSSESSIVE FILTER|PRIMARY_OUTPUT|7|1115154|313863|184489336617|1115154 / 184490451771|313863 / 313870|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|0.999976|12 / 184490451771 (0.000000%)|15 / 313870 (0.004779%)|0.999960|0.999957|0.999957|
|
||||||
|
|2|HUNSPELL ENGLISH LUCENE FILTER|ANY_CANDIDATE|0.581603|1978852 / 184490451771 (0.001073%)|262641 / 313870 (83.678274%)|0.030370|0.043712|0.064174|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|0.999962|0.999952|1.000000|0.999976|1.000000|0.000000|
|
||||||
|
|2|HUNSPELL ENGLISH LUCENE FILTER|ANY_CANDIDATE|0.025235|0.163217|0.999989|0.581603|0.999988|0.000012|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|0.999960|0.999957|0.999954|0.999914|0.999957|0.999957|
|
||||||
|
|2|HUNSPELL ENGLISH LUCENE FILTER|ANY_CANDIDATE|0.030370|0.043712|0.077961|0.022344|0.064178|0.064174|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|HUNSPELL ENGLISH LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|313855|12|15|184490451759|12 / 184490451771|15 / 313870|
|
||||||
|
|2|HUNSPELL ENGLISH LUCENE FILTER|ANY_CANDIDATE|51229|1978852|262641|184488472919|1978852 / 184490451771|262641 / 313870|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999945|11482166 / 184490451771 (0.006224%)|15 / 313870 (0.004779%)|0.033039|0.051834|0.163107|
|
||||||
|
|2|HUNSPELL ENGLISH LUCENE FILTER|ALL_CANDIDATES|0.581603|2008917 / 184490451771 (0.001089%)|262641 / 313870 (83.678274%)|0.029943|0.043158|0.063704|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.026607|0.999952|0.999938|0.999945|0.999938|0.000062|
|
||||||
|
|2|HUNSPELL ENGLISH LUCENE FILTER|ALL_CANDIDATES|0.024867|0.163217|0.999989|0.581603|0.999988|0.000012|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.033039|0.051834|0.120237|0.026607|0.163112|0.163107|
|
||||||
|
|2|HUNSPELL ENGLISH LUCENE FILTER|ALL_CANDIDATES|0.029943|0.043158|0.077254|0.022055|0.063708|0.063704|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|HUNSPELL ENGLISH LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|313855|11482166|15|184478969605|11482166 / 184490451771|15 / 313870|
|
||||||
|
|2|HUNSPELL ENGLISH LUCENE FILTER|ALL_CANDIDATES|51229|2008917|262641|184488442854|2008917 / 184490451771|262641 / 313870|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|21854|1149874|10332280|29208|4.808384%|1355|2838145|
|
||||||
|
|HUNSPELL ENGLISH LUCENE FILTER|5227|3134|26931|6837|1.125545%|4|614296|
|
||||||
|
|
||||||
|
### `LOWERCASE_GROUPS_ONLY`
|
||||||
|
|
||||||
|
This mode contains **15 result rows**, **11 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.965820|1148489 / 170474840204 (0.000674%)|21319 / 311891 (6.835401%)|0.239424|0.331902|0.433722|
|
||||||
|
|2|ENGLISH LUCENE PORTER COPIED|PRIMARY_OUTPUT|0.954900|1552702 / 170474840204 (0.000911%)|28130 / 311891 (9.019177%)|0.185277|0.264166|0.374937|
|
||||||
|
|3|ENGLISH LUCENE PORTER FILTER|PRIMARY_OUTPUT|0.954900|1552702 / 170474840204 (0.000911%)|28130 / 311891 (9.019177%)|0.185277|0.264166|0.374937|
|
||||||
|
|4|ENGLISH OPENNLP PORTER|PRIMARY_OUTPUT|0.954900|1552702 / 170474840204 (0.000911%)|28130 / 311891 (9.019177%)|0.185277|0.264166|0.374937|
|
||||||
|
|5|ENGLISH SNOWBALL PORTER2|PRIMARY_OUTPUT|0.954850|1561891 / 170474840204 (0.000916%)|28161 / 311891 (9.029116%)|0.184375|0.263016|0.373964|
|
||||||
|
|6|ENGLISH SNOWBALL ORIGINAL PORTER|PRIMARY_OUTPUT|0.954762|1550615 / 170474840204 (0.000910%)|28216 / 311891 (9.046750%)|0.185431|0.264353|0.375045|
|
||||||
|
|7|ENGLISH PAICE HUSK LANCASTER|PRIMARY_OUTPUT|0.952710|3045870 / 170474840204 (0.001787%)|29493 / 311891 (9.456188%)|0.103633|0.155157|0.277170|
|
||||||
|
|8|ENGLISH LUCENE KSTEM FILTER|PRIMARY_OUTPUT|0.880820|1367069 / 170474840204 (0.000802%)|74340 / 311891 (23.835250%)|0.176477|0.247899|0.335789|
|
||||||
|
|9|ENGLISH LUCENE MINIMAL FILTER|PRIMARY_OUTPUT|0.719516|1120871 / 170474840204 (0.000657%)|174959 / 311891 (56.096200%)|0.128139|0.174470|0.218621|
|
||||||
|
|10|HUNSPELL ENGLISH LUCENE FILTER|PRIMARY_OUTPUT|0.573619|1978041 / 170474840204 (0.001160%)|265965 / 311891 (85.274984%)|0.027312|0.039323|0.057799|
|
||||||
|
|11|ENGLISH LUCENE POSSESSIVE FILTER|PRIMARY_OUTPUT|0.500005|1113773 / 170474840204 (0.000653%)|311886 / 311891 (99.998397%)|0.000005|0.000007|0.000005|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.201918|0.931646|0.999993|0.965820|0.999993|0.000007|
|
||||||
|
|2|ENGLISH LUCENE PORTER COPIED|PRIMARY_OUTPUT|0.154515|0.909808|0.999991|0.954900|0.999991|0.000009|
|
||||||
|
|3|ENGLISH LUCENE PORTER FILTER|PRIMARY_OUTPUT|0.154515|0.909808|0.999991|0.954900|0.999991|0.000009|
|
||||||
|
|4|ENGLISH OPENNLP PORTER|PRIMARY_OUTPUT|0.154515|0.909808|0.999991|0.954900|0.999991|0.000009|
|
||||||
|
|5|ENGLISH SNOWBALL PORTER2|PRIMARY_OUTPUT|0.153731|0.909709|0.999991|0.954850|0.999991|0.000009|
|
||||||
|
|6|ENGLISH SNOWBALL ORIGINAL PORTER|PRIMARY_OUTPUT|0.154651|0.909532|0.999991|0.954762|0.999991|0.000009|
|
||||||
|
|7|ENGLISH PAICE HUSK LANCASTER|PRIMARY_OUTPUT|0.084848|0.905438|0.999982|0.952710|0.999982|0.000018|
|
||||||
|
|8|ENGLISH LUCENE KSTEM FILTER|PRIMARY_OUTPUT|0.148042|0.761647|0.999992|0.880820|0.999992|0.000008|
|
||||||
|
|9|ENGLISH LUCENE MINIMAL FILTER|PRIMARY_OUTPUT|0.108866|0.439038|0.999993|0.719516|0.999992|0.000008|
|
||||||
|
|10|HUNSPELL ENGLISH LUCENE FILTER|PRIMARY_OUTPUT|0.022691|0.147250|0.999988|0.573619|0.999987|0.000013|
|
||||||
|
|11|ENGLISH LUCENE POSSESSIVE FILTER|PRIMARY_OUTPUT|0.000004|0.000016|0.999993|0.500005|0.999992|0.000008|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.239424|0.331902|0.540775|0.198970|0.433723|0.433722|
|
||||||
|
|2|ENGLISH LUCENE PORTER COPIED|PRIMARY_OUTPUT|0.185277|0.264166|0.460049|0.152184|0.374939|0.374937|
|
||||||
|
|3|ENGLISH LUCENE PORTER FILTER|PRIMARY_OUTPUT|0.185277|0.264166|0.460049|0.152184|0.374939|0.374937|
|
||||||
|
|4|ENGLISH OPENNLP PORTER|PRIMARY_OUTPUT|0.185277|0.264166|0.460049|0.152184|0.374939|0.374937|
|
||||||
|
|5|ENGLISH SNOWBALL PORTER2|PRIMARY_OUTPUT|0.184375|0.263016|0.458637|0.151421|0.373966|0.373964|
|
||||||
|
|6|ENGLISH SNOWBALL ORIGINAL PORTER|PRIMARY_OUTPUT|0.185431|0.264353|0.460234|0.152308|0.375047|0.375045|
|
||||||
|
|7|ENGLISH PAICE HUSK LANCASTER|PRIMARY_OUTPUT|0.103633|0.155157|0.308576|0.084103|0.277173|0.277170|
|
||||||
|
|8|ENGLISH LUCENE KSTEM FILTER|PRIMARY_OUTPUT|0.176477|0.247899|0.416437|0.141487|0.335791|0.335789|
|
||||||
|
|9|ENGLISH LUCENE MINIMAL FILTER|PRIMARY_OUTPUT|0.128139|0.174470|0.273277|0.095572|0.218624|0.218621|
|
||||||
|
|10|HUNSPELL ENGLISH LUCENE FILTER|PRIMARY_OUTPUT|0.027312|0.039323|0.070190|0.020056|0.057804|0.057799|
|
||||||
|
|11|ENGLISH LUCENE POSSESSIVE FILTER|PRIMARY_OUTPUT|0.000005|0.000007|0.000011|0.000004|0.000008|0.000005|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.331900|0.993959|0.997731|0.995842|0.995842|
|
||||||
|
|2|ENGLISH LUCENE PORTER COPIED|PRIMARY_OUTPUT|0.264164|0.968645|0.997109|0.982671|0.982671|
|
||||||
|
|3|ENGLISH LUCENE PORTER FILTER|PRIMARY_OUTPUT|0.264164|0.968645|0.997109|0.982671|0.982671|
|
||||||
|
|4|ENGLISH OPENNLP PORTER|PRIMARY_OUTPUT|0.264164|0.968645|0.997109|0.982671|0.982671|
|
||||||
|
|5|ENGLISH SNOWBALL PORTER2|PRIMARY_OUTPUT|0.263014|0.968020|0.997096|0.982343|0.982343|
|
||||||
|
|6|ENGLISH SNOWBALL ORIGINAL PORTER|PRIMARY_OUTPUT|0.264351|0.968894|0.997102|0.982795|0.982795|
|
||||||
|
|7|ENGLISH PAICE HUSK LANCASTER|PRIMARY_OUTPUT|0.155154|0.936077|0.996487|0.965338|0.965338|
|
||||||
|
|8|ENGLISH LUCENE KSTEM FILTER|PRIMARY_OUTPUT|0.247897|0.979822|0.991991|0.985869|0.985869|
|
||||||
|
|9|ENGLISH LUCENE MINIMAL FILTER|PRIMARY_OUTPUT|0.174467|0.994994|0.980520|0.987704|0.987704|
|
||||||
|
|10|HUNSPELL ENGLISH LUCENE FILTER|PRIMARY_OUTPUT|0.039320|0.993066|0.962317|0.977450|0.977450|
|
||||||
|
|11|ENGLISH LUCENE POSSESSIVE FILTER|PRIMARY_OUTPUT|0.000004|0.995605|0.956423|0.975621|0.975621|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|290572|1148489|21319|170473691715|1148489 / 170474840204|21319 / 311891|
|
||||||
|
|2|ENGLISH LUCENE PORTER COPIED|PRIMARY_OUTPUT|283761|1552702|28130|170473287502|1552702 / 170474840204|28130 / 311891|
|
||||||
|
|3|ENGLISH LUCENE PORTER FILTER|PRIMARY_OUTPUT|283761|1552702|28130|170473287502|1552702 / 170474840204|28130 / 311891|
|
||||||
|
|4|ENGLISH OPENNLP PORTER|PRIMARY_OUTPUT|283761|1552702|28130|170473287502|1552702 / 170474840204|28130 / 311891|
|
||||||
|
|5|ENGLISH SNOWBALL PORTER2|PRIMARY_OUTPUT|283730|1561891|28161|170473278313|1561891 / 170474840204|28161 / 311891|
|
||||||
|
|6|ENGLISH SNOWBALL ORIGINAL PORTER|PRIMARY_OUTPUT|283675|1550615|28216|170473289589|1550615 / 170474840204|28216 / 311891|
|
||||||
|
|7|ENGLISH PAICE HUSK LANCASTER|PRIMARY_OUTPUT|282398|3045870|29493|170471794334|3045870 / 170474840204|29493 / 311891|
|
||||||
|
|8|ENGLISH LUCENE KSTEM FILTER|PRIMARY_OUTPUT|237551|1367069|74340|170473473135|1367069 / 170474840204|74340 / 311891|
|
||||||
|
|9|ENGLISH LUCENE MINIMAL FILTER|PRIMARY_OUTPUT|136932|1120871|174959|170473719333|1120871 / 170474840204|174959 / 311891|
|
||||||
|
|10|HUNSPELL ENGLISH LUCENE FILTER|PRIMARY_OUTPUT|45926|1978041|265965|170472862163|1978041 / 170474840204|265965 / 311891|
|
||||||
|
|11|ENGLISH LUCENE POSSESSIVE FILTER|PRIMARY_OUTPUT|5|1113773|311886|170473726431|1113773 / 170474840204|311886 / 311891|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 170474840204 (0.000000%)|0 / 311891 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|2|HUNSPELL ENGLISH LUCENE FILTER|ANY_CANDIDATE|0.581994|1974950 / 170474840204 (0.001158%)|260741 / 311891 (83.600040%)|0.030387|0.043756|0.064341|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|2|HUNSPELL ENGLISH LUCENE FILTER|ANY_CANDIDATE|0.025246|0.164000|0.999988|0.581994|0.999987|0.000013|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|2|HUNSPELL ENGLISH LUCENE FILTER|ANY_CANDIDATE|0.030387|0.043756|0.078123|0.022367|0.064345|0.064341|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|HUNSPELL ENGLISH LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|311891|0|0|170474840204|0 / 170474840204|0 / 311891|
|
||||||
|
|2|HUNSPELL ENGLISH LUCENE FILTER|ANY_CANDIDATE|51150|1974950|260741|170472865254|1974950 / 170474840204|260741 / 311891|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999966|11470018 / 170474840204 (0.006728%)|0 / 311891 (0.000000%)|0.032872|0.051579|0.162697|
|
||||||
|
|2|HUNSPELL ENGLISH LUCENE FILTER|ALL_CANDIDATES|0.581994|2004598 / 170474840204 (0.001176%)|260741 / 311891 (83.600040%)|0.029965|0.043208|0.063875|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.026472|1.000000|0.999933|0.999966|0.999933|0.000067|
|
||||||
|
|2|HUNSPELL ENGLISH LUCENE FILTER|ALL_CANDIDATES|0.024881|0.164000|0.999988|0.581994|0.999987|0.000013|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.032872|0.051579|0.119687|0.026472|0.162702|0.162697|
|
||||||
|
|2|HUNSPELL ENGLISH LUCENE FILTER|ALL_CANDIDATES|0.029965|0.043208|0.077422|0.022081|0.063879|0.063875|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|HUNSPELL ENGLISH LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|311891|11470018|0|170463370186|11470018 / 170474840204|0 / 311891|
|
||||||
|
|2|HUNSPELL ENGLISH LUCENE FILTER|ALL_CANDIDATES|51150|2004598|260741|170472835606|2004598 / 170474840204|260741 / 311891|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|21319|1148489|10321529|28826|4.936720%|1355|2812871|
|
||||||
|
|HUNSPELL ENGLISH LUCENE FILTER|5224|3091|26557|6786|1.162165%|4|590716|
|
||||||
|
|
||||||
|
### Output Policies and Metric Definitions
|
||||||
|
|
||||||
|
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
|
||||||
|
|
||||||
|
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
|
||||||
|
|
||||||
|
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
|
||||||
|
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
|
||||||
|
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
|
||||||
|
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
|
||||||
|
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
|
||||||
|
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
|
||||||
|
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
|
||||||
|
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
|
||||||
|
- Jaccard index: `TP / (TP + FP + FN)`.
|
||||||
|
- Fowlkes–Mallows index: `sqrt(precision * recall)`.
|
||||||
|
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
|
||||||
|
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
|
||||||
|
|
||||||
|
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
|
||||||
|
|
||||||
|
### Provenance
|
||||||
|
|
||||||
|
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
|
||||||
|
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
|
||||||
|
- Evaluation command: `./gradlew stemmingQuality`
|
||||||
|
- Dictionary language: `US_UK`
|
||||||
|
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
|
||||||
|
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
|
||||||
|
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:END -->
|
||||||
|
|||||||
@@ -26,7 +26,7 @@ Radixor stores the preferred transformation for each normalized dictionary word
|
|||||||
|
|
||||||
## Accuracy
|
## Accuracy
|
||||||
|
|
||||||
Accuracy is computed from one deterministic JMH measurement iteration without warmup. The benchmark may execute the full dictionary pass more than once inside that single timed iteration; percentages divide matching counters by evaluated counters from the same iteration.
|
Accuracy is computed from JMH auxiliary counters in the current report. The counters are deterministic for a fixed corpus and stemmer; percentages divide matching counters by evaluated counters from the same report and are not timing metrics.
|
||||||
|
|
||||||
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
||||||
| --- | ---: | ---: | ---: | --- |
|
| --- | ---: | ---: | ---: | --- |
|
||||||
@@ -41,10 +41,10 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
|
|
||||||
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
||||||
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
||||||
| Radixor | `finnishRadixor` | 228.248 | 9.245 | 130.1 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
| Radixor | `finnishRadixor` | 308.076 | 15.529 | 175.6 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
||||||
| Lucene FinnishLightStemFilter | `finnishLuceneFinnishLightStemFilter` | 175.923 | 87.177 | 100.3 | 0.771 | Light Finnish suffix stemmer. |
|
| Lucene FinnishLightStemFilter | `finnishLuceneFinnishLightStemFilter` | 175.250 | 46.995 | 99.9 | 0.569 | Light Finnish suffix stemmer. |
|
||||||
| Official Snowball direct | `snowballDirect[FINNISH]` | 265.579 | 95.687 | 151.4 | 1.164 | Official Snowball generated Java stemmer; direct API. |
|
| Official Snowball direct | `snowballDirect[FINNISH]` | 264.652 | 63.054 | 150.8 | 0.859 | Official Snowball generated Java stemmer; direct API. |
|
||||||
| Lucene SnowballFilter | `luceneSnowballFilter[FINNISH]` | 338.099 | 175.101 | 192.7 | 1.481 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
| Lucene SnowballFilter | `luceneSnowballFilter[FINNISH]` | 374.883 | 238.157 | 213.6 | 1.217 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
||||||
|
|
||||||
## Interpretation Notes
|
## Interpretation Notes
|
||||||
|
|
||||||
@@ -53,3 +53,356 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
||||||
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
||||||
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:START -->
|
||||||
|
|
||||||
|
## Stemming Quality
|
||||||
|
|
||||||
|
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `FI_FI` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
|
||||||
|
|
||||||
|
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
|
||||||
|
|
||||||
|
### Evaluation Scope and Key Findings
|
||||||
|
|
||||||
|
The dictionary resource is `src/main/resources/fi_fi/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
|
||||||
|
|
||||||
|
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.984594** among 4 deterministic stemmers. The runner-up is `SNOWBALL FINNISH LUCENE FILTER` at 0.740353, a difference of 0.244242. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.988068** among 4 deterministic stemmers. The runner-up is `SNOWBALL FINNISH DIRECT` at 0.738400, a difference of 0.249668. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
### `ALL_WORDS`
|
||||||
|
|
||||||
|
This mode contains **6 result rows**, **4 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.984594|731279 / 1641126814491 (0.000045%)|971268 / 31523695 (3.081073%)|0.975128|0.972893|0.972899|
|
||||||
|
|2|SNOWBALL FINNISH LUCENE FILTER|PRIMARY_OUTPUT|0.740353|1922153 / 1641126814491 (0.000117%)|16370057 / 31523695 (51.929372%)|0.758996|0.623613|0.653138|
|
||||||
|
|3|SNOWBALL FINNISH DIRECT|PRIMARY_OUTPUT|0.739729|1544812 / 1641126814491 (0.000094%)|16409363 / 31523695 (52.054060%)|0.769880|0.627374|0.659540|
|
||||||
|
|4|FINNISH LUCENE FINNISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.695969|2223150 / 1641126814491 (0.000135%)|19168306 / 31523695 (60.806025%)|0.687649|0.536000|0.576338|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.976624|0.969189|1.000000|0.984594|0.999999|0.000001|
|
||||||
|
|2|SNOWBALL FINNISH LUCENE FILTER|PRIMARY_OUTPUT|0.887434|0.480706|0.999999|0.740353|0.999989|0.000011|
|
||||||
|
|3|SNOWBALL FINNISH DIRECT|PRIMARY_OUTPUT|0.907269|0.479459|0.999999|0.739729|0.999989|0.000011|
|
||||||
|
|4|FINNISH LUCENE FINNISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.847505|0.391940|0.999999|0.695969|0.999987|0.000013|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.975128|0.972893|0.970667|0.947216|0.972900|0.972899|
|
||||||
|
|2|SNOWBALL FINNISH LUCENE FILTER|PRIMARY_OUTPUT|0.758996|0.623613|0.529216|0.453080|0.653142|0.653138|
|
||||||
|
|3|SNOWBALL FINNISH DIRECT|PRIMARY_OUTPUT|0.769880|0.627374|0.529384|0.457061|0.659544|0.659540|
|
||||||
|
|4|FINNISH LUCENE FINNISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.687649|0.536000|0.439152|0.366120|0.576343|0.576338|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.972892|0.996085|0.993746|0.994914|0.994914|
|
||||||
|
|2|SNOWBALL FINNISH LUCENE FILTER|PRIMARY_OUTPUT|0.623608|0.990718|0.904385|0.945585|0.945585|
|
||||||
|
|3|SNOWBALL FINNISH DIRECT|PRIMARY_OUTPUT|0.627369|0.991872|0.904139|0.945975|0.945975|
|
||||||
|
|4|FINNISH LUCENE FINNISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.535994|0.988126|0.886473|0.934544|0.934544|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|30552427|731279|971268|1641126083212|731279 / 1641126814491|971268 / 31523695|
|
||||||
|
|2|SNOWBALL FINNISH LUCENE FILTER|PRIMARY_OUTPUT|15153638|1922153|16370057|1641124892338|1922153 / 1641126814491|16370057 / 31523695|
|
||||||
|
|3|SNOWBALL FINNISH DIRECT|PRIMARY_OUTPUT|15114332|1544812|16409363|1641125269679|1544812 / 1641126814491|16409363 / 31523695|
|
||||||
|
|4|FINNISH LUCENE FINNISH LIGHT STEM FILTER|PRIMARY_OUTPUT|12355389|2223150|19168306|1641124591341|2223150 / 1641126814491|19168306 / 31523695|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 1641126814491 (0.000000%)|0 / 31523695 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|31523695|0|0|1641126814491|0 / 1641126814491|0 / 31523695|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999999|1683575 / 1641126814491 (0.000103%)|0 / 31523695 (0.000000%)|0.959025|0.973991|0.974320|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.949301|1.000000|0.999999|0.999999|0.999999|0.000001|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.959025|0.973991|0.989432|0.949301|0.974321|0.974320|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|31523695|1683575|0|1641125130916|1683575 / 1641126814491|0 / 31523695|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|971268|731279|952296|57328|3.164291%|6|1876272|
|
||||||
|
|
||||||
|
### `LOWERCASE_GROUPS_ONLY`
|
||||||
|
|
||||||
|
This mode contains **6 result rows**, **4 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.988068|730145 / 1543589444152 (0.000047%)|735305 / 30813833 (2.386282%)|0.976268|0.976219|0.976218|
|
||||||
|
|2|SNOWBALL FINNISH DIRECT|PRIMARY_OUTPUT|0.738400|1513705 / 1543589444152 (0.000098%)|16121763 / 30813833 (52.319888%)|0.768117|0.624934|0.657464|
|
||||||
|
|3|SNOWBALL FINNISH LUCENE FILTER|PRIMARY_OUTPUT|0.738400|1513705 / 1543589444152 (0.000098%)|16121763 / 30813833 (52.319888%)|0.768117|0.624934|0.657464|
|
||||||
|
|4|FINNISH LUCENE FINNISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.694529|1806392 / 1543589444152 (0.000117%)|18825444 / 30813833 (61.094133%)|0.697056|0.537492|0.581469|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.976301|0.976137|1.000000|0.988068|0.999999|0.000001|
|
||||||
|
|2|SNOWBALL FINNISH DIRECT|PRIMARY_OUTPUT|0.906595|0.476801|0.999999|0.738400|0.999989|0.000011|
|
||||||
|
|3|SNOWBALL FINNISH LUCENE FILTER|PRIMARY_OUTPUT|0.906595|0.476801|0.999999|0.738400|0.999989|0.000011|
|
||||||
|
|4|FINNISH LUCENE FINNISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.869053|0.389059|0.999999|0.694529|0.999987|0.000013|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.976268|0.976219|0.976170|0.953543|0.976219|0.976218|
|
||||||
|
|2|SNOWBALL FINNISH DIRECT|PRIMARY_OUTPUT|0.768117|0.624934|0.526744|0.454475|0.657469|0.657464|
|
||||||
|
|3|SNOWBALL FINNISH LUCENE FILTER|PRIMARY_OUTPUT|0.768117|0.624934|0.526744|0.454475|0.657469|0.657464|
|
||||||
|
|4|FINNISH LUCENE FINNISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.697056|0.537492|0.437372|0.367514|0.581474|0.581469|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.976218|0.996000|0.996069|0.996035|0.996035|
|
||||||
|
|2|SNOWBALL FINNISH DIRECT|PRIMARY_OUTPUT|0.624929|0.991732|0.902933|0.945252|0.945252|
|
||||||
|
|3|SNOWBALL FINNISH LUCENE FILTER|PRIMARY_OUTPUT|0.624929|0.991732|0.902933|0.945252|0.945252|
|
||||||
|
|4|FINNISH LUCENE FINNISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.537486|0.989268|0.885294|0.934397|0.934397|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|30078528|730145|735305|1543588714007|730145 / 1543589444152|735305 / 30813833|
|
||||||
|
|2|SNOWBALL FINNISH DIRECT|PRIMARY_OUTPUT|14692070|1513705|16121763|1543587930447|1513705 / 1543589444152|16121763 / 30813833|
|
||||||
|
|3|SNOWBALL FINNISH LUCENE FILTER|PRIMARY_OUTPUT|14692070|1513705|16121763|1543587930447|1513705 / 1543589444152|16121763 / 30813833|
|
||||||
|
|4|FINNISH LUCENE FINNISH LIGHT STEM FILTER|PRIMARY_OUTPUT|11988389|1806392|18825444|1543587637760|1806392 / 1543589444152|18825444 / 30813833|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 1543589444152 (0.000000%)|0 / 30813833 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|30813833|0|0|1543589444152|0 / 1543589444152|0 / 30813833|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999999|1653320 / 1543589444152 (0.000107%)|0 / 30813833 (0.000000%)|0.958843|0.973873|0.974205|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.949077|1.000000|0.999999|0.999999|0.999999|0.000001|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.958843|0.973873|0.989383|0.949077|0.974206|0.974205|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|30813833|1653320|0|1543587790832|1653320 / 1543589444152|0 / 30813833|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|735305|730145|923175|44331|2.523029%|6|1805864|
|
||||||
|
|
||||||
|
### Output Policies and Metric Definitions
|
||||||
|
|
||||||
|
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
|
||||||
|
|
||||||
|
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
|
||||||
|
|
||||||
|
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
|
||||||
|
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
|
||||||
|
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
|
||||||
|
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
|
||||||
|
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
|
||||||
|
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
|
||||||
|
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
|
||||||
|
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
|
||||||
|
- Jaccard index: `TP / (TP + FP + FN)`.
|
||||||
|
- Fowlkes–Mallows index: `sqrt(precision * recall)`.
|
||||||
|
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
|
||||||
|
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
|
||||||
|
|
||||||
|
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
|
||||||
|
|
||||||
|
### Provenance
|
||||||
|
|
||||||
|
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
|
||||||
|
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
|
||||||
|
- Evaluation command: `./gradlew stemmingQuality`
|
||||||
|
- Dictionary language: `FI_FI`
|
||||||
|
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
|
||||||
|
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
|
||||||
|
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:END -->
|
||||||
|
|||||||
@@ -26,11 +26,12 @@ Radixor stores the preferred transformation for each normalized dictionary word
|
|||||||
|
|
||||||
## Accuracy
|
## Accuracy
|
||||||
|
|
||||||
Accuracy is computed from one deterministic JMH measurement iteration without warmup. The benchmark may execute the full dictionary pass more than once inside that single timed iteration; percentages divide matching counters by evaluated counters from the same iteration.
|
Accuracy is computed from JMH auxiliary counters in the current report. The counters are deterministic for a fixed corpus and stemmer; percentages divide matching counters by evaluated counters from the same report and are not timing metrics.
|
||||||
|
|
||||||
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
||||||
| --- | ---: | ---: | ---: | --- |
|
| --- | ---: | ---: | ---: | --- |
|
||||||
| Radixor | 94.831% | 94.859% | 94.734% | Full Radixor dictionary patch-command stemmer. |
|
| Radixor | 94.831% | 94.859% | 94.734% | Full Radixor dictionary patch-command stemmer. |
|
||||||
|
| Lucene HunspellStemFilter | 68.923% | 63.617% | 86.876% | Benchmark-only French Hunspell dictionary compared via Lucene HunspellStemFilter. |
|
||||||
| Lucene FrenchMinimalStemFilter | 11.472% | 6.236% | 29.192% | Minimal suffix reducer; narrow baseline, not a full stemmer. |
|
| Lucene FrenchMinimalStemFilter | 11.472% | 6.236% | 29.192% | Minimal suffix reducer; narrow baseline, not a full stemmer. |
|
||||||
| Lucene SnowballFilter | 8.551% | 5.183% | 19.952% | Lucene TokenFilter integration path around the Snowball algorithm. |
|
| Lucene SnowballFilter | 8.551% | 5.183% | 19.952% | Lucene TokenFilter integration path around the Snowball algorithm. |
|
||||||
| Official Snowball direct | 8.462% | 5.067% | 19.952% | Official Snowball generated Java stemmer; rule-based suffix algorithm. |
|
| Official Snowball direct | 8.462% | 5.067% | 19.952% | Official Snowball generated Java stemmer; rule-based suffix algorithm. |
|
||||||
@@ -42,11 +43,12 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
|
|
||||||
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
||||||
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
||||||
| Radixor | `frenchRadixor` | 38.598 | 5.425 | 105.5 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
| Radixor | `frenchRadixor` | 47.033 | 4.146 | 128.5 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
||||||
| Lucene FrenchMinimalStemFilter | `frenchLuceneFrenchMinimalStemFilter` | 17.657 | 1.956 | 48.2 | 0.457 | Minimal French suffix reducer; narrow baseline. |
|
| Lucene HunspellStemFilter | `luceneHunspellStemFilter` | 1664.935 | 65.928 | 4549.4 | 35.399 | Benchmark-only French Hunspell dictionary compared via Lucene HunspellStemFilter. |
|
||||||
| Lucene FrenchLightStemFilter | `frenchLuceneFrenchLightStemFilter` | 28.742 | 2.391 | 78.5 | 0.745 | Light French suffix stemmer. |
|
| Lucene FrenchMinimalStemFilter | `frenchLuceneFrenchMinimalStemFilter` | 19.234 | 2.098 | 52.6 | 0.409 | Minimal French suffix reducer; narrow baseline. |
|
||||||
| Official Snowball direct | `snowballDirect[FRENCH]` | 104.983 | 10.116 | 286.9 | 2.720 | Official Snowball generated Java stemmer; direct API. |
|
| Lucene FrenchLightStemFilter | `frenchLuceneFrenchLightStemFilter` | 30.560 | 3.680 | 83.5 | 0.650 | Light French suffix stemmer. |
|
||||||
| Lucene SnowballFilter | `luceneSnowballFilter[FRENCH]` | 117.938 | 4.007 | 322.3 | 3.056 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
| Official Snowball direct | `snowballDirect[FRENCH]` | 111.057 | 8.172 | 303.5 | 2.361 | Official Snowball generated Java stemmer; direct API. |
|
||||||
|
| Lucene SnowballFilter | `luceneSnowballFilter[FRENCH]` | 123.648 | 3.500 | 337.9 | 2.629 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
||||||
|
|
||||||
## Interpretation Notes
|
## Interpretation Notes
|
||||||
|
|
||||||
@@ -55,3 +57,398 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
||||||
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
||||||
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:START -->
|
||||||
|
|
||||||
|
## Stemming Quality
|
||||||
|
|
||||||
|
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `FR_FR` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
|
||||||
|
|
||||||
|
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
|
||||||
|
|
||||||
|
### Evaluation Scope and Key Findings
|
||||||
|
|
||||||
|
The dictionary resource is `src/main/resources/fr_fr/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
|
||||||
|
|
||||||
|
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.956992** among 6 deterministic stemmers. The runner-up is `SNOWBALL FRENCH DIRECT` at 0.845262, a difference of 0.111731. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.957224** among 6 deterministic stemmers. The runner-up is `SNOWBALL FRENCH DIRECT` at 0.845414, a difference of 0.111810. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
### `ALL_WORDS`
|
||||||
|
|
||||||
|
This mode contains **10 result rows**, **6 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.956992|318767 / 90396104830 (0.000353%)|469160 / 5454615 (8.601157%)|0.934603|0.926765|0.926851|
|
||||||
|
|2|SNOWBALL FRENCH DIRECT|PRIMARY_OUTPUT|0.845262|1654723 / 90396104830 (0.001831%)|1687975 / 5454615 (30.945814%)|0.693926|0.692653|0.692638|
|
||||||
|
|3|SNOWBALL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.844999|1661388 / 90396104830 (0.001838%)|1690838 / 5454615 (30.998301%)|0.693010|0.691885|0.691869|
|
||||||
|
|4|HUNSPELL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.813742|776728 / 90396104830 (0.000859%)|2031881 / 5454615 (37.250677%)|0.769069|0.709075|0.715131|
|
||||||
|
|5|FRENCH LUCENE FRENCH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.518587|276403 / 90396104830 (0.000306%)|5251833 / 5454615 (96.282377%)|0.137547|0.068348|0.125415|
|
||||||
|
|6|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.516830|160438 / 90396104830 (0.000177%)|5271003 / 5454615 (96.633823%)|0.134400|0.063329|0.134021|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.939903|0.913988|0.999996|0.956992|0.999991|0.000009|
|
||||||
|
|2|SNOWBALL FRENCH DIRECT|PRIMARY_OUTPUT|0.694777|0.690542|0.999982|0.845262|0.999963|0.000037|
|
||||||
|
|3|SNOWBALL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.693763|0.690017|0.999982|0.844999|0.999963|0.000037|
|
||||||
|
|4|HUNSPELL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.815041|0.627493|0.999991|0.813742|0.999969|0.000031|
|
||||||
|
|5|FRENCH LUCENE FRENCH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.423181|0.037176|0.999997|0.518587|0.999939|0.000061|
|
||||||
|
|6|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.533678|0.033662|0.999998|0.516830|0.999940|0.000060|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.934603|0.926765|0.919056|0.863524|0.926855|0.926851|
|
||||||
|
|2|SNOWBALL FRENCH DIRECT|PRIMARY_OUTPUT|0.693926|0.692653|0.691385|0.529816|0.692656|0.692638|
|
||||||
|
|3|SNOWBALL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.693010|0.691885|0.690763|0.528917|0.691887|0.691869|
|
||||||
|
|4|HUNSPELL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.769069|0.709075|0.657765|0.549277|0.715145|0.715131|
|
||||||
|
|5|FRENCH LUCENE FRENCH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.137547|0.068348|0.045472|0.035383|0.125428|0.125415|
|
||||||
|
|6|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.134400|0.063329|0.041424|0.032700|0.134032|0.134021|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.926760|0.988772|0.985214|0.986990|0.986990|
|
||||||
|
|2|SNOWBALL FRENCH DIRECT|PRIMARY_OUTPUT|0.692635|0.959459|0.944948|0.952148|0.952148|
|
||||||
|
|3|SNOWBALL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.691866|0.958698|0.944715|0.951655|0.951655|
|
||||||
|
|4|HUNSPELL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.709060|0.978337|0.913706|0.944918|0.944918|
|
||||||
|
|5|FRENCH LUCENE FRENCH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.068339|0.974110|0.812376|0.885922|0.885922|
|
||||||
|
|6|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.063322|0.984019|0.810979|0.889158|0.889158|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|4985455|318767|469160|90395786063|318767 / 90396104830|469160 / 5454615|
|
||||||
|
|2|SNOWBALL FRENCH DIRECT|PRIMARY_OUTPUT|3766640|1654723|1687975|90394450107|1654723 / 90396104830|1687975 / 5454615|
|
||||||
|
|3|SNOWBALL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|3763777|1661388|1690838|90394443442|1661388 / 90396104830|1690838 / 5454615|
|
||||||
|
|4|HUNSPELL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|3422734|776728|2031881|90395328102|776728 / 90396104830|2031881 / 5454615|
|
||||||
|
|5|FRENCH LUCENE FRENCH LIGHT STEM FILTER|PRIMARY_OUTPUT|202782|276403|5251833|90395828427|276403 / 90396104830|5251833 / 5454615|
|
||||||
|
|6|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|PRIMARY_OUTPUT|183612|160438|5271003|90395944392|160438 / 90396104830|5271003 / 5454615|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|0.999979|12 / 90396104830 (0.000000%)|232 / 5454615 (0.004253%)|0.999990|0.999978|0.999978|
|
||||||
|
|2|HUNSPELL FRENCH LUCENE FILTER|ANY_CANDIDATE|0.830964|745831 / 90396104830 (0.000825%)|1844003 / 5454615 (33.806291%)|0.789019|0.736029|0.740670|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|0.999998|0.999957|1.000000|0.999979|1.000000|0.000000|
|
||||||
|
|2|HUNSPELL FRENCH LUCENE FILTER|ANY_CANDIDATE|0.828798|0.661937|0.999992|0.830964|0.999971|0.000029|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|0.999990|0.999978|0.999966|0.999955|0.999978|0.999978|
|
||||||
|
|2|HUNSPELL FRENCH LUCENE FILTER|ANY_CANDIDATE|0.789019|0.736029|0.689709|0.582315|0.740684|0.740670|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|HUNSPELL FRENCH LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|5454383|12|232|90396104818|12 / 90396104830|232 / 5454615|
|
||||||
|
|2|HUNSPELL FRENCH LUCENE FILTER|ANY_CANDIDATE|3610612|745831|1844003|90395358999|745831 / 90396104830|1844003 / 5454615|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999973|1056255 / 90396104830 (0.001168%)|232 / 5454615 (0.004253%)|0.865853|0.911704|0.915270|
|
||||||
|
|2|HUNSPELL FRENCH LUCENE FILTER|ALL_CANDIDATES|0.830963|1043199 / 90396104830 (0.001154%)|1844003 / 5454615 (33.806291%)|0.750028|0.714377|0.716613|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.837765|0.999957|0.999988|0.999973|0.999988|0.000012|
|
||||||
|
|2|HUNSPELL FRENCH LUCENE FILTER|ALL_CANDIDATES|0.775840|0.661937|0.999988|0.830963|0.999968|0.000032|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.865853|0.911704|0.962682|0.837735|0.915275|0.915270|
|
||||||
|
|2|HUNSPELL FRENCH LUCENE FILTER|ALL_CANDIDATES|0.750028|0.714377|0.681961|0.555666|0.716629|0.716613|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|HUNSPELL FRENCH LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|5454383|1056255|232|90395048575|1056255 / 90396104830|232 / 5454615|
|
||||||
|
|2|HUNSPELL FRENCH LUCENE FILTER|ALL_CANDIDATES|3610612|1043199|1844003|90395061631|1043199 / 90396104830|1844003 / 5454615|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|468928|318755|737488|43040|10.122057%|56|477024|
|
||||||
|
|HUNSPELL FRENCH LUCENE FILTER|187878|30897|266471|13511|3.177489%|4|439015|
|
||||||
|
|
||||||
|
### `LOWERCASE_GROUPS_ONLY`
|
||||||
|
|
||||||
|
This mode contains **10 result rows**, **6 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.957224|315266 / 88712126506 (0.000355%)|465436 / 5440559 (8.554930%)|0.935099|0.927248|0.927334|
|
||||||
|
|2|SNOWBALL FRENCH DIRECT|PRIMARY_OUTPUT|0.845414|1646111 / 88712126506 (0.001856%)|1681970 / 5440559 (30.915389%)|0.694508|0.693130|0.693115|
|
||||||
|
|3|SNOWBALL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.845163|1641925 / 88712126506 (0.001851%)|1684703 / 5440559 (30.965623%)|0.694714|0.693068|0.693055|
|
||||||
|
|4|HUNSPELL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.813617|763305 / 88712126506 (0.000860%)|2028011 / 5440559 (37.275784%)|0.770537|0.709734|0.715938|
|
||||||
|
|5|FRENCH LUCENE FRENCH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.518442|262689 / 88712126506 (0.000296%)|5239869 / 5440559 (96.311225%)|0.137571|0.067985|0.126383|
|
||||||
|
|6|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.516697|147476 / 88712126506 (0.000166%)|5258873 / 5440559 (96.660527%)|0.134439|0.062979|0.135757|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.940408|0.914451|0.999996|0.957224|0.999991|0.000009|
|
||||||
|
|2|SNOWBALL FRENCH DIRECT|PRIMARY_OUTPUT|0.695430|0.690846|0.999981|0.845414|0.999962|0.000038|
|
||||||
|
|3|SNOWBALL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.695815|0.690344|0.999981|0.845163|0.999963|0.000037|
|
||||||
|
|4|HUNSPELL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.817210|0.627242|0.999991|0.813617|0.999969|0.000031|
|
||||||
|
|5|FRENCH LUCENE FRENCH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.433101|0.036888|0.999997|0.518442|0.999938|0.000062|
|
||||||
|
|6|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.551965|0.033395|0.999998|0.516697|0.999939|0.000061|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.935099|0.927248|0.919527|0.864363|0.927338|0.927334|
|
||||||
|
|2|SNOWBALL FRENCH DIRECT|PRIMARY_OUTPUT|0.694508|0.693130|0.691758|0.530374|0.693134|0.693115|
|
||||||
|
|3|SNOWBALL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.694714|0.693068|0.691431|0.530302|0.693074|0.693055|
|
||||||
|
|4|HUNSPELL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.770537|0.709734|0.657826|0.550068|0.715953|0.715938|
|
||||||
|
|5|FRENCH LUCENE FRENCH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.137571|0.067985|0.045148|0.035189|0.126397|0.126383|
|
||||||
|
|6|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.134439|0.062979|0.041121|0.032513|0.135767|0.135757|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.927243|0.988916|0.985550|0.987230|0.987230|
|
||||||
|
|2|SNOWBALL FRENCH DIRECT|PRIMARY_OUTPUT|0.693112|0.959521|0.944537|0.951970|0.951970|
|
||||||
|
|3|SNOWBALL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.693050|0.959566|0.944385|0.951915|0.951915|
|
||||||
|
|4|HUNSPELL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|0.709719|0.979328|0.913162|0.945088|0.945088|
|
||||||
|
|5|FRENCH LUCENE FRENCH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.067976|0.975086|0.811144|0.885591|0.885591|
|
||||||
|
|6|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.062973|0.985086|0.809774|0.888868|0.888868|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|4975123|315266|465436|88711811240|315266 / 88712126506|465436 / 5440559|
|
||||||
|
|2|SNOWBALL FRENCH DIRECT|PRIMARY_OUTPUT|3758589|1646111|1681970|88710480395|1646111 / 88712126506|1681970 / 5440559|
|
||||||
|
|3|SNOWBALL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|3755856|1641925|1684703|88710484581|1641925 / 88712126506|1684703 / 5440559|
|
||||||
|
|4|HUNSPELL FRENCH LUCENE FILTER|PRIMARY_OUTPUT|3412548|763305|2028011|88711363201|763305 / 88712126506|2028011 / 5440559|
|
||||||
|
|5|FRENCH LUCENE FRENCH LIGHT STEM FILTER|PRIMARY_OUTPUT|200690|262689|5239869|88711863817|262689 / 88712126506|5239869 / 5440559|
|
||||||
|
|6|FRENCH LUCENE FRENCH MINIMAL STEM FILTER|PRIMARY_OUTPUT|181686|147476|5258873|88711979030|147476 / 88712126506|5258873 / 5440559|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 88712126506 (0.000000%)|0 / 5440559 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|2|HUNSPELL FRENCH LUCENE FILTER|ANY_CANDIDATE|0.830852|733584 / 88712126506 (0.000827%)|1840476 / 5440559 (33.828803%)|0.790351|0.736648|0.741404|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|2|HUNSPELL FRENCH LUCENE FILTER|ANY_CANDIDATE|0.830724|0.661712|0.999992|0.830852|0.999971|0.000029|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|2|HUNSPELL FRENCH LUCENE FILTER|ANY_CANDIDATE|0.790351|0.736648|0.689779|0.583090|0.741418|0.741404|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|HUNSPELL FRENCH LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|5440559|0|0|88712126506|0 / 88712126506|0 / 5440559|
|
||||||
|
|2|HUNSPELL FRENCH LUCENE FILTER|ANY_CANDIDATE|3600083|733584|1840476|88711392922|733584 / 88712126506|1840476 / 5440559|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999995|938985 / 88712126506 (0.001058%)|0 / 5440559 (0.000000%)|0.878679|0.920560|0.923474|
|
||||||
|
|2|HUNSPELL FRENCH LUCENE FILTER|ALL_CANDIDATES|0.830850|1027635 / 88712126506 (0.001158%)|1840476 / 5440559 (33.828803%)|0.751538|0.715134|0.717460|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.852813|1.000000|0.999989|0.999995|0.999989|0.000011|
|
||||||
|
|2|HUNSPELL FRENCH LUCENE FILTER|ALL_CANDIDATES|0.777939|0.661712|0.999988|0.830850|0.999968|0.000032|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.878679|0.920560|0.966634|0.852813|0.923479|0.923474|
|
||||||
|
|2|HUNSPELL FRENCH LUCENE FILTER|ALL_CANDIDATES|0.751538|0.715134|0.682093|0.556582|0.717476|0.717460|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|HUNSPELL FRENCH LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|5440559|938985|0|88711187521|938985 / 88712126506|0 / 5440559|
|
||||||
|
|2|HUNSPELL FRENCH LUCENE FILTER|ALL_CANDIDATES|3600083|1027635|1840476|88711098871|1027635 / 88712126506|1840476 / 5440559|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|465436|315266|623719|41130|9.764239%|56|468574|
|
||||||
|
|HUNSPELL FRENCH LUCENE FILTER|187535|29721|264330|13437|3.189936%|4|434961|
|
||||||
|
|
||||||
|
### Output Policies and Metric Definitions
|
||||||
|
|
||||||
|
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
|
||||||
|
|
||||||
|
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
|
||||||
|
|
||||||
|
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
|
||||||
|
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
|
||||||
|
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
|
||||||
|
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
|
||||||
|
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
|
||||||
|
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
|
||||||
|
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
|
||||||
|
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
|
||||||
|
- Jaccard index: `TP / (TP + FP + FN)`.
|
||||||
|
- Fowlkes–Mallows index: `sqrt(precision * recall)`.
|
||||||
|
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
|
||||||
|
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
|
||||||
|
|
||||||
|
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
|
||||||
|
|
||||||
|
### Provenance
|
||||||
|
|
||||||
|
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
|
||||||
|
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
|
||||||
|
- Evaluation command: `./gradlew stemmingQuality`
|
||||||
|
- Dictionary language: `FR_FR`
|
||||||
|
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
|
||||||
|
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
|
||||||
|
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:END -->
|
||||||
|
|||||||
@@ -26,16 +26,18 @@ Radixor stores the preferred transformation for each normalized dictionary word
|
|||||||
|
|
||||||
## Accuracy
|
## Accuracy
|
||||||
|
|
||||||
Accuracy is computed from one deterministic JMH measurement iteration without warmup. The benchmark may execute the full dictionary pass more than once inside that single timed iteration; percentages divide matching counters by evaluated counters from the same iteration.
|
Accuracy is computed from JMH auxiliary counters in the current report. The counters are deterministic for a fixed corpus and stemmer; percentages divide matching counters by evaluated counters from the same report and are not timing metrics.
|
||||||
|
|
||||||
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
||||||
| --- | ---: | ---: | ---: | --- |
|
| --- | ---: | ---: | ---: | --- |
|
||||||
| Radixor | 97.455% | 97.973% | 96.476% | Full Radixor dictionary patch-command stemmer. |
|
| Radixor | 92.725% | 92.847% | 92.396% | Full Radixor dictionary patch-command stemmer. |
|
||||||
| Lucene GermanLightStemFilter | 38.583% | 35.800% | 43.849% | Light suffix stemmer; intentionally narrower than a dictionary-derived stemmer. |
|
| Lucene HunspellStemFilter | 47.064% | 29.661% | 93.678% | Benchmark-only German Hunspell dictionary compared via Lucene HunspellStemFilter. |
|
||||||
| Lucene GermanMinimalStemFilter | 37.492% | 38.538% | 35.513% | Minimal suffix reducer; narrow baseline, not a full stemmer. |
|
| CISTEM (German) | 24.675% | 23.724% | 27.222% | Benchmark-only CISTEM implementation. |
|
||||||
| Lucene SnowballFilter | 33.380% | 30.939% | 37.999% | Lucene TokenFilter integration path around the Snowball algorithm. |
|
| Lucene GermanLightStemFilter | 37.434% | 35.465% | 42.707% | Light suffix stemmer; intentionally narrower than a dictionary-derived stemmer. |
|
||||||
| Official Snowball direct | 32.863% | 31.225% | 35.963% | Official Snowball generated Java stemmer; rule-based suffix algorithm. |
|
| Lucene GermanMinimalStemFilter | 27.640% | 24.951% | 34.844% | Minimal suffix reducer; narrow baseline, not a full stemmer. |
|
||||||
| Lucene GermanStemFilter | 26.168% | 24.979% | 28.416% | German Lucene stemming TokenFilter; broader than minimal/light variants. |
|
| Lucene SnowballFilter | 30.956% | 28.853% | 36.589% | Lucene TokenFilter integration path around the Snowball algorithm. |
|
||||||
|
| Official Snowball direct | 30.481% | 29.027% | 34.376% | Official Snowball generated Java stemmer; rule-based suffix algorithm. |
|
||||||
|
| Lucene GermanStemFilter | 21.559% | 19.312% | 27.576% | German Lucene stemming TokenFilter; broader than minimal/light variants. |
|
||||||
|
|
||||||
## Speed
|
## Speed
|
||||||
|
|
||||||
@@ -43,12 +45,14 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
|
|
||||||
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
||||||
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
||||||
| Radixor | `germanRadixor` | 9.518 | 0.338 | 68.2 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
| Radixor | `germanRadixor` | 41.166 | 2.396 | 294.8 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
||||||
| Lucene GermanMinimalStemFilter | `germanLuceneGermanMinimalStemFilter` | 12.020 | 0.460 | 86.1 | 1.263 | Minimal German suffix reduction; narrow baseline. |
|
| CISTEM | `germanCistem` | 248.392 | 12.294 | 1778.8 | 6.034 | Benchmark-only CISTEM implementation. |
|
||||||
| Lucene GermanLightStemFilter | `germanLuceneGermanLightStemFilter` | 12.413 | 1.059 | 88.9 | 1.304 | Light German suffix stemmer; narrower than a dictionary stemmer. |
|
| Lucene HunspellStemFilter | `luceneHunspellStemFilter` | 281.322 | 3.411 | 2014.6 | 6.834 | Benchmark-only German Hunspell dictionary compared via Lucene HunspellStemFilter. |
|
||||||
| Lucene GermanStemFilter | `germanLuceneGermanStemFilter` | 36.644 | 5.667 | 262.4 | 3.850 | Older German stemming TokenFilter with normalization requirements. |
|
| Lucene GermanMinimalStemFilter | `germanLuceneGermanMinimalStemFilter` | 23.562 | 0.969 | 168.7 | 0.572 | Minimal German suffix reduction; narrow baseline. |
|
||||||
| Lucene SnowballFilter | `luceneSnowballFilter[GERMAN]` | 54.846 | 8.710 | 392.8 | 5.762 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
| Lucene GermanLightStemFilter | `germanLuceneGermanLightStemFilter` | 24.410 | 1.034 | 174.8 | 0.593 | Light German suffix stemmer; narrower than a dictionary stemmer. |
|
||||||
| Official Snowball direct | `snowballDirect[GERMAN]` | 52.974 | 7.989 | 379.4 | 5.566 | Official Snowball generated Java stemmer; direct API. |
|
| Lucene GermanStemFilter | `germanLuceneGermanStemFilter` | 71.039 | 4.443 | 508.7 | 1.726 | Older German stemming TokenFilter with normalization requirements. |
|
||||||
|
| Lucene SnowballFilter | `luceneSnowballFilter[GERMAN]` | 105.771 | 9.617 | 757.4 | 2.569 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
||||||
|
| Official Snowball direct | `snowballDirect[GERMAN]` | 100.688 | 9.018 | 721.0 | 2.446 | Official Snowball generated Java stemmer; direct API. |
|
||||||
|
|
||||||
## Interpretation Notes
|
## Interpretation Notes
|
||||||
|
|
||||||
@@ -57,3 +61,418 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
||||||
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
||||||
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:START -->
|
||||||
|
|
||||||
|
## Stemming Quality
|
||||||
|
|
||||||
|
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `DE_DE` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
|
||||||
|
|
||||||
|
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
|
||||||
|
|
||||||
|
### Evaluation Scope and Key Findings
|
||||||
|
|
||||||
|
The dictionary resource is `src/main/resources/de_de/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
|
||||||
|
|
||||||
|
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.907901** among 8 deterministic stemmers. The runner-up is `GERMAN CISTEM` at 0.880770, a difference of 0.027131. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.966157** among 8 deterministic stemmers. The runner-up is `GERMAN CISTEM` at 0.915288, a difference of 0.050869. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
### `ALL_WORDS`
|
||||||
|
|
||||||
|
This mode contains **12 result rows**, **8 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.907901|98192 / 44095245979 (0.000223%)|254903 / 1383872 (18.419550%)|0.897073|0.864768|0.866326|
|
||||||
|
|2|GERMAN CISTEM|PRIMARY_OUTPUT|0.880770|477122 / 44095245979 (0.001082%)|329983 / 1383872 (23.844908%)|0.701852|0.723109|0.724023|
|
||||||
|
|3|SNOWBALL GERMAN DIRECT|PRIMARY_OUTPUT|0.778614|190680 / 44095245979 (0.000432%)|612734 / 1383872 (44.276783%)|0.737064|0.657494|0.668394|
|
||||||
|
|4|SNOWBALL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.771357|295701 / 44095245979 (0.000671%)|632816 / 1383872 (45.727929%)|0.674089|0.617993|0.624014|
|
||||||
|
|5|GERMAN LUCENE GERMAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.756258|205740 / 44095245979 (0.000467%)|674609 / 1383872 (48.747933%)|0.703092|0.617052|0.630292|
|
||||||
|
|6|GERMAN LUCENE GERMAN STEM FILTER|PRIMARY_OUTPUT|0.723772|331871 / 44095245979 (0.000753%)|764518 / 1383872 (55.244849%)|0.596821|0.530474|0.539809|
|
||||||
|
|7|HUNSPELL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.641579|203883 / 44095245979 (0.000462%)|992010 / 1383872 (71.683653%)|0.520145|0.395897|0.431563|
|
||||||
|
|8|GERMAN LUCENE GERMAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.598139|110840 / 44095245979 (0.000251%)|1112246 / 1383872 (80.372029%)|0.466113|0.307558|0.373350|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.919984|0.815804|0.999998|0.907901|0.999992|0.000008|
|
||||||
|
|2|GERMAN CISTEM|PRIMARY_OUTPUT|0.688361|0.761551|0.999989|0.880770|0.999982|0.000018|
|
||||||
|
|3|SNOWBALL GERMAN DIRECT|PRIMARY_OUTPUT|0.801750|0.557232|0.999996|0.778614|0.999982|0.000018|
|
||||||
|
|4|SNOWBALL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.717508|0.542721|0.999993|0.771357|0.999979|0.000021|
|
||||||
|
|5|GERMAN LUCENE GERMAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.775148|0.512521|0.999995|0.756258|0.999980|0.000020|
|
||||||
|
|6|GERMAN LUCENE GERMAN STEM FILTER|PRIMARY_OUTPUT|0.651112|0.447552|0.999992|0.723772|0.999975|0.000025|
|
||||||
|
|7|HUNSPELL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.657768|0.283163|0.999995|0.641579|0.999973|0.000027|
|
||||||
|
|8|GERMAN LUCENE GERMAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.710196|0.196280|0.999997|0.598139|0.999972|0.000028|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.897073|0.864768|0.834709|0.761755|0.866330|0.866326|
|
||||||
|
|2|GERMAN CISTEM|PRIMARY_OUTPUT|0.701852|0.723109|0.745694|0.566304|0.724032|0.724023|
|
||||||
|
|3|SNOWBALL GERMAN DIRECT|PRIMARY_OUTPUT|0.737064|0.657494|0.593429|0.489751|0.668402|0.668394|
|
||||||
|
|4|SNOWBALL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.674089|0.617993|0.570517|0.447171|0.624024|0.624014|
|
||||||
|
|5|GERMAN LUCENE GERMAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.703092|0.617052|0.549774|0.446186|0.630301|0.630292|
|
||||||
|
|6|GERMAN LUCENE GERMAN STEM FILTER|PRIMARY_OUTPUT|0.596821|0.530474|0.477402|0.360983|0.539820|0.539809|
|
||||||
|
|7|HUNSPELL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.520145|0.395897|0.319562|0.246803|0.431574|0.431563|
|
||||||
|
|8|GERMAN LUCENE GERMAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.466113|0.307558|0.229493|0.181725|0.373359|0.373350|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.864764|0.989946|0.975085|0.982460|0.982460|
|
||||||
|
|2|GERMAN CISTEM|PRIMARY_OUTPUT|0.723100|0.974048|0.975147|0.974597|0.974597|
|
||||||
|
|3|SNOWBALL GERMAN DIRECT|PRIMARY_OUTPUT|0.657485|0.983725|0.949324|0.966218|0.966218|
|
||||||
|
|4|SNOWBALL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.617983|0.975845|0.942925|0.959102|0.959102|
|
||||||
|
|5|GERMAN LUCENE GERMAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.617043|0.980753|0.936533|0.958133|0.958133|
|
||||||
|
|6|GERMAN LUCENE GERMAN STEM FILTER|PRIMARY_OUTPUT|0.530462|0.975550|0.942890|0.958942|0.958942|
|
||||||
|
|7|HUNSPELL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.395885|0.980463|0.886873|0.931322|0.931322|
|
||||||
|
|8|GERMAN LUCENE GERMAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.307549|0.983615|0.896264|0.937910|0.937910|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|1128969|98192|254903|44095147787|98192 / 44095245979|254903 / 1383872|
|
||||||
|
|2|GERMAN CISTEM|PRIMARY_OUTPUT|1053889|477122|329983|44094768857|477122 / 44095245979|329983 / 1383872|
|
||||||
|
|3|SNOWBALL GERMAN DIRECT|PRIMARY_OUTPUT|771138|190680|612734|44095055299|190680 / 44095245979|612734 / 1383872|
|
||||||
|
|4|SNOWBALL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|751056|295701|632816|44094950278|295701 / 44095245979|632816 / 1383872|
|
||||||
|
|5|GERMAN LUCENE GERMAN LIGHT STEM FILTER|PRIMARY_OUTPUT|709263|205740|674609|44095040239|205740 / 44095245979|674609 / 1383872|
|
||||||
|
|6|GERMAN LUCENE GERMAN STEM FILTER|PRIMARY_OUTPUT|619354|331871|764518|44094914108|331871 / 44095245979|764518 / 1383872|
|
||||||
|
|7|HUNSPELL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|391862|203883|992010|44095042096|203883 / 44095245979|992010 / 1383872|
|
||||||
|
|8|GERMAN LUCENE GERMAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|271626|110840|1112246|44095135139|110840 / 44095245979|1112246 / 1383872|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|0.959835|1375 / 44095245979 (0.000003%)|111167 / 1383872 (8.033041%)|0.981996|0.957658|0.958475|
|
||||||
|
|2|HUNSPELL GERMAN LUCENE FILTER|ANY_CANDIDATE|0.647474|158403 / 44095245979 (0.000359%)|975697 / 1383872 (70.504859%)|0.559116|0.418544|0.460956|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|0.998921|0.919670|1.000000|0.959835|0.999997|0.000003|
|
||||||
|
|2|HUNSPELL GERMAN LUCENE FILTER|ANY_CANDIDATE|0.720422|0.294951|0.999996|0.647474|0.999974|0.000026|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|0.981996|0.957658|0.934498|0.918757|0.958476|0.958475|
|
||||||
|
|2|HUNSPELL GERMAN LUCENE FILTER|ANY_CANDIDATE|0.559116|0.418544|0.334456|0.264658|0.460966|0.460956|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|HUNSPELL GERMAN LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1272705|1375|111167|44095244604|1375 / 44095245979|111167 / 1383872|
|
||||||
|
|2|HUNSPELL GERMAN LUCENE FILTER|ANY_CANDIDATE|408175|158403|975697|44095087576|158403 / 44095245979|975697 / 1383872|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.959832|244817 / 44095245979 (0.000555%)|111167 / 1383872 (8.033041%)|0.853711|0.877306|0.878234|
|
||||||
|
|2|HUNSPELL GERMAN LUCENE FILTER|ALL_CANDIDATES|0.647473|242551 / 44095245979 (0.000550%)|975697 / 1383872 (70.504859%)|0.511911|0.401234|0.430118|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.838673|0.919670|0.999994|0.959832|0.999992|0.000008|
|
||||||
|
|2|HUNSPELL GERMAN LUCENE FILTER|ALL_CANDIDATES|0.627261|0.294951|0.999994|0.647473|0.999972|0.000028|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.853711|0.877306|0.902242|0.781429|0.878238|0.878234|
|
||||||
|
|2|HUNSPELL GERMAN LUCENE FILTER|ALL_CANDIDATES|0.511911|0.401234|0.329907|0.250965|0.430130|0.430118|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|HUNSPELL GERMAN LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|1272705|244817|111167|44095001162|244817 / 44095245979|111167 / 1383872|
|
||||||
|
|2|HUNSPELL GERMAN LUCENE FILTER|ALL_CANDIDATES|408175|242551|975697|44095003428|242551 / 44095245979|975697 / 1383872|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|143736|96817|146625|48574|16.356314%|8|361016|
|
||||||
|
|HUNSPELL GERMAN LUCENE FILTER|16313|45480|38668|7891|2.657135%|3|305052|
|
||||||
|
|
||||||
|
### `LOWERCASE_GROUPS_ONLY`
|
||||||
|
|
||||||
|
This mode contains **12 result rows**, **8 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.966157|47898 / 11263756342 (0.000425%)|59114 / 873411 (6.768177%)|0.941996|0.938343|0.938358|
|
||||||
|
|2|GERMAN CISTEM|PRIMARY_OUTPUT|0.915288|156784 / 11263756342 (0.001392%)|147964 / 873411 (16.940936%)|0.823934|0.826418|0.826415|
|
||||||
|
|3|SNOWBALL GERMAN DIRECT|PRIMARY_OUTPUT|0.795926|87697 / 11263756342 (0.000779%)|356475 / 873411 (40.814118%)|0.785153|0.699487|0.711329|
|
||||||
|
|4|SNOWBALL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.775641|77653 / 11263756342 (0.000689%)|391910 / 873411 (44.871200%)|0.774111|0.672222|0.688986|
|
||||||
|
|5|GERMAN LUCENE GERMAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.769953|55477 / 11263756342 (0.000493%)|401846 / 873411 (46.008809%)|0.790797|0.673446|0.695023|
|
||||||
|
|6|GERMAN LUCENE GERMAN STEM FILTER|PRIMARY_OUTPUT|0.716810|78723 / 11263756342 (0.000699%)|494677 / 873411 (56.637368%)|0.700519|0.569153|0.599149|
|
||||||
|
|7|HUNSPELL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.659196|84679 / 11263756342 (0.000752%)|595318 / 873411 (68.160122%)|0.598178|0.449922|0.494019|
|
||||||
|
|8|GERMAN LUCENE GERMAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.575691|21214 / 11263756342 (0.000188%)|741190 / 873411 (84.861537%)|0.444545|0.257528|0.361168|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.944446|0.932318|0.999996|0.966157|0.999991|0.000009|
|
||||||
|
|2|GERMAN CISTEM|PRIMARY_OUTPUT|0.822287|0.830591|0.999986|0.915288|0.999973|0.000027|
|
||||||
|
|3|SNOWBALL GERMAN DIRECT|PRIMARY_OUTPUT|0.854958|0.591859|0.999992|0.795926|0.999961|0.000039|
|
||||||
|
|4|SNOWBALL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.861124|0.551288|0.999993|0.775641|0.999958|0.000042|
|
||||||
|
|5|GERMAN LUCENE GERMAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.894739|0.539912|0.999995|0.769953|0.999959|0.000041|
|
||||||
|
|6|GERMAN LUCENE GERMAN STEM FILTER|PRIMARY_OUTPUT|0.827912|0.433626|0.999993|0.716810|0.999949|0.000051|
|
||||||
|
|7|HUNSPELL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.766578|0.318399|0.999992|0.659196|0.999940|0.000060|
|
||||||
|
|8|GERMAN LUCENE GERMAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.861739|0.151385|0.999998|0.575691|0.999932|0.000068|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.941996|0.938343|0.934719|0.883848|0.938363|0.938358|
|
||||||
|
|2|GERMAN CISTEM|PRIMARY_OUTPUT|0.823934|0.826418|0.828917|0.704184|0.826428|0.826415|
|
||||||
|
|3|SNOWBALL GERMAN DIRECT|PRIMARY_OUTPUT|0.785153|0.699487|0.630675|0.537854|0.711347|0.711329|
|
||||||
|
|4|SNOWBALL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.774111|0.672222|0.594035|0.506276|0.689005|0.688986|
|
||||||
|
|5|GERMAN LUCENE GERMAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.790797|0.673446|0.586424|0.507666|0.695040|0.695023|
|
||||||
|
|6|GERMAN LUCENE GERMAN STEM FILTER|PRIMARY_OUTPUT|0.700519|0.569153|0.479277|0.397774|0.599170|0.599149|
|
||||||
|
|7|HUNSPELL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.598178|0.449922|0.360559|0.290258|0.494042|0.494019|
|
||||||
|
|8|GERMAN LUCENE GERMAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.444545|0.257528|0.181270|0.147795|0.361184|0.361168|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.938338|0.994062|0.990664|0.992360|0.992360|
|
||||||
|
|2|GERMAN CISTEM|PRIMARY_OUTPUT|0.826404|0.985936|0.973570|0.979714|0.979714|
|
||||||
|
|3|SNOWBALL GERMAN DIRECT|PRIMARY_OUTPUT|0.699468|0.988418|0.932452|0.959619|0.959619|
|
||||||
|
|4|SNOWBALL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.672202|0.989021|0.919542|0.953017|0.953017|
|
||||||
|
|5|GERMAN LUCENE GERMAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.673427|0.991320|0.915070|0.951670|0.951670|
|
||||||
|
|6|GERMAN LUCENE GERMAN STEM FILTER|PRIMARY_OUTPUT|0.569130|0.988584|0.918718|0.952371|0.952371|
|
||||||
|
|7|HUNSPELL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|0.449897|0.988041|0.865581|0.922766|0.922766|
|
||||||
|
|8|GERMAN LUCENE GERMAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.257511|0.992643|0.854403|0.918349|0.918349|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|814297|47898|59114|11263708444|47898 / 11263756342|59114 / 873411|
|
||||||
|
|2|GERMAN CISTEM|PRIMARY_OUTPUT|725447|156784|147964|11263599558|156784 / 11263756342|147964 / 873411|
|
||||||
|
|3|SNOWBALL GERMAN DIRECT|PRIMARY_OUTPUT|516936|87697|356475|11263668645|87697 / 11263756342|356475 / 873411|
|
||||||
|
|4|SNOWBALL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|481501|77653|391910|11263678689|77653 / 11263756342|391910 / 873411|
|
||||||
|
|5|GERMAN LUCENE GERMAN LIGHT STEM FILTER|PRIMARY_OUTPUT|471565|55477|401846|11263700865|55477 / 11263756342|401846 / 873411|
|
||||||
|
|6|GERMAN LUCENE GERMAN STEM FILTER|PRIMARY_OUTPUT|378734|78723|494677|11263677619|78723 / 11263756342|494677 / 873411|
|
||||||
|
|7|HUNSPELL GERMAN LUCENE FILTER|PRIMARY_OUTPUT|278093|84679|595318|11263671663|84679 / 11263756342|595318 / 873411|
|
||||||
|
|8|GERMAN LUCENE GERMAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|132221|21214|741190|11263735128|21214 / 11263756342|741190 / 873411|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 11263756342 (0.000000%)|0 / 873411 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|2|HUNSPELL GERMAN LUCENE FILTER|ANY_CANDIDATE|0.665363|60996 / 11263756342 (0.000542%)|584547 / 873411 (66.926911%)|0.635466|0.472281|0.522540|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|2|HUNSPELL GERMAN LUCENE FILTER|ANY_CANDIDATE|0.825656|0.330731|0.999995|0.665363|0.999943|0.000057|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|2|HUNSPELL GERMAN LUCENE FILTER|ANY_CANDIDATE|0.635466|0.472281|0.375782|0.309142|0.522561|0.522540|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|HUNSPELL GERMAN LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|873411|0|0|11263756342|0 / 11263756342|0 / 873411|
|
||||||
|
|2|HUNSPELL GERMAN LUCENE FILTER|ANY_CANDIDATE|288864|60996|584547|11263695346|60996 / 11263756342|584547 / 873411|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999996|97544 / 11263756342 (0.000866%)|0 / 873411 (0.000000%)|0.917983|0.947112|0.948436|
|
||||||
|
|2|HUNSPELL GERMAN LUCENE FILTER|ALL_CANDIDATES|0.665361|96545 / 11263756342 (0.000857%)|584547 / 873411 (66.926911%)|0.598050|0.458944|0.497855|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.899538|1.000000|0.999991|0.999996|0.999991|0.000009|
|
||||||
|
|2|HUNSPELL GERMAN LUCENE FILTER|ALL_CANDIDATES|0.749500|0.330731|0.999991|0.665361|0.999940|0.000060|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.917983|0.947112|0.978152|0.899538|0.948440|0.948436|
|
||||||
|
|2|HUNSPELL GERMAN LUCENE FILTER|ALL_CANDIDATES|0.598050|0.458944|0.372338|0.297811|0.497878|0.497855|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|HUNSPELL GERMAN LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|873411|97544|0|11263658798|97544 / 11263756342|0 / 873411|
|
||||||
|
|2|HUNSPELL GERMAN LUCENE FILTER|ALL_CANDIDATES|288864|96545|584547|11263659797|96545 / 11263756342|584547 / 873411|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|59114|47898|49646|14978|9.978814%|8|167157|
|
||||||
|
|HUNSPELL GERMAN LUCENE FILTER|10771|23683|11866|4989|3.323828%|3|155207|
|
||||||
|
|
||||||
|
### Output Policies and Metric Definitions
|
||||||
|
|
||||||
|
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
|
||||||
|
|
||||||
|
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
|
||||||
|
|
||||||
|
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
|
||||||
|
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
|
||||||
|
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
|
||||||
|
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
|
||||||
|
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
|
||||||
|
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
|
||||||
|
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
|
||||||
|
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
|
||||||
|
- Jaccard index: `TP / (TP + FP + FN)`.
|
||||||
|
- Fowlkes–Mallows index: `sqrt(precision * recall)`.
|
||||||
|
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
|
||||||
|
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
|
||||||
|
|
||||||
|
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
|
||||||
|
|
||||||
|
### Provenance
|
||||||
|
|
||||||
|
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
|
||||||
|
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
|
||||||
|
- Evaluation command: `./gradlew stemmingQuality`
|
||||||
|
- Dictionary language: `DE_DE`
|
||||||
|
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
|
||||||
|
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
|
||||||
|
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:END -->
|
||||||
|
|||||||
@@ -26,7 +26,7 @@ Radixor stores the preferred transformation for each normalized dictionary word
|
|||||||
|
|
||||||
## Accuracy
|
## Accuracy
|
||||||
|
|
||||||
Accuracy is computed from one deterministic JMH measurement iteration without warmup. The benchmark may execute the full dictionary pass more than once inside that single timed iteration; percentages divide matching counters by evaluated counters from the same iteration.
|
Accuracy is computed from JMH auxiliary counters in the current report. The counters are deterministic for a fixed corpus and stemmer; percentages divide matching counters by evaluated counters from the same report and are not timing metrics.
|
||||||
|
|
||||||
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
||||||
| --- | ---: | ---: | ---: | --- |
|
| --- | ---: | ---: | ---: | --- |
|
||||||
@@ -41,10 +41,10 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
|
|
||||||
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
||||||
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
||||||
| Radixor | `hungarianRadixor` | 53.844 | 5.619 | 60.0 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
| Radixor | `hungarianRadixor` | 62.232 | 6.412 | 69.4 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
||||||
| Lucene HungarianLightStemFilter | `hungarianLuceneHungarianLightStemFilter` | 85.802 | 4.554 | 95.7 | 1.594 | Light Hungarian suffix stemmer. |
|
| Lucene HungarianLightStemFilter | `hungarianLuceneHungarianLightStemFilter` | 92.813 | 6.929 | 103.5 | 1.491 | Light Hungarian suffix stemmer. |
|
||||||
| Official Snowball direct | `snowballDirect[HUNGARIAN]` | 161.996 | 61.038 | 180.6 | 3.009 | Official Snowball generated Java stemmer; direct API. |
|
| Official Snowball direct | `snowballDirect[HUNGARIAN]` | 157.765 | 13.202 | 175.9 | 2.535 | Official Snowball generated Java stemmer; direct API. |
|
||||||
| Lucene SnowballFilter | `luceneSnowballFilter[HUNGARIAN]` | 185.097 | 44.447 | 206.4 | 3.438 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
| Lucene SnowballFilter | `luceneSnowballFilter[HUNGARIAN]` | 188.863 | 15.880 | 210.6 | 3.035 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
||||||
|
|
||||||
## Interpretation Notes
|
## Interpretation Notes
|
||||||
|
|
||||||
@@ -53,3 +53,356 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
||||||
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
||||||
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:START -->
|
||||||
|
|
||||||
|
## Stemming Quality
|
||||||
|
|
||||||
|
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `HU_HU` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
|
||||||
|
|
||||||
|
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
|
||||||
|
|
||||||
|
### Evaluation Scope and Key Findings
|
||||||
|
|
||||||
|
The dictionary resource is `src/main/resources/hu_hu/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
|
||||||
|
|
||||||
|
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.995491** among 4 deterministic stemmers. The runner-up is `SNOWBALL HUNGARIAN LUCENE FILTER` at 0.822606, a difference of 0.172885. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.996163** among 4 deterministic stemmers. The runner-up is `SNOWBALL HUNGARIAN DIRECT` at 0.821708, a difference of 0.174455. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
### `ALL_WORDS`
|
||||||
|
|
||||||
|
This mode contains **6 result rows**, **4 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.995491|272900 / 419820542893 (0.000065%)|199837 / 22162103 (0.901706%)|0.988376|0.989352|0.989353|
|
||||||
|
|2|SNOWBALL HUNGARIAN LUCENE FILTER|PRIMARY_OUTPUT|0.822606|1792049 / 419820542893 (0.000427%)|7862745 / 22162103 (35.478334%)|0.826288|0.747610|0.757196|
|
||||||
|
|3|SNOWBALL HUNGARIAN DIRECT|PRIMARY_OUTPUT|0.822348|1506056 / 419820542893 (0.000359%)|7874191 / 22162103 (35.529981%)|0.837137|0.752866|0.763681|
|
||||||
|
|4|HUNGARIAN LUCENE HUNGARIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.816668|4132555 / 419820542893 (0.000984%)|8125833 / 22162103 (36.665442%)|0.740018|0.696055|0.699478|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.987727|0.990983|0.999999|0.995491|0.999999|0.000001|
|
||||||
|
|2|SNOWBALL HUNGARIAN LUCENE FILTER|PRIMARY_OUTPUT|0.888633|0.645217|0.999996|0.822606|0.999977|0.000023|
|
||||||
|
|3|SNOWBALL HUNGARIAN DIRECT|PRIMARY_OUTPUT|0.904644|0.644700|0.999996|0.822348|0.999978|0.000022|
|
||||||
|
|4|HUNGARIAN LUCENE HUNGARIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.772547|0.633346|0.999990|0.816668|0.999971|0.000029|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.988376|0.989352|0.990330|0.978929|0.989353|0.989353|
|
||||||
|
|2|SNOWBALL HUNGARIAN LUCENE FILTER|PRIMARY_OUTPUT|0.826288|0.747610|0.682613|0.596947|0.757206|0.757196|
|
||||||
|
|3|SNOWBALL HUNGARIAN DIRECT|PRIMARY_OUTPUT|0.837137|0.752866|0.684009|0.603677|0.763691|0.763681|
|
||||||
|
|4|HUNGARIAN LUCENE HUNGARIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.740018|0.696055|0.657023|0.533807|0.699492|0.699478|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.989352|0.998036|0.997809|0.997922|0.997922|
|
||||||
|
|2|SNOWBALL HUNGARIAN LUCENE FILTER|PRIMARY_OUTPUT|0.747599|0.990687|0.924490|0.956445|0.956445|
|
||||||
|
|3|SNOWBALL HUNGARIAN DIRECT|PRIMARY_OUTPUT|0.752855|0.991948|0.924304|0.956932|0.956932|
|
||||||
|
|4|HUNGARIAN LUCENE HUNGARIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.696040|0.982615|0.926772|0.953877|0.953877|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|21962266|272900|199837|419820269993|272900 / 419820542893|199837 / 22162103|
|
||||||
|
|2|SNOWBALL HUNGARIAN LUCENE FILTER|PRIMARY_OUTPUT|14299358|1792049|7862745|419818750844|1792049 / 419820542893|7862745 / 22162103|
|
||||||
|
|3|SNOWBALL HUNGARIAN DIRECT|PRIMARY_OUTPUT|14287912|1506056|7874191|419819036837|1506056 / 419820542893|7874191 / 22162103|
|
||||||
|
|4|HUNGARIAN LUCENE HUNGARIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|14036270|4132555|8125833|419816410338|4132555 / 419820542893|8125833 / 22162103|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 419820542893 (0.000000%)|0 / 22162103 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|22162103|0|0|419820542893|0 / 419820542893|0 / 22162103|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999999|460158 / 419820542893 (0.000110%)|0 / 22162103 (0.000000%)|0.983661|0.989725|0.989777|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.979659|1.000000|0.999999|0.999999|0.999999|0.000001|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.983661|0.989725|0.995865|0.979659|0.989777|0.989777|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|22162103|460158|0|419820082735|460158 / 419820542893|0 / 22162103|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|199837|272900|187258|12320|1.344473%|5|929326|
|
||||||
|
|
||||||
|
### `LOWERCASE_GROUPS_ONLY`
|
||||||
|
|
||||||
|
This mode contains **6 result rows**, **4 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.996163|272775 / 385870694917 (0.000071%)|164277 / 21411411 (0.767240%)|0.988321|0.989820|0.989822|
|
||||||
|
|2|SNOWBALL HUNGARIAN DIRECT|PRIMARY_OUTPUT|0.821708|1496670 / 385870694917 (0.000388%)|7634885 / 21411411 (35.658019%)|0.834899|0.751079|0.761809|
|
||||||
|
|3|SNOWBALL HUNGARIAN LUCENE FILTER|PRIMARY_OUTPUT|0.821708|1496670 / 385870694917 (0.000388%)|7634885 / 21411411 (35.658019%)|0.834899|0.751079|0.761809|
|
||||||
|
|4|HUNGARIAN LUCENE HUNGARIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.815077|3639046 / 385870694917 (0.000943%)|7918708 / 21411411 (36.983588%)|0.750108|0.700135|0.704477|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.987325|0.992328|0.999999|0.996163|0.999999|0.000001|
|
||||||
|
|2|SNOWBALL HUNGARIAN DIRECT|PRIMARY_OUTPUT|0.902007|0.643420|0.999996|0.821708|0.999976|0.000024|
|
||||||
|
|3|SNOWBALL HUNGARIAN LUCENE FILTER|PRIMARY_OUTPUT|0.902007|0.643420|0.999996|0.821708|0.999976|0.000024|
|
||||||
|
|4|HUNGARIAN LUCENE HUNGARIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.787585|0.630164|0.999991|0.815077|0.999970|0.000030|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.988321|0.989820|0.991323|0.979845|0.989823|0.989822|
|
||||||
|
|2|SNOWBALL HUNGARIAN DIRECT|PRIMARY_OUTPUT|0.834899|0.751079|0.682555|0.601383|0.761820|0.761809|
|
||||||
|
|3|SNOWBALL HUNGARIAN LUCENE FILTER|PRIMARY_OUTPUT|0.834899|0.751079|0.682555|0.601383|0.761820|0.761809|
|
||||||
|
|4|HUNGARIAN LUCENE HUNGARIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.750108|0.700135|0.656404|0.538621|0.704491|0.704477|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.989819|0.997945|0.998273|0.998109|0.998109|
|
||||||
|
|2|SNOWBALL HUNGARIAN DIRECT|PRIMARY_OUTPUT|0.751068|0.991610|0.923288|0.956230|0.956230|
|
||||||
|
|3|SNOWBALL HUNGARIAN LUCENE FILTER|PRIMARY_OUTPUT|0.751068|0.991610|0.923288|0.956230|0.956230|
|
||||||
|
|4|HUNGARIAN LUCENE HUNGARIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.700120|0.983687|0.925487|0.953700|0.953700|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|21247134|272775|164277|385870422142|272775 / 385870694917|164277 / 21411411|
|
||||||
|
|2|SNOWBALL HUNGARIAN DIRECT|PRIMARY_OUTPUT|13776526|1496670|7634885|385869198247|1496670 / 385870694917|7634885 / 21411411|
|
||||||
|
|3|SNOWBALL HUNGARIAN LUCENE FILTER|PRIMARY_OUTPUT|13776526|1496670|7634885|385869198247|1496670 / 385870694917|7634885 / 21411411|
|
||||||
|
|4|HUNGARIAN LUCENE HUNGARIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|13492703|3639046|7918708|385867055871|3639046 / 385870694917|7918708 / 21411411|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 385870694917 (0.000000%)|0 / 21411411 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|21411411|0|0|385870694917|0 / 385870694917|0 / 21411411|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999999|458462 / 385870694917 (0.000119%)|0 / 21411411 (0.000000%)|0.983159|0.989407|0.989462|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.979037|1.000000|0.999999|0.999999|0.999999|0.000001|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.983159|0.989407|0.995736|0.979037|0.989463|0.989462|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|21411411|458462|0|385870236455|458462 / 385870694917|0 / 21411411|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|164277|272775|185687|11153|1.269532%|5|890245|
|
||||||
|
|
||||||
|
### Output Policies and Metric Definitions
|
||||||
|
|
||||||
|
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
|
||||||
|
|
||||||
|
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
|
||||||
|
|
||||||
|
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
|
||||||
|
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
|
||||||
|
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
|
||||||
|
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
|
||||||
|
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
|
||||||
|
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
|
||||||
|
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
|
||||||
|
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
|
||||||
|
- Jaccard index: `TP / (TP + FP + FN)`.
|
||||||
|
- Fowlkes–Mallows index: `sqrt(precision * recall)`.
|
||||||
|
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
|
||||||
|
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
|
||||||
|
|
||||||
|
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
|
||||||
|
|
||||||
|
### Provenance
|
||||||
|
|
||||||
|
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
|
||||||
|
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
|
||||||
|
- Evaluation command: `./gradlew stemmingQuality`
|
||||||
|
- Dictionary language: `HU_HU`
|
||||||
|
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
|
||||||
|
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
|
||||||
|
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:END -->
|
||||||
|
|||||||
@@ -1,12 +1,12 @@
|
|||||||
# Language Benchmark Pages
|
# Language Benchmark Pages
|
||||||
|
|
||||||
This section splits Radixor stemmer benchmark results by language. Each language page lists accuracy first and speed second.
|
This section splits Radixor stemmer benchmark results by language. Each language page preserves the existing exact-root accuracy and runtime-performance results and adds pairwise stemming-quality tables for both dictionary-processing modes.
|
||||||
|
|
||||||
## Reference Pages
|
## Reference Pages
|
||||||
|
|
||||||
| Page | Purpose |
|
| Page | Purpose |
|
||||||
| --- | --- |
|
| --- | --- |
|
||||||
| [Methodology](../reference/methodology.md) | Workload design, normalization, speed metrics, and quality metrics. |
|
| [Methodology](../reference/methodology.md) | Workload design, normalization, speed metrics, and exact-root quality metrics. Pairwise quality definitions are also reproduced on every language page. |
|
||||||
| [Corpora](../reference/corpora.md) | Dictionary sizes and changed-token timing workloads. |
|
| [Corpora](../reference/corpora.md) | Dictionary sizes and changed-token timing workloads. |
|
||||||
| [Environment and reports](../reference/environment.md) | Hardware, JVM, JMH settings, report files, and badge policy. |
|
| [Environment and reports](../reference/environment.md) | Hardware, JVM, JMH settings, report files, and badge policy. |
|
||||||
| [English dictionary coverage](../reference/english-coverage.md) | Quality/speed operating curve for contracted Radixor tries built from 100% down to 10% of English dictionary rows. |
|
| [English dictionary coverage](../reference/english-coverage.md) | Quality/speed operating curve for contracted Radixor tries built from 100% down to 10% of English dictionary rows. |
|
||||||
|
|||||||
@@ -25,7 +25,7 @@ Radixor stores the preferred transformation for each normalized dictionary word
|
|||||||
|
|
||||||
## Accuracy
|
## Accuracy
|
||||||
|
|
||||||
Accuracy is computed from one deterministic JMH measurement iteration without warmup. The benchmark may execute the full dictionary pass more than once inside that single timed iteration; percentages divide matching counters by evaluated counters from the same iteration.
|
Accuracy is computed from JMH auxiliary counters in the current report. The counters are deterministic for a fixed corpus and stemmer; percentages divide matching counters by evaluated counters from the same report and are not timing metrics.
|
||||||
|
|
||||||
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
||||||
| --- | ---: | ---: | ---: | --- |
|
| --- | ---: | ---: | ---: | --- |
|
||||||
@@ -40,10 +40,10 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
|
|
||||||
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
||||||
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
||||||
| Radixor | `italianRadixor` | 23.776 | 9.977 | 74.9 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
| Radixor | `italianRadixor` | 24.491 | 3.128 | 77.1 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
||||||
| Lucene ItalianLightStemFilter | `italianLuceneItalianLightStemFilter` | 14.940 | 1.682 | 47.1 | 0.628 | Light Italian suffix stemmer. |
|
| Lucene ItalianLightStemFilter | `italianLuceneItalianLightStemFilter` | 15.977 | 1.041 | 50.3 | 0.652 | Light Italian suffix stemmer. |
|
||||||
| Official Snowball direct | `snowballDirect[ITALIAN]` | 99.401 | 9.433 | 313.0 | 4.181 | Official Snowball generated Java stemmer; direct API. |
|
| Official Snowball direct | `snowballDirect[ITALIAN]` | 109.526 | 12.572 | 344.9 | 4.472 | Official Snowball generated Java stemmer; direct API. |
|
||||||
| Lucene SnowballFilter | `luceneSnowballFilter[ITALIAN]` | 108.462 | 8.014 | 341.6 | 4.562 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
| Lucene SnowballFilter | `luceneSnowballFilter[ITALIAN]` | 116.260 | 7.459 | 366.1 | 4.747 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
||||||
|
|
||||||
## Interpretation Notes
|
## Interpretation Notes
|
||||||
|
|
||||||
@@ -52,3 +52,356 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
||||||
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
||||||
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:START -->
|
||||||
|
|
||||||
|
## Stemming Quality
|
||||||
|
|
||||||
|
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `IT_IT` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
|
||||||
|
|
||||||
|
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
|
||||||
|
|
||||||
|
### Evaluation Scope and Key Findings
|
||||||
|
|
||||||
|
The dictionary resource is `src/main/resources/it_it/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
|
||||||
|
|
||||||
|
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.996507** among 4 deterministic stemmers. The runner-up is `SNOWBALL ITALIAN DIRECT` at 0.866189, a difference of 0.130318. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.996512** among 4 deterministic stemmers. The runner-up is `SNOWBALL ITALIAN DIRECT` at 0.866205, a difference of 0.130307. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
### `ALL_WORDS`
|
||||||
|
|
||||||
|
This mode contains **6 result rows**, **4 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.996507|124172 / 53638521211 (0.000231%)|42908 / 6143814 (0.698394%)|0.982618|0.986492|0.986512|
|
||||||
|
|2|SNOWBALL ITALIAN DIRECT|PRIMARY_OUTPUT|0.866189|504775 / 53638521211 (0.000941%)|1644164 / 6143814 (26.761292%)|0.859975|0.807240|0.811470|
|
||||||
|
|3|SNOWBALL ITALIAN LUCENE FILTER|PRIMARY_OUTPUT|0.866189|504775 / 53638521211 (0.000941%)|1644164 / 6143814 (26.761292%)|0.859975|0.807240|0.811470|
|
||||||
|
|4|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.508926|10589 / 53638521211 (0.000020%)|6034130 / 6143814 (98.214725%)|0.082782|0.035020|0.127588|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.980053|0.993016|0.999998|0.996507|0.999997|0.000003|
|
||||||
|
|2|SNOWBALL ITALIAN DIRECT|PRIMARY_OUTPUT|0.899134|0.732387|0.999991|0.866189|0.999960|0.000040|
|
||||||
|
|3|SNOWBALL ITALIAN LUCENE FILTER|PRIMARY_OUTPUT|0.899134|0.732387|0.999991|0.866189|0.999960|0.000040|
|
||||||
|
|4|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.911959|0.017853|1.000000|0.508926|0.999887|0.000113|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.982618|0.986492|0.990396|0.973344|0.986513|0.986512|
|
||||||
|
|2|SNOWBALL ITALIAN DIRECT|PRIMARY_OUTPUT|0.859975|0.807240|0.760598|0.676783|0.811489|0.811470|
|
||||||
|
|3|SNOWBALL ITALIAN LUCENE FILTER|PRIMARY_OUTPUT|0.859975|0.807240|0.760598|0.676783|0.811489|0.811470|
|
||||||
|
|4|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.082782|0.035020|0.022207|0.017822|0.127597|0.127588|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.986490|0.995780|0.997113|0.996446|0.996446|
|
||||||
|
|2|SNOWBALL ITALIAN DIRECT|PRIMARY_OUTPUT|0.807220|0.987994|0.933408|0.959925|0.959925|
|
||||||
|
|3|SNOWBALL ITALIAN LUCENE FILTER|PRIMARY_OUTPUT|0.807220|0.987994|0.933408|0.959925|0.959925|
|
||||||
|
|4|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.035016|0.997481|0.737537|0.848037|0.848037|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|6100906|124172|42908|53638397039|124172 / 53638521211|42908 / 6143814|
|
||||||
|
|2|SNOWBALL ITALIAN DIRECT|PRIMARY_OUTPUT|4499650|504775|1644164|53638016436|504775 / 53638521211|1644164 / 6143814|
|
||||||
|
|3|SNOWBALL ITALIAN LUCENE FILTER|PRIMARY_OUTPUT|4499650|504775|1644164|53638016436|504775 / 53638521211|1644164 / 6143814|
|
||||||
|
|4|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|109684|10589|6034130|53638510622|10589 / 53638521211|6034130 / 6143814|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|0.999993|0 / 53638521211 (0.000000%)|80 / 6143814 (0.001302%)|0.999997|0.999993|0.999993|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0.999987|1.000000|0.999993|1.000000|0.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|0.999997|0.999993|0.999990|0.999987|0.999993|0.999993|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|6143734|0|80|53638521211|0 / 53638521211|80 / 6143814|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999992|170950 / 53638521211 (0.000319%)|80 / 6143814 (0.001302%)|0.978222|0.986272|0.986363|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.972928|0.999987|0.999997|0.999992|0.999997|0.000003|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.978222|0.986272|0.994455|0.972916|0.986365|0.986363|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|6143734|170950|80|53638350261|170950 / 53638521211|80 / 6143814|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|42828|124172|46778|6254|1.909321%|4|334175|
|
||||||
|
|
||||||
|
### `LOWERCASE_GROUPS_ONLY`
|
||||||
|
|
||||||
|
This mode contains **6 result rows**, **4 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.996512|124171 / 53611667072 (0.000232%)|42828 / 6142174 (0.697278%)|0.982617|0.986495|0.986515|
|
||||||
|
|2|SNOWBALL ITALIAN DIRECT|PRIMARY_OUTPUT|0.866205|504774 / 53611667072 (0.000942%)|1643522 / 6142174 (26.757985%)|0.859970|0.807252|0.811479|
|
||||||
|
|3|SNOWBALL ITALIAN LUCENE FILTER|PRIMARY_OUTPUT|0.866205|504774 / 53611667072 (0.000942%)|1643522 / 6142174 (26.757985%)|0.859970|0.807252|0.811479|
|
||||||
|
|4|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.508927|10588 / 53611667072 (0.000020%)|6032516 / 6142174 (98.214671%)|0.082784|0.035021|0.127589|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.980048|0.993027|0.999998|0.996512|0.999997|0.000003|
|
||||||
|
|2|SNOWBALL ITALIAN DIRECT|PRIMARY_OUTPUT|0.899114|0.732420|0.999991|0.866205|0.999960|0.000040|
|
||||||
|
|3|SNOWBALL ITALIAN LUCENE FILTER|PRIMARY_OUTPUT|0.899114|0.732420|0.999991|0.866205|0.999960|0.000040|
|
||||||
|
|4|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.911947|0.017853|1.000000|0.508927|0.999887|0.000113|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.982617|0.986495|0.990404|0.973350|0.986516|0.986515|
|
||||||
|
|2|SNOWBALL ITALIAN DIRECT|PRIMARY_OUTPUT|0.859970|0.807252|0.760624|0.676800|0.811498|0.811479|
|
||||||
|
|3|SNOWBALL ITALIAN LUCENE FILTER|PRIMARY_OUTPUT|0.859970|0.807252|0.760624|0.676800|0.811498|0.811479|
|
||||||
|
|4|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.082784|0.035021|0.022208|0.017823|0.127598|0.127589|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.986493|0.995780|0.997115|0.996447|0.996447|
|
||||||
|
|2|SNOWBALL ITALIAN DIRECT|PRIMARY_OUTPUT|0.807232|0.987991|0.933413|0.959927|0.959927|
|
||||||
|
|3|SNOWBALL ITALIAN LUCENE FILTER|PRIMARY_OUTPUT|0.807232|0.987991|0.933413|0.959927|0.959927|
|
||||||
|
|4|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.035017|0.997481|0.737534|0.848035|0.848035|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|6099346|124171|42828|53611542901|124171 / 53611667072|42828 / 6142174|
|
||||||
|
|2|SNOWBALL ITALIAN DIRECT|PRIMARY_OUTPUT|4498652|504774|1643522|53611162298|504774 / 53611667072|1643522 / 6142174|
|
||||||
|
|3|SNOWBALL ITALIAN LUCENE FILTER|PRIMARY_OUTPUT|4498652|504774|1643522|53611162298|504774 / 53611667072|1643522 / 6142174|
|
||||||
|
|4|ITALIAN LUCENE ITALIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|109658|10588|6032516|53611656484|10588 / 53611667072|6032516 / 6142174|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 53611667072 (0.000000%)|0 / 6142174 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|6142174|0|0|53611667072|0 / 53611667072|0 / 6142174|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999998|170949 / 53611667072 (0.000319%)|0 / 6142174 (0.000000%)|0.978219|0.986275|0.986366|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.972922|1.000000|0.999997|0.999998|0.999997|0.000003|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.978219|0.986275|0.994464|0.972922|0.986368|0.986366|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|6142174|170949|0|53611496123|170949 / 53611667072|0 / 6142174|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|42828|124171|46778|6252|1.909188%|4|334089|
|
||||||
|
|
||||||
|
### Output Policies and Metric Definitions
|
||||||
|
|
||||||
|
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
|
||||||
|
|
||||||
|
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
|
||||||
|
|
||||||
|
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
|
||||||
|
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
|
||||||
|
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
|
||||||
|
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
|
||||||
|
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
|
||||||
|
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
|
||||||
|
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
|
||||||
|
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
|
||||||
|
- Jaccard index: `TP / (TP + FP + FN)`.
|
||||||
|
- Fowlkes–Mallows index: `sqrt(precision * recall)`.
|
||||||
|
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
|
||||||
|
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
|
||||||
|
|
||||||
|
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
|
||||||
|
|
||||||
|
### Provenance
|
||||||
|
|
||||||
|
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
|
||||||
|
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
|
||||||
|
- Evaluation command: `./gradlew stemmingQuality`
|
||||||
|
- Dictionary language: `IT_IT`
|
||||||
|
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
|
||||||
|
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
|
||||||
|
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:END -->
|
||||||
|
|||||||
@@ -26,7 +26,7 @@ Radixor stores the preferred transformation for each normalized dictionary word
|
|||||||
|
|
||||||
## Accuracy
|
## Accuracy
|
||||||
|
|
||||||
Accuracy is computed from one deterministic JMH measurement iteration without warmup. The benchmark may execute the full dictionary pass more than once inside that single timed iteration; percentages divide matching counters by evaluated counters from the same iteration.
|
Accuracy is computed from JMH auxiliary counters in the current report. The counters are deterministic for a fixed corpus and stemmer; percentages divide matching counters by evaluated counters from the same report and are not timing metrics.
|
||||||
|
|
||||||
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
||||||
| --- | ---: | ---: | ---: | --- |
|
| --- | ---: | ---: | ---: | --- |
|
||||||
@@ -42,11 +42,11 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
|
|
||||||
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
||||||
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
||||||
| Radixor | `norwegianBokmalRadixor` | 3.235 | 0.147 | 56.4 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
| Radixor | `norwegianBokmalRadixor` | 3.631 | 1.377 | 63.3 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
||||||
| Lucene NorwegianMinimalStemFilter | `norwegianBokmalLuceneNorwegianMinimalStemFilter` | 2.720 | 0.190 | 47.4 | 0.841 | Minimal Norwegian suffix reducer. |
|
| Lucene NorwegianMinimalStemFilter | `norwegianBokmalLuceneNorwegianMinimalStemFilter` | 2.910 | 0.177 | 50.7 | 0.801 | Minimal Norwegian suffix reducer. |
|
||||||
| Lucene NorwegianLightStemFilter | `norwegianBokmalLuceneNorwegianLightStemFilter` | 3.189 | 0.261 | 55.6 | 0.986 | Light Norwegian suffix stemmer. |
|
| Lucene NorwegianLightStemFilter | `norwegianBokmalLuceneNorwegianLightStemFilter` | 3.335 | 0.116 | 58.1 | 0.919 | Light Norwegian suffix stemmer. |
|
||||||
| Official Snowball direct | `snowballDirect[NORWEGIAN_BOKMAL]` | 3.978 | 0.031 | 69.3 | 1.230 | Official Snowball generated Java stemmer; direct API. |
|
| Official Snowball direct | `snowballDirect[NORWEGIAN_BOKMAL]` | 4.277 | 0.082 | 74.5 | 1.178 | Official Snowball generated Java stemmer; direct API. |
|
||||||
| Lucene SnowballFilter | `luceneSnowballFilter[NORWEGIAN_BOKMAL]` | 5.526 | 0.409 | 96.3 | 1.708 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
| Lucene SnowballFilter | `luceneSnowballFilter[NORWEGIAN_BOKMAL]` | 6.077 | 0.208 | 105.9 | 1.674 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
||||||
|
|
||||||
## Interpretation Notes
|
## Interpretation Notes
|
||||||
|
|
||||||
@@ -55,3 +55,366 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
||||||
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
||||||
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:START -->
|
||||||
|
|
||||||
|
## Stemming Quality
|
||||||
|
|
||||||
|
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `NB_NO` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
|
||||||
|
|
||||||
|
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
|
||||||
|
|
||||||
|
### Evaluation Scope and Key Findings
|
||||||
|
|
||||||
|
The dictionary resource is `src/main/resources/nb_no/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
|
||||||
|
|
||||||
|
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.974783** among 5 deterministic stemmers. The runner-up is `SNOWBALL NORWEGIAN BOKMAL DIRECT` at 0.874964, a difference of 0.099819. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.975000** among 5 deterministic stemmers. The runner-up is `SNOWBALL NORWEGIAN BOKMAL DIRECT` at 0.874991, a difference of 0.100009. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
### `ALL_WORDS`
|
||||||
|
|
||||||
|
This mode contains **7 result rows**, **5 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.974783|11482 / 2835618215 (0.000405%)|7170 / 142180 (5.042903%)|0.927078|0.935387|0.935488|
|
||||||
|
|2|SNOWBALL NORWEGIAN BOKMAL DIRECT|PRIMARY_OUTPUT|0.874964|23997 / 2835618215 (0.000846%)|35554 / 142180 (25.006330%)|0.802095|0.781707|0.782399|
|
||||||
|
|3|SNOWBALL NORWEGIAN BOKMAL LUCENE FILTER|PRIMARY_OUTPUT|0.874834|24046 / 2835618215 (0.000848%)|35591 / 142180 (25.032353%)|0.801759|0.781401|0.782091|
|
||||||
|
|4|NORWEGIAN BOKMAL LUCENE NORWEGIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.850006|25171 / 2835618215 (0.000888%)|42651 / 142180 (29.997890%)|0.776381|0.745871|0.747464|
|
||||||
|
|5|NORWEGIAN BOKMAL LUCENE NORWEGIAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.832414|14772 / 2835618215 (0.000521%)|47654 / 142180 (33.516669%)|0.815763|0.751764|0.758263|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.921620|0.949571|0.999996|0.974783|0.999993|0.000007|
|
||||||
|
|2|SNOWBALL NORWEGIAN BOKMAL DIRECT|PRIMARY_OUTPUT|0.816288|0.749937|0.999992|0.874964|0.999979|0.000021|
|
||||||
|
|3|SNOWBALL NORWEGIAN BOKMAL LUCENE FILTER|PRIMARY_OUTPUT|0.815930|0.749676|0.999992|0.874834|0.999979|0.000021|
|
||||||
|
|4|NORWEGIAN BOKMAL LUCENE NORWEGIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.798148|0.700021|0.999991|0.850006|0.999976|0.000024|
|
||||||
|
|5|NORWEGIAN BOKMAL LUCENE NORWEGIAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.864847|0.664833|0.999995|0.832414|0.999978|0.000022|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.927078|0.935387|0.943846|0.878617|0.935491|0.935488|
|
||||||
|
|2|SNOWBALL NORWEGIAN BOKMAL DIRECT|PRIMARY_OUTPUT|0.802095|0.781707|0.762330|0.641641|0.782409|0.782399|
|
||||||
|
|3|SNOWBALL NORWEGIAN BOKMAL LUCENE FILTER|PRIMARY_OUTPUT|0.801759|0.781401|0.762052|0.641229|0.782102|0.782091|
|
||||||
|
|4|NORWEGIAN BOKMAL LUCENE NORWEGIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.776381|0.745871|0.717668|0.594732|0.747476|0.747464|
|
||||||
|
|5|NORWEGIAN BOKMAL LUCENE NORWEGIAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.815763|0.751764|0.697076|0.602261|0.758274|0.758263|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.935384|0.993354|0.994615|0.993984|0.993984|
|
||||||
|
|2|SNOWBALL NORWEGIAN BOKMAL DIRECT|PRIMARY_OUTPUT|0.781696|0.988328|0.971120|0.979648|0.979648|
|
||||||
|
|3|SNOWBALL NORWEGIAN BOKMAL LUCENE FILTER|PRIMARY_OUTPUT|0.781391|0.988295|0.971086|0.979615|0.979615|
|
||||||
|
|4|NORWEGIAN BOKMAL LUCENE NORWEGIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.745859|0.987774|0.965622|0.976573|0.976573|
|
||||||
|
|5|NORWEGIAN BOKMAL LUCENE NORWEGIAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.751753|0.992089|0.962516|0.977079|0.977079|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|135010|11482|7170|2835606733|11482 / 2835618215|7170 / 142180|
|
||||||
|
|2|SNOWBALL NORWEGIAN BOKMAL DIRECT|PRIMARY_OUTPUT|106626|23997|35554|2835594218|23997 / 2835618215|35554 / 142180|
|
||||||
|
|3|SNOWBALL NORWEGIAN BOKMAL LUCENE FILTER|PRIMARY_OUTPUT|106589|24046|35591|2835594169|24046 / 2835618215|35591 / 142180|
|
||||||
|
|4|NORWEGIAN BOKMAL LUCENE NORWEGIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|99529|25171|42651|2835593044|25171 / 2835618215|42651 / 142180|
|
||||||
|
|5|NORWEGIAN BOKMAL LUCENE NORWEGIAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|94526|14772|47654|2835603443|14772 / 2835618215|47654 / 142180|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 2835618215 (0.000000%)|0 / 142180 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|142180|0|0|2835618215|0 / 2835618215|0 / 142180|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999996|20161 / 2835618215 (0.000711%)|0 / 142180 (0.000000%)|0.898118|0.933794|0.935844|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.875811|1.000000|0.999993|0.999996|0.999993|0.000007|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.898118|0.933794|0.972422|0.875811|0.935848|0.935844|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|142180|20161|0|2835598054|20161 / 2835618215|0 / 142180|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|7170|11482|8679|4237|5.626079%|9|79825|
|
||||||
|
|
||||||
|
### `LOWERCASE_GROUPS_ONLY`
|
||||||
|
|
||||||
|
This mode contains **7 result rows**, **5 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.975000|11482 / 2831176784 (0.000406%)|7104 / 142091 (4.999613%)|0.927151|0.935591|0.935695|
|
||||||
|
|2|SNOWBALL NORWEGIAN BOKMAL DIRECT|PRIMARY_OUTPUT|0.874991|23997 / 2831176784 (0.000848%)|35524 / 142091 (25.000880%)|0.802043|0.781698|0.782388|
|
||||||
|
|3|SNOWBALL NORWEGIAN BOKMAL LUCENE FILTER|PRIMARY_OUTPUT|0.874798|23993 / 2831176784 (0.000847%)|35579 / 142091 (25.039587%)|0.801914|0.781464|0.782161|
|
||||||
|
|4|NORWEGIAN BOKMAL LUCENE NORWEGIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.849947|25118 / 2831176784 (0.000887%)|42641 / 142091 (30.009642%)|0.776513|0.745896|0.747500|
|
||||||
|
|5|NORWEGIAN BOKMAL LUCENE NORWEGIAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.832344|14719 / 2831176784 (0.000520%)|47644 / 142091 (33.530625%)|0.815950|0.751796|0.758325|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.921608|0.950004|0.999996|0.975000|0.999993|0.000007|
|
||||||
|
|2|SNOWBALL NORWEGIAN BOKMAL DIRECT|PRIMARY_OUTPUT|0.816205|0.749991|0.999992|0.874991|0.999979|0.000021|
|
||||||
|
|3|SNOWBALL NORWEGIAN BOKMAL LUCENE FILTER|PRIMARY_OUTPUT|0.816153|0.749604|0.999992|0.874798|0.999979|0.000021|
|
||||||
|
|4|NORWEGIAN BOKMAL LUCENE NORWEGIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.798359|0.699904|0.999991|0.849947|0.999976|0.000024|
|
||||||
|
|5|NORWEGIAN BOKMAL LUCENE NORWEGIAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.865169|0.664694|0.999995|0.832344|0.999978|0.000022|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.927151|0.935591|0.944186|0.878976|0.935698|0.935695|
|
||||||
|
|2|SNOWBALL NORWEGIAN BOKMAL DIRECT|PRIMARY_OUTPUT|0.802043|0.781698|0.762360|0.641630|0.782398|0.782388|
|
||||||
|
|3|SNOWBALL NORWEGIAN BOKMAL LUCENE FILTER|PRIMARY_OUTPUT|0.801914|0.781464|0.762031|0.641314|0.782171|0.782161|
|
||||||
|
|4|NORWEGIAN BOKMAL LUCENE NORWEGIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.776513|0.745896|0.717603|0.594765|0.747512|0.747500|
|
||||||
|
|5|NORWEGIAN BOKMAL LUCENE NORWEGIAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.815950|0.751796|0.696995|0.602302|0.758335|0.758325|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.935587|0.993348|0.994694|0.994020|0.994020|
|
||||||
|
|2|SNOWBALL NORWEGIAN BOKMAL DIRECT|PRIMARY_OUTPUT|0.781688|0.988318|0.971127|0.979647|0.979647|
|
||||||
|
|3|SNOWBALL NORWEGIAN BOKMAL LUCENE FILTER|PRIMARY_OUTPUT|0.781454|0.988310|0.971074|0.979616|0.979616|
|
||||||
|
|4|NORWEGIAN BOKMAL LUCENE NORWEGIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.745885|0.987789|0.965603|0.976570|0.976570|
|
||||||
|
|5|NORWEGIAN BOKMAL LUCENE NORWEGIAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.751785|0.992107|0.962494|0.977076|0.977076|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|134987|11482|7104|2831165302|11482 / 2831176784|7104 / 142091|
|
||||||
|
|2|SNOWBALL NORWEGIAN BOKMAL DIRECT|PRIMARY_OUTPUT|106567|23997|35524|2831152787|23997 / 2831176784|35524 / 142091|
|
||||||
|
|3|SNOWBALL NORWEGIAN BOKMAL LUCENE FILTER|PRIMARY_OUTPUT|106512|23993|35579|2831152791|23993 / 2831176784|35579 / 142091|
|
||||||
|
|4|NORWEGIAN BOKMAL LUCENE NORWEGIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|99450|25118|42641|2831151666|25118 / 2831176784|42641 / 142091|
|
||||||
|
|5|NORWEGIAN BOKMAL LUCENE NORWEGIAN MINIMAL STEM FILTER|PRIMARY_OUTPUT|94447|14719|47644|2831162065|14719 / 2831176784|47644 / 142091|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 2831176784 (0.000000%)|0 / 142091 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|142091|0|0|2831176784|0 / 2831176784|0 / 142091|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999996|20161 / 2831176784 (0.000712%)|0 / 142091 (0.000000%)|0.898061|0.933756|0.935808|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.875743|1.000000|0.999993|0.999996|0.999993|0.000007|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.898061|0.933756|0.972405|0.875743|0.935811|0.935808|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|142091|20161|0|2831156623|20161 / 2831176784|0 / 142091|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|7104|11482|8679|4204|5.586637%|9|79733|
|
||||||
|
|
||||||
|
### Output Policies and Metric Definitions
|
||||||
|
|
||||||
|
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
|
||||||
|
|
||||||
|
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
|
||||||
|
|
||||||
|
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
|
||||||
|
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
|
||||||
|
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
|
||||||
|
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
|
||||||
|
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
|
||||||
|
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
|
||||||
|
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
|
||||||
|
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
|
||||||
|
- Jaccard index: `TP / (TP + FP + FN)`.
|
||||||
|
- Fowlkes–Mallows index: `sqrt(precision * recall)`.
|
||||||
|
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
|
||||||
|
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
|
||||||
|
|
||||||
|
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
|
||||||
|
|
||||||
|
### Provenance
|
||||||
|
|
||||||
|
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
|
||||||
|
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
|
||||||
|
- Evaluation command: `./gradlew stemmingQuality`
|
||||||
|
- Dictionary language: `NB_NO`
|
||||||
|
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
|
||||||
|
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
|
||||||
|
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:END -->
|
||||||
|
|||||||
@@ -26,11 +26,11 @@ Radixor stores the preferred transformation for each normalized dictionary word
|
|||||||
|
|
||||||
## Accuracy
|
## Accuracy
|
||||||
|
|
||||||
Accuracy is computed from one deterministic JMH measurement iteration without warmup. The benchmark may execute the full dictionary pass more than once inside that single timed iteration; percentages divide matching counters by evaluated counters from the same iteration.
|
Accuracy is computed from JMH auxiliary counters in the current report. The counters are deterministic for a fixed corpus and stemmer; percentages divide matching counters by evaluated counters from the same report and are not timing metrics.
|
||||||
|
|
||||||
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
||||||
| --- | ---: | ---: | ---: | --- |
|
| --- | ---: | ---: | ---: | --- |
|
||||||
| Radixor | 93.089% | 91.395% | 96.863% | Radixor baseline in the Snowball-language comparison family. |
|
| Radixor | 93.089% | 91.395% | 96.863% | Full Radixor dictionary patch-command stemmer. |
|
||||||
| Official Snowball direct | 60.974% | 60.212% | 62.670% | Official Snowball generated Java stemmer; rule-based suffix algorithm. |
|
| Official Snowball direct | 60.974% | 60.212% | 62.670% | Official Snowball generated Java stemmer; rule-based suffix algorithm. |
|
||||||
| Lucene SnowballFilter | 60.918% | 60.146% | 62.638% | Lucene TokenFilter integration path around the Snowball algorithm. |
|
| Lucene SnowballFilter | 60.918% | 60.146% | 62.638% | Lucene TokenFilter integration path around the Snowball algorithm. |
|
||||||
|
|
||||||
@@ -40,9 +40,9 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
|
|
||||||
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
||||||
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
||||||
| Radixor | `radixor[NORWEGIAN_NYNORSK]` | 0.541 | 0.021 | 39.9 | 1.000 | Radixor baseline for the Snowball-language comparison family. |
|
| Radixor | `radixor[NORWEGIAN_NYNORSK]` | 0.571 | 0.012 | 42.1 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
||||||
| Official Snowball direct | `snowballDirect[NORWEGIAN_NYNORSK]` | 0.855 | 0.008 | 63.0 | 1.580 | Official Snowball generated Java stemmer; direct API. |
|
| Official Snowball direct | `snowballDirect[NORWEGIAN_NYNORSK]` | 0.919 | 0.038 | 67.7 | 1.609 | Official Snowball generated Java stemmer; direct API. |
|
||||||
| Lucene SnowballFilter | `luceneSnowballFilter[NORWEGIAN_NYNORSK]` | 1.221 | 0.025 | 90.0 | 2.258 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
| Lucene SnowballFilter | `luceneSnowballFilter[NORWEGIAN_NYNORSK]` | 1.309 | 0.015 | 96.5 | 2.292 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
||||||
|
|
||||||
## Interpretation Notes
|
## Interpretation Notes
|
||||||
|
|
||||||
@@ -51,3 +51,346 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
||||||
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
||||||
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:START -->
|
||||||
|
|
||||||
|
## Stemming Quality
|
||||||
|
|
||||||
|
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `NN_NO` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
|
||||||
|
|
||||||
|
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
|
||||||
|
|
||||||
|
### Evaluation Scope and Key Findings
|
||||||
|
|
||||||
|
The dictionary resource is `src/main/resources/nn_no/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
|
||||||
|
|
||||||
|
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.935777** among 3 deterministic stemmers. The runner-up is `SNOWBALL NORWEGIAN NYNORSK DIRECT` at 0.858908, a difference of 0.076869. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.935853** among 3 deterministic stemmers. The runner-up is `SNOWBALL NORWEGIAN NYNORSK DIRECT` at 0.859037, a difference of 0.076816. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
### `ALL_WORDS`
|
||||||
|
|
||||||
|
This mode contains **5 result rows**, **3 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.935777|6230 / 166491473 (0.003742%)|3936 / 30652 (12.840924%)|0.822355|0.840152|0.840669|
|
||||||
|
|2|SNOWBALL NORWEGIAN NYNORSK DIRECT|PRIMARY_OUTPUT|0.858908|8274 / 166491473 (0.004970%)|8648 / 30652 (28.213493%)|0.724941|0.722271|0.722234|
|
||||||
|
|3|SNOWBALL NORWEGIAN NYNORSK LUCENE FILTER|PRIMARY_OUTPUT|0.858484|8295 / 166491473 (0.004982%)|8674 / 30652 (28.298317%)|0.724180|0.721477|0.721440|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.810903|0.871591|0.999963|0.935777|0.999939|0.000061|
|
||||||
|
|2|SNOWBALL NORWEGIAN NYNORSK DIRECT|PRIMARY_OUTPUT|0.726732|0.717865|0.999950|0.858908|0.999898|0.000102|
|
||||||
|
|3|SNOWBALL NORWEGIAN NYNORSK LUCENE FILTER|PRIMARY_OUTPUT|0.725993|0.717017|0.999950|0.858484|0.999898|0.000102|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.822355|0.840152|0.858737|0.724364|0.840699|0.840669|
|
||||||
|
|2|SNOWBALL NORWEGIAN NYNORSK DIRECT|PRIMARY_OUTPUT|0.724941|0.722271|0.719621|0.565278|0.722285|0.722234|
|
||||||
|
|3|SNOWBALL NORWEGIAN NYNORSK LUCENE FILTER|PRIMARY_OUTPUT|0.724180|0.721477|0.718794|0.564305|0.721491|0.721440|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.840122|0.983845|0.986802|0.985321|0.985321|
|
||||||
|
|2|SNOWBALL NORWEGIAN NYNORSK DIRECT|PRIMARY_OUTPUT|0.722221|0.980542|0.964998|0.972708|0.972708|
|
||||||
|
|3|SNOWBALL NORWEGIAN NYNORSK LUCENE FILTER|PRIMARY_OUTPUT|0.721426|0.980461|0.964862|0.972599|0.972599|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|26716|6230|3936|166485243|6230 / 166491473|3936 / 30652|
|
||||||
|
|2|SNOWBALL NORWEGIAN NYNORSK DIRECT|PRIMARY_OUTPUT|22004|8274|8648|166483199|8274 / 166491473|8648 / 30652|
|
||||||
|
|3|SNOWBALL NORWEGIAN NYNORSK LUCENE FILTER|PRIMARY_OUTPUT|21978|8295|8674|166483178|8295 / 166491473|8674 / 30652|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 166491473 (0.000000%)|0 / 30652 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|30652|0|0|166491473|0 / 166491473|0 / 30652|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999960|13214 / 166491473 (0.007937%)|0 / 30652 (0.000000%)|0.743562|0.822674|0.835888|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.698764|1.000000|0.999921|0.999960|0.999921|0.000079|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.743562|0.822674|0.920624|0.698764|0.835921|0.835888|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|30652|13214|0|166478259|13214 / 166491473|0 / 30652|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|3936|6230|6984|2404|13.172603%|5|21513|
|
||||||
|
|
||||||
|
### `LOWERCASE_GROUPS_ONLY`
|
||||||
|
|
||||||
|
This mode contains **5 result rows**, **3 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.935853|6230 / 165926276 (0.003755%)|3924 / 30595 (12.825625%)|0.822169|0.840084|0.840609|
|
||||||
|
|2|SNOWBALL NORWEGIAN NYNORSK DIRECT|PRIMARY_OUTPUT|0.859037|8274 / 165926276 (0.004987%)|8624 / 30595 (28.187612%)|0.724757|0.722255|0.722216|
|
||||||
|
|3|SNOWBALL NORWEGIAN NYNORSK LUCENE FILTER|PRIMARY_OUTPUT|0.858661|8274 / 165926276 (0.004987%)|8647 / 30595 (28.262788%)|0.724438|0.721772|0.721734|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.810644|0.871744|0.999962|0.935853|0.999939|0.000061|
|
||||||
|
|2|SNOWBALL NORWEGIAN NYNORSK DIRECT|PRIMARY_OUTPUT|0.726434|0.718124|0.999950|0.859037|0.999898|0.000102|
|
||||||
|
|3|SNOWBALL NORWEGIAN NYNORSK LUCENE FILTER|PRIMARY_OUTPUT|0.726226|0.717372|0.999950|0.858661|0.999898|0.000102|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.822169|0.840084|0.858798|0.724263|0.840639|0.840609|
|
||||||
|
|2|SNOWBALL NORWEGIAN NYNORSK DIRECT|PRIMARY_OUTPUT|0.724757|0.722255|0.719771|0.565258|0.722267|0.722216|
|
||||||
|
|3|SNOWBALL NORWEGIAN NYNORSK LUCENE FILTER|PRIMARY_OUTPUT|0.724438|0.721772|0.719126|0.564666|0.721785|0.721734|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.840054|0.983815|0.986842|0.985326|0.985326|
|
||||||
|
|2|SNOWBALL NORWEGIAN NYNORSK DIRECT|PRIMARY_OUTPUT|0.722204|0.980506|0.965065|0.972724|0.972724|
|
||||||
|
|3|SNOWBALL NORWEGIAN NYNORSK LUCENE FILTER|PRIMARY_OUTPUT|0.721721|0.980506|0.964945|0.972663|0.972663|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|26671|6230|3924|165920046|6230 / 165926276|3924 / 30595|
|
||||||
|
|2|SNOWBALL NORWEGIAN NYNORSK DIRECT|PRIMARY_OUTPUT|21971|8274|8624|165918002|8274 / 165926276|8624 / 30595|
|
||||||
|
|3|SNOWBALL NORWEGIAN NYNORSK LUCENE FILTER|PRIMARY_OUTPUT|21948|8274|8647|165918002|8274 / 165926276|8647 / 30595|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 165926276 (0.000000%)|0 / 30595 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|30595|0|0|165926276|0 / 165926276|0 / 30595|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999960|13214 / 165926276 (0.007964%)|0 / 30595 (0.000000%)|0.743207|0.822402|0.835654|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.698372|1.000000|0.999920|0.999960|0.999920|0.000080|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.743207|0.822402|0.920488|0.698372|0.835687|0.835654|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|30595|13214|0|165913062|13214 / 165926276|0 / 30595|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|3924|6230|6984|2399|13.167572%|5|21477|
|
||||||
|
|
||||||
|
### Output Policies and Metric Definitions
|
||||||
|
|
||||||
|
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
|
||||||
|
|
||||||
|
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
|
||||||
|
|
||||||
|
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
|
||||||
|
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
|
||||||
|
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
|
||||||
|
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
|
||||||
|
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
|
||||||
|
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
|
||||||
|
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
|
||||||
|
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
|
||||||
|
- Jaccard index: `TP / (TP + FP + FN)`.
|
||||||
|
- Fowlkes–Mallows index: `sqrt(precision * recall)`.
|
||||||
|
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
|
||||||
|
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
|
||||||
|
|
||||||
|
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
|
||||||
|
|
||||||
|
### Provenance
|
||||||
|
|
||||||
|
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
|
||||||
|
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
|
||||||
|
- Evaluation command: `./gradlew stemmingQuality`
|
||||||
|
- Dictionary language: `NN_NO`
|
||||||
|
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
|
||||||
|
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
|
||||||
|
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:END -->
|
||||||
|
|||||||
@@ -24,7 +24,7 @@ Radixor stores the preferred transformation for each normalized dictionary word
|
|||||||
|
|
||||||
## Accuracy
|
## Accuracy
|
||||||
|
|
||||||
Accuracy is computed from one deterministic JMH measurement iteration without warmup. The benchmark may execute the full dictionary pass more than once inside that single timed iteration; percentages divide matching counters by evaluated counters from the same iteration.
|
Accuracy is computed from JMH auxiliary counters in the current report. The counters are deterministic for a fixed corpus and stemmer; percentages divide matching counters by evaluated counters from the same report and are not timing metrics.
|
||||||
|
|
||||||
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
||||||
| --- | ---: | ---: | ---: | --- |
|
| --- | ---: | ---: | ---: | --- |
|
||||||
@@ -37,8 +37,8 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
|
|
||||||
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
||||||
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
||||||
| Radixor | `persianRadixor` | 0.231 | 0.021 | 63.6 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
| Radixor | `persianRadixor` | 0.245 | 0.025 | 49.0 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
||||||
| Lucene PersianStemFilter | `persianLucenePersianStemFilter` | 0.443 | 0.012 | 122.0 | 1.918 | Persian suffix stemmer with Lucene normalization in the measured path. |
|
| Lucene PersianStemFilter | `persianLucenePersianStemFilter` | 0.466 | 0.015 | 93.3 | 1.902 | Persian suffix stemmer with Lucene normalization in the measured path. |
|
||||||
|
|
||||||
## Interpretation Notes
|
## Interpretation Notes
|
||||||
|
|
||||||
@@ -47,3 +47,336 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
||||||
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
||||||
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:START -->
|
||||||
|
|
||||||
|
## Stemming Quality
|
||||||
|
|
||||||
|
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `FA_IR` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
|
||||||
|
|
||||||
|
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
|
||||||
|
|
||||||
|
### Evaluation Scope and Key Findings
|
||||||
|
|
||||||
|
The dictionary resource is `src/main/resources/fa_ir/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
|
||||||
|
|
||||||
|
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.974922** among 2 deterministic stemmers. The runner-up is `PERSIAN LUCENE PERSIAN STEM FILTER` at 0.502171, a difference of 0.472751. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.974922** among 2 deterministic stemmers. The runner-up is `PERSIAN LUCENE PERSIAN STEM FILTER` at 0.502171, a difference of 0.472751. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
### `ALL_WORDS`
|
||||||
|
|
||||||
|
This mode contains **4 result rows**, **2 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.974922|8621 / 6748402 (0.127749%)|4812 / 98448 (4.887860%)|0.922566|0.933071|0.932249|
|
||||||
|
|2|PERSIAN LUCENE PERSIAN STEM FILTER|PRIMARY_OUTPUT|0.502171|179 / 6748402 (0.002652%)|98018 / 98448 (99.563221%)|0.021312|0.008682|0.054801|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.915693|0.951121|0.998723|0.974922|0.998038|0.001962|
|
||||||
|
|2|PERSIAN LUCENE PERSIAN STEM FILTER|PRIMARY_OUTPUT|0.706076|0.004368|0.999973|0.502171|0.985658|0.014342|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.922566|0.933071|0.943818|0.874539|0.933239|0.932249|
|
||||||
|
|2|PERSIAN LUCENE PERSIAN STEM FILTER|PRIMARY_OUTPUT|0.021312|0.008682|0.005451|0.004360|0.055534|0.054801|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.932076|0.980343|0.984586|0.982460|0.982460|
|
||||||
|
|2|PERSIAN LUCENE PERSIAN STEM FILTER|PRIMARY_OUTPUT|0.008507|0.985686|0.520347|0.681125|0.681125|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|93636|8621|4812|6739781|8621 / 6748402|4812 / 98448|
|
||||||
|
|2|PERSIAN LUCENE PERSIAN STEM FILTER|PRIMARY_OUTPUT|430|179|98018|6748223|179 / 6748402|98018 / 98448|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 6748402 (0.000000%)|0 / 98448 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|98448|0|0|6748402|0 / 6748402|0 / 98448|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999005|13433 / 6748402 (0.199055%)|0 / 98448 (0.000000%)|0.901585|0.936133|0.937114|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.879935|1.000000|0.998009|0.999005|0.998038|0.001962|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.901585|0.936133|0.973435|0.879935|0.938048|0.937114|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|98448|13433|0|6734969|13433 / 6748402|0 / 98448|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|4812|8621|4812|314|8.484193%|2|4015|
|
||||||
|
|
||||||
|
### `LOWERCASE_GROUPS_ONLY`
|
||||||
|
|
||||||
|
This mode contains **4 result rows**, **2 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.974922|8621 / 6748402 (0.127749%)|4812 / 98448 (4.887860%)|0.922566|0.933071|0.932249|
|
||||||
|
|2|PERSIAN LUCENE PERSIAN STEM FILTER|PRIMARY_OUTPUT|0.502171|179 / 6748402 (0.002652%)|98018 / 98448 (99.563221%)|0.021312|0.008682|0.054801|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.915693|0.951121|0.998723|0.974922|0.998038|0.001962|
|
||||||
|
|2|PERSIAN LUCENE PERSIAN STEM FILTER|PRIMARY_OUTPUT|0.706076|0.004368|0.999973|0.502171|0.985658|0.014342|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.922566|0.933071|0.943818|0.874539|0.933239|0.932249|
|
||||||
|
|2|PERSIAN LUCENE PERSIAN STEM FILTER|PRIMARY_OUTPUT|0.021312|0.008682|0.005451|0.004360|0.055534|0.054801|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.932076|0.980343|0.984586|0.982460|0.982460|
|
||||||
|
|2|PERSIAN LUCENE PERSIAN STEM FILTER|PRIMARY_OUTPUT|0.008507|0.985686|0.520347|0.681125|0.681125|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|93636|8621|4812|6739781|8621 / 6748402|4812 / 98448|
|
||||||
|
|2|PERSIAN LUCENE PERSIAN STEM FILTER|PRIMARY_OUTPUT|430|179|98018|6748223|179 / 6748402|98018 / 98448|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 6748402 (0.000000%)|0 / 98448 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|98448|0|0|6748402|0 / 6748402|0 / 98448|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999005|13433 / 6748402 (0.199055%)|0 / 98448 (0.000000%)|0.901585|0.936133|0.937114|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.879935|1.000000|0.998009|0.999005|0.998038|0.001962|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.901585|0.936133|0.973435|0.879935|0.938048|0.937114|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|98448|13433|0|6734969|13433 / 6748402|0 / 98448|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|4812|8621|4812|314|8.484193%|2|4015|
|
||||||
|
|
||||||
|
### Output Policies and Metric Definitions
|
||||||
|
|
||||||
|
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
|
||||||
|
|
||||||
|
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
|
||||||
|
|
||||||
|
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
|
||||||
|
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
|
||||||
|
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
|
||||||
|
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
|
||||||
|
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
|
||||||
|
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
|
||||||
|
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
|
||||||
|
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
|
||||||
|
- Jaccard index: `TP / (TP + FP + FN)`.
|
||||||
|
- Fowlkes–Mallows index: `sqrt(precision * recall)`.
|
||||||
|
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
|
||||||
|
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
|
||||||
|
|
||||||
|
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
|
||||||
|
|
||||||
|
### Provenance
|
||||||
|
|
||||||
|
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
|
||||||
|
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
|
||||||
|
- Evaluation command: `./gradlew stemmingQuality`
|
||||||
|
- Dictionary language: `FA_IR`
|
||||||
|
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
|
||||||
|
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
|
||||||
|
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:END -->
|
||||||
|
|||||||
@@ -26,11 +26,12 @@ Radixor stores the preferred transformation for each normalized dictionary word
|
|||||||
|
|
||||||
## Accuracy
|
## Accuracy
|
||||||
|
|
||||||
Accuracy is computed from one deterministic JMH measurement iteration without warmup. The benchmark may execute the full dictionary pass more than once inside that single timed iteration; percentages divide matching counters by evaluated counters from the same iteration.
|
Accuracy is computed from JMH auxiliary counters in the current report. The counters are deterministic for a fixed corpus and stemmer; percentages divide matching counters by evaluated counters from the same report and are not timing metrics.
|
||||||
|
|
||||||
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
||||||
| --- | ---: | ---: | ---: | --- |
|
| --- | ---: | ---: | ---: | --- |
|
||||||
| Radixor | 98.837% | 98.744% | 99.359% | Full Radixor dictionary patch-command stemmer. |
|
| Radixor | 98.837% | 98.744% | 99.359% | Full Radixor dictionary patch-command stemmer. |
|
||||||
|
| Lucene HunspellStemFilter | 89.545% | 88.272% | 96.713% | Benchmark-only Polish Hunspell dictionary compared via Lucene HunspellStemFilter. |
|
||||||
| Lucene MorfologikFilter | 87.729% | 86.606% | 94.047% | Dictionary-based path; Morfologik can emit multiple terms. |
|
| Lucene MorfologikFilter | 87.729% | 86.606% | 94.047% | Dictionary-based path; Morfologik can emit multiple terms. |
|
||||||
| Lucene StempelFilter | 70.009% | 69.262% | 74.220% | Lucene TokenFilter integration path for table-driven Polish Stempel. |
|
| Lucene StempelFilter | 70.009% | 69.262% | 74.220% | Lucene TokenFilter integration path for table-driven Polish Stempel. |
|
||||||
| Lucene StempelStemmer direct | 70.009% | 69.262% | 74.220% | Direct table-driven Polish Stempel stemmer API. |
|
| Lucene StempelStemmer direct | 70.009% | 69.262% | 74.220% | Direct table-driven Polish Stempel stemmer API. |
|
||||||
@@ -41,10 +42,11 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
|
|
||||||
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
||||||
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
||||||
| Radixor | `polishRadixor` | 7.760 | 0.240 | 69.1 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
| Radixor | `polishRadixor` | 9.049 | 0.485 | 80.5 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
||||||
| Lucene StempelStemmer direct | `polishLuceneStempelStemmerDirect` | 34.295 | 0.418 | 305.3 | 4.420 | Direct table-driven Polish Stempel stemmer API. |
|
| Lucene HunspellStemFilter | `luceneHunspellStemFilter` | 483.316 | 11.455 | 4301.8 | 53.408 | Benchmark-only Polish Hunspell dictionary compared via Lucene HunspellStemFilter. |
|
||||||
| Lucene StempelFilter | `polishLuceneStempelFilter` | 39.116 | 1.717 | 348.2 | 5.041 | Lucene TokenFilter integration path for table-driven Polish Stempel. |
|
| Lucene StempelStemmer direct | `polishLuceneStempelStemmerDirect` | 41.932 | 1.916 | 373.2 | 4.634 | Direct table-driven Polish Stempel stemmer API. |
|
||||||
| Lucene MorfologikFilter | `polishLuceneMorfologikFilter` | 128.516 | 12.557 | 1143.9 | 16.562 | Dictionary-based Morfologik TokenFilter; may emit multiple terms. |
|
| Lucene StempelFilter | `polishLuceneStempelFilter` | 45.277 | 13.693 | 403.0 | 5.003 | Lucene TokenFilter integration path for table-driven Polish Stempel. |
|
||||||
|
| Lucene MorfologikFilter | `polishLuceneMorfologikFilter` | 135.763 | 31.634 | 1208.4 | 15.002 | Dictionary-based Morfologik TokenFilter; may emit multiple terms. |
|
||||||
|
|
||||||
## Interpretation Notes
|
## Interpretation Notes
|
||||||
|
|
||||||
@@ -53,3 +55,410 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
||||||
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
||||||
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:START -->
|
||||||
|
|
||||||
|
## Stemming Quality
|
||||||
|
|
||||||
|
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `PL_PL` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
|
||||||
|
|
||||||
|
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
|
||||||
|
|
||||||
|
### Evaluation Scope and Key Findings
|
||||||
|
|
||||||
|
The dictionary resource is `src/main/resources/pl_pl/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
|
||||||
|
|
||||||
|
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.990388** among 5 deterministic stemmers. The runner-up is `POLISH LUCENE MORFOLOGIK FILTER` at 0.948154, a difference of 0.042234. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.990579** among 5 deterministic stemmers. The runner-up is `POLISH LUCENE MORFOLOGIK FILTER` at 0.948177, a difference of 0.042402. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
### `ALL_WORDS`
|
||||||
|
|
||||||
|
This mode contains **11 result rows**, **5 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.990388|13669 / 7482478003 (0.000183%)|21547 / 1120967 (1.922180%)|0.986324|0.984237|0.984241|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.948154|99228 / 7482478003 (0.001326%)|116220 / 1120967 (10.367834%)|0.907324|0.903167|0.903179|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|PRIMARY_OUTPUT|0.933222|52652 / 7482478003 (0.000704%)|149705 / 1120967 (13.354987%)|0.930930|0.905656|0.906571|
|
||||||
|
|4|POLISH LUCENE STEMPEL DIRECT|PRIMARY_OUTPUT|0.855748|66669 / 7482478003 (0.000891%)|323394 / 1120967 (28.849556%)|0.871106|0.803515|0.810296|
|
||||||
|
|5|POLISH LUCENE STEMPEL FILTER|PRIMARY_OUTPUT|0.855748|66669 / 7482478003 (0.000891%)|323394 / 1120967 (28.849556%)|0.871106|0.803515|0.810296|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.987720|0.980778|0.999998|0.990388|0.999995|0.000005|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.910118|0.896322|0.999987|0.948154|0.999971|0.000029|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|PRIMARY_OUTPUT|0.948578|0.866450|0.999993|0.933222|0.999973|0.000027|
|
||||||
|
|4|POLISH LUCENE STEMPEL DIRECT|PRIMARY_OUTPUT|0.922858|0.711504|0.999991|0.855748|0.999948|0.000052|
|
||||||
|
|5|POLISH LUCENE STEMPEL FILTER|PRIMARY_OUTPUT|0.922858|0.711504|0.999991|0.855748|0.999948|0.000052|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.986324|0.984237|0.982159|0.968963|0.984243|0.984241|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.907324|0.903167|0.899047|0.823432|0.903193|0.903179|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|PRIMARY_OUTPUT|0.930930|0.905656|0.881718|0.827579|0.906584|0.906571|
|
||||||
|
|4|POLISH LUCENE STEMPEL DIRECT|PRIMARY_OUTPUT|0.871106|0.803515|0.745659|0.671564|0.810320|0.810296|
|
||||||
|
|5|POLISH LUCENE STEMPEL FILTER|PRIMARY_OUTPUT|0.871106|0.803515|0.745659|0.671564|0.810320|0.810296|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.984234|0.996967|0.996469|0.996718|0.996718|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.903153|0.990022|0.977054|0.983495|0.983495|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|PRIMARY_OUTPUT|0.905642|0.994546|0.970520|0.982386|0.982386|
|
||||||
|
|4|POLISH LUCENE STEMPEL DIRECT|PRIMARY_OUTPUT|0.803490|0.991767|0.931069|0.960460|0.960460|
|
||||||
|
|5|POLISH LUCENE STEMPEL FILTER|PRIMARY_OUTPUT|0.803490|0.991767|0.931069|0.960460|0.960460|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|1099420|13669|21547|7482464334|13669 / 7482478003|21547 / 1120967|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|1004747|99228|116220|7482378775|99228 / 7482478003|116220 / 1120967|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|PRIMARY_OUTPUT|971262|52652|149705|7482425351|52652 / 7482478003|149705 / 1120967|
|
||||||
|
|4|POLISH LUCENE STEMPEL DIRECT|PRIMARY_OUTPUT|797573|66669|323394|7482411334|66669 / 7482478003|323394 / 1120967|
|
||||||
|
|5|POLISH LUCENE STEMPEL FILTER|PRIMARY_OUTPUT|797573|66669|323394|7482411334|66669 / 7482478003|323394 / 1120967|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 7482478003 (0.000000%)|0 / 1120967 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|0.987570|85532 / 7482478003 (0.001143%)|27855 / 1120967 (2.484908%)|0.936598|0.950693|0.950985|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|ANY_CANDIDATE|0.963982|42213 / 7482478003 (0.000564%)|80743 / 1120967 (7.202977%)|0.954209|0.944197|0.944333|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|0.927432|0.975151|0.999989|0.987570|0.999985|0.000015|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|ANY_CANDIDATE|0.961002|0.927970|0.999994|0.963982|0.999984|0.000016|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|0.936598|0.950693|0.965218|0.906020|0.950992|0.950985|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|ANY_CANDIDATE|0.954209|0.944197|0.934394|0.894293|0.944342|0.944333|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1120967|0|0|7482478003|0 / 7482478003|0 / 1120967|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|1093112|85532|27855|7482392471|85532 / 7482478003|27855 / 1120967|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|ANY_CANDIDATE|1040224|42213|80743|7482435790|42213 / 7482478003|80743 / 1120967|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999997|38073 / 7482478003 (0.000509%)|0 / 1120967 (0.000000%)|0.973547|0.983301|0.983436|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|0.987566|143096 / 7482478003 (0.001912%)|27855 / 1120967 (2.484908%)|0.901045|0.927476|0.928576|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|ALL_CANDIDATES|0.963980|82745 / 7482478003 (0.001106%)|80743 / 1120967 (7.202977%)|0.926646|0.927142|0.927132|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.967151|1.000000|0.999995|0.999997|0.999995|0.000005|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|0.884246|0.975151|0.999981|0.987566|0.999977|0.000023|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|ALL_CANDIDATES|0.926316|0.927970|0.999989|0.963980|0.999978|0.000022|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.973547|0.983301|0.993253|0.967151|0.983438|0.983436|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|0.901045|0.927476|0.955505|0.864761|0.928587|0.928576|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|ALL_CANDIDATES|0.926646|0.927142|0.927639|0.864180|0.927143|0.927132|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|1120967|38073|0|7482439930|38073 / 7482478003|0 / 1120967|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|1093112|143096|27855|7482334907|143096 / 7482478003|27855 / 1120967|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|ALL_CANDIDATES|1040224|82745|80743|7482395258|82745 / 7482478003|80743 / 1120967|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|HUNSPELL POLISH LUCENE FILTER|68962|10439|30093|11447|9.356634%|6|135231|
|
||||||
|
|POLISH LUCENE MORFOLOGIK FILTER|88365|13696|43868|12873|10.522229%|5|136636|
|
||||||
|
|Radixor|21547|13669|24404|2866|2.342632%|4|125778|
|
||||||
|
|
||||||
|
### `LOWERCASE_GROUPS_ONLY`
|
||||||
|
|
||||||
|
This mode contains **11 result rows**, **5 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.990579|13669 / 7310252699 (0.000187%)|21000 / 1114651 (1.883998%)|0.986350|0.984397|0.984400|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.948177|99224 / 7310252699 (0.001357%)|115513 / 1114651 (10.363154%)|0.906972|0.902966|0.902976|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|PRIMARY_OUTPUT|0.933309|51950 / 7310252699 (0.000711%)|148667 / 1114651 (13.337538%)|0.931269|0.905928|0.906847|
|
||||||
|
|4|POLISH LUCENE STEMPEL DIRECT|PRIMARY_OUTPUT|0.856382|66274 / 7310252699 (0.000907%)|320158 / 1114651 (28.722712%)|0.871591|0.804380|0.811082|
|
||||||
|
|5|POLISH LUCENE STEMPEL FILTER|PRIMARY_OUTPUT|0.856382|66274 / 7310252699 (0.000907%)|320158 / 1114651 (28.722712%)|0.871591|0.804380|0.811082|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.987656|0.981160|0.999998|0.990579|0.999995|0.000005|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.909662|0.896368|0.999986|0.948177|0.999971|0.000029|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|PRIMARY_OUTPUT|0.948965|0.866625|0.999993|0.933309|0.999973|0.000027|
|
||||||
|
|4|POLISH LUCENE STEMPEL DIRECT|PRIMARY_OUTPUT|0.923006|0.712773|0.999991|0.856382|0.999947|0.000053|
|
||||||
|
|5|POLISH LUCENE STEMPEL FILTER|PRIMARY_OUTPUT|0.923006|0.712773|0.999991|0.856382|0.999947|0.000053|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.986350|0.984397|0.982452|0.969274|0.984403|0.984400|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.906972|0.902966|0.898996|0.823098|0.902991|0.902976|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|PRIMARY_OUTPUT|0.931269|0.905928|0.881929|0.828033|0.906861|0.906847|
|
||||||
|
|4|POLISH LUCENE STEMPEL DIRECT|PRIMARY_OUTPUT|0.871591|0.804380|0.746792|0.672772|0.811106|0.811082|
|
||||||
|
|5|POLISH LUCENE STEMPEL FILTER|PRIMARY_OUTPUT|0.871591|0.804380|0.746792|0.672772|0.811106|0.811082|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.984395|0.996926|0.996647|0.996786|0.996786|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.902952|0.989889|0.977012|0.983408|0.983408|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|PRIMARY_OUTPUT|0.905914|0.994584|0.970514|0.982402|0.982402|
|
||||||
|
|4|POLISH LUCENE STEMPEL DIRECT|PRIMARY_OUTPUT|0.804354|0.991711|0.931318|0.960566|0.960566|
|
||||||
|
|5|POLISH LUCENE STEMPEL FILTER|PRIMARY_OUTPUT|0.804354|0.991711|0.931318|0.960566|0.960566|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|1093651|13669|21000|7310239030|13669 / 7310252699|21000 / 1114651|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|999138|99224|115513|7310153475|99224 / 7310252699|115513 / 1114651|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|PRIMARY_OUTPUT|965984|51950|148667|7310200749|51950 / 7310252699|148667 / 1114651|
|
||||||
|
|4|POLISH LUCENE STEMPEL DIRECT|PRIMARY_OUTPUT|794493|66274|320158|7310186425|66274 / 7310252699|320158 / 1114651|
|
||||||
|
|5|POLISH LUCENE STEMPEL FILTER|PRIMARY_OUTPUT|794493|66274|320158|7310186425|66274 / 7310252699|320158 / 1114651|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 7310252699 (0.000000%)|0 / 1114651 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|0.987661|85532 / 7310252699 (0.001170%)|27494 / 1114651 (2.466602%)|0.936331|0.950586|0.950885|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|ANY_CANDIDATE|0.963946|41671 / 7310252699 (0.000570%)|80368 / 1114651 (7.210149%)|0.954406|0.944290|0.944429|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|0.927063|0.975334|0.999988|0.987661|0.999985|0.000015|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|ANY_CANDIDATE|0.961271|0.927899|0.999994|0.963946|0.999983|0.000017|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|0.936331|0.950586|0.965282|0.905826|0.950892|0.950885|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|ANY_CANDIDATE|0.954406|0.944290|0.934386|0.894459|0.944437|0.944429|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1114651|0|0|7310252699|0 / 7310252699|0 / 1114651|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|1087157|85532|27494|7310167167|85532 / 7310252699|27494 / 1114651|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|ANY_CANDIDATE|1034283|41671|80368|7310211028|41671 / 7310252699|80368 / 1114651|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999997|38073 / 7310252699 (0.000521%)|0 / 1114651 (0.000000%)|0.973401|0.983208|0.983344|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|0.987657|143085 / 7310252699 (0.001957%)|27494 / 1114651 (2.466602%)|0.900618|0.927255|0.928372|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|ALL_CANDIDATES|0.963944|81865 / 7310252699 (0.001120%)|80368 / 1114651 (7.210149%)|0.926903|0.927276|0.927265|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.966971|1.000000|0.999995|0.999997|0.999995|0.000005|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|0.883694|0.975334|0.999980|0.987657|0.999977|0.000023|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|ALL_CANDIDATES|0.926654|0.927899|0.999989|0.963944|0.999978|0.000022|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.973401|0.983208|0.993215|0.966971|0.983347|0.983344|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|0.900618|0.927255|0.955516|0.864376|0.928384|0.928372|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|ALL_CANDIDATES|0.926903|0.927276|0.927649|0.864412|0.927276|0.927265|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|1114651|38073|0|7310214626|38073 / 7310252699|0 / 1114651|
|
||||||
|
|2|POLISH LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|1087157|143085|27494|7310109614|143085 / 7310252699|27494 / 1114651|
|
||||||
|
|3|HUNSPELL POLISH LUCENE FILTER|ALL_CANDIDATES|1034283|81865|80368|7310170834|81865 / 7310252699|80368 / 1114651|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|HUNSPELL POLISH LUCENE FILTER|68299|10279|29915|11265|9.315692%|6|133595|
|
||||||
|
|POLISH LUCENE MORFOLOGIK FILTER|88019|13692|43861|12763|10.554476%|5|135105|
|
||||||
|
|Radixor|21000|13669|24404|2780|2.298946%|4|124274|
|
||||||
|
|
||||||
|
### Output Policies and Metric Definitions
|
||||||
|
|
||||||
|
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
|
||||||
|
|
||||||
|
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
|
||||||
|
|
||||||
|
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
|
||||||
|
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
|
||||||
|
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
|
||||||
|
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
|
||||||
|
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
|
||||||
|
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
|
||||||
|
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
|
||||||
|
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
|
||||||
|
- Jaccard index: `TP / (TP + FP + FN)`.
|
||||||
|
- Fowlkes–Mallows index: `sqrt(precision * recall)`.
|
||||||
|
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
|
||||||
|
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
|
||||||
|
|
||||||
|
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
|
||||||
|
|
||||||
|
### Provenance
|
||||||
|
|
||||||
|
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
|
||||||
|
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
|
||||||
|
- Evaluation command: `./gradlew stemmingQuality`
|
||||||
|
- Dictionary language: `PL_PL`
|
||||||
|
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
|
||||||
|
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
|
||||||
|
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:END -->
|
||||||
|
|||||||
@@ -26,7 +26,7 @@ Radixor stores the preferred transformation for each normalized dictionary word
|
|||||||
|
|
||||||
## Accuracy
|
## Accuracy
|
||||||
|
|
||||||
Accuracy is computed from one deterministic JMH measurement iteration without warmup. The benchmark may execute the full dictionary pass more than once inside that single timed iteration; percentages divide matching counters by evaluated counters from the same iteration.
|
Accuracy is computed from JMH auxiliary counters in the current report. The counters are deterministic for a fixed corpus and stemmer; percentages divide matching counters by evaluated counters from the same report and are not timing metrics.
|
||||||
|
|
||||||
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
||||||
| --- | ---: | ---: | ---: | --- |
|
| --- | ---: | ---: | ---: | --- |
|
||||||
@@ -43,12 +43,12 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
|
|
||||||
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
||||||
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
||||||
| Radixor | `portugueseRadixor` | 10.598 | 0.273 | 51.1 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
| Radixor | `portugueseRadixor` | 12.109 | 0.698 | 58.4 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
||||||
| Lucene PortugueseLightStemFilter | `portugueseLucenePortugueseLightStemFilter` | 10.101 | 0.389 | 48.7 | 0.953 | Light Portuguese suffix stemmer. |
|
| Lucene PortugueseLightStemFilter | `portugueseLucenePortugueseLightStemFilter` | 11.172 | 1.870 | 53.8 | 0.923 | Light Portuguese suffix stemmer. |
|
||||||
| Lucene PortugueseMinimalStemFilter | `portugueseLucenePortugueseMinimalStemFilter` | 14.493 | 0.760 | 69.8 | 1.367 | Minimal Portuguese suffix reducer. |
|
| Lucene PortugueseMinimalStemFilter | `portugueseLucenePortugueseMinimalStemFilter` | 16.038 | 1.752 | 77.3 | 1.325 | Minimal Portuguese suffix reducer. |
|
||||||
| Official Snowball direct | `snowballDirect[PORTUGUESE]` | 52.508 | 6.338 | 253.1 | 4.954 | Official Snowball generated Java stemmer; direct API. |
|
| Official Snowball direct | `snowballDirect[PORTUGUESE]` | 53.725 | 5.356 | 258.9 | 4.437 | Official Snowball generated Java stemmer; direct API. |
|
||||||
| Lucene SnowballFilter | `luceneSnowballFilter[PORTUGUESE]` | 54.048 | 1.173 | 260.5 | 5.100 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
| Lucene SnowballFilter | `luceneSnowballFilter[PORTUGUESE]` | 57.457 | 1.182 | 276.9 | 4.745 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
||||||
| Lucene PortugueseStemFilter | `portugueseLucenePortugueseStemFilter` | 141.208 | 12.785 | 680.6 | 13.324 | Portuguese RSLP-style Lucene TokenFilter. |
|
| Lucene PortugueseStemFilter | `portugueseLucenePortugueseStemFilter` | 165.447 | 40.334 | 797.4 | 13.663 | Portuguese RSLP-style Lucene TokenFilter. |
|
||||||
|
|
||||||
## Interpretation Notes
|
## Interpretation Notes
|
||||||
|
|
||||||
@@ -57,3 +57,376 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
||||||
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
||||||
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:START -->
|
||||||
|
|
||||||
|
## Stemming Quality
|
||||||
|
|
||||||
|
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `PT_PT` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
|
||||||
|
|
||||||
|
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
|
||||||
|
|
||||||
|
### Evaluation Scope and Key Findings
|
||||||
|
|
||||||
|
The dictionary resource is `src/main/resources/pt_pt/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
|
||||||
|
|
||||||
|
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.998502** among 6 deterministic stemmers. The runner-up is `SNOWBALL PORTUGUESE DIRECT` at 0.938800, a difference of 0.059702. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.998502** among 6 deterministic stemmers. The runner-up is `SNOWBALL PORTUGUESE DIRECT` at 0.938800, a difference of 0.059702. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
### `ALL_WORDS`
|
||||||
|
|
||||||
|
This mode contains **8 result rows**, **6 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.998502|20678 / 22358203756 (0.000092%)|16444 / 5489060 (0.299578%)|0.996389|0.996620|0.996619|
|
||||||
|
|2|SNOWBALL PORTUGUESE DIRECT|PRIMARY_OUTPUT|0.938800|167230 / 22358203756 (0.000748%)|671821 / 5489060 (12.239272%)|0.947271|0.919888|0.920940|
|
||||||
|
|3|SNOWBALL PORTUGUESE LUCENE FILTER|PRIMARY_OUTPUT|0.938800|167230 / 22358203756 (0.000748%)|671821 / 5489060 (12.239272%)|0.947271|0.919888|0.920940|
|
||||||
|
|4|PORTUGUESE LUCENE PORTUGUESE STEM FILTER|PRIMARY_OUTPUT|0.846459|99075 / 22358203756 (0.000443%)|1685572 / 5489060 (30.707844%)|0.901330|0.809975|0.821750|
|
||||||
|
|5|PORTUGUESE LUCENE PORTUGUESE LIGHT STEM FILTER|PRIMARY_OUTPUT|0.513648|2511 / 22358203756 (0.000011%)|5339230 / 5489060 (97.270389%)|0.122843|0.053118|0.163828|
|
||||||
|
|6|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.503954|598 / 22358203756 (0.000003%)|5445654 / 5489060 (99.209227%)|0.038310|0.015690|0.088308|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.996236|0.997004|0.999999|0.998502|0.999998|0.000002|
|
||||||
|
|2|SNOWBALL PORTUGUESE DIRECT|PRIMARY_OUTPUT|0.966450|0.877607|0.999993|0.938800|0.999962|0.000038|
|
||||||
|
|3|SNOWBALL PORTUGUESE LUCENE FILTER|PRIMARY_OUTPUT|0.966450|0.877607|0.999993|0.938800|0.999962|0.000038|
|
||||||
|
|4|PORTUGUESE LUCENE PORTUGUESE STEM FILTER|PRIMARY_OUTPUT|0.974613|0.692922|0.999996|0.846459|0.999920|0.000080|
|
||||||
|
|5|PORTUGUESE LUCENE PORTUGUESE LIGHT STEM FILTER|PRIMARY_OUTPUT|0.983517|0.027296|1.000000|0.513648|0.999761|0.000239|
|
||||||
|
|6|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.986410|0.007908|1.000000|0.503954|0.999756|0.000244|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.996389|0.996620|0.996850|0.993262|0.996620|0.996619|
|
||||||
|
|2|SNOWBALL PORTUGUESE DIRECT|PRIMARY_OUTPUT|0.947271|0.919888|0.894045|0.851661|0.920958|0.920940|
|
||||||
|
|3|SNOWBALL PORTUGUESE LUCENE FILTER|PRIMARY_OUTPUT|0.947271|0.919888|0.894045|0.851661|0.920958|0.920940|
|
||||||
|
|4|PORTUGUESE LUCENE PORTUGUESE STEM FILTER|PRIMARY_OUTPUT|0.901330|0.809975|0.735434|0.680636|0.821785|0.821750|
|
||||||
|
|5|PORTUGUESE LUCENE PORTUGUESE LIGHT STEM FILTER|PRIMARY_OUTPUT|0.122843|0.053118|0.033885|0.027284|0.163848|0.163828|
|
||||||
|
|6|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.038310|0.015690|0.009865|0.007907|0.088319|0.088308|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.996619|0.999299|0.999347|0.999323|0.999323|
|
||||||
|
|2|SNOWBALL PORTUGUESE DIRECT|PRIMARY_OUTPUT|0.919870|0.996663|0.967924|0.982083|0.982083|
|
||||||
|
|3|SNOWBALL PORTUGUESE LUCENE FILTER|PRIMARY_OUTPUT|0.919870|0.996663|0.967924|0.982083|0.982083|
|
||||||
|
|4|PORTUGUESE LUCENE PORTUGUESE STEM FILTER|PRIMARY_OUTPUT|0.809936|0.996729|0.918475|0.956003|0.956003|
|
||||||
|
|5|PORTUGUESE LUCENE PORTUGUESE LIGHT STEM FILTER|PRIMARY_OUTPUT|0.053105|0.999226|0.720580|0.837330|0.837330|
|
||||||
|
|6|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.015686|0.999664|0.692383|0.818122|0.818122|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|5472616|20678|16444|22358183078|20678 / 22358203756|16444 / 5489060|
|
||||||
|
|2|SNOWBALL PORTUGUESE DIRECT|PRIMARY_OUTPUT|4817239|167230|671821|22358036526|167230 / 22358203756|671821 / 5489060|
|
||||||
|
|3|SNOWBALL PORTUGUESE LUCENE FILTER|PRIMARY_OUTPUT|4817239|167230|671821|22358036526|167230 / 22358203756|671821 / 5489060|
|
||||||
|
|4|PORTUGUESE LUCENE PORTUGUESE STEM FILTER|PRIMARY_OUTPUT|3803488|99075|1685572|22358104681|99075 / 22358203756|1685572 / 5489060|
|
||||||
|
|5|PORTUGUESE LUCENE PORTUGUESE LIGHT STEM FILTER|PRIMARY_OUTPUT|149830|2511|5339230|22358201245|2511 / 22358203756|5339230 / 5489060|
|
||||||
|
|6|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|PRIMARY_OUTPUT|43406|598|5445654|22358203158|598 / 22358203756|5445654 / 5489060|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 22358203756 (0.000000%)|0 / 5489060 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|5489060|0|0|22358203756|0 / 22358203756|0 / 5489060|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999999|38310 / 22358203756 (0.000171%)|0 / 5489060 (0.000000%)|0.994448|0.996522|0.996528|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.993069|1.000000|0.999998|0.999999|0.999998|0.000002|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.994448|0.996522|0.998606|0.993069|0.996528|0.996528|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|5489060|38310|0|22358165446|38310 / 22358203756|0 / 5489060|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|16444|20678|17632|790|0.373542%|3|212297|
|
||||||
|
|
||||||
|
### `LOWERCASE_GROUPS_ONLY`
|
||||||
|
|
||||||
|
This mode contains **8 result rows**, **6 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.998502|20678 / 22358203756 (0.000092%)|16444 / 5489060 (0.299578%)|0.996389|0.996620|0.996619|
|
||||||
|
|2|SNOWBALL PORTUGUESE DIRECT|PRIMARY_OUTPUT|0.938800|167230 / 22358203756 (0.000748%)|671821 / 5489060 (12.239272%)|0.947271|0.919888|0.920940|
|
||||||
|
|3|SNOWBALL PORTUGUESE LUCENE FILTER|PRIMARY_OUTPUT|0.938800|167230 / 22358203756 (0.000748%)|671821 / 5489060 (12.239272%)|0.947271|0.919888|0.920940|
|
||||||
|
|4|PORTUGUESE LUCENE PORTUGUESE STEM FILTER|PRIMARY_OUTPUT|0.846459|99075 / 22358203756 (0.000443%)|1685572 / 5489060 (30.707844%)|0.901330|0.809975|0.821750|
|
||||||
|
|5|PORTUGUESE LUCENE PORTUGUESE LIGHT STEM FILTER|PRIMARY_OUTPUT|0.513648|2511 / 22358203756 (0.000011%)|5339230 / 5489060 (97.270389%)|0.122843|0.053118|0.163828|
|
||||||
|
|6|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.503954|598 / 22358203756 (0.000003%)|5445654 / 5489060 (99.209227%)|0.038310|0.015690|0.088308|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.996236|0.997004|0.999999|0.998502|0.999998|0.000002|
|
||||||
|
|2|SNOWBALL PORTUGUESE DIRECT|PRIMARY_OUTPUT|0.966450|0.877607|0.999993|0.938800|0.999962|0.000038|
|
||||||
|
|3|SNOWBALL PORTUGUESE LUCENE FILTER|PRIMARY_OUTPUT|0.966450|0.877607|0.999993|0.938800|0.999962|0.000038|
|
||||||
|
|4|PORTUGUESE LUCENE PORTUGUESE STEM FILTER|PRIMARY_OUTPUT|0.974613|0.692922|0.999996|0.846459|0.999920|0.000080|
|
||||||
|
|5|PORTUGUESE LUCENE PORTUGUESE LIGHT STEM FILTER|PRIMARY_OUTPUT|0.983517|0.027296|1.000000|0.513648|0.999761|0.000239|
|
||||||
|
|6|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.986410|0.007908|1.000000|0.503954|0.999756|0.000244|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.996389|0.996620|0.996850|0.993262|0.996620|0.996619|
|
||||||
|
|2|SNOWBALL PORTUGUESE DIRECT|PRIMARY_OUTPUT|0.947271|0.919888|0.894045|0.851661|0.920958|0.920940|
|
||||||
|
|3|SNOWBALL PORTUGUESE LUCENE FILTER|PRIMARY_OUTPUT|0.947271|0.919888|0.894045|0.851661|0.920958|0.920940|
|
||||||
|
|4|PORTUGUESE LUCENE PORTUGUESE STEM FILTER|PRIMARY_OUTPUT|0.901330|0.809975|0.735434|0.680636|0.821785|0.821750|
|
||||||
|
|5|PORTUGUESE LUCENE PORTUGUESE LIGHT STEM FILTER|PRIMARY_OUTPUT|0.122843|0.053118|0.033885|0.027284|0.163848|0.163828|
|
||||||
|
|6|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.038310|0.015690|0.009865|0.007907|0.088319|0.088308|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.996619|0.999299|0.999347|0.999323|0.999323|
|
||||||
|
|2|SNOWBALL PORTUGUESE DIRECT|PRIMARY_OUTPUT|0.919870|0.996663|0.967924|0.982083|0.982083|
|
||||||
|
|3|SNOWBALL PORTUGUESE LUCENE FILTER|PRIMARY_OUTPUT|0.919870|0.996663|0.967924|0.982083|0.982083|
|
||||||
|
|4|PORTUGUESE LUCENE PORTUGUESE STEM FILTER|PRIMARY_OUTPUT|0.809936|0.996729|0.918475|0.956003|0.956003|
|
||||||
|
|5|PORTUGUESE LUCENE PORTUGUESE LIGHT STEM FILTER|PRIMARY_OUTPUT|0.053105|0.999226|0.720580|0.837330|0.837330|
|
||||||
|
|6|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.015686|0.999664|0.692383|0.818122|0.818122|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|5472616|20678|16444|22358183078|20678 / 22358203756|16444 / 5489060|
|
||||||
|
|2|SNOWBALL PORTUGUESE DIRECT|PRIMARY_OUTPUT|4817239|167230|671821|22358036526|167230 / 22358203756|671821 / 5489060|
|
||||||
|
|3|SNOWBALL PORTUGUESE LUCENE FILTER|PRIMARY_OUTPUT|4817239|167230|671821|22358036526|167230 / 22358203756|671821 / 5489060|
|
||||||
|
|4|PORTUGUESE LUCENE PORTUGUESE STEM FILTER|PRIMARY_OUTPUT|3803488|99075|1685572|22358104681|99075 / 22358203756|1685572 / 5489060|
|
||||||
|
|5|PORTUGUESE LUCENE PORTUGUESE LIGHT STEM FILTER|PRIMARY_OUTPUT|149830|2511|5339230|22358201245|2511 / 22358203756|5339230 / 5489060|
|
||||||
|
|6|PORTUGUESE LUCENE PORTUGUESE MINIMAL STEM FILTER|PRIMARY_OUTPUT|43406|598|5445654|22358203158|598 / 22358203756|5445654 / 5489060|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 22358203756 (0.000000%)|0 / 5489060 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|5489060|0|0|22358203756|0 / 22358203756|0 / 5489060|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999999|38310 / 22358203756 (0.000171%)|0 / 5489060 (0.000000%)|0.994448|0.996522|0.996528|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.993069|1.000000|0.999998|0.999999|0.999998|0.000002|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.994448|0.996522|0.998606|0.993069|0.996528|0.996528|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|5489060|38310|0|22358165446|38310 / 22358203756|0 / 5489060|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|16444|20678|17632|790|0.373542%|3|212297|
|
||||||
|
|
||||||
|
### Output Policies and Metric Definitions
|
||||||
|
|
||||||
|
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
|
||||||
|
|
||||||
|
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
|
||||||
|
|
||||||
|
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
|
||||||
|
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
|
||||||
|
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
|
||||||
|
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
|
||||||
|
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
|
||||||
|
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
|
||||||
|
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
|
||||||
|
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
|
||||||
|
- Jaccard index: `TP / (TP + FP + FN)`.
|
||||||
|
- Fowlkes–Mallows index: `sqrt(precision * recall)`.
|
||||||
|
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
|
||||||
|
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
|
||||||
|
|
||||||
|
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
|
||||||
|
|
||||||
|
### Provenance
|
||||||
|
|
||||||
|
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
|
||||||
|
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
|
||||||
|
- Evaluation command: `./gradlew stemmingQuality`
|
||||||
|
- Dictionary language: `PT_PT`
|
||||||
|
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
|
||||||
|
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
|
||||||
|
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:END -->
|
||||||
|
|||||||
@@ -26,7 +26,7 @@ Radixor stores the preferred transformation for each normalized dictionary word
|
|||||||
|
|
||||||
## Accuracy
|
## Accuracy
|
||||||
|
|
||||||
Accuracy is computed from one deterministic JMH measurement iteration without warmup. The benchmark may execute the full dictionary pass more than once inside that single timed iteration; percentages divide matching counters by evaluated counters from the same iteration.
|
Accuracy is computed from JMH auxiliary counters in the current report. The counters are deterministic for a fixed corpus and stemmer; percentages divide matching counters by evaluated counters from the same report and are not timing metrics.
|
||||||
|
|
||||||
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
||||||
| --- | ---: | ---: | ---: | --- |
|
| --- | ---: | ---: | ---: | --- |
|
||||||
@@ -41,10 +41,10 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
|
|
||||||
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
||||||
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
||||||
| Radixor | `russianRadixor` | 72.970 | 15.642 | 99.8 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
| Radixor | `russianRadixor` | 89.671 | 3.886 | 122.6 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
||||||
| Lucene RussianLightStemFilter | `russianLuceneRussianLightStemFilter` | 57.900 | 4.404 | 79.2 | 0.793 | Light Russian suffix stemmer. |
|
| Lucene RussianLightStemFilter | `russianLuceneRussianLightStemFilter` | 60.522 | 5.310 | 82.7 | 0.675 | Light Russian suffix stemmer. |
|
||||||
| Official Snowball direct | `snowballDirect[RUSSIAN]` | 99.019 | 14.688 | 135.4 | 1.357 | Official Snowball generated Java stemmer; direct API. |
|
| Official Snowball direct | `snowballDirect[RUSSIAN]` | 106.031 | 9.287 | 145.0 | 1.182 | Official Snowball generated Java stemmer; direct API. |
|
||||||
| Lucene SnowballFilter | `luceneSnowballFilter[RUSSIAN]` | 128.272 | 7.201 | 175.4 | 1.758 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
| Lucene SnowballFilter | `luceneSnowballFilter[RUSSIAN]` | 137.512 | 10.801 | 188.0 | 1.534 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
||||||
|
|
||||||
## Interpretation Notes
|
## Interpretation Notes
|
||||||
|
|
||||||
@@ -53,3 +53,356 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
||||||
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
||||||
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:START -->
|
||||||
|
|
||||||
|
## Stemming Quality
|
||||||
|
|
||||||
|
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `RU_RU` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
|
||||||
|
|
||||||
|
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
|
||||||
|
|
||||||
|
### Evaluation Scope and Key Findings
|
||||||
|
|
||||||
|
The dictionary resource is `src/main/resources/ru_ru/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
|
||||||
|
|
||||||
|
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.989827** among 4 deterministic stemmers. The runner-up is `SNOWBALL RUSSIAN LUCENE FILTER` at 0.834876, a difference of 0.154951. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.989852** among 4 deterministic stemmers. The runner-up is `SNOWBALL RUSSIAN DIRECT` at 0.834854, a difference of 0.154998. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
### `ALL_WORDS`
|
||||||
|
|
||||||
|
This mode contains **6 result rows**, **4 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.989827|155850 / 295576291016 (0.000053%)|266302 / 13089505 (2.034470%)|0.986313|0.983806|0.983814|
|
||||||
|
|2|SNOWBALL RUSSIAN LUCENE FILTER|PRIMARY_OUTPUT|0.834876|3785790 / 295576291016 (0.001281%)|4322616 / 13089505 (33.023525%)|0.692485|0.683786|0.683923|
|
||||||
|
|3|SNOWBALL RUSSIAN DIRECT|PRIMARY_OUTPUT|0.834867|3782908 / 295576291016 (0.001280%)|4322849 / 13089505 (33.025305%)|0.692603|0.683851|0.683989|
|
||||||
|
|4|RUSSIAN LUCENE RUSSIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.617692|321183 / 295576291016 (0.000109%)|10008438 / 13089505 (76.461547%)|0.577011|0.373649|0.461687|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.987992|0.979655|0.999999|0.989827|0.999999|0.000001|
|
||||||
|
|2|SNOWBALL RUSSIAN LUCENE FILTER|PRIMARY_OUTPUT|0.698408|0.669765|0.999987|0.834876|0.999973|0.000027|
|
||||||
|
|3|SNOWBALL RUSSIAN DIRECT|PRIMARY_OUTPUT|0.698563|0.669747|0.999987|0.834867|0.999973|0.000027|
|
||||||
|
|4|RUSSIAN LUCENE RUSSIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.905597|0.235385|0.999999|0.617692|0.999965|0.000035|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.986313|0.983806|0.981311|0.968128|0.983815|0.983814|
|
||||||
|
|2|SNOWBALL RUSSIAN LUCENE FILTER|PRIMARY_OUTPUT|0.692485|0.683786|0.675304|0.519510|0.683936|0.683923|
|
||||||
|
|3|SNOWBALL RUSSIAN DIRECT|PRIMARY_OUTPUT|0.692603|0.683851|0.675318|0.519585|0.684003|0.683989|
|
||||||
|
|4|RUSSIAN LUCENE RUSSIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.577011|0.373649|0.276278|0.229747|0.461696|0.461687|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.983805|0.997699|0.997274|0.997487|0.997487|
|
||||||
|
|2|SNOWBALL RUSSIAN LUCENE FILTER|PRIMARY_OUTPUT|0.683773|0.974131|0.953674|0.963794|0.963794|
|
||||||
|
|3|SNOWBALL RUSSIAN DIRECT|PRIMARY_OUTPUT|0.683838|0.974180|0.953661|0.963811|0.963811|
|
||||||
|
|4|RUSSIAN LUCENE RUSSIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.373638|0.994311|0.870888|0.928516|0.928516|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|12823203|155850|266302|295576135166|155850 / 295576291016|266302 / 13089505|
|
||||||
|
|2|SNOWBALL RUSSIAN LUCENE FILTER|PRIMARY_OUTPUT|8766889|3785790|4322616|295572505226|3785790 / 295576291016|4322616 / 13089505|
|
||||||
|
|3|SNOWBALL RUSSIAN DIRECT|PRIMARY_OUTPUT|8766656|3782908|4322849|295572508108|3782908 / 295576291016|4322849 / 13089505|
|
||||||
|
|4|RUSSIAN LUCENE RUSSIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|3081067|321183|10008438|295575969833|321183 / 295576291016|10008438 / 13089505|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 295576291016 (0.000000%)|13 / 13089505 (0.000099%)|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0.999999|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|0.999999|0.999999|1.000000|1.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|13089492|0|13|295576291016|0 / 295576291016|13 / 13089505|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999999|434710 / 295576291016 (0.000147%)|13 / 13089505 (0.000099%)|0.974119|0.983665|0.983796|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.967857|0.999999|0.999999|0.999999|0.999999|0.000001|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.974119|0.983665|0.993401|0.967856|0.983797|0.983796|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|13089492|434710|13|295575856306|434710 / 295576291016|13 / 13089505|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|266289|155850|278860|19162|2.492190%|4|788492|
|
||||||
|
|
||||||
|
### `LOWERCASE_GROUPS_ONLY`
|
||||||
|
|
||||||
|
This mode contains **6 result rows**, **4 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.989852|155850 / 295000681652 (0.000053%)|265613 / 13087126 (2.029575%)|0.986322|0.983830|0.983838|
|
||||||
|
|2|SNOWBALL RUSSIAN DIRECT|PRIMARY_OUTPUT|0.834854|3782908 / 295000681652 (0.001282%)|4322407 / 13087126 (33.027931%)|0.692561|0.683815|0.683953|
|
||||||
|
|3|SNOWBALL RUSSIAN LUCENE FILTER|PRIMARY_OUTPUT|0.834854|3782908 / 295000681652 (0.001282%)|4322407 / 13087126 (33.027931%)|0.692561|0.683815|0.683953|
|
||||||
|
|4|RUSSIAN LUCENE RUSSIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.617630|318921 / 295000681652 (0.000108%)|10008238 / 13087126 (76.473918%)|0.577038|0.373540|0.461703|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.987991|0.979704|0.999999|0.989852|0.999999|0.000001|
|
||||||
|
|2|SNOWBALL RUSSIAN DIRECT|PRIMARY_OUTPUT|0.698516|0.669721|0.999987|0.834854|0.999973|0.000027|
|
||||||
|
|3|SNOWBALL RUSSIAN LUCENE FILTER|PRIMARY_OUTPUT|0.698516|0.669721|0.999987|0.834854|0.999973|0.000027|
|
||||||
|
|4|RUSSIAN LUCENE RUSSIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.906139|0.235261|0.999999|0.617630|0.999965|0.000035|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.986322|0.983830|0.981350|0.968175|0.983839|0.983838|
|
||||||
|
|2|SNOWBALL RUSSIAN DIRECT|PRIMARY_OUTPUT|0.692561|0.683815|0.675288|0.519544|0.683967|0.683953|
|
||||||
|
|3|SNOWBALL RUSSIAN LUCENE FILTER|PRIMARY_OUTPUT|0.692561|0.683815|0.675288|0.519544|0.683967|0.683953|
|
||||||
|
|4|RUSSIAN LUCENE RUSSIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.577038|0.373540|0.276152|0.229664|0.461713|0.461703|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.983829|0.997697|0.997321|0.997509|0.997509|
|
||||||
|
|2|SNOWBALL RUSSIAN DIRECT|PRIMARY_OUTPUT|0.683802|0.974149|0.953634|0.963782|0.963782|
|
||||||
|
|3|SNOWBALL RUSSIAN LUCENE FILTER|PRIMARY_OUTPUT|0.683802|0.974149|0.953634|0.963782|0.963782|
|
||||||
|
|4|RUSSIAN LUCENE RUSSIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|0.373528|0.994350|0.870767|0.928464|0.928464|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|12821513|155850|265613|295000525802|155850 / 295000681652|265613 / 13087126|
|
||||||
|
|2|SNOWBALL RUSSIAN DIRECT|PRIMARY_OUTPUT|8764719|3782908|4322407|294996898744|3782908 / 295000681652|4322407 / 13087126|
|
||||||
|
|3|SNOWBALL RUSSIAN LUCENE FILTER|PRIMARY_OUTPUT|8764719|3782908|4322407|294996898744|3782908 / 295000681652|4322407 / 13087126|
|
||||||
|
|4|RUSSIAN LUCENE RUSSIAN LIGHT STEM FILTER|PRIMARY_OUTPUT|3078888|318921|10008238|295000362731|318921 / 295000681652|10008238 / 13087126|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 295000681652 (0.000000%)|0 / 13087126 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|13087126|0|0|295000681652|0 / 295000681652|0 / 13087126|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999999|434710 / 295000681652 (0.000147%)|0 / 13087126 (0.000000%)|0.974115|0.983663|0.983794|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.967851|1.000000|0.999999|0.999999|0.999999|0.000001|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.974115|0.983663|0.993401|0.967851|0.983794|0.983794|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|13087126|434710|0|295000246942|434710 / 295000681652|0 / 13087126|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|265613|155850|278860|18991|2.472358%|4|787549|
|
||||||
|
|
||||||
|
### Output Policies and Metric Definitions
|
||||||
|
|
||||||
|
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
|
||||||
|
|
||||||
|
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
|
||||||
|
|
||||||
|
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
|
||||||
|
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
|
||||||
|
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
|
||||||
|
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
|
||||||
|
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
|
||||||
|
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
|
||||||
|
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
|
||||||
|
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
|
||||||
|
- Jaccard index: `TP / (TP + FP + FN)`.
|
||||||
|
- Fowlkes–Mallows index: `sqrt(precision * recall)`.
|
||||||
|
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
|
||||||
|
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
|
||||||
|
|
||||||
|
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
|
||||||
|
|
||||||
|
### Provenance
|
||||||
|
|
||||||
|
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
|
||||||
|
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
|
||||||
|
- Evaluation command: `./gradlew stemmingQuality`
|
||||||
|
- Dictionary language: `RU_RU`
|
||||||
|
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
|
||||||
|
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
|
||||||
|
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:END -->
|
||||||
|
|||||||
@@ -26,11 +26,12 @@ Radixor stores the preferred transformation for each normalized dictionary word
|
|||||||
|
|
||||||
## Accuracy
|
## Accuracy
|
||||||
|
|
||||||
Accuracy is computed from one deterministic JMH measurement iteration without warmup. The benchmark may execute the full dictionary pass more than once inside that single timed iteration; percentages divide matching counters by evaluated counters from the same iteration.
|
Accuracy is computed from JMH auxiliary counters in the current report. The counters are deterministic for a fixed corpus and stemmer; percentages divide matching counters by evaluated counters from the same report and are not timing metrics.
|
||||||
|
|
||||||
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
||||||
| --- | ---: | ---: | ---: | --- |
|
| --- | ---: | ---: | ---: | --- |
|
||||||
| Radixor | 97.459% | 97.544% | 96.891% | Full Radixor dictionary patch-command stemmer. |
|
| Radixor | 97.459% | 97.544% | 96.891% | Full Radixor dictionary patch-command stemmer. |
|
||||||
|
| Lucene HunspellStemFilter | 49.074% | 42.656% | 92.154% | Benchmark-only Spanish Hunspell dictionary compared via Lucene HunspellStemFilter. |
|
||||||
| Lucene SpanishMinimalStemFilter | 17.284% | 5.347% | 97.403% | Minimal suffix reducer; narrow baseline, not a full stemmer. |
|
| Lucene SpanishMinimalStemFilter | 17.284% | 5.347% | 97.403% | Minimal suffix reducer; narrow baseline, not a full stemmer. |
|
||||||
| Lucene SpanishPluralStemFilter | 15.140% | 5.802% | 77.820% | Plural-focused suffix reducer; narrow baseline. |
|
| Lucene SpanishPluralStemFilter | 15.140% | 5.802% | 77.820% | Plural-focused suffix reducer; narrow baseline. |
|
||||||
| Lucene SpanishLightStemFilter | 9.577% | 7.088% | 26.279% | Light suffix stemmer; intentionally narrower than a dictionary-derived stemmer. |
|
| Lucene SpanishLightStemFilter | 9.577% | 7.088% | 26.279% | Light suffix stemmer; intentionally narrower than a dictionary-derived stemmer. |
|
||||||
@@ -43,12 +44,13 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
|
|
||||||
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
||||||
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
||||||
| Radixor | `spanishRadixor` | 64.539 | 3.448 | 80.0 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
| Radixor | `spanishRadixor` | 78.919 | 7.253 | 97.9 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
||||||
| Lucene SpanishMinimalStemFilter | `spanishLuceneSpanishMinimalStemFilter` | 38.288 | 2.382 | 47.5 | 0.593 | Minimal Spanish suffix reducer; narrow baseline. |
|
| Lucene HunspellStemFilter | `luceneHunspellStemFilter` | 2079.041 | 193.548 | 2578.6 | 26.344 | Benchmark-only Spanish Hunspell dictionary compared via Lucene HunspellStemFilter. |
|
||||||
| Lucene SpanishLightStemFilter | `spanishLuceneSpanishLightStemFilter` | 40.855 | 2.407 | 50.7 | 0.633 | Light Spanish suffix stemmer. |
|
| Lucene SpanishMinimalStemFilter | `spanishLuceneSpanishMinimalStemFilter` | 45.596 | 4.639 | 56.6 | 0.578 | Minimal Spanish suffix reducer; narrow baseline. |
|
||||||
| Lucene SpanishPluralStemFilter | `spanishLuceneSpanishPluralStemFilter` | 91.054 | 8.822 | 112.9 | 1.411 | Plural-oriented Spanish suffix reducer. |
|
| Lucene SpanishLightStemFilter | `spanishLuceneSpanishLightStemFilter` | 42.003 | 1.683 | 52.1 | 0.532 | Light Spanish suffix stemmer. |
|
||||||
| Official Snowball direct | `snowballDirect[SPANISH]` | 168.813 | 32.860 | 209.4 | 2.616 | Official Snowball generated Java stemmer; direct API. |
|
| Lucene SpanishPluralStemFilter | `spanishLuceneSpanishPluralStemFilter` | 93.734 | 6.247 | 116.3 | 1.188 | Plural-oriented Spanish suffix reducer. |
|
||||||
| Lucene SnowballFilter | `luceneSnowballFilter[SPANISH]` | 185.626 | 50.297 | 230.2 | 2.876 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
| Official Snowball direct | `snowballDirect[SPANISH]` | 171.995 | 11.035 | 213.3 | 2.179 | Official Snowball generated Java stemmer; direct API. |
|
||||||
|
| Lucene SnowballFilter | `luceneSnowballFilter[SPANISH]` | 211.138 | 17.940 | 261.9 | 2.675 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
||||||
|
|
||||||
## Interpretation Notes
|
## Interpretation Notes
|
||||||
|
|
||||||
@@ -57,3 +59,408 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
||||||
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
||||||
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:START -->
|
||||||
|
|
||||||
|
## Stemming Quality
|
||||||
|
|
||||||
|
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `ES_ES` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
|
||||||
|
|
||||||
|
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
|
||||||
|
|
||||||
|
### Evaluation Scope and Key Findings
|
||||||
|
|
||||||
|
The dictionary resource is `src/main/resources/es_es/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
|
||||||
|
|
||||||
|
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.989295** among 7 deterministic stemmers. The runner-up is `SNOWBALL SPANISH LUCENE FILTER` at 0.652614, a difference of 0.336680. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.989429** among 7 deterministic stemmers. The runner-up is `SNOWBALL SPANISH DIRECT` at 0.652720, a difference of 0.336709. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
### `ALL_WORDS`
|
||||||
|
|
||||||
|
This mode contains **11 result rows**, **7 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.989295|288483 / 379567318110 (0.000076%)|898652 / 41973336 (2.141007%)|0.990105|0.985755|0.985780|
|
||||||
|
|2|SNOWBALL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.652614|2230481 / 379567318110 (0.000588%)|29161643 / 41973336 (69.476591%)|0.627151|0.449411|0.509848|
|
||||||
|
|3|SNOWBALL SPANISH DIRECT|PRIMARY_OUTPUT|0.652614|2228819 / 379567318110 (0.000587%)|29161649 / 41973336 (69.476605%)|0.627192|0.449424|0.509876|
|
||||||
|
|4|HUNSPELL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.615102|536192 / 379567318110 (0.000141%)|32310860 / 41973336 (76.979490%)|0.583708|0.370408|0.466992|
|
||||||
|
|5|SPANISH LUCENE SPANISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.514823|147956 / 379567318110 (0.000039%)|40729019 / 41973336 (97.035458%)|0.130864|0.057387|0.162762|
|
||||||
|
|6|SPANISH LUCENE SPANISH PLURAL STEM FILTER|PRIMARY_OUTPUT|0.503874|58578 / 379567318110 (0.000015%)|41648091 / 41973336 (99.225115%)|0.037377|0.015357|0.081026|
|
||||||
|
|7|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.501768|47859 / 379567318110 (0.000013%)|41824873 / 41973336 (99.646292%)|0.017361|0.007041|0.051714|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.993026|0.978590|0.999999|0.989295|0.999997|0.000003|
|
||||||
|
|2|SNOWBALL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.851718|0.305234|0.999994|0.652614|0.999917|0.000083|
|
||||||
|
|3|SNOWBALL SPANISH DIRECT|PRIMARY_OUTPUT|0.851812|0.305234|0.999994|0.652614|0.999917|0.000083|
|
||||||
|
|4|HUNSPELL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.947425|0.230205|0.999999|0.615102|0.999913|0.000087|
|
||||||
|
|5|SPANISH LUCENE SPANISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.893731|0.029645|1.000000|0.514823|0.999892|0.000108|
|
||||||
|
|6|SPANISH LUCENE SPANISH PLURAL STEM FILTER|PRIMARY_OUTPUT|0.847383|0.007749|1.000000|0.503874|0.999890|0.000110|
|
||||||
|
|7|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.756222|0.003537|1.000000|0.501768|0.999890|0.000110|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.990105|0.985755|0.981443|0.971910|0.985781|0.985780|
|
||||||
|
|2|SNOWBALL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.627151|0.449411|0.350170|0.289832|0.509876|0.509848|
|
||||||
|
|3|SNOWBALL SPANISH DIRECT|PRIMARY_OUTPUT|0.627192|0.449424|0.350173|0.289843|0.509904|0.509876|
|
||||||
|
|4|HUNSPELL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.583708|0.370408|0.271278|0.227301|0.467014|0.466992|
|
||||||
|
|5|SPANISH LUCENE SPANISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.130864|0.057387|0.036752|0.029541|0.162773|0.162762|
|
||||||
|
|6|SPANISH LUCENE SPANISH PLURAL STEM FILTER|PRIMARY_OUTPUT|0.037377|0.015357|0.009664|0.007738|0.081032|0.081026|
|
||||||
|
|7|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.017361|0.007041|0.004416|0.003533|0.051719|0.051714|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.985753|0.995418|0.993266|0.994341|0.994341|
|
||||||
|
|2|SNOWBALL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.449379|0.981386|0.852461|0.912391|0.912391|
|
||||||
|
|3|SNOWBALL SPANISH DIRECT|PRIMARY_OUTPUT|0.449392|0.981406|0.852463|0.912401|0.912401|
|
||||||
|
|4|HUNSPELL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.370381|0.993314|0.790558|0.880414|0.880414|
|
||||||
|
|5|SPANISH LUCENE SPANISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.057381|0.993824|0.756690|0.859195|0.859195|
|
||||||
|
|6|SPANISH LUCENE SPANISH PLURAL STEM FILTER|PRIMARY_OUTPUT|0.015355|0.995442|0.723731|0.838115|0.838115|
|
||||||
|
|7|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.007040|0.995635|0.710610|0.829316|0.829316|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|41074684|288483|898652|379567029627|288483 / 379567318110|898652 / 41973336|
|
||||||
|
|2|SNOWBALL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|12811693|2230481|29161643|379565087629|2230481 / 379567318110|29161643 / 41973336|
|
||||||
|
|3|SNOWBALL SPANISH DIRECT|PRIMARY_OUTPUT|12811687|2228819|29161649|379565089291|2228819 / 379567318110|29161649 / 41973336|
|
||||||
|
|4|HUNSPELL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|9662476|536192|32310860|379566781918|536192 / 379567318110|32310860 / 41973336|
|
||||||
|
|5|SPANISH LUCENE SPANISH LIGHT STEM FILTER|PRIMARY_OUTPUT|1244317|147956|40729019|379567170154|147956 / 379567318110|40729019 / 41973336|
|
||||||
|
|6|SPANISH LUCENE SPANISH PLURAL STEM FILTER|PRIMARY_OUTPUT|325245|58578|41648091|379567259532|58578 / 379567318110|41648091 / 41973336|
|
||||||
|
|7|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|148463|47859|41824873|379567270251|47859 / 379567318110|41824873 / 41973336|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|0.999993|2 / 379567318110 (0.000000%)|626 / 41973336 (0.001491%)|0.999997|0.999993|0.999993|
|
||||||
|
|2|HUNSPELL SPANISH LUCENE FILTER|ANY_CANDIDATE|0.620065|416345 / 379567318110 (0.000110%)|31894218 / 41973336 (75.986855%)|0.600268|0.384195|0.480192|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0.999985|1.000000|0.999993|1.000000|0.000000|
|
||||||
|
|2|HUNSPELL SPANISH LUCENE FILTER|ANY_CANDIDATE|0.960331|0.240131|0.999999|0.620065|0.999915|0.000085|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|0.999997|0.999993|0.999988|0.999985|0.999993|0.999993|
|
||||||
|
|2|HUNSPELL SPANISH LUCENE FILTER|ANY_CANDIDATE|0.600268|0.384195|0.282504|0.237773|0.480214|0.480192|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|HUNSPELL SPANISH LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|41972710|2|626|379567318108|2 / 379567318110|626 / 41973336|
|
||||||
|
|2|HUNSPELL SPANISH LUCENE FILTER|ANY_CANDIDATE|10079118|416345|31894218|379566901765|416345 / 379567318110|31894218 / 41973336|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999991|1349800 / 379567318110 (0.000356%)|626 / 41973336 (0.001491%)|0.974915|0.984168|0.984289|
|
||||||
|
|2|HUNSPELL SPANISH LUCENE FILTER|ALL_CANDIDATES|0.620065|888077 / 379567318110 (0.000234%)|31894218 / 41973336 (75.986855%)|0.587073|0.380771|0.469749|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.968843|0.999985|0.999996|0.999991|0.999996|0.000004|
|
||||||
|
|2|HUNSPELL SPANISH LUCENE FILTER|ALL_CANDIDATES|0.919024|0.240131|0.999998|0.620065|0.999914|0.000086|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.974915|0.984168|0.993598|0.968829|0.984291|0.984289|
|
||||||
|
|2|HUNSPELL SPANISH LUCENE FILTER|ALL_CANDIDATES|0.587073|0.380771|0.281759|0.235156|0.469773|0.469749|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|HUNSPELL SPANISH LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|41972710|1349800|626|379565968310|1349800 / 379567318110|626 / 41973336|
|
||||||
|
|2|HUNSPELL SPANISH LUCENE FILTER|ALL_CANDIDATES|10079118|888077|31894218|379566430033|888077 / 379567318110|31894218 / 41973336|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|HUNSPELL SPANISH LUCENE FILTER|416642|119847|351885|17877|2.051686%|5|890999|
|
||||||
|
|Radixor|898026|288481|1061317|42637|4.893313%|21|916797|
|
||||||
|
|
||||||
|
### `LOWERCASE_GROUPS_ONLY`
|
||||||
|
|
||||||
|
This mode contains **11 result rows**, **7 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.989429|276044 / 377860669765 (0.000073%)|885033 / 41863370 (2.114099%)|0.990385|0.986031|0.986056|
|
||||||
|
|2|SNOWBALL SPANISH DIRECT|PRIMARY_OUTPUT|0.652720|2201196 / 377860669765 (0.000583%)|29076352 / 41863370 (69.455354%)|0.627946|0.449839|0.510450|
|
||||||
|
|3|SNOWBALL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.652720|2201196 / 377860669765 (0.000583%)|29076352 / 41863370 (69.455354%)|0.627946|0.449839|0.510450|
|
||||||
|
|4|HUNSPELL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.614999|531181 / 377860669765 (0.000141%)|32234855 / 41863370 (77.000144%)|0.583531|0.370163|0.466854|
|
||||||
|
|5|SPANISH LUCENE SPANISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.514832|146613 / 377860669765 (0.000039%)|40621522 / 41863370 (97.033569%)|0.130949|0.057424|0.162875|
|
||||||
|
|6|SPANISH LUCENE SPANISH PLURAL STEM FILTER|PRIMARY_OUTPUT|0.503877|57716 / 377860669765 (0.000015%)|41538714 / 41863370 (99.224487%)|0.037409|0.015370|0.081139|
|
||||||
|
|7|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.501770|47148 / 377860669765 (0.000012%)|41715144 / 41863370 (99.645929%)|0.017379|0.007049|0.051824|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.993309|0.978859|0.999999|0.989429|0.999997|0.000003|
|
||||||
|
|2|SNOWBALL SPANISH DIRECT|PRIMARY_OUTPUT|0.853138|0.305446|0.999994|0.652720|0.999917|0.000083|
|
||||||
|
|3|SNOWBALL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.853138|0.305446|0.999994|0.652720|0.999917|0.000083|
|
||||||
|
|4|HUNSPELL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.947717|0.229999|0.999999|0.614999|0.999913|0.000087|
|
||||||
|
|5|SPANISH LUCENE SPANISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.894406|0.029664|1.000000|0.514832|0.999892|0.000108|
|
||||||
|
|6|SPANISH LUCENE SPANISH PLURAL STEM FILTER|PRIMARY_OUTPUT|0.849058|0.007755|1.000000|0.503877|0.999890|0.000110|
|
||||||
|
|7|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.758678|0.003541|1.000000|0.501770|0.999889|0.000111|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.990385|0.986031|0.981715|0.972447|0.986057|0.986056|
|
||||||
|
|2|SNOWBALL SPANISH DIRECT|PRIMARY_OUTPUT|0.627946|0.449839|0.350441|0.290188|0.510478|0.510450|
|
||||||
|
|3|SNOWBALL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.627946|0.449839|0.350441|0.290188|0.510478|0.510450|
|
||||||
|
|4|HUNSPELL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.583531|0.370163|0.271053|0.227117|0.466876|0.466854|
|
||||||
|
|5|SPANISH LUCENE SPANISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.130949|0.057424|0.036775|0.029561|0.162886|0.162875|
|
||||||
|
|6|SPANISH LUCENE SPANISH PLURAL STEM FILTER|PRIMARY_OUTPUT|0.037409|0.015370|0.009672|0.007744|0.081145|0.081139|
|
||||||
|
|7|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.017379|0.007049|0.004421|0.003537|0.051829|0.051824|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.986029|0.995464|0.993323|0.994392|0.994392|
|
||||||
|
|2|SNOWBALL SPANISH DIRECT|PRIMARY_OUTPUT|0.449806|0.981469|0.852556|0.912482|0.912482|
|
||||||
|
|3|SNOWBALL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.449806|0.981469|0.852556|0.912482|0.912482|
|
||||||
|
|4|HUNSPELL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|0.370136|0.993362|0.790500|0.880396|0.880396|
|
||||||
|
|5|SPANISH LUCENE SPANISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.057417|0.993866|0.756725|0.859234|0.859234|
|
||||||
|
|6|SPANISH LUCENE SPANISH PLURAL STEM FILTER|PRIMARY_OUTPUT|0.015368|0.995484|0.723753|0.838145|0.838145|
|
||||||
|
|7|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.007048|0.995676|0.710626|0.829341|0.829341|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|40978337|276044|885033|377860393721|276044 / 377860669765|885033 / 41863370|
|
||||||
|
|2|SNOWBALL SPANISH DIRECT|PRIMARY_OUTPUT|12787018|2201196|29076352|377858468569|2201196 / 377860669765|29076352 / 41863370|
|
||||||
|
|3|SNOWBALL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|12787018|2201196|29076352|377858468569|2201196 / 377860669765|29076352 / 41863370|
|
||||||
|
|4|HUNSPELL SPANISH LUCENE FILTER|PRIMARY_OUTPUT|9628515|531181|32234855|377860138584|531181 / 377860669765|32234855 / 41863370|
|
||||||
|
|5|SPANISH LUCENE SPANISH LIGHT STEM FILTER|PRIMARY_OUTPUT|1241848|146613|40621522|377860523152|146613 / 377860669765|40621522 / 41863370|
|
||||||
|
|6|SPANISH LUCENE SPANISH PLURAL STEM FILTER|PRIMARY_OUTPUT|324656|57716|41538714|377860612049|57716 / 377860669765|41538714 / 41863370|
|
||||||
|
|7|SPANISH LUCENE SPANISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|148226|47148|41715144|377860622617|47148 / 377860669765|41715144 / 41863370|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 377860669765 (0.000000%)|0 / 41863370 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|2|HUNSPELL SPANISH LUCENE FILTER|ANY_CANDIDATE|0.619928|412198 / 377860669765 (0.000109%)|31822108 / 41863370 (76.014205%)|0.600000|0.383864|0.479978|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|2|HUNSPELL SPANISH LUCENE FILTER|ANY_CANDIDATE|0.960568|0.239858|0.999999|0.619928|0.999915|0.000085|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|2|HUNSPELL SPANISH LUCENE FILTER|ANY_CANDIDATE|0.600000|0.383864|0.282205|0.237519|0.480000|0.479978|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|HUNSPELL SPANISH LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|41863370|0|0|377860669765|0 / 377860669765|0 / 41863370|
|
||||||
|
|2|HUNSPELL SPANISH LUCENE FILTER|ANY_CANDIDATE|10041262|412198|31822108|377860257567|412198 / 377860669765|31822108 / 41863370|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999998|1255381 / 377860669765 (0.000332%)|0 / 41863370 (0.000000%)|0.976572|0.985228|0.985334|
|
||||||
|
|2|HUNSPELL SPANISH LUCENE FILTER|ALL_CANDIDATES|0.619928|878949 / 377860669765 (0.000233%)|31822108 / 41863370 (76.014205%)|0.586905|0.380469|0.469606|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.970885|1.000000|0.999997|0.999998|0.999997|0.000003|
|
||||||
|
|2|HUNSPELL SPANISH LUCENE FILTER|ALL_CANDIDATES|0.919512|0.239858|0.999998|0.619928|0.999913|0.000087|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.976572|0.985228|0.994038|0.970885|0.985335|0.985334|
|
||||||
|
|2|HUNSPELL SPANISH LUCENE FILTER|ALL_CANDIDATES|0.586905|0.380469|0.281467|0.234926|0.469630|0.469606|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|HUNSPELL SPANISH LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|41863370|1255381|0|377859414384|1255381 / 377860669765|0 / 41863370|
|
||||||
|
|2|HUNSPELL SPANISH LUCENE FILTER|ALL_CANDIDATES|10041262|878949|31822108|377859790816|878949 / 377860669765|31822108 / 41863370|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|HUNSPELL SPANISH LUCENE FILTER|412747|118983|347768|17807|2.048262%|5|888962|
|
||||||
|
|Radixor|885033|276044|979337|42403|4.877434%|21|914127|
|
||||||
|
|
||||||
|
### Output Policies and Metric Definitions
|
||||||
|
|
||||||
|
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
|
||||||
|
|
||||||
|
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
|
||||||
|
|
||||||
|
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
|
||||||
|
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
|
||||||
|
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
|
||||||
|
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
|
||||||
|
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
|
||||||
|
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
|
||||||
|
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
|
||||||
|
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
|
||||||
|
- Jaccard index: `TP / (TP + FP + FN)`.
|
||||||
|
- Fowlkes–Mallows index: `sqrt(precision * recall)`.
|
||||||
|
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
|
||||||
|
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
|
||||||
|
|
||||||
|
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
|
||||||
|
|
||||||
|
### Provenance
|
||||||
|
|
||||||
|
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
|
||||||
|
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
|
||||||
|
- Evaluation command: `./gradlew stemmingQuality`
|
||||||
|
- Dictionary language: `ES_ES`
|
||||||
|
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
|
||||||
|
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
|
||||||
|
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:END -->
|
||||||
|
|||||||
@@ -26,7 +26,7 @@ Radixor stores the preferred transformation for each normalized dictionary word
|
|||||||
|
|
||||||
## Accuracy
|
## Accuracy
|
||||||
|
|
||||||
Accuracy is computed from one deterministic JMH measurement iteration without warmup. The benchmark may execute the full dictionary pass more than once inside that single timed iteration; percentages divide matching counters by evaluated counters from the same iteration.
|
Accuracy is computed from JMH auxiliary counters in the current report. The counters are deterministic for a fixed corpus and stemmer; percentages divide matching counters by evaluated counters from the same report and are not timing metrics.
|
||||||
|
|
||||||
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
||||||
| --- | ---: | ---: | ---: | --- |
|
| --- | ---: | ---: | ---: | --- |
|
||||||
@@ -42,11 +42,11 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
|
|
||||||
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
||||||
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
||||||
| Radixor | `swedishRadixor` | 4.916 | 0.525 | 57.3 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
| Radixor | `swedishRadixor` | 5.489 | 0.355 | 64.0 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
||||||
| Lucene SwedishMinimalStemFilter | `swedishLuceneSwedishMinimalStemFilter` | 4.453 | 0.530 | 51.9 | 0.906 | Minimal Swedish suffix reducer. |
|
| Lucene SwedishMinimalStemFilter | `swedishLuceneSwedishMinimalStemFilter` | 4.630 | 0.130 | 54.0 | 0.843 | Minimal Swedish suffix reducer. |
|
||||||
| Lucene SwedishLightStemFilter | `swedishLuceneSwedishLightStemFilter` | 4.523 | 0.151 | 52.8 | 0.920 | Light Swedish suffix stemmer. |
|
| Lucene SwedishLightStemFilter | `swedishLuceneSwedishLightStemFilter` | 4.876 | 0.328 | 56.9 | 0.888 | Light Swedish suffix stemmer. |
|
||||||
| Official Snowball direct | `snowballDirect[SWEDISH]` | 7.075 | 0.541 | 82.5 | 1.439 | Official Snowball generated Java stemmer; direct API. |
|
| Official Snowball direct | `snowballDirect[SWEDISH]` | 7.517 | 0.072 | 87.7 | 1.370 | Official Snowball generated Java stemmer; direct API. |
|
||||||
| Lucene SnowballFilter | `luceneSnowballFilter[SWEDISH]` | 9.379 | 0.056 | 109.4 | 1.908 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
| Lucene SnowballFilter | `luceneSnowballFilter[SWEDISH]` | 9.793 | 0.338 | 114.2 | 1.784 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
||||||
|
|
||||||
## Interpretation Notes
|
## Interpretation Notes
|
||||||
|
|
||||||
@@ -55,3 +55,366 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
||||||
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
||||||
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:START -->
|
||||||
|
|
||||||
|
## Stemming Quality
|
||||||
|
|
||||||
|
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `SV_SE` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
|
||||||
|
|
||||||
|
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
|
||||||
|
|
||||||
|
### Evaluation Scope and Key Findings
|
||||||
|
|
||||||
|
The dictionary resource is `src/main/resources/sv_se/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
|
||||||
|
|
||||||
|
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.974636** among 5 deterministic stemmers. The runner-up is `SNOWBALL SWEDISH DIRECT` at 0.807534, a difference of 0.167101. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.974584** among 5 deterministic stemmers. The runner-up is `SNOWBALL SWEDISH DIRECT` at 0.807599, a difference of 0.166985. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
### `ALL_WORDS`
|
||||||
|
|
||||||
|
This mode contains **7 result rows**, **5 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.974636|24473 / 4812155436 (0.000509%)|19546 / 385342 (5.072377%)|0.939665|0.943246|0.943260|
|
||||||
|
|2|SNOWBALL SWEDISH DIRECT|PRIMARY_OUTPUT|0.807534|67105 / 4812155436 (0.001394%)|148325 / 385342 (38.491781%)|0.739832|0.687540|0.692339|
|
||||||
|
|3|SNOWBALL SWEDISH LUCENE FILTER|PRIMARY_OUTPUT|0.799307|64262 / 4812155436 (0.001335%)|154666 / 385342 (40.137333%)|0.736940|0.678180|0.684227|
|
||||||
|
|4|SWEDISH LUCENE SWEDISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.796072|40227 / 4812155436 (0.000836%)|157161 / 385342 (40.784809%)|0.781991|0.698068|0.709491|
|
||||||
|
|5|SWEDISH LUCENE SWEDISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.783685|45941 / 4812155436 (0.000955%)|166707 / 385342 (43.262089%)|0.757232|0.672808|0.684713|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.937292|0.949276|0.999995|0.974636|0.999991|0.000009|
|
||||||
|
|2|SNOWBALL SWEDISH DIRECT|PRIMARY_OUTPUT|0.779348|0.615082|0.999986|0.807534|0.999955|0.000045|
|
||||||
|
|3|SNOWBALL SWEDISH LUCENE FILTER|PRIMARY_OUTPUT|0.782117|0.598627|0.999987|0.799307|0.999955|0.000045|
|
||||||
|
|4|SWEDISH LUCENE SWEDISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.850127|0.592152|0.999992|0.796072|0.999959|0.000041|
|
||||||
|
|5|SWEDISH LUCENE SWEDISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.826360|0.567379|0.999990|0.783685|0.999956|0.000044|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.939665|0.943246|0.946855|0.892588|0.943265|0.943260|
|
||||||
|
|2|SNOWBALL SWEDISH DIRECT|PRIMARY_OUTPUT|0.739832|0.687540|0.642152|0.523856|0.692361|0.692339|
|
||||||
|
|3|SNOWBALL SWEDISH LUCENE FILTER|PRIMARY_OUTPUT|0.736940|0.678180|0.628098|0.513065|0.684249|0.684227|
|
||||||
|
|4|SWEDISH LUCENE SWEDISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.781991|0.698068|0.630412|0.536179|0.709510|0.709491|
|
||||||
|
|5|SWEDISH LUCENE SWEDISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.757232|0.672808|0.605321|0.506941|0.684733|0.684713|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.943241|0.992631|0.993395|0.993013|0.993013|
|
||||||
|
|2|SNOWBALL SWEDISH DIRECT|PRIMARY_OUTPUT|0.687518|0.984860|0.942685|0.963311|0.963311|
|
||||||
|
|3|SNOWBALL SWEDISH LUCENE FILTER|PRIMARY_OUTPUT|0.678157|0.985207|0.939659|0.961894|0.961894|
|
||||||
|
|4|SWEDISH LUCENE SWEDISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.698048|0.988493|0.944582|0.966038|0.966038|
|
||||||
|
|5|SWEDISH LUCENE SWEDISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.672787|0.986795|0.942303|0.964036|0.964036|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|365796|24473|19546|4812130963|24473 / 4812155436|19546 / 385342|
|
||||||
|
|2|SNOWBALL SWEDISH DIRECT|PRIMARY_OUTPUT|237017|67105|148325|4812088331|67105 / 4812155436|148325 / 385342|
|
||||||
|
|3|SNOWBALL SWEDISH LUCENE FILTER|PRIMARY_OUTPUT|230676|64262|154666|4812091174|64262 / 4812155436|154666 / 385342|
|
||||||
|
|4|SWEDISH LUCENE SWEDISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|228181|40227|157161|4812115209|40227 / 4812155436|157161 / 385342|
|
||||||
|
|5|SWEDISH LUCENE SWEDISH LIGHT STEM FILTER|PRIMARY_OUTPUT|218635|45941|166707|4812109495|45941 / 4812155436|166707 / 385342|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 4812155436 (0.000000%)|0 / 385342 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|385342|0|0|4812155436|0 / 4812155436|0 / 385342|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999995|47848 / 4812155436 (0.000994%)|0 / 385342 (0.000000%)|0.909640|0.941544|0.943152|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.889545|1.000000|0.999990|0.999995|0.999990|0.000010|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.909640|0.941544|0.975768|0.889545|0.943157|0.943152|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|385342|47848|0|4812107588|47848 / 4812155436|0 / 385342|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|19546|24473|23375|5767|5.878216%|5|104148|
|
||||||
|
|
||||||
|
### `LOWERCASE_GROUPS_ONLY`
|
||||||
|
|
||||||
|
This mode contains **7 result rows**, **5 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.974584|24473 / 4789911577 (0.000511%)|19546 / 384563 (5.082652%)|0.939544|0.943132|0.943146|
|
||||||
|
|2|SNOWBALL SWEDISH DIRECT|PRIMARY_OUTPUT|0.807599|67105 / 4789911577 (0.001401%)|147975 / 384563 (38.478741%)|0.739645|0.687500|0.692274|
|
||||||
|
|3|SNOWBALL SWEDISH LUCENE FILTER|PRIMARY_OUTPUT|0.799355|64262 / 4789911577 (0.001342%)|154316 / 384563 (40.127625%)|0.736744|0.678122|0.684143|
|
||||||
|
|4|SWEDISH LUCENE SWEDISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.795947|40227 / 4789911577 (0.000840%)|156939 / 384563 (40.809698%)|0.781694|0.697790|0.709212|
|
||||||
|
|5|SWEDISH LUCENE SWEDISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.783598|45941 / 4789911577 (0.000959%)|166437 / 384563 (43.279515%)|0.756945|0.672575|0.684469|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.937167|0.949173|0.999995|0.974584|0.999991|0.000009|
|
||||||
|
|2|SNOWBALL SWEDISH DIRECT|PRIMARY_OUTPUT|0.779037|0.615213|0.999986|0.807599|0.999955|0.000045|
|
||||||
|
|3|SNOWBALL SWEDISH LUCENE FILTER|PRIMARY_OUTPUT|0.781800|0.598724|0.999987|0.799355|0.999954|0.000046|
|
||||||
|
|4|SWEDISH LUCENE SWEDISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.849816|0.591903|0.999992|0.795947|0.999959|0.000041|
|
||||||
|
|5|SWEDISH LUCENE SWEDISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.826025|0.567205|0.999990|0.783598|0.999956|0.000044|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.939544|0.943132|0.946748|0.892384|0.943151|0.943146|
|
||||||
|
|2|SNOWBALL SWEDISH DIRECT|PRIMARY_OUTPUT|0.739645|0.687500|0.642223|0.523810|0.692296|0.692274|
|
||||||
|
|3|SNOWBALL SWEDISH LUCENE FILTER|PRIMARY_OUTPUT|0.736744|0.678122|0.628142|0.512999|0.684165|0.684143|
|
||||||
|
|4|SWEDISH LUCENE SWEDISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.781694|0.697790|0.630152|0.535851|0.709231|0.709212|
|
||||||
|
|5|SWEDISH LUCENE SWEDISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.756945|0.672575|0.605126|0.506676|0.684489|0.684469|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.943127|0.992612|0.993378|0.992995|0.992995|
|
||||||
|
|2|SNOWBALL SWEDISH DIRECT|PRIMARY_OUTPUT|0.687478|0.984821|0.942695|0.963298|0.963298|
|
||||||
|
|3|SNOWBALL SWEDISH LUCENE FILTER|PRIMARY_OUTPUT|0.678100|0.985169|0.939661|0.961877|0.961877|
|
||||||
|
|4|SWEDISH LUCENE SWEDISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|0.697770|0.988463|0.944528|0.965996|0.965996|
|
||||||
|
|5|SWEDISH LUCENE SWEDISH LIGHT STEM FILTER|PRIMARY_OUTPUT|0.672553|0.986761|0.942265|0.964000|0.964000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|365017|24473|19546|4789887104|24473 / 4789911577|19546 / 384563|
|
||||||
|
|2|SNOWBALL SWEDISH DIRECT|PRIMARY_OUTPUT|236588|67105|147975|4789844472|67105 / 4789911577|147975 / 384563|
|
||||||
|
|3|SNOWBALL SWEDISH LUCENE FILTER|PRIMARY_OUTPUT|230247|64262|154316|4789847315|64262 / 4789911577|154316 / 384563|
|
||||||
|
|4|SWEDISH LUCENE SWEDISH MINIMAL STEM FILTER|PRIMARY_OUTPUT|227624|40227|156939|4789871350|40227 / 4789911577|156939 / 384563|
|
||||||
|
|5|SWEDISH LUCENE SWEDISH LIGHT STEM FILTER|PRIMARY_OUTPUT|218126|45941|166437|4789865636|45941 / 4789911577|166437 / 384563|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 4789911577 (0.000000%)|0 / 384563 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|384563|0|0|4789911577|0 / 4789911577|0 / 384563|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999995|47848 / 4789911577 (0.000999%)|0 / 384563 (0.000000%)|0.909473|0.941433|0.943047|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.889346|1.000000|0.999990|0.999995|0.999990|0.000010|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.909473|0.941433|0.975720|0.889346|0.943051|0.943047|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|384563|47848|0|4789863729|47848 / 4789911577|0 / 384563|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|19546|24473|23375|5767|5.891848%|5|103921|
|
||||||
|
|
||||||
|
### Output Policies and Metric Definitions
|
||||||
|
|
||||||
|
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
|
||||||
|
|
||||||
|
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
|
||||||
|
|
||||||
|
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
|
||||||
|
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
|
||||||
|
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
|
||||||
|
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
|
||||||
|
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
|
||||||
|
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
|
||||||
|
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
|
||||||
|
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
|
||||||
|
- Jaccard index: `TP / (TP + FP + FN)`.
|
||||||
|
- Fowlkes–Mallows index: `sqrt(precision * recall)`.
|
||||||
|
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
|
||||||
|
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
|
||||||
|
|
||||||
|
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
|
||||||
|
|
||||||
|
### Provenance
|
||||||
|
|
||||||
|
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
|
||||||
|
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
|
||||||
|
- Evaluation command: `./gradlew stemmingQuality`
|
||||||
|
- Dictionary language: `SV_SE`
|
||||||
|
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
|
||||||
|
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
|
||||||
|
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:END -->
|
||||||
|
|||||||
@@ -26,11 +26,12 @@ Radixor stores the preferred transformation for each normalized dictionary word
|
|||||||
|
|
||||||
## Accuracy
|
## Accuracy
|
||||||
|
|
||||||
Accuracy is computed from one deterministic JMH measurement iteration without warmup. The benchmark may execute the full dictionary pass more than once inside that single timed iteration; percentages divide matching counters by evaluated counters from the same iteration.
|
Accuracy is computed from JMH auxiliary counters in the current report. The counters are deterministic for a fixed corpus and stemmer; percentages divide matching counters by evaluated counters from the same report and are not timing metrics.
|
||||||
|
|
||||||
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
||||||
| --- | ---: | ---: | ---: | --- |
|
| --- | ---: | ---: | ---: | --- |
|
||||||
| Radixor | 99.307% | 99.365% | 99.062% | Full Radixor dictionary patch-command stemmer. |
|
| Radixor | 99.307% | 99.365% | 99.062% | Full Radixor dictionary patch-command stemmer. |
|
||||||
|
| Lucene HunspellStemFilter | 86.815% | 83.759% | 99.866% | Benchmark-only Ukrainian Hunspell dictionary compared via Lucene HunspellStemFilter. |
|
||||||
| Lucene MorfologikFilter | 92.362% | 90.637% | 99.732% | Dictionary-based path; Morfologik can emit multiple terms. |
|
| Lucene MorfologikFilter | 92.362% | 90.637% | 99.732% | Dictionary-based path; Morfologik can emit multiple terms. |
|
||||||
| Morfologik direct | 92.362% | 90.637% | 99.732% | Direct dictionary lookup; first returned stem is used for quality when no ranking weight is exposed. |
|
| Morfologik direct | 92.362% | 90.637% | 99.732% | Direct dictionary lookup; first returned stem is used for quality when no ranking weight is exposed. |
|
||||||
|
|
||||||
@@ -40,9 +41,10 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
|
|
||||||
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
||||||
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
||||||
| Radixor | `ukrainianRadixor` | 0.605 | 0.056 | 47.4 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
| Radixor | `ukrainianRadixor` | 0.682 | 0.057 | 53.5 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
||||||
| Morfologik direct | `ukrainianMorfologikDirect` | 8.106 | 0.040 | 635.7 | 13.408 | Direct Morfologik dictionary lookup; first returned stem is used for quality. |
|
| Lucene HunspellStemFilter | `luceneHunspellStemFilter` | 43.527 | 1.207 | 3413.3 | 63.799 | Benchmark-only Ukrainian Hunspell dictionary compared via Lucene HunspellStemFilter. |
|
||||||
| Lucene MorfologikFilter | `ukrainianLuceneMorfologikFilter` | 14.684 | 5.214 | 1151.5 | 24.287 | Dictionary-based Morfologik TokenFilter; may emit multiple terms. |
|
| Morfologik direct | `ukrainianMorfologikDirect` | 8.680 | 0.073 | 680.7 | 12.723 | Direct Morfologik dictionary lookup; first returned stem is used for quality. |
|
||||||
|
| Lucene MorfologikFilter | `ukrainianLuceneMorfologikFilter` | 14.575 | 0.248 | 1143.0 | 21.364 | Dictionary-based Morfologik TokenFilter; may emit multiple terms. |
|
||||||
|
|
||||||
## Interpretation Notes
|
## Interpretation Notes
|
||||||
|
|
||||||
@@ -51,3 +53,422 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
||||||
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
||||||
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:START -->
|
||||||
|
|
||||||
|
## Stemming Quality
|
||||||
|
|
||||||
|
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `UK_UA` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
|
||||||
|
|
||||||
|
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
|
||||||
|
|
||||||
|
### Evaluation Scope and Key Findings
|
||||||
|
|
||||||
|
The dictionary resource is `src/main/resources/uk_ua/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
|
||||||
|
|
||||||
|
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.995343** among 4 deterministic stemmers. The runner-up is `UKRAINIAN LUCENE MORFOLOGIK FILTER` at 0.928768, a difference of 0.066575. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.995342** among 4 deterministic stemmers. The runner-up is `UKRAINIAN LUCENE MORFOLOGIK FILTER` at 0.928751, a difference of 0.066591. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
### `ALL_WORDS`
|
||||||
|
|
||||||
|
This mode contains **12 result rows**, **4 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.995343|880 / 101387550 (0.000868%)|608 / 65340 (0.930517%)|0.987406|0.988637|0.988632|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.928768|828 / 101387550 (0.000817%)|9308 / 65340 (14.245485%)|0.956896|0.917054|0.919223|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|PRIMARY_OUTPUT|0.928646|828 / 101387550 (0.000817%)|9324 / 65340 (14.269972%)|0.956832|0.916912|0.919090|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|PRIMARY_OUTPUT|0.885793|794 / 101387550 (0.000783%)|14924 / 65340 (22.840526%)|0.933008|0.865139|0.871499|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.986588|0.990695|0.999991|0.995343|0.999985|0.000015|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.985438|0.857545|0.999992|0.928768|0.999900|0.000100|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|PRIMARY_OUTPUT|0.985434|0.857300|0.999992|0.928646|0.999900|0.000100|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|PRIMARY_OUTPUT|0.984495|0.771595|0.999992|0.885793|0.999845|0.000155|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.987406|0.988637|0.989871|0.977529|0.988639|0.988632|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.956896|0.917054|0.880397|0.846814|0.919270|0.919223|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|PRIMARY_OUTPUT|0.956832|0.916912|0.880190|0.846572|0.919137|0.919090|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|PRIMARY_OUTPUT|0.933008|0.865139|0.806475|0.762331|0.871568|0.871499|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.988630|0.997994|0.998266|0.998130|0.998130|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.917004|0.997990|0.971000|0.984310|0.984310|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|PRIMARY_OUTPUT|0.916862|0.997990|0.970876|0.984246|0.984246|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|PRIMARY_OUTPUT|0.865063|0.998114|0.949804|0.973360|0.973360|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|64732|880|608|101386670|880 / 101387550|608 / 65340|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|56032|828|9308|101386722|828 / 101387550|9308 / 65340|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|PRIMARY_OUTPUT|56016|828|9324|101386722|828 / 101387550|9324 / 65340|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|PRIMARY_OUTPUT|50416|794|14924|101386756|794 / 101387550|14924 / 65340|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 101387550 (0.000000%)|0 / 65340 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|0.962151|122 / 101387550 (0.000120%)|4946 / 65340 (7.569636%)|0.982323|0.959732|0.960413|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|ANY_CANDIDATE|0.962029|122 / 101387550 (0.000120%)|4962 / 65340 (7.594123%)|0.982267|0.959599|0.960286|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ANY_CANDIDATE|0.927570|326 / 101387550 (0.000322%)|9465 / 65340 (14.485767%)|0.962884|0.919443|0.922008|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|0.997984|0.924304|0.999999|0.962151|0.999950|0.000050|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|ANY_CANDIDATE|0.997983|0.924059|0.999999|0.962029|0.999950|0.000050|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ANY_CANDIDATE|0.994199|0.855142|0.999997|0.927570|0.999903|0.000097|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|0.982323|0.959732|0.938156|0.922581|0.960438|0.960413|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|ANY_CANDIDATE|0.982267|0.959599|0.937954|0.922337|0.960310|0.960286|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ANY_CANDIDATE|0.962884|0.919443|0.879752|0.850897|0.922053|0.922008|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|65340|0|0|101387550|0 / 101387550|0 / 65340|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|60394|122|4946|101387428|122 / 101387550|4946 / 65340|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|ANY_CANDIDATE|60378|122|4962|101387428|122 / 101387550|4962 / 65340|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ANY_CANDIDATE|55875|326|9465|101387224|326 / 101387550|9465 / 65340|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999993|1490 / 101387550 (0.001470%)|0 / 65340 (0.000000%)|0.982084|0.988727|0.988782|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|0.962145|1368 / 101387550 (0.001349%)|4946 / 65340 (7.569636%)|0.966650|0.950323|0.950669|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|ALL_CANDIDATES|0.962023|1368 / 101387550 (0.001349%)|4962 / 65340 (7.594123%)|0.966592|0.950191|0.950541|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ALL_CANDIDATES|0.927565|1271 / 101387550 (0.001254%)|9465 / 65340 (14.485767%)|0.950501|0.912349|0.914347|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.977705|1.000000|0.999985|0.999993|0.999985|0.000015|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|0.977850|0.924304|0.999987|0.962145|0.999938|0.000062|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|ALL_CANDIDATES|0.977845|0.924059|0.999987|0.962023|0.999938|0.000062|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ALL_CANDIDATES|0.977759|0.855142|0.999987|0.927565|0.999894|0.000106|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.982084|0.988727|0.995460|0.977705|0.988789|0.988782|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|0.966650|0.950323|0.934539|0.905349|0.950700|0.950669|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|ALL_CANDIDATES|0.966592|0.950191|0.934337|0.905109|0.950571|0.950541|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ALL_CANDIDATES|0.950501|0.912349|0.877142|0.838825|0.914398|0.914347|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|65340|1490|0|101386060|1490 / 101387550|0 / 65340|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|60394|1368|4946|101386182|1368 / 101387550|4946 / 65340|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|ALL_CANDIDATES|60378|1368|4962|101386182|1368 / 101387550|4962 / 65340|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ALL_CANDIDATES|55875|1271|9465|101386279|1271 / 101387550|9465 / 65340|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|HUNSPELL UKRAINIAN LUCENE FILTER|5459|468|477|1322|9.280449%|6|15740|
|
||||||
|
|UKRAINIAN LUCENE MORFOLOGIK FILTER|4362|706|540|2207|15.493155%|6|16937|
|
||||||
|
|UKRAINIAN MORFOLOGIK DIRECT|4362|706|540|2207|15.493155%|6|16937|
|
||||||
|
|Radixor|608|880|610|190|1.333801%|2|14435|
|
||||||
|
|
||||||
|
### `LOWERCASE_GROUPS_ONLY`
|
||||||
|
|
||||||
|
This mode contains **12 result rows**, **4 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.995342|880 / 101259406 (0.000869%)|608 / 65324 (0.930745%)|0.987403|0.988634|0.988629|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.928751|828 / 101259406 (0.000818%)|9308 / 65324 (14.248974%)|0.956884|0.917032|0.919202|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|PRIMARY_OUTPUT|0.928751|828 / 101259406 (0.000818%)|9308 / 65324 (14.248974%)|0.956884|0.917032|0.919202|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|PRIMARY_OUTPUT|0.885796|794 / 101259406 (0.000784%)|14920 / 65324 (22.839998%)|0.933007|0.865141|0.871500|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.986585|0.990693|0.999991|0.995342|0.999985|0.000015|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.985434|0.857510|0.999992|0.928751|0.999900|0.000100|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|PRIMARY_OUTPUT|0.985434|0.857510|0.999992|0.928751|0.999900|0.000100|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|PRIMARY_OUTPUT|0.984492|0.771600|0.999992|0.885796|0.999845|0.000155|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.987403|0.988634|0.989868|0.977524|0.988636|0.988629|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.956884|0.917032|0.880367|0.846777|0.919249|0.919202|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|PRIMARY_OUTPUT|0.956884|0.917032|0.880367|0.846777|0.919249|0.919202|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|PRIMARY_OUTPUT|0.933007|0.865141|0.806479|0.762334|0.871570|0.871500|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.988627|0.997992|0.998264|0.998128|0.998128|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|0.916982|0.997988|0.970978|0.984298|0.984298|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|PRIMARY_OUTPUT|0.916982|0.997988|0.970978|0.984298|0.984298|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|PRIMARY_OUTPUT|0.865065|0.998113|0.949788|0.973351|0.973351|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|64716|880|608|101258526|880 / 101259406|608 / 65324|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|PRIMARY_OUTPUT|56016|828|9308|101258578|828 / 101259406|9308 / 65324|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|PRIMARY_OUTPUT|56016|828|9308|101258578|828 / 101259406|9308 / 65324|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|PRIMARY_OUTPUT|50404|794|14920|101258612|794 / 101259406|14920 / 65324|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 101259406 (0.000000%)|0 / 65324 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|0.962142|122 / 101259406 (0.000120%)|4946 / 65324 (7.571490%)|0.982318|0.959722|0.960404|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|ANY_CANDIDATE|0.962142|122 / 101259406 (0.000120%)|4946 / 65324 (7.571490%)|0.982318|0.959722|0.960404|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ANY_CANDIDATE|0.927552|326 / 101259406 (0.000322%)|9465 / 65324 (14.489315%)|0.962874|0.919422|0.921988|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|0.997983|0.924285|0.999999|0.962142|0.999950|0.000050|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|ANY_CANDIDATE|0.997983|0.924285|0.999999|0.962142|0.999950|0.000050|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ANY_CANDIDATE|0.994198|0.855107|0.999997|0.927552|0.999903|0.000097|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|0.982318|0.959722|0.938141|0.922562|0.960428|0.960404|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|ANY_CANDIDATE|0.982318|0.959722|0.938141|0.922562|0.960428|0.960404|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ANY_CANDIDATE|0.962874|0.919422|0.879722|0.850861|0.922033|0.921988|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|65324|0|0|101259406|0 / 101259406|0 / 65324|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ANY_CANDIDATE|60378|122|4946|101259284|122 / 101259406|4946 / 65324|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|ANY_CANDIDATE|60378|122|4946|101259284|122 / 101259406|4946 / 65324|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ANY_CANDIDATE|55859|326|9465|101259080|326 / 101259406|9465 / 65324|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999993|1490 / 101259406 (0.001471%)|0 / 65324 (0.000000%)|0.982079|0.988724|0.988779|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|0.962136|1368 / 101259406 (0.001351%)|4946 / 65324 (7.571490%)|0.966642|0.950311|0.950657|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|ALL_CANDIDATES|0.962136|1368 / 101259406 (0.001351%)|4946 / 65324 (7.571490%)|0.966642|0.950311|0.950657|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ALL_CANDIDATES|0.927547|1271 / 101259406 (0.001255%)|9465 / 65324 (14.489315%)|0.950487|0.912326|0.914325|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.977699|1.000000|0.999985|0.999993|0.999985|0.000015|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|0.977845|0.924285|0.999986|0.962136|0.999938|0.000062|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|ALL_CANDIDATES|0.977845|0.924285|0.999986|0.962136|0.999938|0.000062|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ALL_CANDIDATES|0.977752|0.855107|0.999987|0.927547|0.999894|0.000106|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.982079|0.988724|0.995459|0.977699|0.988787|0.988779|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|0.966642|0.950311|0.934522|0.905326|0.950688|0.950657|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|ALL_CANDIDATES|0.966642|0.950311|0.934522|0.905326|0.950688|0.950657|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ALL_CANDIDATES|0.950487|0.912326|0.877111|0.838787|0.914376|0.914325|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|65324|1490|0|101257916|1490 / 101259406|0 / 65324|
|
||||||
|
|2|UKRAINIAN LUCENE MORFOLOGIK FILTER|ALL_CANDIDATES|60378|1368|4946|101258038|1368 / 101259406|4946 / 65324|
|
||||||
|
|3|UKRAINIAN MORFOLOGIK DIRECT|ALL_CANDIDATES|60378|1368|4946|101258038|1368 / 101259406|4946 / 65324|
|
||||||
|
|4|HUNSPELL UKRAINIAN LUCENE FILTER|ALL_CANDIDATES|55859|1271|9465|101258135|1271 / 101259406|9465 / 65324|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|HUNSPELL UKRAINIAN LUCENE FILTER|5455|468|477|1321|9.279292%|6|15730|
|
||||||
|
|UKRAINIAN LUCENE MORFOLOGIK FILTER|4362|706|540|2207|15.502950%|6|16928|
|
||||||
|
|UKRAINIAN MORFOLOGIK DIRECT|4362|706|540|2207|15.502950%|6|16928|
|
||||||
|
|Radixor|608|880|610|190|1.334645%|2|14426|
|
||||||
|
|
||||||
|
### Output Policies and Metric Definitions
|
||||||
|
|
||||||
|
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
|
||||||
|
|
||||||
|
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
|
||||||
|
|
||||||
|
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
|
||||||
|
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
|
||||||
|
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
|
||||||
|
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
|
||||||
|
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
|
||||||
|
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
|
||||||
|
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
|
||||||
|
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
|
||||||
|
- Jaccard index: `TP / (TP + FP + FN)`.
|
||||||
|
- Fowlkes–Mallows index: `sqrt(precision * recall)`.
|
||||||
|
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
|
||||||
|
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
|
||||||
|
|
||||||
|
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
|
||||||
|
|
||||||
|
### Provenance
|
||||||
|
|
||||||
|
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
|
||||||
|
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
|
||||||
|
- Evaluation command: `./gradlew stemmingQuality`
|
||||||
|
- Dictionary language: `UK_UA`
|
||||||
|
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
|
||||||
|
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
|
||||||
|
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:END -->
|
||||||
|
|||||||
@@ -25,11 +25,11 @@ Radixor stores the preferred transformation for each normalized dictionary word
|
|||||||
|
|
||||||
## Accuracy
|
## Accuracy
|
||||||
|
|
||||||
Accuracy is computed from one deterministic JMH measurement iteration without warmup. The benchmark may execute the full dictionary pass more than once inside that single timed iteration; percentages divide matching counters by evaluated counters from the same iteration.
|
Accuracy is computed from JMH auxiliary counters in the current report. The counters are deterministic for a fixed corpus and stemmer; percentages divide matching counters by evaluated counters from the same report and are not timing metrics.
|
||||||
|
|
||||||
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
| Stemmer | All exact | Changed exact | Root preserved | Note |
|
||||||
| --- | ---: | ---: | ---: | --- |
|
| --- | ---: | ---: | ---: | --- |
|
||||||
| Radixor | 98.930% | 98.343% | 100.000% | Radixor baseline in the Snowball-language comparison family. |
|
| Radixor | 98.930% | 98.343% | 100.000% | Full Radixor dictionary patch-command stemmer. |
|
||||||
| Lucene SnowballFilter | 2.837% | 2.558% | 3.346% | Lucene TokenFilter integration path around the Snowball algorithm. |
|
| Lucene SnowballFilter | 2.837% | 2.558% | 3.346% | Lucene TokenFilter integration path around the Snowball algorithm. |
|
||||||
| Official Snowball direct | 2.837% | 2.558% | 3.346% | Official Snowball generated Java stemmer; rule-based suffix algorithm. |
|
| Official Snowball direct | 2.837% | 2.558% | 3.346% | Official Snowball generated Java stemmer; rule-based suffix algorithm. |
|
||||||
|
|
||||||
@@ -39,9 +39,9 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
|
|
||||||
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
| Stemmer | Benchmark method | Score ms/op | Error ms | ns/token | Relative vs Radixor | Note |
|
||||||
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
| --- | --- | ---: | ---: | ---: | ---: | --- |
|
||||||
| Radixor | `radixor[YIDDISH]` | 0.236 | 0.004 | 85.1 | 1.000 | Radixor baseline for the Snowball-language comparison family. |
|
| Radixor | `radixor[YIDDISH]` | 0.254 | 0.004 | 50.7 | 1.000 | Full Radixor dictionary patch-command stemmer. |
|
||||||
| Official Snowball direct | `snowballDirect[YIDDISH]` | 1.432 | 0.193 | 515.7 | 6.058 | Official Snowball generated Java stemmer; direct API. |
|
| Official Snowball direct | `snowballDirect[YIDDISH]` | 1.537 | 0.220 | 307.3 | 6.058 | Official Snowball generated Java stemmer; direct API. |
|
||||||
| Lucene SnowballFilter | `luceneSnowballFilter[YIDDISH]` | 1.595 | 0.068 | 574.6 | 6.749 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
| Lucene SnowballFilter | `luceneSnowballFilter[YIDDISH]` | 1.714 | 0.120 | 342.8 | 6.756 | Lucene TokenFilter path around Snowball; includes TokenStream overhead. |
|
||||||
|
|
||||||
## Interpretation Notes
|
## Interpretation Notes
|
||||||
|
|
||||||
@@ -50,3 +50,346 @@ Speed uses JMH average time, 3 warmup iterations, 5 measurement iterations, 1 fo
|
|||||||
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
- Lucene TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
||||||
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
- Morfologik rows are dictionary-based and can emit multiple terms for one input token. Quality rows use the first returned term when no ranking weight is available.
|
||||||
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
- Snowball rows are rule-based generated suffix stemmers; they are useful algorithmic baselines, not dictionary-root equivalence guarantees.
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:START -->
|
||||||
|
|
||||||
|
## Stemming Quality
|
||||||
|
|
||||||
|
Runtime performance and linguistic grouping quality are independent dimensions. This section evaluates language `YI` using the complete validated stemming-quality result matrix. Every usable dictionary row is one gold-standard group of forms expected to share a morphological family or lemma. Exact equality with a predetermined lemma is not required. Same-row pairs are positive pairs; pairs from different rows are negative pairs.
|
||||||
|
|
||||||
|
`ALL_WORDS` includes every valid group and its original forms. `LOWERCASE_GROUPS_ONLY` excludes an entire group when any Unicode code point is uppercase or titlecase; retained words are not lowercased or otherwise rewritten. This isolates case-handling effects without changing retained inputs. [Download the complete machine-readable result snapshot](../data/stemming-quality.csv).
|
||||||
|
|
||||||
|
### Evaluation Scope and Key Findings
|
||||||
|
|
||||||
|
The dictionary resource is `src/main/resources/yi/stemmer.gz`. The following findings compare only deterministic `PRIMARY_OUTPUT` rows over identical included groups; candidate policies are reported separately as capability analyses.
|
||||||
|
|
||||||
|
- **ALL_WORDS:** `Radixor` ranks first by balanced accuracy at **0.988241** among 3 deterministic stemmers. The runner-up is `SNOWBALL YIDDISH DIRECT` at 0.890988, a difference of 0.097253. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
- **LOWERCASE_GROUPS_ONLY:** `Radixor` ranks first by balanced accuracy at **0.988241** among 3 deterministic stemmers. The runner-up is `SNOWBALL YIDDISH DIRECT` at 0.890988, a difference of 0.097253. This rank does not imply leadership in throughput or every secondary metric.
|
||||||
|
### `ALL_WORDS`
|
||||||
|
|
||||||
|
This mode contains **5 result rows**, **3 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.988241|195 / 6392909 (0.003050%)|149 / 6344 (2.348676%)|0.970881|0.972986|0.972965|
|
||||||
|
|2|SNOWBALL YIDDISH DIRECT|PRIMARY_OUTPUT|0.890988|1151 / 6392909 (0.018004%)|1382 / 6344 (21.784363%)|0.805624|0.796661|0.796600|
|
||||||
|
|3|SNOWBALL YIDDISH LUCENE FILTER|PRIMARY_OUTPUT|0.890988|1151 / 6392909 (0.018004%)|1382 / 6344 (21.784363%)|0.805624|0.796661|0.796600|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.969484|0.976513|0.999969|0.988241|0.999946|0.000054|
|
||||||
|
|2|SNOWBALL YIDDISH DIRECT|PRIMARY_OUTPUT|0.811713|0.782156|0.999820|0.890988|0.999604|0.000396|
|
||||||
|
|3|SNOWBALL YIDDISH LUCENE FILTER|PRIMARY_OUTPUT|0.811713|0.782156|0.999820|0.890988|0.999604|0.000396|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.970881|0.972986|0.975099|0.947393|0.972992|0.972965|
|
||||||
|
|2|SNOWBALL YIDDISH DIRECT|PRIMARY_OUTPUT|0.805624|0.796661|0.787894|0.662041|0.796798|0.796600|
|
||||||
|
|3|SNOWBALL YIDDISH LUCENE FILTER|PRIMARY_OUTPUT|0.805624|0.796661|0.787894|0.662041|0.796798|0.796600|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.972959|0.995691|0.996142|0.995917|0.995917|
|
||||||
|
|2|SNOWBALL YIDDISH DIRECT|PRIMARY_OUTPUT|0.796462|0.982919|0.962014|0.972354|0.972354|
|
||||||
|
|3|SNOWBALL YIDDISH LUCENE FILTER|PRIMARY_OUTPUT|0.796462|0.982919|0.962014|0.972354|0.972354|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|6195|195|149|6392714|195 / 6392909|149 / 6344|
|
||||||
|
|2|SNOWBALL YIDDISH DIRECT|PRIMARY_OUTPUT|4962|1151|1382|6391758|1151 / 6392909|1382 / 6344|
|
||||||
|
|3|SNOWBALL YIDDISH LUCENE FILTER|PRIMARY_OUTPUT|4962|1151|1382|6391758|1151 / 6392909|1382 / 6344|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 6392909 (0.000000%)|0 / 6344 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|6344|0|0|6392909|0 / 6392909|0 / 6344|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999970|389 / 6392909 (0.006085%)|0 / 6344 (0.000000%)|0.953240|0.970253|0.970653|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.942225|1.000000|0.999939|0.999970|0.999939|0.000061|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.953240|0.970253|0.987885|0.942225|0.970683|0.970653|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|6344|389|0|6392520|389 / 6392909|0 / 6344|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|149|195|194|89|2.487423%|3|3676|
|
||||||
|
|
||||||
|
### `LOWERCASE_GROUPS_ONLY`
|
||||||
|
|
||||||
|
This mode contains **5 result rows**, **3 evaluated stemmers**, and **3 output policies**. Applied-row and form counts are shown per row because adapters share the language corpus but policy rows remain independently auditable. Rankings are separated by output policy and ordered by unrounded balanced accuracy, followed by MCC, F1, over-stemming rate, over-stemming count, under-stemming rate, and stemmer. Balanced accuracy is a navigation metric, not a universally authoritative quality score.
|
||||||
|
|
||||||
|
#### `PRIMARY_OUTPUT` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.988241|195 / 6392909 (0.003050%)|149 / 6344 (2.348676%)|0.970881|0.972986|0.972965|
|
||||||
|
|2|SNOWBALL YIDDISH DIRECT|PRIMARY_OUTPUT|0.890988|1151 / 6392909 (0.018004%)|1382 / 6344 (21.784363%)|0.805624|0.796661|0.796600|
|
||||||
|
|3|SNOWBALL YIDDISH LUCENE FILTER|PRIMARY_OUTPUT|0.890988|1151 / 6392909 (0.018004%)|1382 / 6344 (21.784363%)|0.805624|0.796661|0.796600|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.969484|0.976513|0.999969|0.988241|0.999946|0.000054|
|
||||||
|
|2|SNOWBALL YIDDISH DIRECT|PRIMARY_OUTPUT|0.811713|0.782156|0.999820|0.890988|0.999604|0.000396|
|
||||||
|
|3|SNOWBALL YIDDISH LUCENE FILTER|PRIMARY_OUTPUT|0.811713|0.782156|0.999820|0.890988|0.999604|0.000396|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.970881|0.972986|0.975099|0.947393|0.972992|0.972965|
|
||||||
|
|2|SNOWBALL YIDDISH DIRECT|PRIMARY_OUTPUT|0.805624|0.796661|0.787894|0.662041|0.796798|0.796600|
|
||||||
|
|3|SNOWBALL YIDDISH LUCENE FILTER|PRIMARY_OUTPUT|0.805624|0.796661|0.787894|0.662041|0.796798|0.796600|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|0.972959|0.995691|0.996142|0.995917|0.995917|
|
||||||
|
|2|SNOWBALL YIDDISH DIRECT|PRIMARY_OUTPUT|0.796462|0.982919|0.962014|0.972354|0.972354|
|
||||||
|
|3|SNOWBALL YIDDISH LUCENE FILTER|PRIMARY_OUTPUT|0.796462|0.982919|0.962014|0.972354|0.972354|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|PRIMARY_OUTPUT|6195|195|149|6392714|195 / 6392909|149 / 6344|
|
||||||
|
|2|SNOWBALL YIDDISH DIRECT|PRIMARY_OUTPUT|4962|1151|1382|6391758|1151 / 6392909|1382 / 6344|
|
||||||
|
|3|SNOWBALL YIDDISH LUCENE FILTER|PRIMARY_OUTPUT|4962|1151|1382|6391758|1151 / 6392909|1382 / 6344|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ANY_CANDIDATE` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|0 / 6392909 (0.000000%)|0 / 6344 (0.000000%)|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|0.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|1.000000|1.000000|1.000000|1.000000|1.000000|1.000000|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ANY_CANDIDATE|6344|0|0|6392909|0 / 6392909|0 / 6344|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### `ALL_CANDIDATES` ranking
|
||||||
|
|
||||||
|
<div class="quality-table quality-table--compact" role="region" aria-label="Compact stemming-quality ranking; scroll horizontally for additional columns" tabindex="0" markdown="1">
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Balanced accuracy | Over-stemming | Under-stemming | F0.5 | F1 | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.999970|389 / 6392909 (0.006085%)|0 / 6344 (0.000000%)|0.953240|0.970253|0.970653|
|
||||||
|
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Classification metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Precision | Recall | Specificity | Balanced accuracy | Pairwise accuracy | Error rate |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.942225|1.000000|0.999939|0.999970|0.999939|0.000061|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Pair-relation metrics</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | F0.5 | F1 | F2 | Jaccard | Fowlkes–Mallows | MCC |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|0.953240|0.970253|0.987885|0.942225|0.970683|0.970653|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Partition metrics (PRIMARY_OUTPUT only)</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | Adjusted Rand Index | Homogeneity | Completeness | V-measure | Normalized mutual information |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|n/a|n/a|n/a|n/a|n/a|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
<details class="quality-details" markdown="1"><summary>Raw pair counts</summary>
|
||||||
|
|
||||||
|
| Rank | Stemmer | Output policy | TP | FP | FN | TN | Over error / possible | Under error / possible |
|
||||||
|
|---:|---|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
|1|Radixor|ALL_CANDIDATES|6344|389|0|6392520|389 / 6392909|0 / 6344|
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
#### Multi-output analysis
|
||||||
|
|
||||||
|
Alternative candidates are capability analyses, not replacements for the deterministic comparison.
|
||||||
|
|
||||||
|
| Stemmer | Under pairs repaired | Best-case over pairs avoided | All-candidate collisions added | Multi-candidate forms | Multi-candidate share | Maximum candidates | Total candidate assignments |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
|Radixor|149|195|194|89|2.487423%|3|3676|
|
||||||
|
|
||||||
|
### Output Policies and Metric Definitions
|
||||||
|
|
||||||
|
`PRIMARY_OUTPUT` uses one deterministic stem per form and therefore defines a strict partition. `ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound: a same-group pair succeeds when candidates intersect, while a different-group pair succeeds when a non-colliding selection exists. Candidate choices may differ between pairs, so this is not deterministic runtime behaviour and need not represent one globally consistent assignment. `ALL_CANDIDATES` activates every returned candidate; forms are related when candidate sets intersect. Alternatives can reduce under-stemming but can introduce cross-group collisions, and the resulting relation can overlap and need not be a partition.
|
||||||
|
|
||||||
|
For each row, `TP = underPossiblePairs - underErrorPairs`, `FN = underErrorPairs`, `FP = overErrorPairs`, and `TN = overPossiblePairs - overErrorPairs`. TP and FN concern same-group pairs; FP and TN concern different-group pairs. Consequently, under-stemming and over-stemming use different denominators. Undefined values are rendered as `n/a`.
|
||||||
|
|
||||||
|
- Under-stemming rate: `FN / (TP + FN)`, the false-negative rate over same-group pairs.
|
||||||
|
- Over-stemming rate: `FP / (TN + FP)`, the false-positive rate over different-group pairs.
|
||||||
|
- Pairwise precision: `TP / (TP + FP)`, the fraction of predicted conflations that are gold-standard positive pairs.
|
||||||
|
- Pairwise recall: `TP / (TP + FN)`, the fraction of gold-standard positive pairs successfully connected.
|
||||||
|
- Pairwise specificity: `TN / (TN + FP)`, the fraction of different-group pairs correctly separated.
|
||||||
|
- Balanced accuracy: `(recall + specificity) / 2`. It gives equal weight to positive and negative pair classes and is less dominated by the large true-negative class than ordinary accuracy. It does not replace the raw errors or other metrics.
|
||||||
|
- Pairwise F-beta: `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`. F0.5 emphasizes precision and penalizes over-stemming more; F1 weights precision and recall equally; F2 emphasizes recall and penalizes under-stemming more.
|
||||||
|
- MCC: `(TP * TN - FP * FN) / sqrt((TP + FP) * (TP + FN) * (TN + FP) * (TN + FN))`. It uses all confusion counts and remains useful under class imbalance, except when its denominator is degenerate.
|
||||||
|
- Jaccard index: `TP / (TP + FP + FN)`.
|
||||||
|
- Fowlkes–Mallows index: `sqrt(precision * recall)`.
|
||||||
|
- Pairwise accuracy: `(TP + TN) / (TP + TN + FP + FN)`. It can be dominated by true-negative cross-group pairs.
|
||||||
|
- Pairwise error rate: `(FP + FN) / (TP + TN + FP + FN)`.
|
||||||
|
|
||||||
|
Adjusted Rand Index uses the gold/predicted contingency table and chance correction. Homogeneity is `1 - H(gold | predicted) / H(gold)`; completeness is `1 - H(predicted | gold) / H(predicted)`; V-measure is their harmonic mean; normalized mutual information uses the arithmetic-mean entropy normalization `MI / ((H(gold) + H(predicted)) / 2)`. These partition-only metrics apply to `PRIMARY_OUTPUT`; candidate-relation rows show `n/a`.
|
||||||
|
|
||||||
|
### Provenance
|
||||||
|
|
||||||
|
- Authoritative source: `docs/benchmarks/data/stemming-quality.csv`
|
||||||
|
- Source SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
|
||||||
|
- Evaluation command: `./gradlew stemmingQuality`
|
||||||
|
- Dictionary language: `YI`
|
||||||
|
- Processing modes: `ALL_WORDS`, `LOWERCASE_GROUPS_ONLY`
|
||||||
|
- Stemmer versions and transitive artifacts: resolved by the repository's JMH Gradle configuration and `gradle.lockfile`
|
||||||
|
- Radixor version, Git revision, generation date, JDK version, operating system, and dictionary revision: not recorded in the authoritative CSV
|
||||||
|
|
||||||
|
<!-- STEMMING-QUALITY:END -->
|
||||||
|
|||||||
@@ -4,20 +4,28 @@ Implemented benchmark methods are documented on the per-language pages under [La
|
|||||||
|
|
||||||
## Included Candidate Families
|
## Included Candidate Families
|
||||||
|
|
||||||
The current benchmark pages include Radixor baselines, Lucene language filters where the language matches a bundled Radixor resource, Lucene Stempel and Morfologik paths where applicable, official Snowball Java stemmers where same-language comparison is available, and selected English-specific non-Lucene baselines such as OpenNLP Porter and Paice/Husk Lancaster.
|
The current benchmark pages include Radixor baselines, Lucene language filters where the language matches a bundled Radixor resource, Lucene Stempel and Morfologik paths where applicable, official Snowball Java stemmers where same-language comparison is available, benchmark-only CISTEM German stemmer evaluation, benchmark-only Hunspell comparisons, and selected English-specific non-Lucene baselines such as OpenNLP Porter and Paice/Husk Lancaster.
|
||||||
|
|
||||||
|
Benchmark-only Hunspell comparisons use bundled benchmark dictionaries and the Lucene HunspellStemFilter adapter over the selected language token streams.
|
||||||
|
The CISTEM candidate is implemented in `src/jmh/java/org/egothor/stemmer/benchmark/Cistem.java` and follows the original MIT-licensed upstream implementation from Leonie Weissweiler's CISTEM project.
|
||||||
|
CISTEM German gold-standard files are not vendored in this repository. The Gradle JMH resource preparation tasks download `goldstandard1.txt` and `goldstandard2.txt` from the upstream CISTEM repository into generated build resources.
|
||||||
|
|
||||||
Direct stemmer APIs and Lucene TokenFilter paths are documented separately on language pages. TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
Direct stemmer APIs and Lucene TokenFilter paths are documented separately on language pages. TokenFilter rows include TokenStream, attribute, and required normalization overhead. Direct rows measure exposed direct APIs.
|
||||||
|
|
||||||
|
For the benchmark refresh used in this documentation build:
|
||||||
|
|
||||||
|
- Hunspell families are included in `HunspellStemmerComparisonBenchmark` (speed) and `HunspellStemmerComparisonBenchmarkQuality` (quality for all benchmark languages in this corpus). The legacy
|
||||||
|
`EnglishHunspellStemmerComparisonBenchmarkQuality` result is retained for continuity.
|
||||||
|
- CISTEM quality is present in the published per-language results under `GERMAN_CISTEM`. CISTEM speed is present as `germanCistem` in `MultiLanguageStemmerComparisonBenchmark`.
|
||||||
|
|
||||||
## Evaluated But Skipped Candidates
|
## Evaluated But Skipped Candidates
|
||||||
|
|
||||||
| Candidate | Language | Link/source | Reason skipped |
|
| Candidate | Language | Link/source | Reason skipped |
|
||||||
| --- | --- | --- | --- |
|
| --- | --- | --- | --- |
|
||||||
| Lucene Arabic, Bulgarian, Bengali, Sorani, Greek, Galician, Hindi, Indonesian, Latvian, Telugu filters | Various | `lucene-analysis-common` | No bundled same-language Radixor resource in this repository snapshot. |
|
| Lucene Arabic, Bulgarian, Bengali, Sorani, Greek, Galician, Hindi, Indonesian, Latvian, Telugu filters | Various | `lucene-analysis-common` | No bundled same-language Radixor resource in this repository snapshot. |
|
||||||
| Lucene analyzer-only paths | Multiple | Lucene analyzers | Full analyzers mix tokenization, stop-word handling, and other behavior; direct filters are used where available. |
|
| Lucene analyzer-only paths | Multiple | Lucene analyzers | Full analyzers mix tokenization, stop-word handling, and other behavior; direct filters are used where available. |
|
||||||
| Lucene HunspellStemFilter | Multiple | `lucene-analysis-common` | Requires external Hunspell dictionaries not resolved as benchmark-only resources here. |
|
|
||||||
| Lucene StemmerOverrideFilter | Multiple | `lucene-analysis-common` | Override map facility, not a stemmer algorithm. |
|
| Lucene StemmerOverrideFilter | Multiple | `lucene-analysis-common` | Override map facility, not a stemmer algorithm. |
|
||||||
| Additional Snowball Lovins | English | Official Snowball Java distribution | No Lovins Java stemmer was present in the selected Snowball Java distribution. |
|
| Additional Snowball Lovins | English | Official Snowball Java distribution | No Lovins Java stemmer was present in the selected Snowball Java distribution. |
|
||||||
| Lemur Project Krovetz Stemmer | English | Lemur Project | Lucene KStem represents the Krovetz-style path without adding separate dependency and license risk. |
|
| Lemur Project Krovetz Stemmer | English | Lemur Project | Lucene KStem represents the Krovetz-style path without adding separate dependency and license risk. |
|
||||||
| Smile Lancaster / Paice-Husk | English | Smile NLP | Smile is large for one stemmer; Paice/Husk is included through a smaller benchmark-only generated path. |
|
| Smile Lancaster / Paice-Husk | English | Smile NLP | Smile is large for one stemmer; Paice/Husk is included through a smaller benchmark-only generated path. |
|
||||||
| CISTEM German stemmer | German | `https://github.com/LeonieWeissweiler/CISTEM` | Clean benchmark-only Java integration was not completed in this phase. |
|
|
||||||
| `stemmerEval` reference repository | Multiple | `https://github.com/endredy/stemmerEval` | Used only as a candidate reference; no code or data copied. |
|
| `stemmerEval` reference repository | Multiple | `https://github.com/endredy/stemmerEval` | Used only as a candidate reference; no code or data copied. |
|
||||||
|
|||||||
@@ -6,16 +6,16 @@ This benchmark is the clearest demonstration of the Radixor quality/speed envelo
|
|||||||
|
|
||||||
| Used rows | Actual row ratio | All exact | Changed exact | Root preserved | Speed ms/op | Error ms | ns/token |
|
| Used rows | Actual row ratio | All exact | Changed exact | Root preserved | Speed ms/op | Error ms | ns/token |
|
||||||
| ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: |
|
| ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: |
|
||||||
| 100% | 100.000% | 97.478% | 97.197% | 97.552% | 23.113 | 7.065 | 109.8 |
|
| 100% | 100.000% | 97.478% | 97.197% | 97.552% | 28.578 | 7.571 | 135.8 |
|
||||||
| 90% | 90.000% | 97.047% | 94.913% | 97.613% | 21.270 | 9.914 | 101.0 |
|
| 90% | 90.000% | 97.047% | 94.913% | 97.613% | 26.612 | 9.227 | 126.4 |
|
||||||
| 80% | 80.000% | 96.635% | 92.768% | 97.661% | 19.170 | 6.609 | 91.1 |
|
| 80% | 80.000% | 96.635% | 92.768% | 97.661% | 23.331 | 8.106 | 110.8 |
|
||||||
| 70% | 70.000% | 96.209% | 90.565% | 97.705% | 20.857 | 6.734 | 99.1 |
|
| 70% | 70.000% | 96.209% | 90.565% | 97.705% | 22.362 | 1.957 | 106.2 |
|
||||||
| 60% | 60.000% | 95.750% | 88.384% | 97.703% | 14.975 | 1.215 | 71.1 |
|
| 60% | 60.000% | 95.750% | 88.384% | 97.703% | 16.497 | 2.026 | 78.4 |
|
||||||
| 50% | 50.000% | 95.262% | 86.107% | 97.690% | 15.249 | 1.078 | 72.4 |
|
| 50% | 50.000% | 95.262% | 86.107% | 97.690% | 16.035 | 0.986 | 76.2 |
|
||||||
| 40% | 40.000% | 94.753% | 83.855% | 97.643% | 15.323 | 2.340 | 72.8 |
|
| 40% | 40.000% | 94.753% | 83.855% | 97.643% | 16.459 | 0.664 | 78.2 |
|
||||||
| 30% | 30.000% | 94.208% | 81.651% | 97.537% | 16.778 | 2.643 | 79.7 |
|
| 30% | 30.000% | 94.208% | 81.651% | 97.537% | 19.566 | 0.758 | 92.9 |
|
||||||
| 20% | 20.000% | 93.633% | 79.366% | 97.416% | 18.929 | 3.241 | 89.9 |
|
| 20% | 20.000% | 93.633% | 79.366% | 97.416% | 14.616 | 0.487 | 69.4 |
|
||||||
| 10% | 10.000% | 92.868% | 76.516% | 97.204% | 19.124 | 1.883 | 90.9 |
|
| 10% | 10.000% | 92.868% | 76.516% | 97.204% | 18.093 | 3.147 | 86.0 |
|
||||||
|
|
||||||
## Column Meanings
|
## Column Meanings
|
||||||
|
|
||||||
|
|||||||
@@ -4,11 +4,11 @@ The values below are environment-specific and must not be read as universal perf
|
|||||||
|
|
||||||
| Item | Value |
|
| Item | Value |
|
||||||
| --- | --- |
|
| --- | --- |
|
||||||
| Benchmark date | 2026-07-03 |
|
| Benchmark date | 2026-07-06 (Europe/Prague) |
|
||||||
| Focused comparison command family | JMH jar runs limited to `EnglishStemmerComparisonBenchmark`, `MultiLanguageStemmerComparisonBenchmark`, and `SnowballLanguageStemmerComparisonBenchmark`; Radixor exact-root metrics were recomputed deterministically against the same contracted loaders |
|
| Focused comparison command family | `./gradlew jmh -Pjmh.includes='.*StemmerComparisonBenchmark.*' --no-daemon` |
|
||||||
| English coverage command | `./gradlew jmh -Pjmh.includes='.*EnglishRadixorDictionaryCoverageBenchmark.*' --no-daemon` |
|
| English coverage command | `./gradlew jmh -Pjmh.includes='.*EnglishRadixorDictionaryCoverageBenchmark.*' --no-daemon` |
|
||||||
| Speed result reports | `build/reports/jmh/contracted/english-comparison.csv`, `multilanguage-speed.csv`, `snowball-language-speed.csv` |
|
| Speed result reports | `build/reports/jmh/stemmer-comparison-2026-07-06.csv`, `build/reports/jmh/stemmer-comparison-2026-07-06.txt`, `build/reports/jmh/english-coverage-2026-07-06.csv`, and `build/reports/jmh/english-coverage-2026-07-06.txt` |
|
||||||
| Accuracy result reports | Deterministic Radixor exact-root pass over bundled dictionaries; non-Radixor quality rows retained from the existing published quality suite |
|
| Accuracy result reports | `build/reports/jmh/stemmer-comparison-2026-07-06.csv`, `build/reports/jmh/english-coverage-2026-07-06.csv`, and deterministic Radixor exact-root accounting over the same bundled language corpora |
|
||||||
| Final comparison JMH scope | Stemmer comparison benchmarks only; internal `FrequencyTrie*` microbenchmarks were not run |
|
| Final comparison JMH scope | Stemmer comparison benchmarks only; internal `FrequencyTrie*` microbenchmarks were not run |
|
||||||
| Coverage JMH scope | English Radixor dictionary coverage benchmark only |
|
| Coverage JMH scope | English Radixor dictionary coverage benchmark only |
|
||||||
| JMH version | 1.37 |
|
| JMH version | 1.37 |
|
||||||
@@ -16,15 +16,19 @@ The values below are environment-specific and must not be read as universal perf
|
|||||||
| Score unit | `ns/op` |
|
| Score unit | `ns/op` |
|
||||||
| Speed warmup | 3 iterations, 1 s each |
|
| Speed warmup | 3 iterations, 1 s each |
|
||||||
| Speed measurement | 5 iterations, 1 s each |
|
| Speed measurement | 5 iterations, 1 s each |
|
||||||
| Accuracy warmup | none for deterministic exact-root accounting |
|
| Accuracy warmup | 3 JMH warmup iterations were applied by the Gradle invocation; timing scores from quality methods are not interpreted |
|
||||||
| Accuracy measurement | 1 deterministic measurement iteration; counters only, not speed interpretation |
|
| Accuracy measurement | 5 JMH measurement samples; documentation uses deterministic auxiliary counter ratios from the same report |
|
||||||
| Fork count in generated report files | 1 |
|
| Fork count in generated report files | 1 |
|
||||||
| Default fork policy for accuracy-only benchmark classes | `@Fork(0)` for future default runs because accuracy counters are deterministic and not interpreted as speed |
|
| Default fork policy for accuracy-only benchmark classes | `@Fork(0)` for future default runs because accuracy counters are deterministic and not interpreted as speed |
|
||||||
| Thread count | 1 |
|
| Thread count | 1 |
|
||||||
| JVM reported by JMH | OpenJDK 64-Bit Server VM, 25.0.3+9 |
|
| JVM reported by JMH | JDK 25.0.3, OpenJDK 64-Bit Server VM, 25.0.3+9 |
|
||||||
|
| Java runtime | OpenJDK Runtime Environment, Red Hat build 25.0.3+9 |
|
||||||
| JVM invoker | `/usr/lib/jvm/java-25-openjdk/bin/java` |
|
| JVM invoker | `/usr/lib/jvm/java-25-openjdk/bin/java` |
|
||||||
| Operating system | Linux 7.0.13-200.fc44.x86_64 |
|
| Operating system | Fedora Linux 44 (MATE-Compiz) |
|
||||||
| CPU | AMD Ryzen 5 7600 6-Core Processor |
|
| Kernel | Linux 7.0.12-201.fc44.x86_64 |
|
||||||
|
| Architecture | x86_64 |
|
||||||
|
| CPU | AMD Ryzen 5 8600G w/ Radeon 760M Graphics |
|
||||||
|
| Physical cores | 6 |
|
||||||
| Logical CPUs | 12 |
|
| Logical CPUs | 12 |
|
||||||
|
|
||||||
## Contracted Trie Baseline
|
## Contracted Trie Baseline
|
||||||
@@ -35,12 +39,10 @@ All Radixor rows in the refreshed benchmark tables use contracted compiled patch
|
|||||||
|
|
||||||
Generated local report files for this benchmark update:
|
Generated local report files for this benchmark update:
|
||||||
|
|
||||||
- `build/reports/jmh/contracted/english-comparison.csv`
|
- `build/reports/jmh/stemmer-comparison-2026-07-06.csv`
|
||||||
- `build/reports/jmh/contracted/english-comparison.txt`
|
- `build/reports/jmh/stemmer-comparison-2026-07-06.txt`
|
||||||
- `build/reports/jmh/contracted/multilanguage-speed.csv`
|
- `build/reports/jmh/english-coverage-2026-07-06.csv`
|
||||||
- `build/reports/jmh/contracted/multilanguage-speed.txt`
|
- `build/reports/jmh/english-coverage-2026-07-06.txt`
|
||||||
- `build/reports/jmh/contracted/snowball-language-speed.csv`
|
|
||||||
- `build/reports/jmh/contracted/snowball-language-speed.txt`
|
|
||||||
|
|
||||||
JMH TXT and CSV reports are still published as benchmark artifacts. They are not converted into a Porter speed badge.
|
JMH TXT and CSV reports are still published as benchmark artifacts. They are not converted into a Porter speed badge.
|
||||||
|
|
||||||
|
|||||||
84
docs/benchmarks/reference/linguistic-quality.md
Normal file
84
docs/benchmarks/reference/linguistic-quality.md
Normal file
@@ -0,0 +1,84 @@
|
|||||||
|
# Linguistic Quality Methodology
|
||||||
|
|
||||||
|
This evaluation measures agreement between the relation predicted by a stemmer and the gold-standard relation represented by Radixor dictionary groups. It does not require a generated stem to equal one predetermined lemma string. Runtime performance and linguistic quality are separate measurements.
|
||||||
|
|
||||||
|
## Scope and fair-comparison rules
|
||||||
|
|
||||||
|
The authoritative Radixor language universe is the reconciliation of registered default model descriptors and `StemmerPatchTrieLoader.Language`. Radixor is evaluated for every reconciled language. Optional models are separate comparison rows. A third-party adapter is evaluated only for languages supported by its tested implementation and having a compatible Radixor dictionary; unsupported combinations are absent rather than assigned zero quality.
|
||||||
|
|
||||||
|
Model identity is part of the candidate identity. Default Polish means `pl-pl-unimorph`; optional PoliMorf means `pl-pl-polimorf`. Results for those inputs must not be combined or relabeled, and historical snapshots cannot acquire a newer model identity retroactively.
|
||||||
|
|
||||||
|
Within one language and dictionary mode, every adapter receives the same original included forms. Exact duplicates are removed only within one dictionary row. Identical surface forms in different rows remain distinct entries. Candidate strings use exact `String.equals`, with no evaluation-only lowercasing, normalization, accent removal, or gold-label-aware selection. Adapter preprocessing and lifecycle match the JMH comparison path.
|
||||||
|
|
||||||
|
## Gold-standard pairs
|
||||||
|
|
||||||
|
Every usable dictionary row is a gold-standard equivalence group. An unordered pair from the same row is positive; a pair from different rows is negative. For group size `n`, `C2(n) = n * (n - 1) / 2`.
|
||||||
|
|
||||||
|
- `TP = underPossiblePairs - underErrorPairs`: same-group pairs correctly related.
|
||||||
|
- `FN = underErrorPairs`: same-group pairs incorrectly separated.
|
||||||
|
- `FP = overErrorPairs`: different-group pairs incorrectly related.
|
||||||
|
- `TN = overPossiblePairs - overErrorPairs`: different-group pairs correctly separated.
|
||||||
|
|
||||||
|
Under-stemming is the false-negative relation among same-group pairs. Over-stemming is the false-positive relation among different-group pairs. Their percentages use different denominators and must not be added or averaged without an explicitly defined composite.
|
||||||
|
|
||||||
|
## Dictionary-processing modes
|
||||||
|
|
||||||
|
- `ALL_WORDS` includes every valid group and preserves every original form.
|
||||||
|
- `LOWERCASE_GROUPS_ONLY` excludes an entire group if any Unicode code point is uppercase or titlecase. Retained forms are not converted to lowercase. Digits, punctuation, combining marks, and characters without case distinctions do not exclude a group by themselves.
|
||||||
|
|
||||||
|
## Output policies
|
||||||
|
|
||||||
|
`PRIMARY_OUTPUT` uses the adapter's deterministic primary stem. It defines a strict predicted partition and is the principal direct comparison between implementations.
|
||||||
|
|
||||||
|
`ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound. Same-group pairs succeed when candidate sets intersect. Different-group pairs avoid an error whenever a non-colliding candidate selection exists. Selection may differ between pairs, so this policy is not deterministic runtime behaviour and may not correspond to one globally realizable assignment.
|
||||||
|
|
||||||
|
`ALL_CANDIDATES` treats every returned candidate as active. Two forms are related when their candidate sets intersect. Alternatives can recover same-group relationships while introducing cross-group collisions. This overlapping relation need not be transitive or form a partition.
|
||||||
|
|
||||||
|
Candidate-aware policies are reported as capability analyses. They are not mixed into the principal `PRIMARY_OUTPUT` ranking.
|
||||||
|
|
||||||
|
## Relation metrics
|
||||||
|
|
||||||
|
Undefined denominators produce `n/a`, never zero, `NaN`, or infinity. Metrics are calculated from unrounded raw counts and displayed with six decimals.
|
||||||
|
|
||||||
|
| Metric | Formula | Range and interpretation | Sensitivity and applicability |
|
||||||
|
| --- | --- | --- | --- |
|
||||||
|
| Under-stemming rate | `FN / (TP + FN)` | `[0, 1]`; lower is better. False-negative rate over same-group pairs. | Sensitive to splitting large gold groups. All policies. |
|
||||||
|
| Over-stemming rate | `FP / (TN + FP)` | `[0, 1]`; lower is better. False-positive rate over different-group pairs. | The denominator is usually very large. All policies. |
|
||||||
|
| Precision | `TP / (TP + FP)` | `[0, 1]`; higher is better. Fraction of predicted relations that are gold-positive. | Penalizes over-stemming. All policies, with oracle-assisted interpretation for `ANY_CANDIDATE`. |
|
||||||
|
| Recall | `TP / (TP + FN)` | `[0, 1]`; higher is better. Fraction of gold-positive pairs recovered. | Equivalent to one minus the under-stemming rate. All policies. |
|
||||||
|
| Specificity | `TN / (TN + FP)` | `[0, 1]`; higher is better. Fraction of negative pairs separated. | Sensitive to cross-group collisions. All policies. |
|
||||||
|
| Balanced accuracy | `(recall + specificity) / 2` | `[0, 1]`; higher is better. Equal weight for positive and negative classes. | Primary navigation metric; less dominated by TN than ordinary accuracy, but not uniquely authoritative. |
|
||||||
|
| Pairwise accuracy | `(TP + TN) / (TP + TN + FP + FN)` | `[0, 1]`; higher is better. | Can be dominated by the very large TN class and is not the default ranking metric. |
|
||||||
|
| Pairwise error rate | `(FP + FN) / (TP + TN + FP + FN)` | `[0, 1]`; lower is better. | Also sensitive to the number of negative pairs. |
|
||||||
|
| F0.5 | `1.25 TP / (1.25 TP + 0.25 FN + FP)` | `[0, 1]`; higher is better. | Gives greater weight to precision and over-stemming avoidance. |
|
||||||
|
| F1 | `2 TP / (2 TP + FN + FP)` | `[0, 1]`; higher is better. | Equal precision/recall emphasis. |
|
||||||
|
| F2 | `5 TP / (5 TP + 4 FN + FP)` | `[0, 1]`; higher is better. | Gives greater weight to recall and under-stemming avoidance. |
|
||||||
|
| Jaccard | `TP / (TP + FP + FN)` | `[0, 1]`; higher is better. | Excludes TN. All policies. |
|
||||||
|
| Fowlkes–Mallows | `sqrt(precision * recall)` | `[0, 1]`; higher is better. | Geometric balance of precision and recall. All policies. |
|
||||||
|
| MCC | `(TP TN - FP FN) / sqrt((TP+FP)(TP+FN)(TN+FP)(TN+FN))` | `[-1, 1]`; higher is better. Uses all four counts. | Informative under imbalance; undefined for a zero product denominator. All policies with policy-specific interpretation. |
|
||||||
|
|
||||||
|
The general F-beta formula is `((1 + betaSquared) * TP) / (((1 + betaSquared) * TP) + (betaSquared * FN) + FP)`.
|
||||||
|
|
||||||
|
## Partition-only metrics
|
||||||
|
|
||||||
|
These metrics apply only to `PRIMARY_OUTPUT`. Candidate relations are not forced into artificial partitions.
|
||||||
|
|
||||||
|
- Adjusted Rand Index is the Rand agreement corrected for agreement expected from the gold/predicted contingency-table marginals. Its usual range is `[-1, 1]`, with `1` indicating identical partitions.
|
||||||
|
- Homogeneity is `1 - H(gold | predicted) / H(gold)`, in `[0, 1]`; each predicted cluster ideally contains one gold group.
|
||||||
|
- Completeness is `1 - H(predicted | gold) / H(predicted)`, in `[0, 1]`; each gold group ideally maps to one predicted cluster.
|
||||||
|
- V-measure is the harmonic mean of homogeneity and completeness, in `[0, 1]`.
|
||||||
|
- Normalized mutual information uses arithmetic-mean entropy normalization: `MI / ((H(gold) + H(predicted)) / 2)`, in `[0, 1]` under this implementation.
|
||||||
|
|
||||||
|
Entropy zero cases follow the evaluator's explicit perfect/undefined conventions. Language tables render inapplicable candidate-policy values as `n/a`.
|
||||||
|
|
||||||
|
## Aggregation and ranking
|
||||||
|
|
||||||
|
Macro metrics average defined per-language values, giving each language equal weight. Micro metrics sum TP, FP, FN, and TN before calculating a metric. Cross-stemmer aggregate comparisons require the exact common supported-language intersection; unsupported languages are not zero-filled.
|
||||||
|
|
||||||
|
Language tables sort by unrounded balanced accuracy, then MCC, F1, over-stemming rate, over-stemming error count, under-stemming rate, stemmer name, and stable policy order. Display rounding never controls rank.
|
||||||
|
|
||||||
|
Multiple metrics and Pearson/Spearman correlation datasets are published because metric suitability and correlation remain analytical questions. Strong correlation does not establish equivalence.
|
||||||
|
|
||||||
|
## Limitations
|
||||||
|
|
||||||
|
Dictionary groups encode the available annotation, not every linguistic distinction. Homographs may occur in different groups, singleton rows contribute no under-stemming pair, and group size affects pair counts. `ANY_CANDIDATE` is optimistic; `ALL_CANDIDATES` measures an overlapping graph; neither is a deterministic global assignment. Results characterize the tested versions, adapters, dictionaries, and preprocessing, not every deployment or domain.
|
||||||
@@ -1,6 +1,6 @@
|
|||||||
# Benchmark Methodology
|
# Benchmark Methodology
|
||||||
|
|
||||||
The stemmer comparison suite measures Radixor and Java stemmers on the same language and deterministic Radixor dictionary-derived data. Published Radixor rows in this refresh use contracted compiled patch tries, where uniform preferred-command subtrees are collapsed into accepting leaves before the trie is frozen for lookup. For each language, the bundled dictionary resource stores the expected root as the first tab-separated field on a line and its surface forms on the same line. Every single-token field on that line can therefore be paired with the same expected root.
|
The stemmer comparison suite measures Radixor and Java stemmers on the same language and deterministic Radixor model dictionary-derived data. Published Radixor rows in this refresh use contracted compiled patch tries, where uniform preferred-command subtrees are collapsed into accepting leaves before the trie is frozen for lookup. For each language, the registered default model resource stores the expected root as the first tab-separated field on a line and its surface forms on the same line. Every single-token field on that line can therefore be paired with the same expected root.
|
||||||
|
|
||||||
Published stemmer comparison results must come only from benchmark classes matching `.*StemmerComparisonBenchmark.*`. Internal `FrequencyTrie*` microbenchmarks are not part of those results.
|
Published stemmer comparison results must come only from benchmark classes matching `.*StemmerComparisonBenchmark.*`. Internal `FrequencyTrie*` microbenchmarks are not part of those results.
|
||||||
|
|
||||||
@@ -21,11 +21,9 @@ timePerChangedTokenNs = JMH score ns/op / changedTimingTokenCount
|
|||||||
|
|
||||||
This is necessary because Radixor dictionaries have different token counts by language.
|
This is necessary because Radixor dictionaries have different token counts by language.
|
||||||
|
|
||||||
## Quality And Search Interpretation
|
## Exact-root quality and interpretation
|
||||||
|
|
||||||
Radixor speed must be interpreted together with exact-root quality. A slower Radixor row must not be read as a simple performance weakness when Radixor is also the row with accuracy close to 100% and competing stemmers are much lower.
|
Runtime and exact-root agreement must be interpreted separately. Light, minimal, possessive, and aggressive rule-based implementations deliberately address different scopes and may achieve lower latency by performing fewer transformations. A throughput advantage does not establish higher linguistic quality, and higher dictionary agreement does not establish lower operational cost.
|
||||||
|
|
||||||
Many fast light, minimal, possessive, or aggressive rule-based stemmers are fast because they do much less linguistic work. The measured Radixor cost buys dictionary-trained precision, and that precision is what improves search quality when queries and indexed text are reduced to the same intended roots.
|
|
||||||
|
|
||||||
The [English dictionary coverage benchmark](english-coverage.md) shows this operating curve explicitly: contracted tries reduce lookup cost in uniform regions, while reduced dictionary coverage still lowers changed-form precision.
|
The [English dictionary coverage benchmark](english-coverage.md) shows this operating curve explicitly: contracted tries reduce lookup cost in uniform regions, while reduced dictionary coverage still lowers changed-form precision.
|
||||||
|
|
||||||
@@ -35,7 +33,7 @@ Radixor is measured over dictionary tokens from its own resources: lower-case wi
|
|||||||
|
|
||||||
Lucene TokenFilter paths include required normalization in the measured pipeline. Examples include lower-case normalization for filters requiring lower-case input, German normalization before German light/minimal stemming, and Persian decimal, Arabic, and Persian normalization before Persian stemming. No ASCII folding is applied to Czech or Polish paths, because those Lucene stemmers are diacritic-aware or dictionary/table-backed for those languages. TokenFilter throughput methods materialize each emitted `CharTermAttribute` as a `String` before passing it to the JMH `Blackhole`, so output consumption is easier to inspect and closer to the direct stemmer methods.
|
Lucene TokenFilter paths include required normalization in the measured pipeline. Examples include lower-case normalization for filters requiring lower-case input, German normalization before German light/minimal stemming, and Persian decimal, Arabic, and Persian normalization before Persian stemming. No ASCII folding is applied to Czech or Polish paths, because those Lucene stemmers are diacritic-aware or dictionary/table-backed for those languages. TokenFilter throughput methods materialize each emitted `CharTermAttribute` as a `String` before passing it to the JMH `Blackhole`, so output consumption is easier to inspect and closer to the direct stemmer methods.
|
||||||
|
|
||||||
For right-to-left Radixor languages, patch application uses the traversal direction stored in trie metadata. This is required because static backward patch application is not correct for all bundled languages.
|
For right-to-left Radixor languages, patch application uses the traversal direction stored in trie metadata. This is required because static backward patch application is not correct for all registered language models.
|
||||||
|
|
||||||
## Quality Metric
|
## Quality Metric
|
||||||
|
|
||||||
@@ -56,4 +54,7 @@ rootPreservedPercent = rootPreservedMatches / rootEvaluatedTokens * 100
|
|||||||
|
|
||||||
Morfologik can emit multiple terms for one input token. The quality benchmark uses the first emitted term for exact-root accounting when no ranking weight is exposed. Throughput benchmarks for Morfologik TokenFilter paths consume all emitted terms.
|
Morfologik can emit multiple terms for one input token. The quality benchmark uses the first emitted term for exact-root accounting when no ranking weight is exposed. Throughput benchmarks for Morfologik TokenFilter paths consume all emitted terms.
|
||||||
|
|
||||||
Quality reports intentionally use one deterministic measurement iteration without warmup, because exact-root agreement is not a timing metric and repeated precision passes would only duplicate the same counters.
|
Quality reports use JMH auxiliary counter rows. Exact-root accounting is deterministic for a fixed corpus and stemmer, so repeated measurement samples duplicate the same counters; documentation uses the counter ratios and does not interpret quality benchmark timing scores.
|
||||||
|
|
||||||
|
Pairwise over-stemming, under-stemming, candidate-aware policies, balanced accuracy, and partition comparison are a separate analytical evaluation. See [Linguistic Quality Methodology](linguistic-quality.md); exact-root accuracy must not be interpreted as the complement of pairwise under-stemming.
|
||||||
|
Default rows use `Language.defaultModelId()`. Optional variants require a separate model field; `pl-pl-unimorph` and `pl-pl-polimorf` must never share an ambiguous Polish label. The benchmark runtime receives each resource exactly once from its individual model JAR through direct JMH runtime dependencies. See [Model Selection and Loading](../../model-selection-and-loading.md).
|
||||||
|
|||||||
83
docs/benchmarks/reference/reproducibility.md
Normal file
83
docs/benchmarks/reference/reproducibility.md
Normal file
@@ -0,0 +1,83 @@
|
|||||||
|
# Reproducibility and Raw Data
|
||||||
|
|
||||||
|
## Published quality snapshot
|
||||||
|
|
||||||
|
- Machine-readable CSV: [stemming-quality.csv](../data/stemming-quality.csv)
|
||||||
|
- SHA-256 record: [stemming-quality.sha256](../data/stemming-quality.sha256)
|
||||||
|
- SHA-256: `5a93a6ab60e46489737cd649eb1ac48182114b9038f7f20195ab9d1c1fc0dd28`
|
||||||
|
- Complete scenarios: 308
|
||||||
|
- Authoritative language universe: 20 languages
|
||||||
|
- Language-page scenarios: 302 across 19 existing benchmark pages
|
||||||
|
|
||||||
|
The six remaining scenarios are the three Radixor policies in two modes for `HE_IL`. Hebrew is present in the complete result snapshot but has no existing language benchmark page.
|
||||||
|
|
||||||
|
The CSV contains raw TP, FP, FN, and TN counts; raw over/under numerators and denominators; candidate statistics; relation metrics; and partition-only metrics. Documentation is regenerated from this file rather than manually transcribed.
|
||||||
|
|
||||||
|
## Commands
|
||||||
|
|
||||||
|
```bash
|
||||||
|
./gradlew stemmingQuality
|
||||||
|
./gradlew publishStemmingQualityDocumentation
|
||||||
|
./gradlew verifyStemmingQualityDocumentation
|
||||||
|
./gradlew test
|
||||||
|
./gradlew prepareMkDocsSource
|
||||||
|
mkdocs build --strict --config-file build/mkdocs/mkdocs.yml
|
||||||
|
```
|
||||||
|
|
||||||
|
`stemmingQuality` performs the expensive complete evaluation and is intentionally not attached to `test` or `check`. It prepares JMH third-party dependencies automatically and writes:
|
||||||
|
|
||||||
|
- `build/reports/stemming-quality/stemming-quality.csv`
|
||||||
|
- `build/reports/stemming-quality/stemming-quality.md`
|
||||||
|
- `build/reports/stemming-quality/metric-correlations-pearson.csv`
|
||||||
|
- `build/reports/stemming-quality/metric-correlations-spearman.csv`
|
||||||
|
|
||||||
|
Audit mode is enabled with `-PstemmingQualityAudit=true`. Language, stemmer, dictionary-mode, output-policy, and ranking filters are documented on the central [stemming-quality page](../../stemming-quality.md). Filtered reports use separate filenames and cannot be accepted as publication sources.
|
||||||
|
|
||||||
|
`publishStemmingQualityDocumentation` validates the complete build CSV, copies a versioned documentation snapshot, and replaces only marked generated sections. `verifyStemmingQualityDocumentation` re-renders from the checked-in snapshot and fails on changed values, ordering, missing pages, duplicate keys, arithmetic inconsistencies, policy violations, or stale sections.
|
||||||
|
|
||||||
|
The model catalog and rendered site are build outputs under `build/`. They are generated for publication and are never maintained in Git.
|
||||||
|
|
||||||
|
For new measurements, record language, stable model ID, model artifact version, descriptor checksum, source dictionary identity/version, core revision, and benchmark configuration. JMH resolves the required default models and optional PoliMorf directly from their individual model JARs; these benchmark-only dependencies are not transitive to ordinary users.
|
||||||
|
|
||||||
|
Current model descriptors also record the official repository, dataset, license, attribution,
|
||||||
|
verification date, transformations, and source-revision status. Exact historical revisions were
|
||||||
|
not recorded for the legacy UniMorph imports; that limitation is disclosed with
|
||||||
|
`not-recorded-in-legacy-import` rather than reconstructed. Future imports must record the exact
|
||||||
|
upstream revision and source-archive checksum. This reproducibility limitation does not replace or
|
||||||
|
weaken the packaged license and attribution requirements.
|
||||||
|
|
||||||
|
Each UniMorph-derived model artifact carries its own notice with the canonical CC BY-SA 3.0 URI,
|
||||||
|
upstream attribution, transformations, ShareAlike statement, and Leo Galambos contribution notice.
|
||||||
|
The full CC legal text is not duplicated or presented as a root-project license. PoliMorf retains
|
||||||
|
its separately packaged BSD-2-Clause license.
|
||||||
|
|
||||||
|
For a future full PoliMorf measurement, also record the startup heap separately from benchmark parameters. Complete runtime construction is currently verified with a dedicated 6 GiB maximum heap; this limit is neither a retained-trie measurement nor a setting applied to ordinary JMH runs.
|
||||||
|
|
||||||
|
The Pages workflow publishes that staged documentation together with Javadoc, JUnit, PMD, JaCoCo, PIT, representative JMH, SBOM, optional dependency-check output, badge metadata, and retained build history. Its filesystem merge explicitly preserves the `builds/` tree in the separate `gh-pages` publication branch, so documentation regeneration cannot erase durable report URLs.
|
||||||
|
|
||||||
|
## Performance benchmark reproduction
|
||||||
|
|
||||||
|
The JMH comparison command family is:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
./gradlew jmh -Pjmh.includes='.*StemmerComparisonBenchmark.*' --no-daemon
|
||||||
|
```
|
||||||
|
|
||||||
|
The exact JMH configuration, hardware, operating system, and JDK captured for the published performance tables are listed in [Environment and reports](environment.md). Quality and performance reports are separate datasets and are not combined into an undocumented scalar.
|
||||||
|
|
||||||
|
## Recorded and unavailable provenance
|
||||||
|
|
||||||
|
The performance documentation records its 2026-07-06 environment, JDK 25.0.3, operating system, and hardware. The quality CSV records the evaluated identifiers and counts but does not embed the Radixor Git revision, generation date, JDK, operating system, model ID, dictionary content hash, or immutable upstream revisions for every downloaded source. These fields are explicitly unavailable for this historical snapshot and are not reconstructed from filesystem timestamps. In particular, the snapshot predates the optional PoliMorf integration and must not be relabeled as `pl-pl-polimorf`.
|
||||||
|
|
||||||
|
Dependency versions that are reproducible from repository configuration include Apache Lucene 10.5.0, Morfologik 2.1.9, the Ukrainian dictionary artifact 4.9.1, and JMH 1.37. Other upstream branches or downloaded dictionary revisions should be pinned and embedded in a future result schema.
|
||||||
|
|
||||||
|
## Correlation and audit data
|
||||||
|
|
||||||
|
Pearson and Spearman files are generated from unrounded metric values in cohorts separated by dictionary mode and output policy. A missing coefficient means too few observations, undefined input, or zero variance. Correlation is descriptive and does not demonstrate that two metrics are scientifically interchangeable.
|
||||||
|
|
||||||
|
Audit reports preserve original multilingual forms and identify high-contributing dictionary groups. They are build artifacts rather than checked-in publication data because of their size. No documentation value is manually altered after generation.
|
||||||
|
|
||||||
|
## JMH badge compatibility
|
||||||
|
|
||||||
|
The quality documentation generator does not invoke JMH, change JMH result formats, or modify badge tooling. Existing JMH result paths and historical badge-compatible inputs remain independent. The repository currently publishes coverage and mutation badge metadata and retains JMH TXT/CSV artifacts as documented in [Environment and reports](environment.md).
|
||||||
|
See [Model Selection and Loading](../../model-selection-and-loading.md), [Stemmer Models](../../stemmer-models.md), and the generated [model catalog](../../stemmer-model-catalog.md) for current model identities.
|
||||||
30
docs/benchmarks/reference/tested-stemmers.md
Normal file
30
docs/benchmarks/reference/tested-stemmers.md
Normal file
@@ -0,0 +1,30 @@
|
|||||||
|
# Tested Stemmer Inventory
|
||||||
|
|
||||||
|
The JMH adapter registry is authoritative for evaluated implementations and language mappings. Names below describe the implementation actually invoked, not an abstract algorithm in every possible implementation. Unsupported language combinations are omitted rather than scored as failures.
|
||||||
|
|
||||||
|
| Family or implementation | Upstream / attribution | Tested version or revision | Evaluated scope | Output capability and adapter behaviour | Interpretation notes |
|
||||||
|
| --- | --- | --- | --- | --- | --- |
|
||||||
|
| Radixor | Egothor / Radixor project | Current repository revision; exact revision was not embedded in the quality CSV | All 20 reconciled default model languages; 19 have benchmark pages | Deterministic preferred patch via `get`; ranked distinct alternatives via `getAll`; primary is always included | Model-dictionary-derived compiled patch trie. Default rows use each language's stable default model ID. |
|
||||||
|
| Apache Lucene language stem filters | Apache Lucene project | 10.5.0 | Adapter-declared language-specific subsets | TokenFilter lifecycle and language normalization match JMH; normally single-output | Light, minimal, possessive, and language stem filters deliberately implement different scopes. Narrow scope is not a defect. |
|
||||||
|
| Apache Lucene SnowballFilter | Apache Lucene project using Snowball algorithms | Lucene 10.5.0 | Snowball-supported subset of Radixor languages | Single primary token emitted through the Lucene TokenFilter path | Includes TokenStream overhead and required normalization. |
|
||||||
|
| Official Snowball Java | Snowball project | Repository preparation downloads the configured upstream Java distribution; an immutable revision was not recorded in the quality CSV | Same-language adapter subset | Direct generated Java API; single output | Rule-based suffix algorithms provide broad baselines rather than dictionary-root guarantees. |
|
||||||
|
| Lucene Stempel | Apache Lucene / Polish stemming tables | Lucene 10.5.0 | Polish | Direct and TokenFilter paths where registered; single primary output | Table-driven Polish implementation. |
|
||||||
|
| Morfologik | Morfologik project; Lucene integration by Apache Lucene | Morfologik 2.1.9, Lucene integration 10.5.0; Ukrainian dictionary artifact 4.9.1 | Registered Polish and Ukrainian paths | Deterministic first lemma for primary comparison; all distinct lemma strings for candidate policies | Several analyses may share a lemma and are deduplicated by exact string equality. |
|
||||||
|
| Hunspell via Lucene | Hunspell dictionaries from the `wooorm/dictionaries` repository; adapter by Apache Lucene | Lucene 10.5.0; dictionary repository revision was not recorded | Configured German, English, Spanish, French, Dutch, Polish, and Ukrainian dictionaries | First emitted stem is primary; all distinct stems at the token position are candidates | Dictionary content and affix rules differ by language. |
|
||||||
|
| CISTEM | Leonie Weissweiler, CISTEM project | Upstream `master` source path used by preparation; immutable commit not recorded | German | Single output | German stemming algorithm; benchmark-only implementation and gold-standard preparation remain under JMH infrastructure. |
|
||||||
|
| OpenNLP Porter | Apache OpenNLP project | Version resolved by `gradle/opennlp-benchmarks.gradle` and `gradle.lockfile` | English | Direct single output | Porter-family English baseline. |
|
||||||
|
| Lucene Porter source copy | Apache Lucene project | 10.5.0 source artifact | English | Package-isolated benchmark-only generated source; single output | Generated into the JMH build tree, never production code. |
|
||||||
|
| Paice/Husk Lancaster | Upstream Java implementation from `Hopper262/paice-husk-stemmer` | Configured upstream branch/revision in `gradle/paicehusk-benchmarks.gradle`; immutable commit not recorded | English | Direct single output | Aggressive rule-based English baseline; benchmark-only generated source. |
|
||||||
|
|
||||||
|
## Preprocessing and lifecycle
|
||||||
|
|
||||||
|
The quality evaluator calls the same adapter matrix used by JMH. Each language mapping is explicit. Retained dictionary forms are not evaluation-lowercased or normalized. Where an implementation requires preprocessing, such as Lucene German or Persian normalization, that operation is part of its documented adapter path. Stateful TokenFilters are reset through the same sequential lifecycle used by the benchmark and are not invoked concurrently.
|
||||||
|
|
||||||
|
Candidate sets are non-null, non-empty, contain the deterministic primary output, contain no null strings, and are deduplicated using exact Java string equality. Gold-standard group identity never selects, removes, or ranks a candidate.
|
||||||
|
|
||||||
|
## Coverage fairness
|
||||||
|
|
||||||
|
Radixor coverage is derived from registered default descriptors reconciled with language enumeration. Third-party coverage is the intersection of that universe with actual adapter support. Absence therefore means “not supported or not configured for this language,” not “zero quality.” Optional `pl-pl-polimorf` is a separate model row and does not replace default `pl-pl-unimorph`. Consult each language page for the exact evaluated rows.
|
||||||
|
|
||||||
|
Project authors and organizations are named only where repository configuration or source notices establish attribution. No broader authorship or license claim is inferred when metadata was not captured.
|
||||||
|
The JMH runtime configuration directly includes optional models needed for controlled comparisons; ordinary users do not receive these benchmark-only dependencies transitively. Historical rows retain their original model inputs. See [Model Selection and Loading](../../model-selection-and-loading.md).
|
||||||
@@ -1,267 +1,104 @@
|
|||||||
# Built-in Languages
|
# Built-in Languages and Default Models
|
||||||
|
|
||||||
Radixor ships with a curated set of bundled stemmer dictionaries that can be loaded directly from the library distribution. These resources are intended to provide an immediately usable baseline for evaluation, prototyping, integration, and general-purpose stemming workloads, while still fitting naturally into workflows where the bundled baseline is later refined, extended, or replaced with custom lexical data.
|
“Supported language” means that Radixor defines a language enum value and publishes a corresponding default model artifact. It does not mean that a dictionary is embedded in the core JAR. Applications add model artifacts explicitly or use the optional standard pack.
|
||||||
|
|
||||||
## Overview
|
The language enum carries language identity, writing direction, a legacy resource-directory name, and the stable default model ID. A model descriptor carries the independently versioned model identity and resource. See [Model Selection and Loading](model-selection-and-loading.md) for the API and the generated [model catalog](stemmer-model-catalog.md) for versions, provenance, checksums, and sizes.
|
||||||
|
|
||||||
Bundled dictionaries are exposed through:
|
## Defaults and variants
|
||||||
|
|
||||||
```java
|
| Language | Enum | Default model ID | Default artifact | Optional variants |
|
||||||
org.egothor.stemmer.StemmerPatchTrieLoader.Language
|
|---|---|---|---|---|
|
||||||
```
|
| Czech | `CS_CZ` | `cs-cz-default` | `org.egothor:radixor-model-cs-cz-default` | — |
|
||||||
|
| Danish | `DA_DK` | `da-dk-default` | `org.egothor:radixor-model-da-dk-default` | — |
|
||||||
|
| German | `DE_DE` | `de-de-default` | `org.egothor:radixor-model-de-de-default` | — |
|
||||||
|
| Spanish | `ES_ES` | `es-es-default` | `org.egothor:radixor-model-es-es-default` | — |
|
||||||
|
| Persian | `FA_IR` | `fa-ir-default` | `org.egothor:radixor-model-fa-ir-default` | — |
|
||||||
|
| Finnish | `FI_FI` | `fi-fi-default` | `org.egothor:radixor-model-fi-fi-default` | — |
|
||||||
|
| French | `FR_FR` | `fr-fr-default` | `org.egothor:radixor-model-fr-fr-default` | — |
|
||||||
|
| Hebrew | `HE_IL` | `he-il-default` | `org.egothor:radixor-model-he-il-default` | — |
|
||||||
|
| Hungarian | `HU_HU` | `hu-hu-default` | `org.egothor:radixor-model-hu-hu-default` | — |
|
||||||
|
| Italian | `IT_IT` | `it-it-default` | `org.egothor:radixor-model-it-it-default` | — |
|
||||||
|
| Norwegian Bokmål | `NB_NO` | `nb-no-default` | `org.egothor:radixor-model-nb-no-default` | — |
|
||||||
|
| Dutch | `NL_NL` | `nl-nl-default` | `org.egothor:radixor-model-nl-nl-default` | — |
|
||||||
|
| Norwegian Nynorsk | `NN_NO` | `nn-no-default` | `org.egothor:radixor-model-nn-no-default` | — |
|
||||||
|
| Polish | `PL_PL` | `pl-pl-unimorph` | `org.egothor:radixor-model-pl-pl-unimorph` | `pl-pl-polimorf` / `org.egothor:radixor-model-pl-pl-polimorf` |
|
||||||
|
| Portuguese | `PT_PT` | `pt-pt-default` | `org.egothor:radixor-model-pt-pt-default` | — |
|
||||||
|
| Russian | `RU_RU` | `ru-ru-default` | `org.egothor:radixor-model-ru-ru-default` | — |
|
||||||
|
| Swedish | `SV_SE` | `sv-se-default` | `org.egothor:radixor-model-sv-se-default` | — |
|
||||||
|
| Ukrainian | `UK_UA` | `uk-ua-default` | `org.egothor:radixor-model-uk-ua-default` | — |
|
||||||
|
| English | `US_UK` | `us-uk-default` | `org.egothor:radixor-model-us-uk-default` | — |
|
||||||
|
| Yiddish | `YI` | `yi-default` | `org.egothor:radixor-model-yi-default` | — |
|
||||||
|
|
||||||
Each bundled dictionary is packaged with the library as a compressed UTF-8 text resource. When loaded through the runtime API, the resource is parsed by `StemmerDictionaryParser`, transformed into patch-command mappings, and compiled into a read-only `FrequencyTrie<CompiledPatchCommand>` by `StemmerPatchTrieLoader`.
|
The maintained table deliberately avoids duplicating mutable provenance and checksum fields. Those values come from module metadata and are generated into the model catalog.
|
||||||
|
|
||||||
The bundled language definition also carries a language-level right-to-left flag. That flag is used by the loader to derive the `WordTraversalDirection` used for both trie-key construction and patch-command generation. In practice, left-to-right bundled languages use historical backward Egothor traversal, while right-to-left bundled languages use forward traversal over the stored form.
|
## The Polish dual-model case
|
||||||
|
|
||||||
## Supported bundled languages
|
`PL_PL` represents Polish. It is not an alias for either source dictionary.
|
||||||
|
|
||||||
The following bundled language identifiers are currently available:
|
- `loadCompiled(Language.PL_PL, ...)` resolves `pl-pl-unimorph`.
|
||||||
|
- `registry.require("pl-pl-polimorf")` resolves the optional PoliMorf model.
|
||||||
|
- `StemmerPatchTrieLoader.loadCompiled("pl-pl-polimorf", true, reductionMode)` constructs its compiled trie explicitly; complete construction is verified with a dedicated 6 GiB test heap.
|
||||||
|
- Both artifacts may be present and loaded independently.
|
||||||
|
- Adding PoliMorf does not change the language default.
|
||||||
|
- Radixor does not merge their dictionaries or outputs automatically.
|
||||||
|
|
||||||
| Language | Enum constant | Writing direction | Notes | Benchmark page |
|
UniMorph and PoliMorf have different lexical sources and provenance. Applications should compare outputs with application-specific regression tests before changing an explicit model choice.
|
||||||
|---|---|---:|---|---|
|
|
||||||
| Czech | `CS_CZ` | LTR | Bundled general-purpose dictionary | [Czech](benchmarks/languages/czech.md) |
|
|
||||||
| Danish | `DA_DK` | LTR | Bundled general-purpose dictionary | [Danish](benchmarks/languages/danish.md) |
|
|
||||||
| German | `DE_DE` | LTR | Bundled general-purpose dictionary | [German](benchmarks/languages/german.md) |
|
|
||||||
| Spanish | `ES_ES` | LTR | Bundled general-purpose dictionary | [Spanish](benchmarks/languages/spanish.md) |
|
|
||||||
| Persian | `FA_IR` | RTL | Bundled dictionary uses forward traversal over the stored form | [Persian](benchmarks/languages/persian.md) |
|
|
||||||
| Finnish | `FI_FI` | LTR | Bundled general-purpose dictionary | [Finnish](benchmarks/languages/finnish.md) |
|
|
||||||
| French | `FR_FR` | LTR | Bundled general-purpose dictionary | [French](benchmarks/languages/french.md) |
|
|
||||||
| Hebrew | `HE_IL` | RTL | Bundled dictionary uses forward traversal over the stored form | No same-language external benchmark in this run |
|
|
||||||
| Hungarian | `HU_HU` | LTR | Bundled general-purpose dictionary | [Hungarian](benchmarks/languages/hungarian.md) |
|
|
||||||
| Italian | `IT_IT` | LTR | Bundled general-purpose dictionary | [Italian](benchmarks/languages/italian.md) |
|
|
||||||
| Norwegian Bokmål | `NB_NO` | LTR | Bundled general-purpose dictionary | [Norwegian Bokmal](benchmarks/languages/norwegian-bokmal.md) |
|
|
||||||
| Dutch | `NL_NL` | LTR | Bundled general-purpose dictionary | [Dutch](benchmarks/languages/dutch.md) |
|
|
||||||
| Norwegian Nynorsk | `NN_NO` | LTR | Bundled general-purpose dictionary | [Norwegian Nynorsk](benchmarks/languages/norwegian-nynorsk.md) |
|
|
||||||
| Polish | `PL_PL` | LTR | Bundled general-purpose dictionary | [Polish](benchmarks/languages/polish.md) |
|
|
||||||
| Portuguese | `PT_PT` | LTR | Bundled general-purpose dictionary | [Portuguese](benchmarks/languages/portuguese.md) |
|
|
||||||
| Russian | `RU_RU` | LTR | Bundled general-purpose dictionary | [Russian](benchmarks/languages/russian.md) |
|
|
||||||
| Swedish | `SV_SE` | LTR | Bundled general-purpose dictionary | [Swedish](benchmarks/languages/swedish.md) |
|
|
||||||
| Ukrainian | `UK_UA` | LTR | Bundled general-purpose dictionary | [Ukrainian](benchmarks/languages/ukrainian.md) |
|
|
||||||
| English | `US_UK` | LTR | Bundled general-purpose dictionary | [English](benchmarks/languages/english.md) |
|
|
||||||
| Yiddish | `YI` | RTL | Bundled dictionary uses forward traversal over the stored form | [Yiddish](benchmarks/languages/yiddish.md) |
|
|
||||||
|
|
||||||
## Basic usage
|
## Dependency patterns
|
||||||
|
|
||||||
Load a bundled dictionary like this:
|
Minimal English:
|
||||||
|
|
||||||
```java
|
```groovy
|
||||||
import java.io.IOException;
|
dependencies {
|
||||||
|
implementation 'org.egothor:radixor:<radixor-version>'
|
||||||
import org.egothor.stemmer.CompiledPatchCommand;
|
runtimeOnly 'org.egothor:radixor-model-us-uk-default:1.0.0'
|
||||||
import org.egothor.stemmer.FrequencyTrie;
|
|
||||||
import org.egothor.stemmer.ReductionMode;
|
|
||||||
import org.egothor.stemmer.StemmerPatchTrieLoader;
|
|
||||||
|
|
||||||
public final class BuiltInExample {
|
|
||||||
|
|
||||||
private BuiltInExample() {
|
|
||||||
throw new AssertionError("No instances.");
|
|
||||||
}
|
|
||||||
|
|
||||||
public static void main(final String[] arguments) throws IOException {
|
|
||||||
final FrequencyTrie<CompiledPatchCommand> trie = StemmerPatchTrieLoader.loadCompiled(
|
|
||||||
StemmerPatchTrieLoader.Language.US_UK,
|
|
||||||
true,
|
|
||||||
ReductionMode.MERGE_SUBTREES_WITH_EQUIVALENT_RANKED_GET_ALL_RESULTS);
|
|
||||||
|
|
||||||
System.out.println(trie.traversalDirection());
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
This call loads the bundled dictionary resource for the selected language, parses its lexical entries, derives patch-command mappings, and compiles the result into a read-only trie.
|
All documented defaults:
|
||||||
|
|
||||||
## Example: stemming with a bundled dictionary
|
```groovy
|
||||||
|
dependencies {
|
||||||
```java
|
implementation 'org.egothor:radixor:<radixor-version>'
|
||||||
import java.io.IOException;
|
runtimeOnly 'org.egothor:radixor-models-standard:<catalog-version>'
|
||||||
|
|
||||||
import org.egothor.stemmer.CompiledPatchCommand;
|
|
||||||
import org.egothor.stemmer.FrequencyTrie;
|
|
||||||
import org.egothor.stemmer.ReductionMode;
|
|
||||||
import org.egothor.stemmer.StemmerPatchTrieLoader;
|
|
||||||
|
|
||||||
public final class EnglishExample {
|
|
||||||
|
|
||||||
private EnglishExample() {
|
|
||||||
throw new AssertionError("No instances.");
|
|
||||||
}
|
|
||||||
|
|
||||||
public static void main(final String[] arguments) throws IOException {
|
|
||||||
final FrequencyTrie<CompiledPatchCommand> trie = StemmerPatchTrieLoader.loadCompiled(
|
|
||||||
StemmerPatchTrieLoader.Language.US_UK,
|
|
||||||
true,
|
|
||||||
ReductionMode.MERGE_SUBTREES_WITH_EQUIVALENT_RANKED_GET_ALL_RESULTS);
|
|
||||||
|
|
||||||
final String word = "running";
|
|
||||||
final CompiledPatchCommand patch = trie.get(word);
|
|
||||||
final String stem = patch == null ? word : patch.apply(word);
|
|
||||||
|
|
||||||
System.out.println(word + " -> " + stem);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
`CompiledPatchCommand` values are compiled with the traversal direction used when the trie and its patch commands were produced.
|
The standard pack is metadata-only and excludes optional PoliMorf.
|
||||||
|
|
||||||
## Traversal behavior and right-to-left languages
|
Every individual model artifact carries its own provenance and licensing material. UniMorph
|
||||||
|
models carry different model-specific CC BY-SA 3.0 notices because their official language
|
||||||
|
repositories identify different lexical sources and contributors. Each notice preserves upstream
|
||||||
|
attribution and records the Radixor transformations and Leo Galambos contribution statement.
|
||||||
|
Legacy imports disclose when an exact historical revision was not recorded; this is a
|
||||||
|
reproducibility limitation, not a claim that the source or license is unknown.
|
||||||
|
|
||||||
Bundled dictionaries are not all processed identically.
|
## Loading a language default
|
||||||
|
|
||||||
For traditional left-to-right suffix-oriented resources, Radixor preserves historical Egothor behavior and traverses logical word characters backward. That means trie paths are constructed from the logical end of the stored word toward its beginning, and patch commands are interpreted with the same backward traversal model.
|
|
||||||
|
|
||||||
For bundled right-to-left languages such as Persian, Hebrew, and Yiddish, Radixor uses forward traversal over the stored form. In those cases:
|
|
||||||
|
|
||||||
- trie keys are traversed from the logical beginning of the stored form,
|
|
||||||
- patch commands are generated in that same forward direction,
|
|
||||||
- compiled patch-command application uses `WordTraversalDirection.FORWARD`, which is naturally captured when `loadCompiled(...)` creates `CompiledPatchCommand` values.
|
|
||||||
|
|
||||||
This design keeps the traversal policy explicit and consistent across dictionary loading, trie lookup, binary persistence, builder reconstruction, and patch application.
|
|
||||||
|
|
||||||
## Reduction behavior
|
|
||||||
|
|
||||||
Bundled dictionaries can be compiled using any supported `ReductionMode`. The reduction configuration controls how semantically equivalent subtrees are merged during trie compilation, while preserving the contract of the selected mode.
|
|
||||||
|
|
||||||
Typical entry points are:
|
|
||||||
|
|
||||||
- `StemmerPatchTrieLoader.loadCompiled(language, storeOriginal, reductionMode)`
|
|
||||||
- `StemmerPatchTrieLoader.loadCompiled(language, storeOriginal, reductionSettings)`
|
|
||||||
|
|
||||||
For most users, `ReductionMode.MERGE_SUBTREES_WITH_EQUIVALENT_RANKED_GET_ALL_RESULTS` is the most conservative general-purpose choice because it preserves ranked `getAll(...)` behavior.
|
|
||||||
|
|
||||||
Compiled bundled dictionaries also use internal uniform-subtree contraction. If a whole subtree
|
|
||||||
would return the same preferred patch command, Radixor stores that subtree as an accepting leaf and
|
|
||||||
removes the deeper branches. This is the contracted trie representation used by the published
|
|
||||||
benchmark tables and is independent of the public reduction mode selected by the caller.
|
|
||||||
|
|
||||||
## Intended role of bundled dictionaries
|
|
||||||
|
|
||||||
Bundled dictionaries should be understood as practical default resources.
|
|
||||||
|
|
||||||
They are a good fit when:
|
|
||||||
|
|
||||||
- a supported language is already available,
|
|
||||||
- immediate usability matters,
|
|
||||||
- a reasonable baseline is sufficient,
|
|
||||||
- the goal is evaluation, prototyping, or straightforward integration.
|
|
||||||
|
|
||||||
They are also well suited to staged refinement workflows in which a bundled base is loaded first, then extended with domain-specific vocabulary, and finally persisted as a custom binary artifact.
|
|
||||||
|
|
||||||
## Character representation
|
|
||||||
|
|
||||||
Bundled dictionaries are ordinary UTF-8 lexical resources. The parser reads them as text, the trie stores standard Java strings, and the patch-command model operates on general character sequences.
|
|
||||||
|
|
||||||
This is important for two reasons:
|
|
||||||
|
|
||||||
1. the built-in resources are not limited to ASCII-only processing,
|
|
||||||
2. the traversal model is orthogonal to character encoding and script choice.
|
|
||||||
|
|
||||||
In other words, right-to-left handling in the loader is about logical traversal strategy, not about introducing a separate character model.
|
|
||||||
|
|
||||||
## When to prefer custom dictionaries
|
|
||||||
|
|
||||||
A custom dictionary is usually the better choice when:
|
|
||||||
|
|
||||||
- domain-specific vocabulary materially affects stemming quality,
|
|
||||||
- lexical coverage must be controlled more precisely,
|
|
||||||
- a stronger lexical resource is available than the bundled baseline,
|
|
||||||
- operational requirements demand an explicitly curated, versioned artifact.
|
|
||||||
|
|
||||||
Typical examples include:
|
|
||||||
|
|
||||||
- technical terminology,
|
|
||||||
- biomedical language,
|
|
||||||
- legal or financial vocabulary,
|
|
||||||
- organization-specific product and process names,
|
|
||||||
- dictionaries maintained with project-specific validation rules.
|
|
||||||
|
|
||||||
## Production recommendation
|
|
||||||
|
|
||||||
For production systems, the most robust workflow is usually:
|
|
||||||
|
|
||||||
1. start from a bundled dictionary when it is suitable,
|
|
||||||
2. extend it with domain-specific forms if needed,
|
|
||||||
3. rebuild it into a binary artifact,
|
|
||||||
4. deploy that compiled binary artifact,
|
|
||||||
5. load it at runtime through `loadBinaryCompiled(...)`.
|
|
||||||
|
|
||||||
This avoids repeated startup parsing and makes the deployed stemming behavior explicit, reproducible, and versionable.
|
|
||||||
|
|
||||||
## Example refinement workflow
|
|
||||||
|
|
||||||
```java
|
```java
|
||||||
import java.io.IOException;
|
final FrequencyTrie<CompiledPatchCommand> trie =
|
||||||
import java.nio.file.Path;
|
StemmerPatchTrieLoader.loadCompiled(
|
||||||
|
|
||||||
import org.egothor.stemmer.FrequencyTrie;
|
|
||||||
import org.egothor.stemmer.FrequencyTrieBuilders;
|
|
||||||
import org.egothor.stemmer.PatchCommandEncoder;
|
|
||||||
import org.egothor.stemmer.ReductionMode;
|
|
||||||
import org.egothor.stemmer.ReductionSettings;
|
|
||||||
import org.egothor.stemmer.StemmerPatchTrieBinaryIO;
|
|
||||||
import org.egothor.stemmer.StemmerPatchTrieLoader;
|
|
||||||
|
|
||||||
public final class BundledRefinementExample {
|
|
||||||
|
|
||||||
private BundledRefinementExample() {
|
|
||||||
throw new AssertionError("No instances.");
|
|
||||||
}
|
|
||||||
|
|
||||||
public static void main(final String[] arguments) throws IOException {
|
|
||||||
final FrequencyTrie<String> base = StemmerPatchTrieLoader.load(
|
|
||||||
StemmerPatchTrieLoader.Language.US_UK,
|
StemmerPatchTrieLoader.Language.US_UK,
|
||||||
true,
|
true,
|
||||||
ReductionMode.MERGE_SUBTREES_WITH_EQUIVALENT_RANKED_GET_ALL_RESULTS);
|
ReductionMode.MERGE_SUBTREES_WITH_EQUIVALENT_RANKED_GET_ALL_RESULTS);
|
||||||
|
|
||||||
final FrequencyTrie.Builder<String> builder = FrequencyTrieBuilders.copyOf(
|
|
||||||
base,
|
|
||||||
String[]::new,
|
|
||||||
ReductionSettings.withDefaults(
|
|
||||||
ReductionMode.MERGE_SUBTREES_WITH_EQUIVALENT_RANKED_GET_ALL_RESULTS));
|
|
||||||
|
|
||||||
final PatchCommandEncoder encoder = PatchCommandEncoder.builder()
|
|
||||||
.traversalDirection(base.traversalDirection())
|
|
||||||
.build();
|
|
||||||
|
|
||||||
builder.put("microservices", encoder.encode("microservices", "microservice"));
|
|
||||||
|
|
||||||
final FrequencyTrie<String> compiled = builder.build();
|
|
||||||
|
|
||||||
StemmerPatchTrieBinaryIO.write(compiled, Path.of("english-custom.radixor.gz"));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
```
|
```
|
||||||
|
|
||||||
The reconstructed builder preserves the traversal direction of the source trie, so refinements remain semantically aligned with the original bundled dictionary.
|
The call discovers the default descriptor from the runtime classpath, verifies its compressed resource, parses the GZip UTF-8 dictionary, and constructs a read-only trie. A missing default throws `StemmerModelNotFoundException`; there is no arbitrary fallback.
|
||||||
|
|
||||||
## Extending language support
|
## Writing direction
|
||||||
|
|
||||||
The built-in set is intentionally a practical baseline rather than a closed catalog. Additional languages, stronger lexical coverage, and improved dictionaries for currently supported languages are all natural extension paths.
|
Persian, Hebrew, and Yiddish declare right-to-left language metadata and use forward traversal over stored forms. Other defaults use historical backward Egothor traversal. This setting must remain aligned across dictionary parsing, trie lookup, patch generation, persistence, and application. Model identity remains separate from writing direction.
|
||||||
|
|
||||||
What matters most is not only the number of entries, but the quality, consistency, maintainability, and operational usefulness of the lexical resource being added.
|
## Custom and persisted alternatives
|
||||||
|
|
||||||
## Related API surface
|
Registered model artifacts are a convenient reproducible baseline. Applications may instead load caller-owned textual dictionaries or persist compiled `.radixor.gz` tries. Those paths are distinct from model artifact discovery:
|
||||||
|
|
||||||
The following types are typically involved when working with bundled dictionaries:
|
- a model `stemmer.gz` is a compressed textual dictionary plus descriptor/index metadata;
|
||||||
|
- a `.radixor.gz` created by the binary writer is a persisted compiled trie;
|
||||||
|
- a source dictionary is upstream input, not automatically a valid model artifact.
|
||||||
|
|
||||||
- `StemmerPatchTrieLoader`
|
See [Dictionary Format](dictionary-format.md), [CLI Compilation](cli-compilation.md), and [Stemmer Models](stemmer-models.md).
|
||||||
- `StemmerPatchTrieLoader.Language`
|
|
||||||
- `FrequencyTrie`
|
|
||||||
- `PatchCommandEncoder`
|
|
||||||
- `WordTraversalDirection`
|
|
||||||
- `ReductionMode`
|
|
||||||
- `ReductionSettings`
|
|
||||||
- `StemmerPatchTrieBinaryIO`
|
|
||||||
- `FrequencyTrieBuilders`
|
|
||||||
|
|
||||||
## Next steps
|
## Benchmark interpretation
|
||||||
|
|
||||||
- [Quick start](quick-start.md)
|
Benchmark rows must identify the Radixor model ID used. Default rows use the default IDs above. Optional Polish PoliMorf comparisons must be labeled `pl-pl-polimorf`; they are not interchangeable with the historical default Polish row. Continue with [Benchmarking](benchmarking.md) and [Reproducibility](benchmarks/reference/reproducibility.md).
|
||||||
- [Dictionary format](dictionary-format.md)
|
|
||||||
- [CLI compilation](cli-compilation.md)
|
|
||||||
- [Programmatic usage](programmatic-usage.md)
|
|
||||||
|
|
||||||
## Summary
|
|
||||||
|
|
||||||
Radixor’s built-in language support provides immediate usability, a professionally defined baseline API, and a practical starting point for custom refinement. The bundled set now includes both left-to-right and right-to-left languages, and the library models that distinction explicitly through `WordTraversalDirection` so that trie construction, lookup, and patch application remain consistent.
|
|
||||||
|
|||||||
@@ -2,6 +2,8 @@
|
|||||||
|
|
||||||
Radixor provides a command-line compiler for turning line-oriented dictionary files into compact binary stemmer artifacts.
|
Radixor provides a command-line compiler for turning line-oriented dictionary files into compact binary stemmer artifacts.
|
||||||
|
|
||||||
|
The CLI output is not a model JAR. A model artifact contains a compressed textual dictionary, descriptor, index, checksum, and license so the runtime registry can discover and compile it. The CLI instead emits an already compiled binary trie for direct `loadBinaryCompiled(...)` use. Choose the model-module workflow when independently published classpath discovery is required; choose the CLI when the application owns a compiled binary asset.
|
||||||
|
|
||||||
This is the preferred preparation workflow when stemming should run against an already compiled artifact rather than against raw dictionary input. The CLI reads the dictionary, derives patch commands, builds a mutable trie, applies the selected subtree reduction strategy, and writes the final compiled trie in the project binary format under GZip compression. The result is a deployment-ready `.radixor.gz` file that can be loaded directly by application code.
|
This is the preferred preparation workflow when stemming should run against an already compiled artifact rather than against raw dictionary input. The CLI reads the dictionary, derives patch commands, builds a mutable trie, applies the selected subtree reduction strategy, and writes the final compiled trie in the project binary format under GZip compression. The result is a deployment-ready `.radixor.gz` file that can be loaded directly by application code.
|
||||||
|
|
||||||
## What the CLI does
|
## What the CLI does
|
||||||
@@ -17,6 +19,10 @@ The `Compile` tool performs the following steps:
|
|||||||
|
|
||||||
This workflow is intentionally aligned with the same dictionary semantics used elsewhere in the library. Remarks introduced by `#` or `//` are supported through the shared dictionary parser.
|
This workflow is intentionally aligned with the same dictionary semantics used elsewhere in the library. Remarks introduced by `#` or `//` are supported through the shared dictionary parser.
|
||||||
|
|
||||||
|
## Create a registered custom model instead
|
||||||
|
|
||||||
|
To publish or deploy a custom dictionary through `StemmerModelRegistry`, do not merely rename CLI output to `stemmer.gz`. Create `models/<model-id>`, preserve the textual dictionary as a GZip module input, provide source metadata and a license, apply `org.egothor.radixor.model`, and run the model validation tasks. The resulting JAR has an index, descriptor, namespaced textual dictionary, checksum, and license. Detailed packaging is documented in [Stemmer Models](stemmer-models.md); selection is documented in [Model Selection and Loading](model-selection-and-loading.md).
|
||||||
|
|
||||||
## Basic usage
|
## Basic usage
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
@@ -220,6 +226,8 @@ The ranked `getAll()` mode is the safest default. The unordered and dominant mod
|
|||||||
|
|
||||||
Compilation is usually a one-time step and is generally fast. The more important operational consideration is memory usage during preparation, because the dictionary-derived mutable structure exists before reduction compacts it into the final read-only trie. This is especially relevant for very large source dictionaries.
|
Compilation is usually a one-time step and is generally fast. The more important operational consideration is memory usage during preparation, because the dictionary-derived mutable structure exists before reduction compacts it into the final read-only trie. This is especially relevant for very large source dictionaries.
|
||||||
|
|
||||||
|
The complete PoliMorf model is the current exceptional case: registered-model verification uses `runtimeModelIntegrationTest` with a 6 GiB maximum heap, configurable through `-PradixorLargeModelMaxHeap=<size>`. This setting applies only to that isolated test process, not the Gradle daemon or ordinary tests.
|
||||||
|
|
||||||
## Example workflow
|
## Example workflow
|
||||||
|
|
||||||
### 1. Prepare a dictionary
|
### 1. Prepare a dictionary
|
||||||
@@ -282,3 +290,5 @@ The CLI and the programmatic API implement the same conceptual preparation step.
|
|||||||
- [Quick start](quick-start.md)
|
- [Quick start](quick-start.md)
|
||||||
- [Programmatic usage](programmatic-usage.md)
|
- [Programmatic usage](programmatic-usage.md)
|
||||||
- [Architecture and reduction](architecture-and-reduction.md)
|
- [Architecture and reduction](architecture-and-reduction.md)
|
||||||
|
!!! note "Radixor 4 model artifacts"
|
||||||
|
Language dictionaries are independently versioned runtime model artifacts, not resources embedded in `radixor`. Language-based APIs resolve deterministic defaults through `StemmerModelRegistry`; see [Stemmer Models](stemmer-models.md).
|
||||||
|
|||||||
@@ -37,7 +37,7 @@ This API is expected to remain supportable across future versions. The preferred
|
|||||||
|
|
||||||
Examples of likely additive evolution include:
|
Examples of likely additive evolution include:
|
||||||
|
|
||||||
- additional bundled language resources,
|
- additional independently versioned language models,
|
||||||
- fuller support for diacritics or native-script language resources,
|
- fuller support for diacritics or native-script language resources,
|
||||||
- expanded documentation and operational tooling,
|
- expanded documentation and operational tooling,
|
||||||
- new convenience methods that do not break existing code.
|
- new convenience methods that do not break existing code.
|
||||||
@@ -83,6 +83,8 @@ Compiled `FrequencyTrie` instances are immutable and thread-safe for concurrent
|
|||||||
|
|
||||||
Serialized patch-command strings remain the stable stored representation used by textual dictionaries and binary artifacts. Runtime stemming should use `CompiledPatchCommand` values produced by `StemmerPatchTrieLoader.loadCompiled(...)`, `StemmerPatchTrieLoader.loadBinaryCompiled(...)`, or `PatchCommandEncoder.compile(...)`.
|
Serialized patch-command strings remain the stable stored representation used by textual dictionaries and binary artifacts. Runtime stemming should use `CompiledPatchCommand` values produced by `StemmerPatchTrieLoader.loadCompiled(...)`, `StemmerPatchTrieLoader.loadBinaryCompiled(...)`, or `PatchCommandEncoder.compile(...)`.
|
||||||
|
|
||||||
|
Language-default, descriptor, and stable model-ID `loadCompiled` entry points share the same compiled-value conversion. Explicit model IDs never fall back to a language default. Model loading is not cached, and construction-memory requirements are model-dependent; the unusually large PoliMorf input is verified separately with a 6 GiB maximum heap.
|
||||||
|
|
||||||
The historical `PatchCommandEncoder.apply(...)` and String-based `applyTo(...)` overloads remain compatibility APIs during the 2.x transition, but they are deprecated because they reparse the patch-command string on each application. See [Migration and Backward Compatibility](migration-and-backward-compatibility.md) for old and new code examples.
|
The historical `PatchCommandEncoder.apply(...)` and String-based `applyTo(...)` overloads remain compatibility APIs during the 2.x transition, but they are deprecated because they reparse the patch-command string on each application. See [Migration and Backward Compatibility](migration-and-backward-compatibility.md) for old and new code examples.
|
||||||
|
|
||||||
Compiled buffer-oriented `CompiledPatchCommand.applyTo(...)` overloads use caller-owned output storage. They do not retain output arrays and report insufficient capacity with `CompiledPatchCommand.APPLY_INSUFFICIENT_CAPACITY`.
|
Compiled buffer-oriented `CompiledPatchCommand.applyTo(...)` overloads use caller-owned output storage. They do not retain output arrays and report insufficient capacity with `CompiledPatchCommand.APPLY_INSUFFICIENT_CAPACITY`.
|
||||||
@@ -110,7 +112,7 @@ The following kinds of change are generally compatible with the project’s dire
|
|||||||
|
|
||||||
- improved internal data structures,
|
- improved internal data structures,
|
||||||
- changes inside `org.egothor.stemmer.trie`,
|
- changes inside `org.egothor.stemmer.trie`,
|
||||||
- expanded bundled dictionaries,
|
- expanded model dictionaries,
|
||||||
- additional supported languages,
|
- additional supported languages,
|
||||||
- improved native-script handling,
|
- improved native-script handling,
|
||||||
- better benchmarks, tests, and reports,
|
- better benchmarks, tests, and reports,
|
||||||
@@ -122,11 +124,11 @@ The project should be able to improve substantially while keeping the main user-
|
|||||||
|
|
||||||
Some areas should be treated as stable in intent but still approached carefully when changed.
|
Some areas should be treated as stable in intent but still approached carefully when changed.
|
||||||
|
|
||||||
### Bundled dictionary contents
|
### Independently versioned model contents
|
||||||
|
|
||||||
Bundled resources are versioned project data, not immutable language standards. Their contents may improve over time.
|
Model resources are independently versioned project data, not immutable language standards. Their contents may improve over time.
|
||||||
|
|
||||||
That means stemming outcomes can legitimately change when bundled dictionaries are refined or expanded. Such changes are compatible with the project’s direction, but they should still be understood as behavior changes at the lexical-resource level.
|
That means stemming outcomes can legitimately change when a model artifact is updated. Such changes are separate from core compatibility and should be reviewed as lexical-resource behavior changes.
|
||||||
|
|
||||||
### Binary format evolution
|
### Binary format evolution
|
||||||
|
|
||||||
@@ -159,7 +161,7 @@ Users should avoid depending on:
|
|||||||
- internal trie package details,
|
- internal trie package details,
|
||||||
- undocumented internal classes or intermediate representations,
|
- undocumented internal classes or intermediate representations,
|
||||||
- incidental internal ordering outside documented lookup semantics,
|
- incidental internal ordering outside documented lookup semantics,
|
||||||
- assumptions that bundled dictionary contents will never evolve,
|
- assumptions that a model's dictionary contents will never evolve across model versions,
|
||||||
- assumptions that internal binary-format details are frozen forever.
|
- assumptions that internal binary-format details are frozen forever.
|
||||||
|
|
||||||
If a behavior is important to your integration, it should ideally be documented at the public API or project-documentation level rather than inferred from internal implementation details.
|
If a behavior is important to your integration, it should ideally be documented at the public API or project-documentation level rather than inferred from internal implementation details.
|
||||||
|
|||||||
@@ -2,7 +2,7 @@
|
|||||||
|
|
||||||
High-quality dictionaries are one of the most valuable ways to improve **Radixor**.
|
High-quality dictionaries are one of the most valuable ways to improve **Radixor**.
|
||||||
|
|
||||||
The project already includes practical bundled dictionaries for common use, but the long-term quality and language reach of the stemmer depend heavily on the quality of its lexical resources. Contributions are therefore welcome not only in the form of code changes, but also in the form of well-prepared dictionary data for existing or additional languages.
|
The project already publishes practical model dictionaries for common use, but long-term quality and language reach depend heavily on lexical-resource quality. Contributions may provide well-prepared model inputs for existing or additional languages.
|
||||||
|
|
||||||
This document explains what makes a dictionary contribution useful, how to structure it, and how to prepare it so that it integrates cleanly with the project.
|
This document explains what makes a dictionary contribution useful, how to structure it, and how to prepare it so that it integrates cleanly with the project.
|
||||||
|
|
||||||
@@ -52,7 +52,7 @@ For full format details, see [Dictionary format](dictionary-format.md).
|
|||||||
|
|
||||||
The most useful dictionary contributions generally fall into one of four categories.
|
The most useful dictionary contributions generally fall into one of four categories.
|
||||||
|
|
||||||
### 1. Stronger dictionaries for already bundled languages
|
### 1. Stronger models for already supported languages
|
||||||
|
|
||||||
Improving lexical quality for already supported languages is often more valuable than merely expanding the language list. Better coverage, cleaner canonicalization, and improved consistency directly improve practical stemming outcomes.
|
Improving lexical quality for already supported languages is often more valuable than merely expanding the language list. Better coverage, cleaner canonicalization, and improved consistency directly improve practical stemming outcomes.
|
||||||
|
|
||||||
@@ -68,7 +68,7 @@ That convention belongs to the supplied dictionaries, not to the underlying algo
|
|||||||
|
|
||||||
### 4. Domain-quality refinements
|
### 4. Domain-quality refinements
|
||||||
|
|
||||||
Some contributions may be more appropriate as curated domain extensions than as replacements for a general-purpose bundled dictionary. These are still useful when they are clearly scoped and operationally coherent.
|
Some contributions may be more appropriate as curated domain extensions than as replacements for a general-purpose default model. These are still useful when clearly scoped and operationally coherent.
|
||||||
|
|
||||||
## Normalization guidance
|
## Normalization guidance
|
||||||
|
|
||||||
@@ -139,6 +139,14 @@ A dictionary should read like a curated lexical resource, not like an unfiltered
|
|||||||
|
|
||||||
## Practical preparation workflow
|
## Practical preparation workflow
|
||||||
|
|
||||||
|
Before conversion, record the official source project and repository, exact revision or release,
|
||||||
|
source-archive checksum, retrieval date, dataset license and URI, supplied attribution, and any
|
||||||
|
required upstream notice. Add a model-specific notice describing every material transformation and
|
||||||
|
the license applied to the derived data, including its canonical URI. Record any protectable
|
||||||
|
Radixor-specific contribution without claiming ownership over the upstream data. A legacy model
|
||||||
|
may disclose that its historical revision was not recorded; new imports must record an exact
|
||||||
|
revision and source-archive checksum rather than using that sentinel.
|
||||||
|
|
||||||
A disciplined dictionary contribution should typically follow this path:
|
A disciplined dictionary contribution should typically follow this path:
|
||||||
|
|
||||||
1. prepare or normalize the lexical source,
|
1. prepare or normalize the lexical source,
|
||||||
@@ -183,7 +191,7 @@ This note does not need to be long. It simply needs to make the resource intelli
|
|||||||
|
|
||||||
## Bundled-resource expectations
|
## Bundled-resource expectations
|
||||||
|
|
||||||
Not every useful dictionary must automatically become a bundled language resource.
|
Not every useful dictionary must automatically become a published default model.
|
||||||
|
|
||||||
To be suitable for bundling, a dictionary should generally be:
|
To be suitable for bundling, a dictionary should generally be:
|
||||||
|
|
||||||
|
|||||||
@@ -2,6 +2,27 @@
|
|||||||
|
|
||||||
Radixor uses a simple line-oriented dictionary format designed for practical stemming workflows. The textual source format is tab-separated values, meaning that columns are separated by the tab character.
|
Radixor uses a simple line-oriented dictionary format designed for practical stemming workflows. The textual source format is tab-separated values, meaning that columns are separated by the tab character.
|
||||||
|
|
||||||
|
## Source text, model resource, and compiled trie
|
||||||
|
|
||||||
|
Three artifacts must not be confused:
|
||||||
|
|
||||||
|
| Artifact | Representation | Consumer |
|
||||||
|
|---|---|---|
|
||||||
|
| Source textual dictionary | Plain UTF-8 tab-separated rows | Authors, parser, CLI, or model preparation |
|
||||||
|
| Registered model resource | The same Radixor dictionary bytes under GZip, accompanied by index, descriptor, checksum, and license | `StemmerModelRegistry` and `StemmerPatchTrieLoader` |
|
||||||
|
| Persisted compiled trie | GZip-compressed Radixor binary format, commonly `.radixor.gz` | `loadBinaryCompiled(...)` |
|
||||||
|
|
||||||
|
The model file named `stemmer.gz` is not Java serialization and is not a pre-instantiated or persisted trie. It is compressed textual dictionary input parsed when the model is loaded.
|
||||||
|
|
||||||
|
Consequently, compressed size is not a construction-memory estimate. The PoliMorf resource is 12,624,997 bytes compressed and 68,093,680 bytes decompressed, while full parsing, trie construction, reduction, and patch compilation require a dedicated verification JVM with a 6 GiB maximum heap.
|
||||||
|
|
||||||
|
Comment headers in maintained model inputs summarize provenance but do not replace packaged legal
|
||||||
|
material. Each UniMorph-derived artifact includes a language-specific notice describing its
|
||||||
|
official repository, lexical source, upstream attribution, CC BY-SA 3.0 canonical URI, ShareAlike
|
||||||
|
status, Radixor transformations, and Leo Galambos's protectable model-data contributions. The
|
||||||
|
notice does not claim ownership over the underlying data. GZip packaging and descriptor/checksum
|
||||||
|
generation are disclosed transformations; the in-memory trie is a Radixor runtime structure.
|
||||||
|
|
||||||
Each logical line describes one canonical stem and zero or more known word variants that should reduce to that stem. The format is intentionally lightweight, easy to maintain in source control, and directly consumable both by the programmatic loader and by the CLI compiler.
|
Each logical line describes one canonical stem and zero or more known word variants that should reduce to that stem. The format is intentionally lightweight, easy to maintain in source control, and directly consumable both by the programmatic loader and by the CLI compiler.
|
||||||
|
|
||||||
## Core structure
|
## Core structure
|
||||||
@@ -129,7 +150,11 @@ run running runs ran
|
|||||||
|
|
||||||
## Character set, compression, and normalization
|
## Character set, compression, and normalization
|
||||||
|
|
||||||
Dictionary files are read as UTF-8 text. Files loaded through `StemmerPatchTrieLoader.load(Path, ...)` may be either plain UTF-8 text or GZip-compressed UTF-8 text; the loader detects GZip input from the stream header instead of relying on the file extension. Bundled dictionaries are stored as GZip resources and are decoded as UTF-8 after decompression.
|
Dictionary files are read as UTF-8 text. Files loaded through `StemmerPatchTrieLoader.load(Path, ...)` may be either plain UTF-8 text or GZip-compressed UTF-8 text; the loader detects GZip input from the stream header instead of relying on the file extension. Registered model dictionaries are stored as GZip resources and are decoded as UTF-8 after decompression.
|
||||||
|
|
||||||
|
## Turn a dictionary into a model artifact
|
||||||
|
|
||||||
|
An arbitrary classpath copy is not a discoverable model. A model module places immutable input and its license under `models/<model-id>/src/modelInput/`, declares metadata and an independent version, and applies the model convention plugin. The build validates the input, copies identical bytes into a generated namespaced resource, generates `META-INF/radixor/models.index` and a descriptor, records SHA-256, and packages licensing material. See [Stemmer Models](stemmer-models.md#create-or-update-a-model-module) for the complete procedure and [Model Selection and Loading](model-selection-and-loading.md) for runtime use.
|
||||||
|
|
||||||
The parser and trie are not restricted to ASCII. Dictionary items are ordinary Java `String` values, and trie traversal works over Java `char` sequences. This supports Latin-script data with diacritics, Cyrillic data, Hebrew, Persian, Yiddish, and other scripts represented in UTF-8, subject to the normal Java `String` model and the project’s traversal configuration.
|
The parser and trie are not restricted to ASCII. Dictionary items are ordinary Java `String` values, and trie traversal works over Java `char` sequences. This supports Latin-script data with diacritics, Cyrillic data, Hebrew, Persian, Yiddish, and other scripts represented in UTF-8, subject to the normal Java `String` model and the project’s traversal configuration.
|
||||||
|
|
||||||
@@ -235,3 +260,5 @@ To understand how those dictionary lines are transformed into compiled runtime a
|
|||||||
- [CLI compilation](cli-compilation.md)
|
- [CLI compilation](cli-compilation.md)
|
||||||
- [Programmatic usage](programmatic-usage.md)
|
- [Programmatic usage](programmatic-usage.md)
|
||||||
- [Architecture and reduction](architecture-and-reduction.md)
|
- [Architecture and reduction](architecture-and-reduction.md)
|
||||||
|
!!! note "Radixor 4 model artifacts"
|
||||||
|
Language dictionaries are independently versioned runtime model artifacts, not resources embedded in `radixor`. Language-based APIs resolve deterministic defaults through `StemmerModelRegistry`; see [Stemmer Models](stemmer-models.md).
|
||||||
|
|||||||
@@ -1,14 +1,14 @@
|
|||||||
# Fast Track
|
# Fast Track
|
||||||
|
|
||||||
This page is the shortest path from an empty Java project to a working Radixor stemmer.
|
This page is the shortest path from an empty Java project to a working Radixor stemmer.
|
||||||
It deliberately uses a bundled dictionary and the preferred compiled-command runtime API, so the
|
It deliberately uses an external model artifact and the preferred compiled-command runtime API, so the
|
||||||
first result does not require writing a dictionary, running the CLI compiler, or understanding
|
first result does not require writing a dictionary, running the CLI compiler, or understanding
|
||||||
reduction internals.
|
reduction internals.
|
||||||
|
|
||||||
Use this page when the goal is:
|
Use this page when the goal is:
|
||||||
|
|
||||||
- add the dependency,
|
- add the dependency,
|
||||||
- load a bundled language resource,
|
- load a registered language model,
|
||||||
- stem a token,
|
- stem a token,
|
||||||
- know where to go next.
|
- know where to go next.
|
||||||
|
|
||||||
@@ -23,14 +23,14 @@ groupId: org.egothor
|
|||||||
artifactId: radixor
|
artifactId: radixor
|
||||||
```
|
```
|
||||||
|
|
||||||
Use the current published version from Maven Central. The snippets below use `3.0.0`; replace it
|
Radixor 4 is not yet represented by a published release in this working tree. Replace the version placeholder with the reviewed release you deploy.
|
||||||
with the version you deploy if a newer release is available.
|
|
||||||
|
|
||||||
For a Gradle project:
|
For a Gradle project:
|
||||||
|
|
||||||
```kotlin
|
```kotlin
|
||||||
dependencies {
|
dependencies {
|
||||||
implementation("org.egothor:radixor:3.0.0")
|
implementation("org.egothor:radixor:<radixor-version>")
|
||||||
|
runtimeOnly("org.egothor:radixor-model-us-uk-default:1.0.0")
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
@@ -40,17 +40,23 @@ For a Maven project:
|
|||||||
<dependency>
|
<dependency>
|
||||||
<groupId>org.egothor</groupId>
|
<groupId>org.egothor</groupId>
|
||||||
<artifactId>radixor</artifactId>
|
<artifactId>radixor</artifactId>
|
||||||
<version>3.0.0</version>
|
<version>${radixor.version}</version>
|
||||||
|
</dependency>
|
||||||
|
<dependency>
|
||||||
|
<groupId>org.egothor</groupId>
|
||||||
|
<artifactId>radixor-model-us-uk-default</artifactId>
|
||||||
|
<version>1.0.0</version>
|
||||||
|
<scope>runtime</scope>
|
||||||
</dependency>
|
</dependency>
|
||||||
```
|
```
|
||||||
|
|
||||||
Radixor targets modern Java and has a dependency-light runtime core. The project documentation and
|
Radixor targets modern Java and has a dependency-light runtime core. The project documentation and
|
||||||
benchmarks assume a current JDK; Java 21 or newer is the practical baseline for current releases.
|
benchmarks assume a current JDK; Java 21 or newer is the practical baseline for current releases.
|
||||||
|
|
||||||
## 2. Load A Bundled Dictionary
|
## 2. Load An External Model Dictionary
|
||||||
|
|
||||||
The fastest path is to use a bundled dictionary through `StemmerPatchTrieLoader.Language`.
|
The fastest path is to use a registered model through `StemmerPatchTrieLoader.Language`.
|
||||||
This example uses the bundled English resource, `US_UK`.
|
This example uses `US_UK`, whose default ID is `us-uk-default`; the runtime model dependency above must be present.
|
||||||
|
|
||||||
```java
|
```java
|
||||||
import java.io.IOException;
|
import java.io.IOException;
|
||||||
@@ -81,12 +87,11 @@ public final class RadixorFirstStem {
|
|||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
The loaded `FrequencyTrie<CompiledPatchCommand>` is immutable and can be shared across request
|
The loaded `FrequencyTrie<CompiledPatchCommand>` has no mutating API. Load it once during application startup, publish it safely through application-owned lifecycle code, and reuse it for indexing and query processing.
|
||||||
threads. Load it once during application startup and reuse it for indexing and query processing.
|
|
||||||
|
|
||||||
## 3. Choose A Language Resource
|
## 3. Choose a Language Default or Explicit Model
|
||||||
|
|
||||||
Bundled dictionaries are exposed as enum constants. Common examples:
|
Language defaults are exposed as enum constants. Common examples:
|
||||||
|
|
||||||
| Language | Enum constant |
|
| Language | Enum constant |
|
||||||
| --- | --- |
|
| --- | --- |
|
||||||
@@ -102,6 +107,8 @@ Bundled dictionaries are exposed as enum constants. Common examples:
|
|||||||
The full list, writing-direction notes, and benchmark links are in
|
The full list, writing-direction notes, and benchmark links are in
|
||||||
[Built-in Languages](built-in-languages.md).
|
[Built-in Languages](built-in-languages.md).
|
||||||
|
|
||||||
|
Polish has two models. `Language.PL_PL` selects `pl-pl-unimorph`; load the alternative explicitly with `StemmerPatchTrieLoader.loadCompiled("pl-pl-polimorf", true, reductionMode)`, or retain a registry and pass `registry.require("pl-pl-polimorf")` to the descriptor overload. See [Model Selection and Loading](model-selection-and-loading.md). Full PoliMorf construction requires substantially more startup heap than ordinary models; the repository verifies it in a dedicated 6 GiB test JVM.
|
||||||
|
|
||||||
## 4. Use The Same Stemmer On Both Sides
|
## 4. Use The Same Stemmer On Both Sides
|
||||||
|
|
||||||
For search, use the same Radixor configuration during indexing and query processing. A typical
|
For search, use the same Radixor configuration during indexing and query processing. A typical
|
||||||
@@ -118,7 +125,7 @@ limited to lookup and patch application.
|
|||||||
|
|
||||||
## 5. Next Step For Production
|
## 5. Next Step For Production
|
||||||
|
|
||||||
The fast path compiles a bundled dictionary during startup. That is convenient for evaluation and
|
The fast path parses and compiles a registered model dictionary during startup. That is convenient for evaluation and
|
||||||
small services. For larger deployments, compile once, persist a `.radixor.gz` artifact, and load
|
small services. For larger deployments, compile once, persist a `.radixor.gz` artifact, and load
|
||||||
that binary artifact at runtime.
|
that binary artifact at runtime.
|
||||||
|
|
||||||
@@ -126,5 +133,6 @@ Continue with:
|
|||||||
|
|
||||||
- [Integration Deep Dive](integration-deep-dive.md) for production lifecycle guidance.
|
- [Integration Deep Dive](integration-deep-dive.md) for production lifecycle guidance.
|
||||||
- [Loading and Building Stemmers](programmatic-loading-and-building.md) for all loading APIs.
|
- [Loading and Building Stemmers](programmatic-loading-and-building.md) for all loading APIs.
|
||||||
- [Built-in Languages](built-in-languages.md) for bundled resources and dictionary locations.
|
- [Model Selection and Loading](model-selection-and-loading.md) for model dependencies, variants, and failures.
|
||||||
|
- [Built-in Languages](built-in-languages.md) for defaults and optional variants.
|
||||||
- [Benchmarking](benchmarking.md) for speed and quality interpretation.
|
- [Benchmarking](benchmarking.md) for speed and quality interpretation.
|
||||||
|
|||||||
@@ -28,12 +28,26 @@ Radixor delivers:
|
|||||||
|
|
||||||
Radixor is intended for teams that require consistent stemming quality at scale, while retaining the ability to evolve lexical resources after compilation and to handle ambiguous reductions with greater precision than traditional single-stem pipelines allow.
|
Radixor is intended for teams that require consistent stemming quality at scale, while retaining the ability to evolve lexical resources after compilation and to handle ambiguous reductions with greater precision than traditional single-stem pipelines allow.
|
||||||
|
|
||||||
|
## Add the core and model data
|
||||||
|
|
||||||
|
The core `org.egothor:radixor` JAR contains no language dictionary. A minimal application adds one model; broad deployments may use the optional standard pack:
|
||||||
|
|
||||||
|
```groovy
|
||||||
|
dependencies {
|
||||||
|
implementation 'org.egothor:radixor:<radixor-version>'
|
||||||
|
runtimeOnly 'org.egothor:radixor-model-pl-pl-unimorph:1.0.0'
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
`StemmerPatchTrieLoader.loadCompiled(Language.PL_PL, ...)` resolves the default `pl-pl-unimorph`. `pl-pl-polimorf` is a separate optional model selected by stable model ID. Follow [Model Selection and Loading](model-selection-and-loading.md) for runnable examples or choose artifacts from the generated [model catalog](stemmer-model-catalog.md).
|
||||||
|
|
||||||
## Start here
|
## Start here
|
||||||
|
|
||||||
- Read [Fast Track](fast-track.md) when you want the shortest path to a working bundled stemmer.
|
- Read [Fast Track](fast-track.md) when you want the shortest path to a working bundled stemmer.
|
||||||
|
- Use [Model Selection and Loading](model-selection-and-loading.md) for default, explicit, dual-model, and ClassLoader examples.
|
||||||
- Use [Integration Deep Dive](integration-deep-dive.md) when you are wiring Radixor into a real application or search pipeline.
|
- Use [Integration Deep Dive](integration-deep-dive.md) when you are wiring Radixor into a real application or search pipeline.
|
||||||
- Read [Quick Start](quick-start.md) for the broader developer walkthrough after the first result works.
|
- Read [Quick Start](quick-start.md) for the broader developer walkthrough after the first result works.
|
||||||
- Use [Built-in Languages](built-in-languages.md) to find the bundled dictionaries exposed by Radixor.
|
- Use [Built-in Languages](built-in-languages.md) to interpret language defaults and optional model variants.
|
||||||
- Review [Benchmarking](benchmarking.md) and [Benchmark Results](benchmarks/index.md) for reproducible performance and quality methodology.
|
- Review [Benchmarking](benchmarking.md) and [Benchmark Results](benchmarks/index.md) for reproducible performance and quality methodology.
|
||||||
- Open [CI Reports](reports.md) to inspect published build artifacts and quality metrics.
|
- Open [CI Reports](reports.md) to inspect published build artifacts and quality metrics.
|
||||||
- See the historical paper: [*Lemmatizer for Document Information Retrieval Systems in JAVA*](https://www.researchgate.net/publication/221512865_Lemmatizer_for_Document_Information_Retrieval_Systems_in_JAVA).
|
- See the historical paper: [*Lemmatizer for Document Information Retrieval Systems in JAVA*](https://www.researchgate.net/publication/221512865_Lemmatizer_for_Document_Information_Retrieval_Systems_in_JAVA).
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
# Integration Deep Dive
|
# Integration Deep Dive
|
||||||
|
|
||||||
This page explains how to integrate Radixor into a real Java application after the first
|
This page explains how to integrate Radixor into a real Java application after the first
|
||||||
fast-track experiment works. It covers dependencies, bundled dictionaries, runtime lifecycle,
|
fast-track experiment works. It covers dependencies, external model artifacts, runtime lifecycle,
|
||||||
deployment artifacts, and the decisions that matter in search or text-processing systems.
|
deployment artifacts, and the decisions that matter in search or text-processing systems.
|
||||||
|
|
||||||
## Integration Model
|
## Integration Model
|
||||||
@@ -15,9 +15,10 @@ Radixor has two separate phases:
|
|||||||
|
|
||||||
The practical rule is simple: compile rarely, stem often.
|
The practical rule is simple: compile rarely, stem often.
|
||||||
|
|
||||||
For production systems, prefer a startup-owned or dependency-injected singleton
|
For production systems, prefer a startup-owned or dependency-injected
|
||||||
`FrequencyTrie<CompiledPatchCommand>` per language/configuration. The trie is immutable after
|
`FrequencyTrie<CompiledPatchCommand>` per language/configuration. The compiled structure has no
|
||||||
construction and is suitable for concurrent reads.
|
mutating API. The project does not currently publish a formal cross-thread safety guarantee, so
|
||||||
|
applications should use normal safe-publication practices when sharing a loaded trie.
|
||||||
|
|
||||||
## Dependency Coordinates
|
## Dependency Coordinates
|
||||||
|
|
||||||
@@ -31,7 +32,8 @@ Gradle:
|
|||||||
|
|
||||||
```kotlin
|
```kotlin
|
||||||
dependencies {
|
dependencies {
|
||||||
implementation("org.egothor:radixor:3.0.0")
|
implementation("org.egothor:radixor:<radixor-version>")
|
||||||
|
runtimeOnly("org.egothor:radixor-models-standard:<catalog-version>")
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
@@ -41,11 +43,17 @@ Maven:
|
|||||||
<dependency>
|
<dependency>
|
||||||
<groupId>org.egothor</groupId>
|
<groupId>org.egothor</groupId>
|
||||||
<artifactId>radixor</artifactId>
|
<artifactId>radixor</artifactId>
|
||||||
<version>3.0.0</version>
|
<version>${radixor.version}</version>
|
||||||
|
</dependency>
|
||||||
|
<dependency>
|
||||||
|
<groupId>org.egothor</groupId>
|
||||||
|
<artifactId>radixor-models-standard</artifactId>
|
||||||
|
<version>${model.catalog.version}</version>
|
||||||
|
<scope>runtime</scope>
|
||||||
</dependency>
|
</dependency>
|
||||||
```
|
```
|
||||||
|
|
||||||
Replace `3.0.0` with the current release selected for your deployment.
|
Replace the example versions with the independently selected core and catalog releases for your deployment.
|
||||||
|
|
||||||
The core Java module is:
|
The core Java module is:
|
||||||
|
|
||||||
@@ -61,30 +69,15 @@ module example.search {
|
|||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
## Bundled Dictionaries
|
## Runtime Model Artifacts
|
||||||
|
|
||||||
Radixor ships bundled dictionaries inside the library artifact. The public API exposes them through:
|
The core ships no language dictionary. Add one or more `radixor-model-<model-id>` artifacts, or the optional metadata-only standard pack. Each model JAR contains an indexed descriptor and a namespaced GZip dictionary. `StemmerPatchTrieLoader.Language` represents language properties and a stable default model ID; it does not own embedded data.
|
||||||
|
|
||||||
```java
|
The standard option is specifically a POM-only runtime dependency aggregate, not an all-model binary JAR. It resolves one default model JAR per language and excludes optional PoliMorf. The separate POM-only `radixor-models-bom` manages recommended versions without adding runtime artifacts. Repository tests and JMH attach individual model projects directly to non-production configurations, so neither path changes the root publication's dependency graph.
|
||||||
StemmerPatchTrieLoader.Language
|
|
||||||
```
|
|
||||||
|
|
||||||
The physical resources are packaged as compressed UTF-8 dictionaries under resource directories
|
For minimal deployments choose only required model artifacts. For multiple Polish variants add both `pl-pl-unimorph` and `pl-pl-polimorf`, retain UniMorph as the language default, and request PoliMorf explicitly. See [Model Selection and Loading](model-selection-and-loading.md) for complete dependencies and [Built-in Languages](built-in-languages.md) for mappings.
|
||||||
such as:
|
|
||||||
|
|
||||||
```text
|
Use `loadCompiled("pl-pl-polimorf", true, reductionMode)` for direct exact selection, or discover once and call `loadCompiled(descriptor, true, reductionMode)`. Neither form caches the trie. Complete PoliMorf startup is memory-intensive and is verified with a dedicated 6 GiB heap; construct it once during application initialization and retain the immutable result.
|
||||||
us_uk/stemmer.gz
|
|
||||||
de_de/stemmer.gz
|
|
||||||
fr_fr/stemmer.gz
|
|
||||||
pl_pl/stemmer.gz
|
|
||||||
```
|
|
||||||
|
|
||||||
Treat those resource paths as implementation details. Application code should load bundled
|
|
||||||
dictionaries through `StemmerPatchTrieLoader.Language`, because the enum also carries the language
|
|
||||||
metadata needed for correct traversal.
|
|
||||||
|
|
||||||
See [Built-in Languages](built-in-languages.md) for the complete language list, writing-direction
|
|
||||||
notes, and links to per-language benchmark pages.
|
|
||||||
|
|
||||||
## Minimal Service Wrapper
|
## Minimal Service Wrapper
|
||||||
|
|
||||||
@@ -126,7 +119,7 @@ searchable.
|
|||||||
|
|
||||||
For a controlled deployment, compile once and deploy the binary artifact:
|
For a controlled deployment, compile once and deploy the binary artifact:
|
||||||
|
|
||||||
1. choose a bundled or custom dictionary,
|
1. choose a registered model resource or caller-owned custom dictionary,
|
||||||
2. optionally extend it with domain vocabulary,
|
2. optionally extend it with domain vocabulary,
|
||||||
3. compile a contracted trie,
|
3. compile a contracted trie,
|
||||||
4. persist it as `.radixor.gz`,
|
4. persist it as `.radixor.gz`,
|
||||||
@@ -173,9 +166,9 @@ Use Radixor consistently across indexing and querying:
|
|||||||
For multilingual content, do not run every token through every language. Route text by field,
|
For multilingual content, do not run every token through every language. Route text by field,
|
||||||
document metadata, or language detection before stemming.
|
document metadata, or language detection before stemming.
|
||||||
|
|
||||||
## Choosing Bundled Versus Custom Dictionaries
|
## Choosing Registered Versus Custom Dictionaries
|
||||||
|
|
||||||
Start with bundled dictionaries when:
|
Start with registered model artifacts when:
|
||||||
|
|
||||||
- the language is supported,
|
- the language is supported,
|
||||||
- the application needs a strong baseline quickly,
|
- the application needs a strong baseline quickly,
|
||||||
@@ -218,7 +211,7 @@ use [Benchmark Results](benchmarks/index.md) for the detailed reference tree.
|
|||||||
Before production rollout:
|
Before production rollout:
|
||||||
|
|
||||||
- dependency version is pinned,
|
- dependency version is pinned,
|
||||||
- language resource and reduction mode are documented,
|
- language, model ID, model artifact version, checksum, and reduction mode are documented,
|
||||||
- indexing and query pipelines use the same stemming configuration,
|
- indexing and query pipelines use the same stemming configuration,
|
||||||
- custom artifacts are versioned and reproducible,
|
- custom artifacts are versioned and reproducible,
|
||||||
- fallback behavior for unknown tokens is explicit,
|
- fallback behavior for unknown tokens is explicit,
|
||||||
@@ -231,5 +224,6 @@ Before production rollout:
|
|||||||
- [Quick Start](quick-start.md)
|
- [Quick Start](quick-start.md)
|
||||||
- [Built-in Languages](built-in-languages.md)
|
- [Built-in Languages](built-in-languages.md)
|
||||||
- [Programmatic Usage](programmatic-usage.md)
|
- [Programmatic Usage](programmatic-usage.md)
|
||||||
|
- [Model Selection and Loading](model-selection-and-loading.md)
|
||||||
- [CLI Compilation](cli-compilation.md)
|
- [CLI Compilation](cli-compilation.md)
|
||||||
- [Benchmarking](benchmarking.md)
|
- [Benchmarking](benchmarking.md)
|
||||||
|
|||||||
@@ -1,6 +1,149 @@
|
|||||||
# Migration and Backward Compatibility
|
# Migration and Backward Compatibility
|
||||||
|
|
||||||
This page describes the migration from repeated serialized patch-command application to compiled patch commands.
|
## Radixor 3.x to 4.x architecture migration
|
||||||
|
|
||||||
|
Radixor 3.x published algorithm classes and language dictionaries together as `org.egothor:radixor`. Radixor 4 keeps that established coordinate for the algorithmic core but removes every dictionary from the core JAR. Applications must now choose independently versioned model artifacts. This is deliberately source-compatible where practical and deliberately different at runtime.
|
||||||
|
|
||||||
|
### Before and after: dependencies
|
||||||
|
|
||||||
|
| Deployment | 3.x | 4.x |
|
||||||
|
|---|---|---|
|
||||||
|
| Core | `org.egothor:radixor:<3.x-version>` included dictionaries | `org.egothor:radixor:<radixor-version>` contains code only |
|
||||||
|
| Minimal Polish | No separate data dependency | Add `radixor-model-pl-pl-unimorph:1.0.0` |
|
||||||
|
| All defaults | Implicitly embedded | Add optional `radixor-models-standard:<catalog-version>` |
|
||||||
|
| Optional Polish variant | Not independently selectable | Add and explicitly select `radixor-model-pl-pl-polimorf:1.0.0` |
|
||||||
|
|
||||||
|
Gradle, preserving the previous Polish default:
|
||||||
|
|
||||||
|
```groovy
|
||||||
|
dependencies {
|
||||||
|
implementation 'org.egothor:radixor:<radixor-version>'
|
||||||
|
runtimeOnly 'org.egothor:radixor-model-pl-pl-unimorph:1.0.0'
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Gradle, broad default coverage:
|
||||||
|
|
||||||
|
```groovy
|
||||||
|
dependencies {
|
||||||
|
implementation 'org.egothor:radixor:<radixor-version>'
|
||||||
|
runtimeOnly 'org.egothor:radixor-models-standard:<catalog-version>'
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Maven, preserving the Polish default:
|
||||||
|
|
||||||
|
```xml
|
||||||
|
<dependency>
|
||||||
|
<groupId>org.egothor</groupId>
|
||||||
|
<artifactId>radixor</artifactId>
|
||||||
|
<version>${radixor.version}</version>
|
||||||
|
</dependency>
|
||||||
|
<dependency>
|
||||||
|
<groupId>org.egothor</groupId>
|
||||||
|
<artifactId>radixor-model-pl-pl-unimorph</artifactId>
|
||||||
|
<version>1.0.0</version>
|
||||||
|
<scope>runtime</scope>
|
||||||
|
</dependency>
|
||||||
|
```
|
||||||
|
|
||||||
|
### Before and after: API behavior
|
||||||
|
|
||||||
|
Language-oriented calls remain source-compatible:
|
||||||
|
|
||||||
|
```java
|
||||||
|
final FrequencyTrie<CompiledPatchCommand> polish =
|
||||||
|
StemmerPatchTrieLoader.loadCompiled(
|
||||||
|
StemmerPatchTrieLoader.Language.PL_PL,
|
||||||
|
true,
|
||||||
|
ReductionMode.MERGE_SUBTREES_WITH_EQUIVALENT_RANKED_GET_ALL_RESULTS);
|
||||||
|
```
|
||||||
|
|
||||||
|
In 4.x this call creates a registry and resolves `Language.PL_PL.defaultModelId()`, which is `pl-pl-unimorph`. Source compatibility does not imply runtime classpath compatibility: the call fails with `StemmerModelNotFoundException` unless that model is visible.
|
||||||
|
|
||||||
|
Explicit selection enables multiple variants:
|
||||||
|
|
||||||
|
```java
|
||||||
|
final StemmerModelRegistry registry = StemmerModelRegistry.fromContextClassLoader();
|
||||||
|
final StemmerModelDescriptor polimorf = registry.require("pl-pl-polimorf");
|
||||||
|
final FrequencyTrie<CompiledPatchCommand> trie =
|
||||||
|
StemmerPatchTrieLoader.loadCompiled(
|
||||||
|
polimorf,
|
||||||
|
true,
|
||||||
|
ReductionMode.MERGE_SUBTREES_WITH_EQUIVALENT_RANKED_GET_ALL_RESULTS);
|
||||||
|
```
|
||||||
|
|
||||||
|
The existing `load(String, ...)` overload means a filesystem path. The compiled `loadCompiled(String, boolean, ReductionMode)` overload now means a stable model ID; use the `Path` overload for a filesystem dictionary. Descriptor-based compiled loading avoids rediscovery when an application retains a registry. See [Model Selection and Loading](model-selection-and-loading.md) for complete examples.
|
||||||
|
|
||||||
|
### Polish migration scenarios
|
||||||
|
|
||||||
|
1. **Preserve previous default behavior:** add `radixor-model-pl-pl-unimorph` and keep using `Language.PL_PL`.
|
||||||
|
2. **Use PoliMorf:** add `radixor-model-pl-pl-polimorf` and call `registry.require("pl-pl-polimorf")`.
|
||||||
|
3. **Deploy both:** add both runtime artifacts and load each descriptor by ID. They are not merged.
|
||||||
|
4. **Verify selection:** compare `registry.requireDefault(Language.PL_PL).id()` with `pl-pl-unimorph` through normal application control flow or a JUnit assertion, and inspect `registry.findByLanguage(Language.PL_PL)`.
|
||||||
|
5. **Diagnose absence:** read the exact `StemmerModelNotFoundException` message, then inspect the production `runtimeClasspath` rather than changing dependency order.
|
||||||
|
|
||||||
|
UniMorph and PoliMorf are not interchangeable quality datasets. They can differ in vocabulary, provenance, licensing, and stemming outputs.
|
||||||
|
|
||||||
|
Model migration does not erase source obligations. Each migrated UniMorph artifact packages its
|
||||||
|
language-specific notice with upstream attribution, Radixor modifications and contribution
|
||||||
|
statement, ShareAlike terms, and the canonical CC BY-SA 3.0 URI. The original imports did not
|
||||||
|
record exact UniMorph commits, so descriptors use
|
||||||
|
`source.revision=not-recorded-in-legacy-import` and disclose that fact. Future model imports must
|
||||||
|
record an exact upstream revision and source-archive checksum.
|
||||||
|
|
||||||
|
### Compatibility table
|
||||||
|
|
||||||
|
| Dimension | 4.x migration status |
|
||||||
|
|---|---|
|
||||||
|
| Source compatibility | Language-oriented loader signatures remain; external model dependencies are new |
|
||||||
|
| Binary compatibility | Removing resources is a major-version boundary; review all deployed artifacts |
|
||||||
|
| Runtime classpath | At least one selected model JAR is required |
|
||||||
|
| Model format | Descriptor format `radixor-dictionary-tsv-gzip` version `1` is validated by the registry |
|
||||||
|
| Model IDs | Stable runtime identities, independent of artifact discovery order |
|
||||||
|
| Core Maven coordinate | Remains `org.egothor:radixor` |
|
||||||
|
| Release versions | Core, each model, upstream source, format, and catalog versions evolve separately |
|
||||||
|
|
||||||
|
### Upgrade checklist
|
||||||
|
|
||||||
|
- Update the core dependency.
|
||||||
|
- Choose individual model artifacts or the standard pack.
|
||||||
|
- Put resource-only model dependencies on the production runtime classpath.
|
||||||
|
- Verify `Language.defaultModelId()` mappings used by the application.
|
||||||
|
- Inspect shaded, minimized, plugin, or modular packaging for indexes and resources.
|
||||||
|
- Run application-level vocabulary and output regression tests.
|
||||||
|
- Track model artifact versions and checksums separately from the core version.
|
||||||
|
|
||||||
|
### Roll back model choice
|
||||||
|
|
||||||
|
To return from optional PoliMorf to the default UniMorph behavior, add or retain `radixor-model-pl-pl-unimorph`, stop requesting `pl-pl-polimorf`, and load `Language.PL_PL` or explicitly request `pl-pl-unimorph`. Do not change the language constant. Remove the unused PoliMorf runtime dependency after verifying no explicit lookup still needs it.
|
||||||
|
|
||||||
|
Rolling the whole application back to 3.x instead requires restoring the reviewed 3.x core dependency and removing 4.x model assumptions. Do not combine 3.x embedded resources with the 4.x registry architecture.
|
||||||
|
|
||||||
|
Core, model, and catalog releases are independent:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
git tag -a "release@4.0.0" -m "Release Radixor 4.0.0"
|
||||||
|
git tag -a "model/pl-pl-polimorf@1.0.0" -m "Release Polish PoliMorf model 1.0.0"
|
||||||
|
git tag -a "models-catalog@2026.1" -m "Release Radixor model catalog 2026.1"
|
||||||
|
```
|
||||||
|
|
||||||
|
A core tag publishes only the root `org.egothor:radixor` software artifacts, never model JARs. A model tag validates and publishes exactly its matching module, never core, standard, BOM, JMH, or the multilingual quality suite. A catalog tag publishes only BOM and standard aggregate metadata. Local model dry-run:
|
||||||
|
|
||||||
|
The catalog artifacts are POM-only: `radixor-models-standard` carries runtime dependencies on the 20 defaults, while `radixor-models-bom` carries dependency-management constraints for all 21 individual models. Neither publishes an empty binary, sources, or Javadoc JAR. This Maven BOM is distinct from the root CycloneDX SBOM report under `build/reports/sbom/`.
|
||||||
|
|
||||||
|
```bash
|
||||||
|
./tools/parse-model-release-tag.sh "model/pl-pl-polimorf@1.0.0" .
|
||||||
|
./gradlew --no-daemon :models:pl-pl-polimorf:check
|
||||||
|
./gradlew --no-daemon :models:pl-pl-polimorf:validateModelRelease -PmodelReleaseVersion=1.0.0
|
||||||
|
./gradlew --no-daemon :models:pl-pl-polimorf:packageModelReleaseCandidate -PmodelReleaseVersion=1.0.0
|
||||||
|
```
|
||||||
|
|
||||||
|
Model format compatibility is descriptor-level and does not alter migrated bytes. Version 1 is `radixor-dictionary-tsv-gzip`. Model versions come from each module's `model-version.txt` or the matching explicit release property; catalog version comes from `models/catalog-version.txt`; only core uses Git-derived `release@` versioning.
|
||||||
|
|
||||||
|
The model catalog used by the published documentation is generated under `build/mkdocs-source/`. Neither generated Markdown nor rendered MkDocs output belongs in Git.
|
||||||
|
|
||||||
|
The remainder of this page describes the earlier migration from repeated serialized patch-command application to compiled patch commands.
|
||||||
|
|
||||||
## Summary
|
## Summary
|
||||||
|
|
||||||
|
|||||||
277
docs/model-selection-and-loading.md
Normal file
277
docs/model-selection-and-loading.md
Normal file
@@ -0,0 +1,277 @@
|
|||||||
|
# Model Selection and Loading
|
||||||
|
|
||||||
|
Radixor separates executable stemming code from language data. The core artifact supplies dictionary parsing, trie construction, patch commands, lookup, and the model registry. A model artifact supplies one indexed descriptor, one GZip-compressed Radixor dictionary, and its licensing material. The core JAR contains no language dictionary.
|
||||||
|
|
||||||
|
```text
|
||||||
|
Application
|
||||||
|
-> org.egothor:radixor (algorithmic core)
|
||||||
|
-> StemmerModelRegistry
|
||||||
|
-> indexed model descriptor
|
||||||
|
-> namespaced stemmer.gz resource
|
||||||
|
-> checksum verification and dictionary parsing
|
||||||
|
-> FrequencyTrie construction
|
||||||
|
-> patch lookup and stemming
|
||||||
|
```
|
||||||
|
|
||||||
|
## Language and model ID
|
||||||
|
|
||||||
|
These identifiers answer different questions:
|
||||||
|
|
||||||
|
| Concept | Example | Meaning |
|
||||||
|
|---|---|---|
|
||||||
|
| Language | `Language.PL_PL` | Polish as a linguistic identity |
|
||||||
|
| Model ID | `pl-pl-unimorph` | One concrete Polish model configuration |
|
||||||
|
| Model ID | `pl-pl-polimorf` | A different concrete Polish model configuration |
|
||||||
|
| Default model | `PL_PL -> pl-pl-unimorph` | The model selected by the language convenience API |
|
||||||
|
|
||||||
|
One language can have several models. `Language.PL_PL` is neither UniMorph nor PoliMorf. `loadCompiled(Language.PL_PL, ...)` resolves the stable default ID declared by `Language.defaultModelId()`. An explicit lookup requests exactly one ID. Registry ordering never changes either decision.
|
||||||
|
|
||||||
|
Licensing follows the selected artifact. Radixor Java software is BSD-3-Clause; UniMorph-derived
|
||||||
|
model data carries a model-specific CC BY-SA 3.0 notice, while PoliMorf carries its separate
|
||||||
|
BSD-2-Clause license. The UniMorph notice preserves upstream attribution and identifies the
|
||||||
|
Radixor transformations and limited protectable contributions without claiming the underlying data.
|
||||||
|
|
||||||
|
## Choose runtime dependencies
|
||||||
|
|
||||||
|
Radixor 4 is an architectural migration that is not yet represented by a published release in this working tree, so core and catalog versions below use placeholders. Every source-controlled model currently has model version `1.0.0`.
|
||||||
|
|
||||||
|
### Core plus the default Polish model
|
||||||
|
|
||||||
|
```groovy
|
||||||
|
dependencies {
|
||||||
|
implementation 'org.egothor:radixor:<radixor-version>'
|
||||||
|
runtimeOnly 'org.egothor:radixor-model-pl-pl-unimorph:1.0.0'
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
### Core plus optional PoliMorf
|
||||||
|
|
||||||
|
```groovy
|
||||||
|
dependencies {
|
||||||
|
implementation 'org.egothor:radixor:<radixor-version>'
|
||||||
|
runtimeOnly 'org.egothor:radixor-model-pl-pl-polimorf:1.0.0'
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
This dependency makes `pl-pl-polimorf` discoverable; it does not change the default for `PL_PL`.
|
||||||
|
|
||||||
|
### Both Polish models
|
||||||
|
|
||||||
|
```groovy
|
||||||
|
dependencies {
|
||||||
|
implementation 'org.egothor:radixor:<radixor-version>'
|
||||||
|
runtimeOnly 'org.egothor:radixor-model-pl-pl-unimorph:1.0.0'
|
||||||
|
runtimeOnly 'org.egothor:radixor-model-pl-pl-polimorf:1.0.0'
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
### Standard defaults
|
||||||
|
|
||||||
|
```groovy
|
||||||
|
dependencies {
|
||||||
|
implementation 'org.egothor:radixor:<radixor-version>'
|
||||||
|
runtimeOnly 'org.egothor:radixor-models-standard:<catalog-version>'
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
The standard aggregate is POM-only. Its POM supplies exactly one default model per supported language as transitive runtime dependencies and excludes optional PoliMorf. It publishes no empty binary JAR.
|
||||||
|
|
||||||
|
### BOM-managed versions
|
||||||
|
|
||||||
|
```groovy
|
||||||
|
dependencies {
|
||||||
|
implementation 'org.egothor:radixor:<radixor-version>'
|
||||||
|
implementation platform('org.egothor:radixor-models-bom:<catalog-version>')
|
||||||
|
runtimeOnly 'org.egothor:radixor-model-pl-pl-unimorph'
|
||||||
|
runtimeOnly 'org.egothor:radixor-model-pl-pl-polimorf'
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Equivalent Maven dependencies use ordinary runtime scope:
|
||||||
|
|
||||||
|
```xml
|
||||||
|
<dependency>
|
||||||
|
<groupId>org.egothor</groupId>
|
||||||
|
<artifactId>radixor</artifactId>
|
||||||
|
<version>${radixor.version}</version>
|
||||||
|
</dependency>
|
||||||
|
<dependency>
|
||||||
|
<groupId>org.egothor</groupId>
|
||||||
|
<artifactId>radixor-model-pl-pl-unimorph</artifactId>
|
||||||
|
<version>1.0.0</version>
|
||||||
|
<scope>runtime</scope>
|
||||||
|
</dependency>
|
||||||
|
```
|
||||||
|
|
||||||
|
Use `implementation` for the core because application code imports its API. Models normally use `runtimeOnly` because they provide resources rather than Java types. Tests with a deliberately isolated model set use `testRuntimeOnly`. The repository attaches every default model and optional PoliMorf directly to `jmhRuntimeOnly`; test and quality configurations likewise use direct non-production model dependencies. No benchmark aggregate artifact exists, and no model dependency enters the root published POM.
|
||||||
|
|
||||||
|
## Load the documented default
|
||||||
|
|
||||||
|
Dependency prerequisite: core plus `radixor-model-pl-pl-unimorph` (or the standard pack).
|
||||||
|
|
||||||
|
```java
|
||||||
|
import org.egothor.stemmer.CompiledPatchCommand;
|
||||||
|
import org.egothor.stemmer.FrequencyTrie;
|
||||||
|
import org.egothor.stemmer.ReductionMode;
|
||||||
|
import org.egothor.stemmer.ReductionSettings;
|
||||||
|
import org.egothor.stemmer.StemmerPatchTrieLoader;
|
||||||
|
|
||||||
|
final FrequencyTrie<CompiledPatchCommand> polish =
|
||||||
|
StemmerPatchTrieLoader.loadCompiled(
|
||||||
|
StemmerPatchTrieLoader.Language.PL_PL,
|
||||||
|
true,
|
||||||
|
ReductionSettings.withDefaults(
|
||||||
|
ReductionMode.MERGE_SUBTREES_WITH_EQUIVALENT_RANKED_GET_ALL_RESULTS));
|
||||||
|
|
||||||
|
final String word = "koty";
|
||||||
|
final CompiledPatchCommand patch = polish.get(word);
|
||||||
|
final String stem = patch == null ? word : patch.apply(word);
|
||||||
|
```
|
||||||
|
|
||||||
|
The loader creates a registry from the thread context class loader, resolves `PL_PL` to `pl-pl-unimorph`, verifies the compressed resource checksum, decompresses and parses the UTF-8 dictionary, constructs the trie, and compiles its patch commands. It does not load a serialized Java object. If the default artifact is absent, `StemmerModelNotFoundException` names the missing ID and suggested Maven artifact.
|
||||||
|
|
||||||
|
## Load PoliMorf explicitly
|
||||||
|
|
||||||
|
Dependency prerequisite: core plus `radixor-model-pl-pl-polimorf`.
|
||||||
|
|
||||||
|
```java
|
||||||
|
import org.egothor.stemmer.CompiledPatchCommand;
|
||||||
|
import org.egothor.stemmer.FrequencyTrie;
|
||||||
|
import org.egothor.stemmer.ReductionMode;
|
||||||
|
import org.egothor.stemmer.StemmerModelDescriptor;
|
||||||
|
import org.egothor.stemmer.StemmerModelRegistry;
|
||||||
|
import org.egothor.stemmer.StemmerPatchTrieLoader;
|
||||||
|
|
||||||
|
final StemmerModelRegistry registry = StemmerModelRegistry.fromContextClassLoader();
|
||||||
|
final StemmerModelDescriptor descriptor = registry.require("pl-pl-polimorf");
|
||||||
|
|
||||||
|
final FrequencyTrie<CompiledPatchCommand> polish =
|
||||||
|
StemmerPatchTrieLoader.loadCompiled(
|
||||||
|
descriptor,
|
||||||
|
true,
|
||||||
|
ReductionMode.MERGE_SUBTREES_WITH_EQUIVALENT_RANKED_GET_ALL_RESULTS);
|
||||||
|
|
||||||
|
final String word = "koty";
|
||||||
|
final CompiledPatchCommand patch = polish.get(word);
|
||||||
|
final String stem = patch == null ? word : patch.apply(word);
|
||||||
|
```
|
||||||
|
|
||||||
|
The equivalent direct model-ID form is:
|
||||||
|
|
||||||
|
```java
|
||||||
|
final FrequencyTrie<CompiledPatchCommand> polimorf =
|
||||||
|
StemmerPatchTrieLoader.loadCompiled(
|
||||||
|
"pl-pl-polimorf",
|
||||||
|
true,
|
||||||
|
ReductionMode.MERGE_SUBTREES_WITH_EQUIVALENT_RANKED_GET_ALL_RESULTS);
|
||||||
|
```
|
||||||
|
|
||||||
|
`require("pl-pl-polimorf")` and the direct overload are deterministic because registry keys are stable model IDs. Discovery order is sorted, duplicate IDs are rejected, and no “first Polish model on the classpath” fallback exists. Both overloads return compiled patch-command values and perform complete integrity checking, parsing, reduction, and trie construction.
|
||||||
|
|
||||||
|
!!! warning "PoliMorf startup memory"
|
||||||
|
Full construction of the PoliMorf model is memory-intensive. Radixor verifies it in one isolated JVM with a task-specific maximum heap of 6 GiB. Two measured verification runs completed full construction in 23.7 seconds and 23.5 seconds, producing 358,993 canonical trie nodes; the complete Gradle processes peaked at approximately 6.23 GiB resident memory. The compressed model is only 12,624,997 bytes (68,093,680 bytes decompressed), so JAR size is not a proxy for construction-time heap. Applications loading the complete model must provision sufficient startup heap. Radixor does not currently expose a measured retained-heap value, so do not infer one from the process peak.
|
||||||
|
|
||||||
|
## Use both Polish models
|
||||||
|
|
||||||
|
Dependency prerequisite: both Polish model artifacts.
|
||||||
|
|
||||||
|
```java
|
||||||
|
final StemmerModelRegistry registry = StemmerModelRegistry.fromContextClassLoader();
|
||||||
|
|
||||||
|
final StemmerModelDescriptor unimorph = registry.require("pl-pl-unimorph");
|
||||||
|
final StemmerModelDescriptor polimorf = registry.require("pl-pl-polimorf");
|
||||||
|
final StemmerModelDescriptor defaultPolish =
|
||||||
|
registry.requireDefault(StemmerPatchTrieLoader.Language.PL_PL);
|
||||||
|
|
||||||
|
if (!"pl-pl-unimorph".equals(defaultPolish.id())) {
|
||||||
|
throw new IllegalStateException(
|
||||||
|
"Unexpected default Polish model: " + defaultPolish.id());
|
||||||
|
}
|
||||||
|
|
||||||
|
final FrequencyTrie<CompiledPatchCommand> unimorphTrie =
|
||||||
|
StemmerPatchTrieLoader.loadCompiled(unimorph, true, reductionMode);
|
||||||
|
final FrequencyTrie<CompiledPatchCommand> polimorfTrie =
|
||||||
|
StemmerPatchTrieLoader.loadCompiled(polimorf, true, reductionMode);
|
||||||
|
```
|
||||||
|
|
||||||
|
The descriptors and tries coexist independently. The models are not merged, and adding PoliMorf does not alter the language default. An application that compares, votes across, or merges model outputs must implement that higher-level policy explicitly.
|
||||||
|
|
||||||
|
## Discover available models
|
||||||
|
|
||||||
|
```java
|
||||||
|
final StemmerModelRegistry registry = StemmerModelRegistry.fromContextClassLoader();
|
||||||
|
|
||||||
|
for (final StemmerModelDescriptor model : registry.models()) {
|
||||||
|
System.out.printf("%s %s %s %s/%d descriptor=%s%n",
|
||||||
|
model.id(), model.language(), model.version(),
|
||||||
|
model.format(), model.formatVersion(), model.source());
|
||||||
|
}
|
||||||
|
|
||||||
|
final java.util.List<StemmerModelDescriptor> polishModels =
|
||||||
|
registry.findByLanguage(StemmerPatchTrieLoader.Language.PL_PL);
|
||||||
|
```
|
||||||
|
|
||||||
|
Both lists use stable model-ID order. The public descriptor API exposes ID, model artifact version, language, display name, runtime resource, default flag, format, format version, checksum, and descriptor source URL. Packaged provenance properties such as `source.name` and `source.version` are not currently exposed as typed descriptor accessors; consult the generated [model catalog](stemmer-model-catalog.md) for them.
|
||||||
|
|
||||||
|
## Use an explicit ClassLoader
|
||||||
|
|
||||||
|
```java
|
||||||
|
final ClassLoader pluginLoader = plugin.getClass().getClassLoader();
|
||||||
|
final StemmerModelRegistry pluginModels =
|
||||||
|
StemmerModelRegistry.fromClassLoader(pluginLoader);
|
||||||
|
final StemmerModelDescriptor model = pluginModels.require("pl-pl-polimorf");
|
||||||
|
```
|
||||||
|
|
||||||
|
`fromContextClassLoader()` uses the current thread context loader, falling back to Radixor's defining loader when the context loader is `null`. `fromClassLoader(loader)` searches only what that loader can expose through `getResources(...)` and ordinary resource lookup. Plugin containers, application servers, and isolated tests can therefore observe different model sets. Pass a non-null loader and retain the registry associated with that deployment scope.
|
||||||
|
|
||||||
|
## Error handling
|
||||||
|
|
||||||
|
```java
|
||||||
|
try {
|
||||||
|
final StemmerModelRegistry registry = StemmerModelRegistry.fromContextClassLoader();
|
||||||
|
final StemmerModelDescriptor model = registry.require("pl-pl-polimorf");
|
||||||
|
// Load and cache the trie during application startup.
|
||||||
|
} catch (final StemmerModelNotFoundException exception) {
|
||||||
|
// Missing runtime dependency or model hidden from this ClassLoader.
|
||||||
|
throw exception;
|
||||||
|
} catch (final DuplicateStemmerModelException exception) {
|
||||||
|
// Conflicting artifacts or a fat JAR duplicated one stable ID.
|
||||||
|
throw exception;
|
||||||
|
} catch (final UnsupportedStemmerModelFormatException exception) {
|
||||||
|
// The model format or format version is not supported by this core.
|
||||||
|
throw exception;
|
||||||
|
} catch (final StemmerModelIntegrityException exception) {
|
||||||
|
// Malformed descriptor/index, missing resource, wrong language, or checksum failure.
|
||||||
|
throw exception;
|
||||||
|
} catch (final java.io.IOException exception) {
|
||||||
|
// Classpath enumeration or resource I/O failed.
|
||||||
|
throw new java.io.UncheckedIOException(exception);
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Malformed metadata does not have a separate public exception: it is reported as `StemmerModelIntegrityException`. Missing explicit and default models both use `StemmerModelNotFoundException`; the default diagnostic additionally names the language and expected default ID. Never swallow these failures or choose an arbitrary model.
|
||||||
|
|
||||||
|
## Lifecycle and concurrency
|
||||||
|
|
||||||
|
`StemmerModelRegistry` copies discovered descriptors into an unmodifiable map, returns immutable list copies, and has no mutating API. `StemmerModelDescriptor` is final with final fields. These objects are safe to retain after discovery. Registry discovery is not globally cached: every call enumerates indexes and parses descriptors again. Model loading is also not cached: every call reads, hashes, decompresses, parses, and builds a new trie.
|
||||||
|
|
||||||
|
Compiled tries are immutable and thread-safe for concurrent reads. Load a registry and the required tries once during application startup, publish them safely, and reuse them. The loader does not cache model tries; do not repeatedly discover and compile models per token. When comparing both Polish models, account for the memory of two independent tries and avoid constructing them concurrently unless the deployment is sized for that peak.
|
||||||
|
|
||||||
|
## Troubleshooting
|
||||||
|
|
||||||
|
| Symptom | Meaning | Action |
|
||||||
|
|---|---|---|
|
||||||
|
| `No default model '...' is available` | The default artifact is absent from the selected loader | Add the named model as a runtime dependency and inspect `runtimeClasspath` |
|
||||||
|
| `No model 'pl-pl-polimorf' is available` | Explicit optional model is absent or invisible | Add `radixor-model-pl-pl-polimorf` to runtime, not only tests |
|
||||||
|
| Duplicate model ID | Two resources declare one stable ID | Remove the duplicate artifact or fix fat-JAR resource duplication; do not reorder the classpath |
|
||||||
|
| Checksum mismatch | Descriptor and compressed bytes differ | Replace the corrupted or incorrectly repackaged artifact |
|
||||||
|
| Unsupported format | Core supports neither the format name nor version | Use a compatible core/model pair; do not bypass validation |
|
||||||
|
| Works in tests, fails in production | The model is probably `testRuntimeOnly` | Inspect `./gradlew dependencies --configuration runtimeClasspath` |
|
||||||
|
| Visible with one loader only | Class loaders expose different resources | Call `fromClassLoader(...)` with the loader that owns the model JAR |
|
||||||
|
| PoliMorf is installed but language loading uses UniMorph | Expected default behavior | Select `pl-pl-polimorf` explicitly |
|
||||||
|
| Dependency minimization removed the model | Resource-only dependency was treated as unused | Preserve the model JAR, index, descriptor, license, and dictionary |
|
||||||
|
| Shaded JAR fails or reports duplicates | Indexes/resources were dropped or duplicated | Inspect with `jar tf app.jar | grep -E 'models.index|stemmer.gz'`; configure deterministic resource merging without duplicating IDs |
|
||||||
|
|
||||||
|
Useful Gradle diagnostics include `./gradlew dependencyInsight --dependency radixor-model --configuration runtimeClasspath` and `./gradlew dependencies --configuration testRuntimeClasspath`. Classpath order is not a remediation mechanism.
|
||||||
|
|
||||||
|
Continue with [Programmatic Usage](programmatic-usage.md), [Stemmer Models](stemmer-models.md), [Built-in Languages](built-in-languages.md), the generated [model catalog](stemmer-model-catalog.md), and [Architecture](architecture.md).
|
||||||
@@ -2,9 +2,9 @@
|
|||||||
|
|
||||||
This document explains how to acquire a compiled Radixor stemmer in Java.
|
This document explains how to acquire a compiled Radixor stemmer in Java.
|
||||||
|
|
||||||
## Load a bundled language dictionary
|
## Load a registered default model
|
||||||
|
|
||||||
Bundled language resources are simple to use and compile directly into a `FrequencyTrie<CompiledPatchCommand>` during loading.
|
Language-oriented entry points resolve a registered default model and compile its GZip textual dictionary into a `FrequencyTrie<CompiledPatchCommand>`. The corresponding model JAR must be on the runtime classpath; the core contains no dictionary.
|
||||||
|
|
||||||
```java
|
```java
|
||||||
import java.io.IOException;
|
import java.io.IOException;
|
||||||
@@ -14,9 +14,9 @@ import org.egothor.stemmer.FrequencyTrie;
|
|||||||
import org.egothor.stemmer.ReductionMode;
|
import org.egothor.stemmer.ReductionMode;
|
||||||
import org.egothor.stemmer.StemmerPatchTrieLoader;
|
import org.egothor.stemmer.StemmerPatchTrieLoader;
|
||||||
|
|
||||||
public final class BundledLanguageExample {
|
public final class RegisteredLanguageModelExample {
|
||||||
|
|
||||||
private BundledLanguageExample() {
|
private RegisteredLanguageModelExample() {
|
||||||
throw new AssertionError("No instances.");
|
throw new AssertionError("No instances.");
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -31,14 +31,16 @@ public final class BundledLanguageExample {
|
|||||||
|
|
||||||
The `storeOriginal` flag controls whether the canonical stem is inserted as a no-op patch entry for the stem itself.
|
The `storeOriginal` flag controls whether the canonical stem is inserted as a no-op patch entry for the stem itself.
|
||||||
|
|
||||||
Bundled `loadCompiled(...)` entry points build the runtime trie with the same contracted
|
Language-oriented `loadCompiled(...)` entry points build the runtime trie with the same contracted
|
||||||
representation used by the published benchmarks. During compilation, uniform preferred-command
|
representation used by the published benchmarks. During compilation, uniform preferred-command
|
||||||
subtrees are collapsed into accepting leaves, so lookup can stop before consuming the entire input
|
subtrees are collapsed into accepting leaves, so lookup can stop before consuming the entire input
|
||||||
when the remaining characters cannot change the selected patch command.
|
when the remaining characters cannot change the selected patch command.
|
||||||
|
|
||||||
## Load a textual dictionary
|
## Load a textual dictionary
|
||||||
|
|
||||||
Loading from a dictionary file follows the same preparation model as bundled resources, but the source comes from your own file or path. The input may be plain UTF-8 text or GZip-compressed UTF-8 text; the loader detects GZip data from the stream header. The textual format is tab-separated values, meaning that columns are separated by the tab character. Each non-empty logical line starts with the stem column and may contain zero or more variant columns. Input case normalization is controlled by `CaseProcessingMode` (default: `LOWERCASE_WITH_LOCALE_ROOT`), trailing remarks introduced by `#` or `//` are ignored, and dictionary items containing embedded whitespace are currently ignored with warning-level diagnostics.
|
Loading from a dictionary file follows the same trie preparation model as registered model resources, but the source comes from your own file or path and bypasses registry metadata. The input may be plain UTF-8 text or GZip-compressed UTF-8 text; the loader detects GZip data from the stream header. The textual format is tab-separated values, meaning that columns are separated by the tab character. Each non-empty logical line starts with the stem column and may contain zero or more variant columns. Input case normalization is controlled by `CaseProcessingMode` (default: `LOWERCASE_WITH_LOCALE_ROOT`), trailing remarks introduced by `#` or `//` are ignored, and dictionary items containing embedded whitespace are currently ignored with warning-level diagnostics.
|
||||||
|
|
||||||
|
For explicit model IDs, multiple variants, and ClassLoader control, see [Model Selection and Loading](model-selection-and-loading.md).
|
||||||
|
|
||||||
```java
|
```java
|
||||||
import java.io.IOException;
|
import java.io.IOException;
|
||||||
|
|||||||
@@ -1,80 +1,133 @@
|
|||||||
# Programmatic Usage
|
# Programmatic Usage
|
||||||
|
|
||||||
This document provides the programmatic entry point to **Radixor**.
|
Radixor code and model data are separate runtime components. Every example on this page requires `org.egothor:radixor:<radixor-version>` as an `implementation` dependency and at least one model JAR as a runtime dependency. The core JAR contains no `stemmer.gz`.
|
||||||
|
|
||||||
Radixor follows a clear lifecycle:
|
For complete dependency patterns, lifecycle guidance, and troubleshooting, use [Model Selection and Loading](model-selection-and-loading.md). The generated [model catalog](stemmer-model-catalog.md) records the current artifacts, versions, checksums, and provenance.
|
||||||
|
|
||||||
1. acquire a compiled stemmer,
|
## 1. Minimal use: the Polish default
|
||||||
2. query it for patch commands,
|
|
||||||
3. apply those commands to produce stems,
|
|
||||||
4. reopen and extend the compiled structure when needed.
|
|
||||||
|
|
||||||
## Conceptual model
|
Dependency prerequisite:
|
||||||
|
|
||||||
Radixor is dictionary-driven, but runtime stemming does not operate by scanning raw dictionary files. A source dictionary is parsed as a sequence of canonical stems and their known variants. Each variant is converted into a compact patch command that transforms the variant into the stem, while the stem itself may optionally be stored as a canonical no-op patch. The mutable trie is then reduced into a compiled read-only structure that stores ordered values and their counts at addressed nodes.
|
```groovy
|
||||||
|
implementation 'org.egothor:radixor:<radixor-version>'
|
||||||
Two consequences matter for developers:
|
runtimeOnly 'org.egothor:radixor-model-pl-pl-unimorph:1.0.0'
|
||||||
|
|
||||||
- the quality and coverage of stemming behavior depend on dictionary richness,
|
|
||||||
- runtime usage is based on compiled patch-command lookup rather than on direct dictionary traversal.
|
|
||||||
|
|
||||||
This is why Radixor can generalize beyond explicitly listed forms and why compiled artifacts are well suited for deployment.
|
|
||||||
|
|
||||||
## Documentation map
|
|
||||||
|
|
||||||
The programmatic API is easier to understand when split by developer task:
|
|
||||||
|
|
||||||
- [Fast Track](fast-track.md) gives the shortest dependency-to-first-stem path for a new Java project.
|
|
||||||
- [Integration Deep Dive](integration-deep-dive.md) explains production integration, deployment artifacts, search-pipeline usage, and operational decisions.
|
|
||||||
- [Loading and Building Stemmers](programmatic-loading-and-building.md) explains how to acquire a compiled stemmer from bundled resources, textual dictionaries, binary artifacts, or direct builder usage.
|
|
||||||
- [Lookup Edge Optimization](lookup-edge-optimization.md) explains dense child lookup tuning and the speed/memory trade-off when materializing compiled tries.
|
|
||||||
- [Querying and Ambiguity Handling](programmatic-querying-and-ambiguity.md) explains `get(...)`, `getAll(...)`, `getEntries(...)`, patch application, and the practical meaning of reduction modes.
|
|
||||||
- [Extending and Persisting Compiled Tries](programmatic-extending-and-persistence.md) explains how to reopen compiled tries, add new lexical data, rebuild them, and store them as binary artifacts.
|
|
||||||
|
|
||||||
## Core types
|
|
||||||
|
|
||||||
The main types involved in programmatic usage are:
|
|
||||||
|
|
||||||
- `FrequencyTrie.Builder<V>` for mutable construction and extension,
|
|
||||||
- `FrequencyTrie<V>` for the compiled read-only trie,
|
|
||||||
- `PatchCommandEncoder` for creating serialized patch commands,
|
|
||||||
- `CompiledPatchCommand` for repeated runtime patch application,
|
|
||||||
- `StemmerPatchTrieLoader` for loading bundled or textual dictionaries,
|
|
||||||
- `StemmerPatchTrieBinaryIO` for reading and writing compressed binary artifacts,
|
|
||||||
- `FrequencyTrieBuilders` for reconstructing a mutable builder from a compiled trie,
|
|
||||||
- `ReductionMode` and `ReductionSettings` for controlling compilation semantics.
|
|
||||||
|
|
||||||
## Java module system (JPMS)
|
|
||||||
|
|
||||||
The core artifact is published as an explicit JPMS module:
|
|
||||||
|
|
||||||
```java
|
|
||||||
module org.egothor.radixor;
|
|
||||||
```
|
```
|
||||||
|
|
||||||
A named consuming module uses:
|
```java
|
||||||
|
import org.egothor.stemmer.CompiledPatchCommand;
|
||||||
|
import org.egothor.stemmer.FrequencyTrie;
|
||||||
|
import org.egothor.stemmer.ReductionMode;
|
||||||
|
import org.egothor.stemmer.StemmerPatchTrieLoader;
|
||||||
|
|
||||||
|
final FrequencyTrie<CompiledPatchCommand> trie =
|
||||||
|
StemmerPatchTrieLoader.loadCompiled(
|
||||||
|
StemmerPatchTrieLoader.Language.PL_PL,
|
||||||
|
true,
|
||||||
|
ReductionMode.MERGE_SUBTREES_WITH_EQUIVALENT_RANKED_GET_ALL_RESULTS);
|
||||||
|
|
||||||
|
final String word = "koty";
|
||||||
|
final CompiledPatchCommand patch = trie.get(word);
|
||||||
|
final String stem = patch == null ? word : patch.apply(word);
|
||||||
|
```
|
||||||
|
|
||||||
|
`Language.PL_PL` resolves to `pl-pl-unimorph`. The loader creates the registry internally through the thread context class loader.
|
||||||
|
|
||||||
|
## 2. Explicit model selection
|
||||||
|
|
||||||
|
Dependency prerequisite: replace or supplement the default dependency with `runtimeOnly 'org.egothor:radixor-model-pl-pl-polimorf:1.0.0'`.
|
||||||
|
|
||||||
```java
|
```java
|
||||||
module example.consumer {
|
final StemmerModelRegistry registry = StemmerModelRegistry.fromContextClassLoader();
|
||||||
requires org.egothor.radixor;
|
final StemmerModelDescriptor polimorf = registry.require("pl-pl-polimorf");
|
||||||
|
final FrequencyTrie<CompiledPatchCommand> trie =
|
||||||
|
StemmerPatchTrieLoader.loadCompiled(
|
||||||
|
polimorf,
|
||||||
|
true,
|
||||||
|
ReductionMode.MERGE_SUBTREES_WITH_EQUIVALENT_RANKED_GET_ALL_RESULTS);
|
||||||
|
```
|
||||||
|
|
||||||
|
The stable model-ID overload performs the same exact selection without a separately retained registry:
|
||||||
|
|
||||||
|
```java
|
||||||
|
final FrequencyTrie<CompiledPatchCommand> trie =
|
||||||
|
StemmerPatchTrieLoader.loadCompiled(
|
||||||
|
"pl-pl-polimorf",
|
||||||
|
true,
|
||||||
|
ReductionMode.MERGE_SUBTREES_WITH_EQUIVALENT_RANKED_GET_ALL_RESULTS);
|
||||||
|
```
|
||||||
|
|
||||||
|
## 3. Multiple variants for one language
|
||||||
|
|
||||||
|
Dependency prerequisite: both `radixor-model-pl-pl-unimorph:1.0.0` and `radixor-model-pl-pl-polimorf:1.0.0` at runtime.
|
||||||
|
|
||||||
|
```java
|
||||||
|
final StemmerModelRegistry registry = StemmerModelRegistry.fromContextClassLoader();
|
||||||
|
final StemmerModelDescriptor unimorph = registry.require("pl-pl-unimorph");
|
||||||
|
final StemmerModelDescriptor polimorf = registry.require("pl-pl-polimorf");
|
||||||
|
|
||||||
|
final FrequencyTrie<CompiledPatchCommand> unimorphTrie =
|
||||||
|
StemmerPatchTrieLoader.loadCompiled(unimorph, true, reductionMode);
|
||||||
|
final FrequencyTrie<CompiledPatchCommand> polimorfTrie =
|
||||||
|
StemmerPatchTrieLoader.loadCompiled(polimorf, true, reductionMode);
|
||||||
|
|
||||||
|
final StemmerModelDescriptor defaultPolish =
|
||||||
|
registry.requireDefault(StemmerPatchTrieLoader.Language.PL_PL);
|
||||||
|
if (!"pl-pl-unimorph".equals(defaultPolish.id())) {
|
||||||
|
throw new IllegalStateException(
|
||||||
|
"Unexpected default Polish model: " + defaultPolish.id());
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
The core module is standalone and can be consumed directly as a normal Java module.
|
The tries remain independent. Radixor does not merge models or infer an alternative default from classpath order.
|
||||||
|
|
||||||
## Recommended reading order
|
## 4. Discovery
|
||||||
|
|
||||||
For most developers, the best order is:
|
Dependency prerequisite: whichever model artifacts the application intends to discover.
|
||||||
|
|
||||||
1. [Fast Track](fast-track.md)
|
```java
|
||||||
2. [Integration Deep Dive](integration-deep-dive.md)
|
final StemmerModelRegistry registry = StemmerModelRegistry.fromContextClassLoader();
|
||||||
3. [Loading and Building Stemmers](programmatic-loading-and-building.md)
|
|
||||||
4. [Querying and Ambiguity Handling](programmatic-querying-and-ambiguity.md)
|
|
||||||
5. [Extending and Persisting Compiled Tries](programmatic-extending-and-persistence.md)
|
|
||||||
|
|
||||||
## Next steps
|
for (final StemmerModelDescriptor descriptor : registry.models()) {
|
||||||
|
System.out.printf("%s %s %s %s/%d%n",
|
||||||
|
descriptor.id(), descriptor.language(), descriptor.version(),
|
||||||
|
descriptor.format(), descriptor.formatVersion());
|
||||||
|
}
|
||||||
|
|
||||||
- [Quick Start](quick-start.md)
|
final java.util.List<StemmerModelDescriptor> polish =
|
||||||
- [CLI compilation](cli-compilation.md)
|
registry.findByLanguage(StemmerPatchTrieLoader.Language.PL_PL);
|
||||||
- [Dictionary format](dictionary-format.md)
|
```
|
||||||
- [Architecture and reduction](architecture-and-reduction.md)
|
|
||||||
|
Results use deterministic model-ID order. See [Built-in Languages](built-in-languages.md) for default interpretation and the generated [catalog](stemmer-model-catalog.md) for provenance.
|
||||||
|
|
||||||
|
## 5. Advanced ClassLoader selection
|
||||||
|
|
||||||
|
Dependency prerequisite: the model JAR must be visible to the selected loader.
|
||||||
|
|
||||||
|
```java
|
||||||
|
final ClassLoader applicationLoader = application.getClass().getClassLoader();
|
||||||
|
final StemmerModelRegistry isolatedRegistry =
|
||||||
|
StemmerModelRegistry.fromClassLoader(applicationLoader);
|
||||||
|
```
|
||||||
|
|
||||||
|
This form is useful for plugin containers, isolated application servers, and tests. It can discover a different set from the thread context loader. See [ClassLoader troubleshooting](model-selection-and-loading.md#troubleshooting).
|
||||||
|
|
||||||
|
## 6. Error handling
|
||||||
|
|
||||||
|
Dependency prerequisite: none beyond core; this example demonstrates an absent optional model.
|
||||||
|
|
||||||
|
```java
|
||||||
|
try {
|
||||||
|
StemmerModelRegistry.fromContextClassLoader().require("pl-pl-polimorf");
|
||||||
|
} catch (final StemmerModelNotFoundException exception) {
|
||||||
|
System.err.println(exception.getMessage());
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Missing models never produce an empty trie or arbitrary fallback. Duplicate IDs, unsupported formats, malformed descriptors, missing resources, and checksum mismatches are also fatal. The full exception mapping and remediation table are in [Model Selection and Loading](model-selection-and-loading.md#error-handling).
|
||||||
|
|
||||||
|
## Continue into the trie API
|
||||||
|
|
||||||
|
- [Loading and Building Stemmers](programmatic-loading-and-building.md)
|
||||||
|
- [Querying and Ambiguity Handling](programmatic-querying-and-ambiguity.md)
|
||||||
|
- [Extending and Persisting Compiled Tries](programmatic-extending-and-persistence.md)
|
||||||
|
- [Architecture](architecture.md)
|
||||||
|
|||||||
@@ -4,10 +4,23 @@ This guide introduces the fastest practical path to using **Radixor**.
|
|||||||
|
|
||||||
If you are new to Radixor and want the shortest possible path to a first working stem, start with
|
If you are new to Radixor and want the shortest possible path to a first working stem, start with
|
||||||
[Fast Track](fast-track.md). This Quick Start is a broader developer walkthrough: it introduces the
|
[Fast Track](fast-track.md). This Quick Start is a broader developer walkthrough: it introduces the
|
||||||
main loading options, query methods, artifact workflow, and metadata model.
|
main loading options, query methods, artifact workflow, and metadata model. For model-ID selection and failures, use [Model Selection and Loading](model-selection-and-loading.md).
|
||||||
|
|
||||||
Radixor separates preparation from runtime usage. Source dictionaries are used to derive patch commands and reduce them into a compact read-only trie. Runtime stemming then operates on that compiled structure rather than on the original dictionary text. A richer dictionary usually improves the quality and coverage of inferred transformations, including transformations that are applicable to words not explicitly present in the source material. The reduction step also removes a large amount of redundant lexical information, which is why very large dictionaries can still produce compact runtime artifacts. These artifacts can be persisted and loaded directly when needed.
|
Radixor separates preparation from runtime usage. Source dictionaries are used to derive patch commands and reduce them into a compact read-only trie. Runtime stemming then operates on that compiled structure rather than on the original dictionary text. A richer dictionary usually improves the quality and coverage of inferred transformations, including transformations that are applicable to words not explicitly present in the source material. The reduction step also removes a large amount of redundant lexical information, which is why very large dictionaries can still produce compact runtime artifacts. These artifacts can be persisted and loaded directly when needed.
|
||||||
|
|
||||||
|
From version 4 onward, the core and models are explicit dependencies:
|
||||||
|
|
||||||
|
```groovy
|
||||||
|
dependencies {
|
||||||
|
implementation 'org.egothor:radixor:<radixor-version>'
|
||||||
|
runtimeOnly 'org.egothor:radixor-models-standard:<catalog-version>'
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
The core JAR contains no dictionary. Replace the standard pack with `runtimeOnly 'org.egothor:radixor-model-us-uk-default:1.0.0'` for the minimal English example below. For Polish, `Language.PL_PL` resolves `pl-pl-unimorph`; installing optional `pl-pl-polimorf` does not select it automatically.
|
||||||
|
|
||||||
|
Explicit PoliMorf loading uses `StemmerPatchTrieLoader.loadCompiled("pl-pl-polimorf", true, reductionMode)`. Its complete dictionary is supported, but construction is exceptional enough that repository verification runs it separately with a 6 GiB maximum heap. See [Model Selection and Loading](model-selection-and-loading.md#load-polimorf-explicitly) for the complete dependency and Java example.
|
||||||
|
|
||||||
A practical workflow usually consists of two independent phases:
|
A practical workflow usually consists of two independent phases:
|
||||||
|
|
||||||
1. obtain a compiled stemmer,
|
1. obtain a compiled stemmer,
|
||||||
@@ -17,9 +30,9 @@ A practical workflow usually consists of two independent phases:
|
|||||||
|
|
||||||
A compiled stemmer can be obtained in three common ways.
|
A compiled stemmer can be obtained in three common ways.
|
||||||
|
|
||||||
### Use a bundled language dictionary
|
### Use an external language model
|
||||||
|
|
||||||
Radixor ships with bundled dictionaries for a set of supported languages. These resources are line-oriented dictionaries stored with the library and compiled into a `FrequencyTrie<CompiledPatchCommand>` when loaded through the runtime API. The loader can also store the canonical stem itself as a no-op patch command. Compiled trie artifacts now persist self-describing metadata, including the traversal direction and compilation reduction settings used to build the artifact.
|
Language dictionaries are independently versioned model JARs discovered by `StemmerModelRegistry`. The root `org.egothor:radixor` JAR contains no dictionary bytes. The loader compiles a selected model into a `FrequencyTrie<CompiledPatchCommand>`; compiled trie artifacts retain self-describing traversal and reduction metadata.
|
||||||
|
|
||||||
```java
|
```java
|
||||||
import java.io.IOException;
|
import java.io.IOException;
|
||||||
@@ -29,9 +42,9 @@ import org.egothor.stemmer.FrequencyTrie;
|
|||||||
import org.egothor.stemmer.ReductionMode;
|
import org.egothor.stemmer.ReductionMode;
|
||||||
import org.egothor.stemmer.StemmerPatchTrieLoader;
|
import org.egothor.stemmer.StemmerPatchTrieLoader;
|
||||||
|
|
||||||
public final class BundledStemmerExample {
|
public final class RegisteredModelExample {
|
||||||
|
|
||||||
private BundledStemmerExample() {
|
private RegisteredModelExample() {
|
||||||
throw new AssertionError("No instances.");
|
throw new AssertionError("No instances.");
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -251,3 +264,5 @@ Dictionary compilation is usually a one-time preparation step and is generally f
|
|||||||
Every compiled trie artifact stores a `TrieMetadata` descriptor together with the immutable trie payload. That metadata currently records the binary format version, the `WordTraversalDirection`, the `ReductionSettings` used during compilation, the declared `DiacriticProcessingMode`, and the selected `CaseProcessingMode`. Traversal, case processing, and diacritic processing are applied during runtime lookup (`get`, `getAll`), and case/diacritic processing are also applied during dictionary insertion when a trie is built.
|
Every compiled trie artifact stores a `TrieMetadata` descriptor together with the immutable trie payload. That metadata currently records the binary format version, the `WordTraversalDirection`, the `ReductionSettings` used during compilation, the declared `DiacriticProcessingMode`, and the selected `CaseProcessingMode`. Traversal, case processing, and diacritic processing are applied during runtime lookup (`get`, `getAll`), and case/diacritic processing are also applied during dictionary insertion when a trie is built.
|
||||||
|
|
||||||
`DiacriticProcessingMode.AS_IS` keeps dictionary keys and lookup keys unchanged. `DiacriticProcessingMode.REMOVE` strips diacritics from dictionary keys and lookup keys (for Czech diacritics and broad European Latin-script variants). `DiacriticProcessingMode.AS_IS_AND_STRIPPED_FALLBACK` is currently not supported and raises an `UnsupportedOperationException`.
|
`DiacriticProcessingMode.AS_IS` keeps dictionary keys and lookup keys unchanged. `DiacriticProcessingMode.REMOVE` strips diacritics from dictionary keys and lookup keys (for Czech diacritics and broad European Latin-script variants). `DiacriticProcessingMode.AS_IS_AND_STRIPPED_FALLBACK` is currently not supported and raises an `UnsupportedOperationException`.
|
||||||
|
!!! note "Radixor 4 model artifacts"
|
||||||
|
Language dictionaries are independently versioned runtime model artifacts, not resources embedded in `radixor`. Language-based APIs resolve deterministic defaults through `StemmerModelRegistry`; see [Stemmer Models](stemmer-models.md).
|
||||||
|
|||||||
@@ -2,6 +2,8 @@
|
|||||||
|
|
||||||
Radixor publishes durable build outputs to GitHub Pages from qualifying runs of `.github/workflows/pages.yml`.
|
Radixor publishes durable build outputs to GitHub Pages from qualifying runs of `.github/workflows/pages.yml`.
|
||||||
|
|
||||||
|
The workflow builds maintained MkDocs documentation and the generated model catalog from the staged source tree under `build/mkdocs-source/`. It then merges the rendered site into the separate `gh-pages` publication worktree while preserving `builds/`. The main branch stores neither generated Markdown nor rendered site output. The publication retains the ten newest numbered report sets and maintains `builds/latest/` as a stable alias.
|
||||||
|
|
||||||
This page is the central entry point for published project artifacts, including build summaries, API documentation, test and quality reports, benchmark outputs, and software composition materials. It is intended both for routine project inspection and for linking stable report surfaces from external references such as the README, release notes, or development workflows.
|
This page is the central entry point for published project artifacts, including build summaries, API documentation, test and quality reports, benchmark outputs, and software composition materials. It is intended both for routine project inspection and for linking stable report surfaces from external references such as the README, release notes, or development workflows.
|
||||||
|
|
||||||
## Stable entry points
|
## Stable entry points
|
||||||
|
|||||||
192
docs/stemmer-models.md
Normal file
192
docs/stemmer-models.md
Normal file
@@ -0,0 +1,192 @@
|
|||||||
|
# Stemmer Models
|
||||||
|
|
||||||
|
This page defines the model artifact and its maintenance lifecycle. Application developers should begin with [Model Selection and Loading](model-selection-and-loading.md); the generated [model catalog](stemmer-model-catalog.md) is the detailed inventory.
|
||||||
|
|
||||||
|
## Terminology
|
||||||
|
|
||||||
|
| Term | Definition |
|
||||||
|
|---|---|
|
||||||
|
| Radixor core | Java parsing, patch-command, trie, registry, descriptor, and loader code in `org.egothor:radixor` |
|
||||||
|
| Language | Locale-level identity such as `PL_PL`; not a dictionary or model |
|
||||||
|
| Model ID | Stable identity of one concrete model, such as `pl-pl-unimorph` |
|
||||||
|
| Model artifact | Independently versioned JAR containing one descriptor, one runtime dictionary, and licensing material |
|
||||||
|
| Source dictionary | Upstream lexical or morphological source recorded in provenance |
|
||||||
|
| Runtime dictionary | GZip-compressed UTF-8 Radixor tab-separated data consumed during trie construction |
|
||||||
|
| Compiled trie | In-memory lookup structure built by the loader; not the `stemmer.gz` resource |
|
||||||
|
| Default model | Stable ID selected by a language-oriented loader call |
|
||||||
|
| Optional model | Discoverable only when installed and selected explicitly; PoliMorf is optional for Polish |
|
||||||
|
|
||||||
|
Core version, model artifact version, catalog version, source dictionary version, and model format version are separate compatibility axes. Updating Java code need not republish unchanged model bytes; updating one model need not release core or every other model.
|
||||||
|
|
||||||
|
## Model artifact identity and layout
|
||||||
|
|
||||||
|
A module named `models/<model-id>` publishes:
|
||||||
|
|
||||||
|
```text
|
||||||
|
org.egothor:radixor-model-<model-id>:<model-version>
|
||||||
|
```
|
||||||
|
|
||||||
|
The built PoliMorf JAR has this effective tree:
|
||||||
|
|
||||||
|
```text
|
||||||
|
META-INF/
|
||||||
|
LICENSES/PoliMorf-BSD-2-Clause.txt
|
||||||
|
MANIFEST.MF
|
||||||
|
radixor/
|
||||||
|
models.index
|
||||||
|
models/pl-pl-polimorf.properties
|
||||||
|
org/egothor/stemmer/models/pl-pl-polimorf/stemmer.gz
|
||||||
|
```
|
||||||
|
|
||||||
|
Each UniMorph-derived model instead contains one model-specific
|
||||||
|
`META-INF/NOTICE/<model-id>-data.txt`. That notice records the upstream attribution, the Radixor
|
||||||
|
transformations and contribution statement, the ShareAlike distribution terms, and the canonical
|
||||||
|
CC BY-SA 3.0 URI. The repository has no root CC license directory because CC BY-SA applies to
|
||||||
|
these model-data artifacts, not to the BSD-3-Clause Radixor Java software. PoliMorf retains only
|
||||||
|
its BSD-2-Clause data license.
|
||||||
|
|
||||||
|
`models.index` contains the descriptor path. The descriptor contains the exact resource path. No Java provider class is required, and model modules do not compile against a core API.
|
||||||
|
|
||||||
|
## Discovery and integrity
|
||||||
|
|
||||||
|
`StemmerModelRegistry` asks the selected `ClassLoader` for every `META-INF/radixor/models.index`. It sorts index URLs, validates each non-comment entry, loads the named descriptors, sorts descriptors by model ID, and rejects duplicate IDs. It does not scan arbitrary JAR entries.
|
||||||
|
|
||||||
|
Descriptor parsing verifies:
|
||||||
|
|
||||||
|
- the model-ID syntax;
|
||||||
|
- required nonblank runtime properties;
|
||||||
|
- a known `Language` enum name;
|
||||||
|
- format `radixor-dictionary-tsv-gzip` and format version `1`;
|
||||||
|
- the exact namespaced resource path;
|
||||||
|
- presence of the runtime resource;
|
||||||
|
- a lowercase 64-character SHA-256 value.
|
||||||
|
|
||||||
|
Loading then reads the compressed resource bytes through the descriptor's discovering class loader, compares their SHA-256 digest, opens GZip, parses UTF-8 Radixor dictionary rows, and constructs a trie. Duplicate-ID and checksum checks make selection independent of classpath order.
|
||||||
|
|
||||||
|
## Descriptor fields
|
||||||
|
|
||||||
|
The convention plugin generates these fields:
|
||||||
|
|
||||||
|
| Property | Role | Meaning |
|
||||||
|
|---|---|---|
|
||||||
|
| `model.id` | Authoritative runtime identity | Stable model ID |
|
||||||
|
| `model.version` | Authoritative artifact identity | Independently managed model version |
|
||||||
|
| `model.language` | Authoritative selection metadata | Existing `Language` enum value |
|
||||||
|
| `model.displayName` | Display metadata | Human-readable name |
|
||||||
|
| `model.resource` | Authoritative loading metadata | Namespaced GZip resource |
|
||||||
|
| `model.default` | Catalog/build declaration | Whether the module declares itself a default; runtime language selection uses `Language.defaultModelId()` |
|
||||||
|
| `model.format` | Authoritative compatibility metadata | `radixor-dictionary-tsv-gzip` |
|
||||||
|
| `model.formatVersion` | Authoritative compatibility metadata | Currently `1` |
|
||||||
|
| `model.sha256` | Authoritative integrity metadata | Digest of the compressed source bytes |
|
||||||
|
| `model.rightToLeft` | Processing metadata | Language direction recorded by the build |
|
||||||
|
| `model.caseProcessing` | Processing metadata | `LOWERCASE_WITH_LOCALE_ROOT` |
|
||||||
|
| `model.diacriticProcessing` | Processing metadata | `AS_IS` |
|
||||||
|
| `model.storeOriginal` | Processing metadata | Currently `true` |
|
||||||
|
| `source.name` | Provenance | Source dictionary name |
|
||||||
|
| `source.version` | Provenance | Upstream version or the legacy-import sentinel |
|
||||||
|
| `source.project` | Provenance | Upstream project |
|
||||||
|
| `source.repository` | Provenance | Official language repository |
|
||||||
|
| `source.dataset` | Provenance | Upstream dataset and lexical-source identity |
|
||||||
|
| `source.revision` | Provenance | Exact revision or `not-recorded-in-legacy-import` |
|
||||||
|
| `source.revisionStatus` | Provenance | `recorded` or `not-recorded-in-legacy-import` |
|
||||||
|
| `source.license` | Provenance | SPDX or project license reference |
|
||||||
|
| `source.licenseUri` | Provenance | Canonical license URI |
|
||||||
|
| `source.attribution` | Provenance | Attribution supplied by the official source |
|
||||||
|
| `source.verificationDate` | Provenance | Date the maintained upstream information was checked |
|
||||||
|
| `transformations.summary` | Provenance | Material Radixor conversion operations |
|
||||||
|
| `compiler.radixorVersion` | Provenance | Compiler lineage recorded by the plugin |
|
||||||
|
| `compiler.radixorCommit` | Provenance | Commit when available; currently `unavailable` |
|
||||||
|
| `statistics.groups` | Provenance/statistics | Currently `unavailable` |
|
||||||
|
| `statistics.forms` | Provenance/statistics | Currently `unavailable` |
|
||||||
|
|
||||||
|
The current registry consumes the authoritative `model.*` identity, format, resource, and checksum fields. Processing and provenance fields remain packaged for audit and catalog generation but are not all exposed as typed `StemmerModelDescriptor` accessors. The generated catalog is the supported documentation view of source name, version, license, checksum, and size.
|
||||||
|
|
||||||
|
## Immutable input to runtime model
|
||||||
|
|
||||||
|
The packaging sequence is:
|
||||||
|
|
||||||
|
```text
|
||||||
|
models/<id>/src/modelInput/stemmer.gz
|
||||||
|
-> validate GZip, strict UTF-8, rows, metadata, version, and license
|
||||||
|
-> copy identical bytes into build/generated/modelResources
|
||||||
|
-> generate descriptor, index, and packaged license
|
||||||
|
-> package radixor-model-<id>-<version>.jar
|
||||||
|
-> discover from the application's runtime classpath
|
||||||
|
-> verify checksum, parse dictionary, and build a trie
|
||||||
|
```
|
||||||
|
|
||||||
|
Application runtime never reads `src/modelInput` from a source checkout.
|
||||||
|
|
||||||
|
For PoliMorf, the immutable input is exactly:
|
||||||
|
|
||||||
|
`models/pl-pl-polimorf/src/modelInput/stemmer.gz`
|
||||||
|
|
||||||
|
Its required upstream license is:
|
||||||
|
|
||||||
|
`models/pl-pl-polimorf/src/modelInput/LICENSE-BSD-2-Clause.txt`
|
||||||
|
|
||||||
|
The final runtime resource is exactly:
|
||||||
|
|
||||||
|
`org/egothor/stemmer/models/pl-pl-polimorf/stemmer.gz`
|
||||||
|
|
||||||
|
## Aggregate projects
|
||||||
|
|
||||||
|
| Project | Published coordinate | Contents and purpose |
|
||||||
|
|---|---|---|
|
||||||
|
| `models/standard` | `org.egothor:radixor-models-standard:<catalog-version>` | POM-only aggregate with one transitive runtime default per language; excludes PoliMorf |
|
||||||
|
| `models/bom` | `org.egothor:radixor-models-bom:<catalog-version>` | POM-only Maven dependency-management constraints for all individual published model versions |
|
||||||
|
|
||||||
|
Neither catalog artifact publishes a binary, sources, or Javadoc JAR. The standard aggregate resolves model JARs because its POM contains runtime dependencies. Importing the BOM only manages versions and resolves no model by itself. JMH, tests, and quality evaluation depend directly on individual model projects through non-production Gradle configurations.
|
||||||
|
|
||||||
|
The Maven dependency BOM is not a software bill of materials. The root `cyclonedxDirectBom` task generates the project-wide CycloneDX SBOM under `build/reports/sbom/`; it does not write into `models/bom/build/`.
|
||||||
|
|
||||||
|
`models/build/` is an ignored Gradle output directory for the implicit lifecycle parent `:models`, not a source module. CycloneDX direct tasks exposed on subprojects by the root plugin are disabled, so the supported build does not write an SBOM there. Aggregate model reports are owned by the root project under `build/reports/models/`; individual model reports and publication files stay under `models/<model-id>/build/`.
|
||||||
|
|
||||||
|
## Create or update a model module
|
||||||
|
|
||||||
|
1. Choose a stable lowercase model ID matching the module directory.
|
||||||
|
2. Add `models/<id>/model-version.txt`; do not derive it from core.
|
||||||
|
3. Apply `org.egothor.radixor.model` in the module build script.
|
||||||
|
4. Declare `modelId`, `language`, `displayName`, `defaultModel`, repository, dataset, revision and status, license URI, attribution, verification date, and transformations.
|
||||||
|
5. Put immutable `stemmer.gz` and a model-specific `NOTICE-model-data.txt` under `src/modelInput/`. The notice must identify the applicable data license and canonical URI, upstream attribution, transformations, and derived-data contributions without implying that the core software uses that license.
|
||||||
|
6. Add the module ID and its `default` or `optional` build-topology role to `models/model-projects.properties`. `settings.gradle`, verification, standard membership, BOM constraints, tests, and JMH all consume that list; descriptor metadata remains authoritative for model identity and language properties.
|
||||||
|
7. Run:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
./gradlew --no-daemon :models:<model-id>:validateModelInput
|
||||||
|
./gradlew --no-daemon :models:<model-id>:prepareModelResources
|
||||||
|
./gradlew --no-daemon :models:<model-id>:verifyModelDescriptor
|
||||||
|
./gradlew --no-daemon :models:<model-id>:verifyModelJar
|
||||||
|
./gradlew --no-daemon :models:<model-id>:check
|
||||||
|
./gradlew --no-daemon runtimeModelIntegrationTest -PmodelId=<model-id>
|
||||||
|
```
|
||||||
|
|
||||||
|
Validation fails for missing inputs, notices, attribution, repository, revision status, Radixor contribution and transformation disclosures, ShareAlike and no-endorsement statements, notice byte identity, unsafe or mismatched ID, invalid semantic version, invalid GZip/UTF-8, invalid dictionary rows, checksum mismatch, wrong packaged path, duplicate dictionaries, or dictionaries in sources/Javadoc artifacts. The explicit legacy revision sentinel is valid; an absent revision or status is not. The PoliMorf module separately validates its complete BSD-2-Clause license and attribution.
|
||||||
|
|
||||||
|
Copying an arbitrary `stemmer.gz` into an application is insufficient: registry discovery requires an index, a valid descriptor, namespaced resource, checksum, version, language, format declaration, and licensing material.
|
||||||
|
|
||||||
|
## Release boundaries
|
||||||
|
|
||||||
|
| Tag | Publishes | Does not publish |
|
||||||
|
|---|---|---|
|
||||||
|
| `release@<core-version>` | Root `org.egothor:radixor` software artifacts | Model JARs, standard pack, or BOM |
|
||||||
|
| `model/<model-id>@<model-version>` | Exactly the matching independently versioned model | Core, other models, standard pack, BOM, JMH, or full quality suite |
|
||||||
|
| `models-catalog@<catalog-version>` | Standard aggregate and models BOM | Individual model JARs or core |
|
||||||
|
|
||||||
|
Local validation for PoliMorf 1.0.0 is:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
./tools/parse-model-release-tag.sh "model/pl-pl-polimorf@1.0.0" .
|
||||||
|
./gradlew --no-daemon :models:pl-pl-polimorf:check
|
||||||
|
./gradlew --no-daemon runtimeModelIntegrationTest -PmodelId=pl-pl-polimorf
|
||||||
|
./gradlew --no-daemon :models:pl-pl-polimorf:validateModelRelease \
|
||||||
|
-PmodelReleaseVersion=1.0.0
|
||||||
|
./gradlew --no-daemon :models:pl-pl-polimorf:packageModelReleaseCandidate \
|
||||||
|
-PmodelReleaseVersion=1.0.0
|
||||||
|
```
|
||||||
|
|
||||||
|
`runtimeModelIntegrationTest` uses an isolated JVM, defaults to a 6 GiB maximum heap, and can be overridden with `-PradixorLargeModelMaxHeap=10g`. For PoliMorf, `validateModelRelease` depends on this complete runtime construction and real stemming smoke verification in addition to descriptor, checksum, license, and package validation. The generic release workflow still selects and publishes only the requested model. The commands above are local validation only; repository owners control tags and publication.
|
||||||
|
|
||||||
|
## Documentation and troubleshooting
|
||||||
|
|
||||||
|
`prepareMkDocsSource` generates the catalog only at `build/mkdocs-source/stemmer-model-catalog.md`; generated Markdown and rendered site content are not tracked. For runtime failures, dependency inspection, ClassLoader isolation, and fat-JAR guidance, see [Model Selection and Loading](model-selection-and-loading.md#troubleshooting).
|
||||||
87
docs/stemming-quality.md
Normal file
87
docs/stemming-quality.md
Normal file
@@ -0,0 +1,87 @@
|
|||||||
|
# Stemming quality evaluation
|
||||||
|
|
||||||
|
The explicit `stemmingQuality` analysis measures agreement between stemmer outputs and gold-standard equivalence classes represented by registered multilingual model dictionary rows. Dictionary text remains unchanged; reports and diagnostics use English.
|
||||||
|
|
||||||
|
JMH adapters, registries, third-party versions, language mappings, and preparation remain in `src/jmh`. The evaluator, reports, audits, and tests reside in the standard `src/test` source set. The former `src/stemmingQualityTest` source set was removed, and neither analytical nor JMH classes enter the production JAR.
|
||||||
|
|
||||||
|
## Language and adapter coverage
|
||||||
|
|
||||||
|
The authoritative Radixor universe is the validated one-to-one reconciliation of every `StemmerPatchTrieLoader.Language` value with its registered default model descriptor. All 20 current values have exactly one documented default. Optional comparison models, including `pl-pl-polimorf`, are identified separately and never replace default benchmark rows. Third-party combinations come only from explicit JMH adapter metadata.
|
||||||
|
|
||||||
|
Default Polish evaluation is therefore `Radixor` with model `pl-pl-unimorph`. A future PoliMorf evaluation is a distinct `Radixor` / `pl-pl-polimorf` row. Evaluation classpaths receive individual models through direct non-production Gradle dependencies; ordinary applications inherit none of them from the core.
|
||||||
|
|
||||||
|
Complete PoliMorf trie construction and deterministic stemming smoke fixtures are runtime-verified separately. That functional verification is not a linguistic-quality measurement and does not justify rewriting the historical quality snapshot.
|
||||||
|
|
||||||
|
The expected matrix is constructed before evaluation from stemmer, language, dictionary mode, and supported output policy. Generation fails on missing, duplicate, unexpected, or stale keys.
|
||||||
|
|
||||||
|
## Dictionary groups and modes
|
||||||
|
|
||||||
|
Every usable parsed row is one gold-standard group. Exact duplicate strings are removed only within that row; identical forms in different rows remain distinct. `ALL_WORDS` preserves every valid form. `LOWERCASE_GROUPS_ONLY` excludes a complete group containing an uppercase or titlecase Unicode code point. Retained words are not lowercased or normalized by the evaluator.
|
||||||
|
|
||||||
|
## Output policies
|
||||||
|
|
||||||
|
`PRIMARY_OUTPUT` uses the deterministic JMH output and defines a strict partition.
|
||||||
|
|
||||||
|
For multi-output adapters, `C(w)` is the immutable, sorted, exactly deduplicated candidate set. It is non-null, non-empty, contains no null, and contains the primary output. Radixor obtains alternatives through `getAll`. The repository's Morphologik lookups can return distinct lemma strings and are multi-output. Configured Hunspell filters can emit several stems at one token position. Other adapters emit only primary rows.
|
||||||
|
|
||||||
|
`ANY_CANDIDATE` is an optimistic oracle-assisted pairwise upper bound. A same-group pair succeeds when its sets intersect. A cross-group pair is an error only when both sets are the same singleton; otherwise unequal candidates can be selected for that pair. Choices may vary between pairs and need not form one realizable global assignment.
|
||||||
|
|
||||||
|
`ALL_CANDIDATES` activates every candidate. Two forms are related when their sets intersect, for both same-group and cross-group pairs. This relation can overlap and need not be transitive. A pair sharing several candidates is counted once.
|
||||||
|
|
||||||
|
The evaluator verifies:
|
||||||
|
|
||||||
|
```text
|
||||||
|
ANY under <= PRIMARY under
|
||||||
|
ALL under <= PRIMARY under
|
||||||
|
ANY under = ALL under
|
||||||
|
ANY over <= PRIMARY over
|
||||||
|
ALL over >= PRIMARY over
|
||||||
|
```
|
||||||
|
|
||||||
|
## Pair definitions and efficient counting
|
||||||
|
|
||||||
|
For `C2(n) = n(n-1)/2`:
|
||||||
|
|
||||||
|
```text
|
||||||
|
underPossible = sum_g C2(n_g)
|
||||||
|
overPossible = C2(N) - sum_g C2(n_g)
|
||||||
|
```
|
||||||
|
|
||||||
|
Under-stemming counts unrelated same-group pairs. Over-stemming counts related cross-group pairs. Primary output uses global and per-group stem frequencies. Candidate sets are canonical signatures counted globally and per group. An inverted candidate-to-signature index discovers intersections, and signature pairs shared through several candidates are deduplicated. `ANY_CANDIDATE` over-stemming uses only equal singleton signatures. All pair arithmetic uses checked `long` operations; complete production word pairs are never enumerated.
|
||||||
|
|
||||||
|
## Confusion and aggregate metrics
|
||||||
|
|
||||||
|
```text
|
||||||
|
TP = underPossible - underError
|
||||||
|
FN = underError
|
||||||
|
FP = overError
|
||||||
|
TN = overPossible - overError
|
||||||
|
```
|
||||||
|
|
||||||
|
Under-stemming is `FN/(TP+FN)` and over-stemming is `FP/(TN+FP)`; their denominators differ. The CSV also publishes precision, recall, specificity, accuracy, balanced accuracy, F0.5, F1, F2, Jaccard, Fowlkes-Mallows, Matthews correlation coefficient, and pairwise error rate. F0.5 emphasizes precision and over-stemming, F1 balances precision and recall, and F2 emphasizes recall and under-stemming. Accuracy and error rate can be dominated by the large cross-group true-negative population. Metrics use raw counts, not rounded rates. Zero denominators produce `n/a` in Markdown and empty CSV fields.
|
||||||
|
|
||||||
|
Only `PRIMARY_OUTPUT` receives partition metrics: Adjusted Rand Index, homogeneity, completeness, V-measure, and normalized mutual information with arithmetic-mean entropy normalization. Candidate policies remain inapplicable rather than being forced into artificial partitions.
|
||||||
|
|
||||||
|
Micro summaries sum confusion counts before calculation. Macro summaries average defined language values and retain coverage counts. Common-language comparisons use the exact language intersection and never score unsupported languages as zero. Rankings are separated by policy and metric; the default F0.5 choice is navigation, not a universal scientific preference.
|
||||||
|
|
||||||
|
Pearson and average-tie-rank Spearman reports use unrounded values and separate dictionary-mode and output-policy cohorts. Fewer than three observations, undefined inputs, and zero variance produce documented missing values. The reports provide reproducible data and make no automatic scientific conclusion.
|
||||||
|
|
||||||
|
## Exact accuracy and pairwise under-stemming
|
||||||
|
|
||||||
|
Exact textual accuracy and pairwise grouping use different denominators. One erroneous form in a 12-form group creates 11 erroneous pairs: with 88 singleton groups, exact accuracy can be 99% while pairwise under-stemming is `11/C2(12) = 16.666667%`. Singleton groups affect word accuracy but add no within-group pairs.
|
||||||
|
|
||||||
|
## Running the analysis
|
||||||
|
|
||||||
|
```bash
|
||||||
|
./gradlew stemmingQuality
|
||||||
|
./gradlew stemmingQuality -PstemmingQualityStemmer=Radixor -PstemmingQualityLanguage=DE_DE -PstemmingQualityMode=ALL_WORDS -PstemmingQualityAudit=true
|
||||||
|
```
|
||||||
|
|
||||||
|
Optional properties are `stemmingQualityLanguage`, `stemmingQualityStemmer`, `stemmingQualityMode`, `stemmingQualityOutputPolicy`, `stemmingQualityRankMetric`, `stemmingQualityAudit`, and `stemmingQualityAuditLimit`. Policies are `PRIMARY_OUTPUT`, `ANY_CANDIDATE`, and `ALL_CANDIDATES`. Filtered reports carry `-filtered` and cannot overwrite complete output.
|
||||||
|
|
||||||
|
Generated files under `build/reports/stemming-quality/` include `stemming-quality.md`, `stemming-quality.csv`, `metric-correlations-pearson.csv`, `metric-correlations-spearman.csv`, and optional audit Markdown.
|
||||||
|
|
||||||
|
## Limitations
|
||||||
|
|
||||||
|
These measurements evaluate agreement with the available dictionary grouping. They do not capture every semantic, morphological, downstream, or dataset-specific property. `ANY_CANDIDATE` is optimistic and may not be globally realizable. `ALL_CANDIDATES` measures an overlap graph rather than a partition. Language coverage must remain visible in cross-stemmer comparisons. No single published metric establishes universal superiority; multiple metrics and their correlations are provided for transparent scientific assessment.
|
||||||
|
Historical checked-in quality results retain their original inputs and claims. The optional PoliMorf model is not attributed to snapshots that predate it. See [Model Selection and Loading](model-selection-and-loading.md) and the generated [model catalog](stemmer-model-catalog.md).
|
||||||
@@ -5,15 +5,15 @@ antlr:antlr:2.7.7=pitest
|
|||||||
com.github.oowekyala.ooxml:nice-xml-messages:3.1=pmd
|
com.github.oowekyala.ooxml:nice-xml-messages:3.1=pmd
|
||||||
com.google.code.gson:gson:2.13.2=pmd
|
com.google.code.gson:gson:2.13.2=pmd
|
||||||
com.google.errorprone:error_prone_annotations:2.41.0=pmd
|
com.google.errorprone:error_prone_annotations:2.41.0=pmd
|
||||||
net.bytebuddy:byte-buddy-agent:1.17.7=jmhRuntimeClasspath,testCompileClasspath,testRuntimeClasspath
|
net.bytebuddy:byte-buddy-agent:1.17.7=testCompileClasspath,testRuntimeClasspath
|
||||||
net.bytebuddy:byte-buddy:1.17.7=jmhRuntimeClasspath,testCompileClasspath,testRuntimeClasspath
|
net.bytebuddy:byte-buddy:1.17.7=testCompileClasspath,testRuntimeClasspath
|
||||||
net.jqwik:jqwik-api:1.9.3=jmhRuntimeClasspath,testCompileClasspath,testRuntimeClasspath
|
net.jqwik:jqwik-api:1.9.3=testCompileClasspath,testRuntimeClasspath
|
||||||
net.jqwik:jqwik-engine:1.9.3=jmhRuntimeClasspath,testRuntimeClasspath
|
net.jqwik:jqwik-engine:1.9.3=testRuntimeClasspath
|
||||||
net.jqwik:jqwik-time:1.9.3=jmhRuntimeClasspath,testCompileClasspath,testRuntimeClasspath
|
net.jqwik:jqwik-time:1.9.3=testCompileClasspath,testRuntimeClasspath
|
||||||
net.jqwik:jqwik-web:1.9.3=jmhRuntimeClasspath,testCompileClasspath,testRuntimeClasspath
|
net.jqwik:jqwik-web:1.9.3=testCompileClasspath,testRuntimeClasspath
|
||||||
net.jqwik:jqwik:1.9.3=jmhRuntimeClasspath,testCompileClasspath,testRuntimeClasspath
|
net.jqwik:jqwik:1.9.3=testCompileClasspath,testRuntimeClasspath
|
||||||
net.sf.jopt-simple:jopt-simple:4.9=pitest
|
net.sf.jopt-simple:jopt-simple:4.9=pitest
|
||||||
net.sf.jopt-simple:jopt-simple:5.0.4=jmh,jmhCompileClasspath,jmhRuntimeClasspath
|
net.sf.jopt-simple:jopt-simple:5.0.4=jmh,jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
|
||||||
net.sf.saxon:Saxon-HE:12.9=pmd
|
net.sf.saxon:Saxon-HE:12.9=pmd
|
||||||
net.sourceforge.pmd:pmd-ant:7.20.0=pmd
|
net.sourceforge.pmd:pmd-ant:7.20.0=pmd
|
||||||
net.sourceforge.pmd:pmd-core:7.20.0=pmd
|
net.sourceforge.pmd:pmd-core:7.20.0=pmd
|
||||||
@@ -22,45 +22,45 @@ org.antlr:antlr4-runtime:4.9.3=pmd
|
|||||||
org.antlr:stringtemplate:3.2.1=pitest
|
org.antlr:stringtemplate:3.2.1=pitest
|
||||||
org.apache.commons:commons-lang3:3.18.0=pitest
|
org.apache.commons:commons-lang3:3.18.0=pitest
|
||||||
org.apache.commons:commons-lang3:3.20.0=pmd
|
org.apache.commons:commons-lang3:3.20.0=pmd
|
||||||
org.apache.commons:commons-math3:3.6.1=jmh,jmhCompileClasspath,jmhRuntimeClasspath
|
org.apache.commons:commons-math3:3.6.1=jmh,jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
|
||||||
org.apache.commons:commons-text:1.14.0=pitest
|
org.apache.commons:commons-text:1.14.0=pitest
|
||||||
org.apache.lucene:lucene-analysis-common:10.5.0=jmhCompileClasspath,jmhRuntimeClasspath
|
org.apache.lucene:lucene-analysis-common:10.5.0=jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
|
||||||
org.apache.lucene:lucene-analysis-morfologik:10.5.0=jmhCompileClasspath,jmhRuntimeClasspath
|
org.apache.lucene:lucene-analysis-morfologik:10.5.0=jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
|
||||||
org.apache.lucene:lucene-analysis-stempel:10.5.0=jmhCompileClasspath,jmhRuntimeClasspath
|
org.apache.lucene:lucene-analysis-stempel:10.5.0=jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
|
||||||
org.apache.lucene:lucene-core:10.5.0=jmhCompileClasspath,jmhRuntimeClasspath
|
org.apache.lucene:lucene-core:10.5.0=jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
|
||||||
org.apache.opennlp:opennlp-tools:2.5.4=jmhCompileClasspath,jmhRuntimeClasspath
|
org.apache.opennlp:opennlp-tools:2.5.4=jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
|
||||||
org.apiguardian:apiguardian-api:1.1.2=jmhRuntimeClasspath,testCompileClasspath,testRuntimeClasspath
|
org.apiguardian:apiguardian-api:1.1.2=testCompileClasspath,testRuntimeClasspath
|
||||||
org.carrot2:morfologik-fsa:2.1.9=jmhCompileClasspath,jmhRuntimeClasspath
|
org.carrot2:morfologik-fsa:2.1.9=jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
|
||||||
org.carrot2:morfologik-polish:2.1.9=jmhRuntimeClasspath
|
org.carrot2:morfologik-polish:2.1.9=jmhRuntimeClasspath,stemmingQualityJmhRuntime
|
||||||
org.carrot2:morfologik-stemming:2.1.9=jmhCompileClasspath,jmhRuntimeClasspath
|
org.carrot2:morfologik-stemming:2.1.9=jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
|
||||||
org.checkerframework:checker-qual:3.52.1=pmd
|
org.checkerframework:checker-qual:3.52.1=pmd
|
||||||
org.jacoco:org.jacoco.agent:0.8.14=jacocoAgent,jacocoAnt
|
org.jacoco:org.jacoco.agent:0.8.14=jacocoAgent,jacocoAnt
|
||||||
org.jacoco:org.jacoco.ant:0.8.14=jacocoAnt
|
org.jacoco:org.jacoco.ant:0.8.14=jacocoAnt
|
||||||
org.jacoco:org.jacoco.core:0.8.14=jacocoAnt
|
org.jacoco:org.jacoco.core:0.8.14=jacocoAnt
|
||||||
org.jacoco:org.jacoco.report:0.8.14=jacocoAnt
|
org.jacoco:org.jacoco.report:0.8.14=jacocoAnt
|
||||||
org.junit.jupiter:junit-jupiter-api:5.14.3=jmhRuntimeClasspath,testCompileClasspath,testRuntimeClasspath
|
org.junit.jupiter:junit-jupiter-api:5.14.3=testCompileClasspath,testRuntimeClasspath
|
||||||
org.junit.jupiter:junit-jupiter-engine:5.14.3=jmhRuntimeClasspath,testRuntimeClasspath
|
org.junit.jupiter:junit-jupiter-engine:5.14.3=testRuntimeClasspath
|
||||||
org.junit.jupiter:junit-jupiter-params:5.14.3=jmhRuntimeClasspath,testCompileClasspath,testRuntimeClasspath
|
org.junit.jupiter:junit-jupiter-params:5.14.3=testCompileClasspath,testRuntimeClasspath
|
||||||
org.junit.jupiter:junit-jupiter:5.14.3=jmhRuntimeClasspath,testCompileClasspath,testRuntimeClasspath
|
org.junit.jupiter:junit-jupiter:5.14.3=testCompileClasspath,testRuntimeClasspath
|
||||||
org.junit.platform:junit-platform-commons:1.14.3=jmhRuntimeClasspath,testCompileClasspath,testRuntimeClasspath
|
org.junit.platform:junit-platform-commons:1.14.3=testCompileClasspath,testRuntimeClasspath
|
||||||
org.junit.platform:junit-platform-engine:1.14.3=jmhRuntimeClasspath,testRuntimeClasspath
|
org.junit.platform:junit-platform-engine:1.14.3=testRuntimeClasspath
|
||||||
org.junit.platform:junit-platform-launcher:1.14.3=jmhRuntimeClasspath,testRuntimeClasspath
|
org.junit.platform:junit-platform-launcher:1.14.3=testRuntimeClasspath
|
||||||
org.junit:junit-bom:5.14.3=jmhRuntimeClasspath,testCompileClasspath,testRuntimeClasspath
|
org.junit:junit-bom:5.14.3=testCompileClasspath,testRuntimeClasspath
|
||||||
org.mockito:mockito-core:5.23.0=jmhRuntimeClasspath,mockitoAgent,testCompileClasspath,testRuntimeClasspath
|
org.mockito:mockito-core:5.23.0=mockitoAgent,testCompileClasspath,testRuntimeClasspath
|
||||||
org.mockito:mockito-junit-jupiter:5.23.0=jmhRuntimeClasspath,testCompileClasspath,testRuntimeClasspath
|
org.mockito:mockito-junit-jupiter:5.23.0=testCompileClasspath,testRuntimeClasspath
|
||||||
org.objenesis:objenesis:3.3=jmhRuntimeClasspath,testRuntimeClasspath
|
org.objenesis:objenesis:3.3=testRuntimeClasspath
|
||||||
org.openjdk.jmh:jmh-core:1.37=jmh,jmhCompileClasspath,jmhRuntimeClasspath
|
org.openjdk.jmh:jmh-core:1.37=jmh,jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
|
||||||
org.openjdk.jmh:jmh-generator-asm:1.37=jmh,jmhCompileClasspath,jmhRuntimeClasspath
|
org.openjdk.jmh:jmh-generator-asm:1.37=jmh,jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
|
||||||
org.openjdk.jmh:jmh-generator-bytecode:1.37=jmh,jmhCompileClasspath,jmhRuntimeClasspath
|
org.openjdk.jmh:jmh-generator-bytecode:1.37=jmh,jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
|
||||||
org.openjdk.jmh:jmh-generator-reflection:1.37=jmh,jmhCompileClasspath,jmhRuntimeClasspath
|
org.openjdk.jmh:jmh-generator-reflection:1.37=jmh,jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
|
||||||
org.opentest4j:opentest4j:1.3.0=jmhRuntimeClasspath,testCompileClasspath,testRuntimeClasspath
|
org.opentest4j:opentest4j:1.3.0=testCompileClasspath,testRuntimeClasspath
|
||||||
org.ow2.asm:asm-analysis:9.9.1=pitest
|
org.ow2.asm:asm-analysis:9.9.1=pitest
|
||||||
org.ow2.asm:asm-commons:9.9=jacocoAnt
|
org.ow2.asm:asm-commons:9.9=jacocoAnt
|
||||||
org.ow2.asm:asm-commons:9.9.1=pitest
|
org.ow2.asm:asm-commons:9.9.1=pitest
|
||||||
org.ow2.asm:asm-tree:9.9=jacocoAnt
|
org.ow2.asm:asm-tree:9.9=jacocoAnt
|
||||||
org.ow2.asm:asm-tree:9.9.1=pitest
|
org.ow2.asm:asm-tree:9.9.1=pitest
|
||||||
org.ow2.asm:asm-util:9.9.1=pitest
|
org.ow2.asm:asm-util:9.9.1=pitest
|
||||||
org.ow2.asm:asm:9.0=jmh,jmhCompileClasspath,jmhRuntimeClasspath
|
org.ow2.asm:asm:9.0=jmh,jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
|
||||||
org.ow2.asm:asm:9.9=jacocoAnt
|
org.ow2.asm:asm:9.9=jacocoAnt
|
||||||
org.ow2.asm:asm:9.9.1=pitest,pmd
|
org.ow2.asm:asm:9.9.1=pitest,pmd
|
||||||
org.pcollections:pcollections:4.0.2=pmd
|
org.pcollections:pcollections:4.0.2=pmd
|
||||||
@@ -70,7 +70,7 @@ org.pitest:pitest-html-report:1.22.1=pitest
|
|||||||
org.pitest:pitest-junit5-plugin:1.2.3=pitest
|
org.pitest:pitest-junit5-plugin:1.2.3=pitest
|
||||||
org.pitest:pitest:1.22.1=pitest
|
org.pitest:pitest:1.22.1=pitest
|
||||||
org.slf4j:jul-to-slf4j:1.7.36=pmd
|
org.slf4j:jul-to-slf4j:1.7.36=pmd
|
||||||
org.slf4j:slf4j-api:2.0.17=jmhCompileClasspath,jmhRuntimeClasspath
|
org.slf4j:slf4j-api:2.0.17=jmhCompileClasspath,jmhRuntimeClasspath,stemmingQualityJmhRuntime
|
||||||
org.xmlresolver:xmlresolver:5.3.3=pmd
|
org.xmlresolver:xmlresolver:5.3.3=pmd
|
||||||
ua.net.nlp:morfologik-ukrainian-search:4.9.1=jmhRuntimeClasspath
|
ua.net.nlp:morfologik-ukrainian-search:4.9.1=jmhRuntimeClasspath,stemmingQualityJmhRuntime
|
||||||
empty=annotationProcessor,compileClasspath,cyclonedxBom,jmhAnnotationProcessor,mainPmdAuxClasspath,runtimeClasspath,testAnnotationProcessor
|
empty=annotationProcessor,compileClasspath,cyclonedxBom,jmhAnnotationProcessor,mainPmdAuxClasspath,runtimeClasspath,testAnnotationProcessor
|
||||||
|
|||||||
69
gradle/cistem-benchmarks.gradle
Normal file
69
gradle/cistem-benchmarks.gradle
Normal file
@@ -0,0 +1,69 @@
|
|||||||
|
def cistemGoldStandardBaseUrl = 'https://raw.githubusercontent.com/LeonieWeissweiler/CISTEM/refs/heads/master/gold_standards'
|
||||||
|
def cistemGoldStandardFiles = [
|
||||||
|
'goldstandard1.txt',
|
||||||
|
'goldstandard2.txt'
|
||||||
|
]
|
||||||
|
def cistemGoldStandardDownloadDirectory = layout.buildDirectory.dir('third-party/cistem-gold-standards')
|
||||||
|
def cistemGoldStandardGeneratedResourcesDirectory = layout.buildDirectory.dir('generated/resources/cistem-gold-standards')
|
||||||
|
|
||||||
|
def cistemGoldStandardDownloadedFiles = cistemGoldStandardFiles.collect { String fileName ->
|
||||||
|
cistemGoldStandardDownloadDirectory.map { it.file(fileName) }
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.register('downloadCistemGoldStandards') {
|
||||||
|
group = 'build setup'
|
||||||
|
description = 'Downloads benchmark-only CISTEM German gold standards.'
|
||||||
|
|
||||||
|
outputs.files(cistemGoldStandardDownloadedFiles)
|
||||||
|
|
||||||
|
doLast {
|
||||||
|
cistemGoldStandardFiles.each { String fileName ->
|
||||||
|
final File targetFile = cistemGoldStandardDownloadDirectory.get().file(fileName).asFile
|
||||||
|
targetFile.parentFile.mkdirs()
|
||||||
|
|
||||||
|
if (!targetFile.exists()) {
|
||||||
|
final URL sourceUrl = new URL("${cistemGoldStandardBaseUrl}/${fileName}")
|
||||||
|
try {
|
||||||
|
sourceUrl.withInputStream { inputStream ->
|
||||||
|
targetFile.withOutputStream { outputStream ->
|
||||||
|
outputStream << inputStream
|
||||||
|
}
|
||||||
|
}
|
||||||
|
} catch (FileNotFoundException exception) {
|
||||||
|
throw new GradleException(
|
||||||
|
"Unable to download CISTEM gold standard ${fileName} from ${sourceUrl}.",
|
||||||
|
exception)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (targetFile.length() <= 0L) {
|
||||||
|
throw new GradleException("Downloaded CISTEM gold standard ${fileName} was empty.")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.register('prepareCistemGoldStandardResources', Copy) {
|
||||||
|
group = 'build setup'
|
||||||
|
description = 'Copies benchmark-only CISTEM German gold standards into the JMH resource output.'
|
||||||
|
|
||||||
|
dependsOn(tasks.named('downloadCistemGoldStandards'))
|
||||||
|
|
||||||
|
from(cistemGoldStandardDownloadDirectory) {
|
||||||
|
include 'goldstandard1.txt'
|
||||||
|
include 'goldstandard2.txt'
|
||||||
|
}
|
||||||
|
into(cistemGoldStandardGeneratedResourcesDirectory)
|
||||||
|
}
|
||||||
|
|
||||||
|
sourceSets {
|
||||||
|
jmh {
|
||||||
|
resources {
|
||||||
|
srcDir(cistemGoldStandardGeneratedResourcesDirectory)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.named('processJmhResources') {
|
||||||
|
dependsOn(tasks.named('prepareCistemGoldStandardResources'))
|
||||||
|
}
|
||||||
121
gradle/hunspell-benchmarks.gradle
Normal file
121
gradle/hunspell-benchmarks.gradle
Normal file
@@ -0,0 +1,121 @@
|
|||||||
|
import org.gradle.plugins.ide.eclipse.model.SourceFolder
|
||||||
|
|
||||||
|
|
||||||
|
def hunspellDictionaryBaseUrl = 'https://raw.githubusercontent.com/wooorm/dictionaries/main/dictionaries'
|
||||||
|
def hunspellDictionaryLanguages = [
|
||||||
|
en: 'English',
|
||||||
|
cs: 'Czech',
|
||||||
|
de: 'German',
|
||||||
|
es: 'Spanish',
|
||||||
|
fr: 'French',
|
||||||
|
nl: 'Dutch',
|
||||||
|
pl: 'Polish',
|
||||||
|
uk: 'Ukrainian'
|
||||||
|
]
|
||||||
|
def hunspellDownloadDirectory = layout.buildDirectory.dir('third-party/hunspell')
|
||||||
|
def hunspellGeneratedResourcesDirectory = layout.buildDirectory.dir('generated/resources/hunspell')
|
||||||
|
def hunspellGeneratedResourcesPath = provider {
|
||||||
|
project.relativePath(hunspellGeneratedResourcesDirectory.get().asFile)
|
||||||
|
}
|
||||||
|
def hunspellEclipseClasspathAttributes = [
|
||||||
|
gradle_scope : 'jmh',
|
||||||
|
gradle_used_by_scope: 'jmh',
|
||||||
|
test : 'true'
|
||||||
|
]
|
||||||
|
def hunspellIsAbsolutePath = { String path ->
|
||||||
|
path.startsWith('/') || path ==~ /^[A-Za-z]:[\\\/].*/
|
||||||
|
}
|
||||||
|
|
||||||
|
def hunspellDownloadedFiles = hunspellDictionaryLanguages.keySet().collectMany { String code ->
|
||||||
|
[
|
||||||
|
hunspellDownloadDirectory.map { it.file("${code}/index.aff") },
|
||||||
|
hunspellDownloadDirectory.map { it.file("${code}/index.dic") },
|
||||||
|
hunspellDownloadDirectory.map { it.file("${code}/license") }
|
||||||
|
]
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.register('downloadHunspellBenchmarkDictionaries') {
|
||||||
|
group = 'build setup'
|
||||||
|
description = 'Downloads benchmark-only Hunspell dictionaries from wooorm/dictionaries.'
|
||||||
|
|
||||||
|
outputs.files(hunspellDownloadedFiles)
|
||||||
|
|
||||||
|
doLast {
|
||||||
|
hunspellDictionaryLanguages.each { String code, String displayName ->
|
||||||
|
['index.aff', 'index.dic', 'license'].each { String fileName ->
|
||||||
|
final File targetFile = hunspellDownloadDirectory.get().file("${code}/${fileName}").asFile
|
||||||
|
targetFile.parentFile.mkdirs()
|
||||||
|
|
||||||
|
if (!targetFile.exists()) {
|
||||||
|
final URL sourceUrl = new URL("${hunspellDictionaryBaseUrl}/${code}/${fileName}")
|
||||||
|
try {
|
||||||
|
sourceUrl.withInputStream { inputStream ->
|
||||||
|
targetFile.withOutputStream { outputStream ->
|
||||||
|
outputStream << inputStream
|
||||||
|
}
|
||||||
|
}
|
||||||
|
} catch (FileNotFoundException exception) {
|
||||||
|
throw new GradleException(
|
||||||
|
"Unable to download Hunspell ${fileName} file for ${displayName} (${code}) from ${sourceUrl}.",
|
||||||
|
exception)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (targetFile.length() <= 0L) {
|
||||||
|
throw new GradleException("Downloaded Hunspell ${fileName} file for ${displayName} was empty.")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.register('prepareHunspellBenchmarkResources', Copy) {
|
||||||
|
group = 'build setup'
|
||||||
|
description = 'Copies benchmark-only Hunspell dictionaries into the JMH resource output.'
|
||||||
|
|
||||||
|
dependsOn(tasks.named('downloadHunspellBenchmarkDictionaries'))
|
||||||
|
|
||||||
|
from(hunspellDownloadDirectory) {
|
||||||
|
include '**/index.aff'
|
||||||
|
include '**/index.dic'
|
||||||
|
include '**/license'
|
||||||
|
into 'hunspell'
|
||||||
|
}
|
||||||
|
into(hunspellGeneratedResourcesDirectory)
|
||||||
|
}
|
||||||
|
|
||||||
|
sourceSets {
|
||||||
|
jmh {
|
||||||
|
resources {
|
||||||
|
srcDir(hunspellGeneratedResourcesDirectory)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.named('processJmhResources') {
|
||||||
|
dependsOn(tasks.named('prepareHunspellBenchmarkResources'))
|
||||||
|
}
|
||||||
|
|
||||||
|
eclipse {
|
||||||
|
classpath {
|
||||||
|
file {
|
||||||
|
whenMerged { classpath ->
|
||||||
|
String generatedPath = hunspellGeneratedResourcesPath.get()
|
||||||
|
|
||||||
|
classpath.entries.removeAll { entry ->
|
||||||
|
entry.hasProperty('path') && (
|
||||||
|
entry.path == generatedPath ||
|
||||||
|
hunspellIsAbsolutePath(entry.path)
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
SourceFolder hunspellEntry = new SourceFolder(generatedPath, null)
|
||||||
|
hunspellEntry.output = 'bin/jmh'
|
||||||
|
hunspellEclipseClasspathAttributes.each { String name, String value ->
|
||||||
|
hunspellEntry.entryAttributes[name] = value
|
||||||
|
}
|
||||||
|
classpath.entries.add(hunspellEntry)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
30
gradle/java-license-header.txt
Normal file
30
gradle/java-license-header.txt
Normal file
@@ -0,0 +1,30 @@
|
|||||||
|
/*******************************************************************************
|
||||||
|
* Copyright (C) 2026, Leo Galambos
|
||||||
|
* All rights reserved.
|
||||||
|
*
|
||||||
|
* Redistribution and use in source and binary forms, with or without
|
||||||
|
* modification, are permitted provided that the following conditions are met:
|
||||||
|
*
|
||||||
|
* 1. Redistributions of source code must retain the above copyright notice,
|
||||||
|
* this list of conditions and the following disclaimer.
|
||||||
|
*
|
||||||
|
* 2. Redistributions in binary form must reproduce the above copyright notice,
|
||||||
|
* this list of conditions and the following disclaimer in the documentation
|
||||||
|
* and/or other materials provided with the distribution.
|
||||||
|
*
|
||||||
|
* 3. Neither the name of the copyright holder nor the names of its contributors
|
||||||
|
* may be used to endorse or promote products derived from this software
|
||||||
|
* without specific prior written permission.
|
||||||
|
*
|
||||||
|
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
||||||
|
* AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||||
|
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
||||||
|
* ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE
|
||||||
|
* LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
|
||||||
|
* CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
|
||||||
|
* SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
|
||||||
|
* INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
|
||||||
|
* CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
|
||||||
|
* ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
||||||
|
* POSSIBILITY OF SUCH DAMAGE.
|
||||||
|
******************************************************************************/
|
||||||
@@ -51,11 +51,6 @@ publishing {
|
|||||||
url = pomLicenseUrl
|
url = pomLicenseUrl
|
||||||
distribution = pomLicenseDistribution
|
distribution = pomLicenseDistribution
|
||||||
}
|
}
|
||||||
license {
|
|
||||||
name = pomStemmerDataLicenseName
|
|
||||||
url = pomStemmerDataLicenseUrl
|
|
||||||
distribution = pomLicenseDistribution
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
developers {
|
developers {
|
||||||
@@ -104,8 +99,6 @@ tasks.register('validateReleaseMetadata') {
|
|||||||
if (pomScmDeveloperConnection == null || pomScmDeveloperConnection.isBlank()) missing.add('pomScmDeveloperConnection')
|
if (pomScmDeveloperConnection == null || pomScmDeveloperConnection.isBlank()) missing.add('pomScmDeveloperConnection')
|
||||||
if (pomLicenseName == null || pomLicenseName.isBlank()) missing.add('pomLicenseName')
|
if (pomLicenseName == null || pomLicenseName.isBlank()) missing.add('pomLicenseName')
|
||||||
if (pomLicenseUrl == null || pomLicenseUrl.isBlank()) missing.add('pomLicenseUrl')
|
if (pomLicenseUrl == null || pomLicenseUrl.isBlank()) missing.add('pomLicenseUrl')
|
||||||
if (pomStemmerDataLicenseName == null || pomStemmerDataLicenseName.isBlank()) missing.add('pomStemmerDataLicenseName')
|
|
||||||
if (pomStemmerDataLicenseUrl == null || pomStemmerDataLicenseUrl.isBlank()) missing.add('pomStemmerDataLicenseUrl')
|
|
||||||
if (signingKey == null || signingKey.isBlank()) missing.add('pomSigningKey / SIGNING_KEY')
|
if (signingKey == null || signingKey.isBlank()) missing.add('pomSigningKey / SIGNING_KEY')
|
||||||
if (signingPassword == null || signingPassword.isBlank()) missing.add('pomSigningPassword / SIGNING_PASSWORD')
|
if (signingPassword == null || signingPassword.isBlank()) missing.add('pomSigningPassword / SIGNING_PASSWORD')
|
||||||
|
|
||||||
|
|||||||
@@ -120,6 +120,11 @@ def transformPaiceHuskSource = { final File sourceFile, final File rulesFile, fi
|
|||||||
|
|
||||||
transformedText = 'package org.egothor.stemmer.benchmark;' + '\n\n' + transformedText
|
transformedText = 'package org.egothor.stemmer.benchmark;' + '\n\n' + transformedText
|
||||||
transformedText = transformedText.replaceFirst(/(?m)^\s*class\s+PaiceHusk\s*\{/, 'public final class PaiceHuskLancasterStemmer {')
|
transformedText = transformedText.replaceFirst(/(?m)^\s*class\s+PaiceHusk\s*\{/, 'public final class PaiceHuskLancasterStemmer {')
|
||||||
|
transformedText = transformedText.replace('new Character(rule.letter)', 'Character.valueOf(rule.letter)')
|
||||||
|
transformedText = transformedText.replace('new Character(stem.charAt(stem.length() - 1))',
|
||||||
|
'Character.valueOf(stem.charAt(stem.length() - 1))')
|
||||||
|
transformedText = transformedText.replaceFirst(/(?m)^(\s*)static HashMap loadRules\(/,
|
||||||
|
'$1@SuppressWarnings("unchecked")\n$1static HashMap loadRules(')
|
||||||
|
|
||||||
final int packageEnd = transformedText.indexOf('\n', transformedText.indexOf('package org.egothor.stemmer.benchmark;'))
|
final int packageEnd = transformedText.indexOf('\n', transformedText.indexOf('package org.egothor.stemmer.benchmark;'))
|
||||||
if (packageEnd >= 0) {
|
if (packageEnd >= 0) {
|
||||||
|
|||||||
@@ -248,11 +248,24 @@
|
|||||||
<sha256 value="a151df1e2e0b48618d8b06a180748a29b3abb39b1b2396f6a1c879a727488c6e" origin="Generated by Gradle"/>
|
<sha256 value="a151df1e2e0b48618d8b06a180748a29b3abb39b1b2396f6a1c879a727488c6e" origin="Generated by Gradle"/>
|
||||||
</artifact>
|
</artifact>
|
||||||
</component>
|
</component>
|
||||||
|
<component group="com.google.errorprone" name="error_prone_annotations" version="2.47.0">
|
||||||
|
<artifact name="error_prone_annotations-2.47.0.jar">
|
||||||
|
<sha256 value="5364bc6f22e72e98195e406a58d3ba1c09ffa11dea0729592cb870dc2de4056d" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
<artifact name="error_prone_annotations-2.47.0.pom">
|
||||||
|
<sha256 value="d80c889a4a6f711f6945fbee79e05ec247b178a567e9d5abf58eb26ebf0a0752" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
</component>
|
||||||
<component group="com.google.errorprone" name="error_prone_parent" version="2.41.0">
|
<component group="com.google.errorprone" name="error_prone_parent" version="2.41.0">
|
||||||
<artifact name="error_prone_parent-2.41.0.pom">
|
<artifact name="error_prone_parent-2.41.0.pom">
|
||||||
<sha256 value="c538388d760a5c1c98dcf06f6ed3cfe5f11a651827db5cbd2ed8288c795cad42" origin="Generated by Gradle"/>
|
<sha256 value="c538388d760a5c1c98dcf06f6ed3cfe5f11a651827db5cbd2ed8288c795cad42" origin="Generated by Gradle"/>
|
||||||
</artifact>
|
</artifact>
|
||||||
</component>
|
</component>
|
||||||
|
<component group="com.google.errorprone" name="error_prone_parent" version="2.47.0">
|
||||||
|
<artifact name="error_prone_parent-2.47.0.pom">
|
||||||
|
<sha256 value="2368a990c7a63095e1d0d44459d5a4092f0eb31f8562bd12cdf0e1c877b6a685" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
</component>
|
||||||
<component group="com.google.guava" name="failureaccess" version="1.0.3">
|
<component group="com.google.guava" name="failureaccess" version="1.0.3">
|
||||||
<artifact name="failureaccess-1.0.3.jar">
|
<artifact name="failureaccess-1.0.3.jar">
|
||||||
<sha256 value="cbfc3906b19b8f55dd7cfd6dfe0aa4532e834250d7f080bd8d211a3e246b59cb" origin="Generated by Gradle"/>
|
<sha256 value="cbfc3906b19b8f55dd7cfd6dfe0aa4532e834250d7f080bd8d211a3e246b59cb" origin="Generated by Gradle"/>
|
||||||
@@ -274,6 +287,14 @@
|
|||||||
<sha256 value="77ed42c8c8b2cebbb93ac9e07543ff6418aa24bdb8517580cf5324e9a6510956" origin="Generated by Gradle"/>
|
<sha256 value="77ed42c8c8b2cebbb93ac9e07543ff6418aa24bdb8517580cf5324e9a6510956" origin="Generated by Gradle"/>
|
||||||
</artifact>
|
</artifact>
|
||||||
</component>
|
</component>
|
||||||
|
<component group="com.google.guava" name="guava" version="33.6.0-jre">
|
||||||
|
<artifact name="guava-33.6.0-jre.jar">
|
||||||
|
<sha256 value="dc573e1fca4fd5454f4a5fd3d7da2df03002876a4175bafc14a95980dd7713b3" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
<artifact name="guava-33.6.0-jre.module">
|
||||||
|
<sha256 value="2baf73ce839ae48e4b9e0083e256b0e58fc3bf8fc78fc3fbe797bbc89011216e" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
</component>
|
||||||
<component group="com.google.guava" name="guava-parent" version="26.0-android">
|
<component group="com.google.guava" name="guava-parent" version="26.0-android">
|
||||||
<artifact name="guava-parent-26.0-android.pom">
|
<artifact name="guava-parent-26.0-android.pom">
|
||||||
<sha256 value="f8698ab46ca996ce889c1afc8ca4f25eb8ac6b034dc898d4583742360016cc04" origin="Generated by Gradle"/>
|
<sha256 value="f8698ab46ca996ce889c1afc8ca4f25eb8ac6b034dc898d4583742360016cc04" origin="Generated by Gradle"/>
|
||||||
@@ -294,6 +315,11 @@
|
|||||||
<sha256 value="68719e687c6e4c9ff3e0fecbef7bd20896f0f4f7b314743ed33c72f962568215" origin="Generated by Gradle"/>
|
<sha256 value="68719e687c6e4c9ff3e0fecbef7bd20896f0f4f7b314743ed33c72f962568215" origin="Generated by Gradle"/>
|
||||||
</artifact>
|
</artifact>
|
||||||
</component>
|
</component>
|
||||||
|
<component group="com.google.guava" name="guava-parent" version="33.6.0-jre">
|
||||||
|
<artifact name="guava-parent-33.6.0-jre.pom">
|
||||||
|
<sha256 value="374bd31f61b1cf612bee9ab2e4d70bbdf77dd85a49b431f809d4fbdc901f2dd4" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
</component>
|
||||||
<component group="com.google.guava" name="listenablefuture" version="9999.0-empty-to-avoid-conflict-with-guava">
|
<component group="com.google.guava" name="listenablefuture" version="9999.0-empty-to-avoid-conflict-with-guava">
|
||||||
<artifact name="listenablefuture-9999.0-empty-to-avoid-conflict-with-guava.jar">
|
<artifact name="listenablefuture-9999.0-empty-to-avoid-conflict-with-guava.jar">
|
||||||
<sha256 value="b372a037d4230aa57fbeffdef30fd6123f9c0c2db85d0aced00c91b974f33f99" origin="Generated by Gradle"/>
|
<sha256 value="b372a037d4230aa57fbeffdef30fd6123f9c0c2db85d0aced00c91b974f33f99" origin="Generated by Gradle"/>
|
||||||
@@ -524,11 +550,24 @@
|
|||||||
<sha256 value="6d849ae7454ab391718e5fc70e2716418ef3ed264472345bd80c6de64e00b6c4" origin="Generated by Gradle"/>
|
<sha256 value="6d849ae7454ab391718e5fc70e2716418ef3ed264472345bd80c6de64e00b6c4" origin="Generated by Gradle"/>
|
||||||
</artifact>
|
</artifact>
|
||||||
</component>
|
</component>
|
||||||
|
<component group="me.champeau.jmh" name="jmh-gradle-plugin" version="0.7.3">
|
||||||
|
<artifact name="jmh-gradle-plugin-0.7.3.jar">
|
||||||
|
<sha256 value="d7097e619541d90e0a970b2a68573e22ad01d2999ee5365d56d59830765bf98f" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
<artifact name="jmh-gradle-plugin-0.7.3.module">
|
||||||
|
<sha256 value="3487d1aba24fe0af527c6d5f78b5f0e8fd64fe9878708b460e6600e39a47bc43" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
</component>
|
||||||
<component group="me.champeau.jmh" name="me.champeau.jmh.gradle.plugin" version="0.7.2">
|
<component group="me.champeau.jmh" name="me.champeau.jmh.gradle.plugin" version="0.7.2">
|
||||||
<artifact name="me.champeau.jmh.gradle.plugin-0.7.2.pom">
|
<artifact name="me.champeau.jmh.gradle.plugin-0.7.2.pom">
|
||||||
<sha256 value="57e0c23ac60945aefb5a0c4a9339bea68a295364ca47c7a9079a032f79013abb" origin="Generated by Gradle"/>
|
<sha256 value="57e0c23ac60945aefb5a0c4a9339bea68a295364ca47c7a9079a032f79013abb" origin="Generated by Gradle"/>
|
||||||
</artifact>
|
</artifact>
|
||||||
</component>
|
</component>
|
||||||
|
<component group="me.champeau.jmh" name="me.champeau.jmh.gradle.plugin" version="0.7.3">
|
||||||
|
<artifact name="me.champeau.jmh.gradle.plugin-0.7.3.pom">
|
||||||
|
<sha256 value="d516226b3b114e4b32d42544d1d2796c732c5465d5dae7cc846be6b23bed8d1d" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
</component>
|
||||||
<component group="net.bytebuddy" name="byte-buddy" version="1.17.7">
|
<component group="net.bytebuddy" name="byte-buddy" version="1.17.7">
|
||||||
<artifact name="byte-buddy-1.17.7.jar">
|
<artifact name="byte-buddy-1.17.7.jar">
|
||||||
<sha256 value="3575dcb8a98faf943d3c1595c47a16047c4fce8a83ebbb26262f1a2f67546357" origin="Generated by Gradle"/>
|
<sha256 value="3575dcb8a98faf943d3c1595c47a16047c4fce8a83ebbb26262f1a2f67546357" origin="Generated by Gradle"/>
|
||||||
@@ -725,6 +764,11 @@
|
|||||||
<sha256 value="524ec4787aff73af6b3a9fafa154c7f1881b648299b663fdbfcadda1286f2353" origin="Generated by Gradle"/>
|
<sha256 value="524ec4787aff73af6b3a9fafa154c7f1881b648299b663fdbfcadda1286f2353" origin="Generated by Gradle"/>
|
||||||
</artifact>
|
</artifact>
|
||||||
</component>
|
</component>
|
||||||
|
<component group="org.apache" name="apache" version="38">
|
||||||
|
<artifact name="apache-38.pom">
|
||||||
|
<sha256 value="9b0a5f28ddfb4b7500a37022bee8245efdd044fb9a3d79fb827550923eccc4b5" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
</component>
|
||||||
<component group="org.apache.commons" name="commons-collections4" version="4.5.0">
|
<component group="org.apache.commons" name="commons-collections4" version="4.5.0">
|
||||||
<artifact name="commons-collections4-4.5.0.jar">
|
<artifact name="commons-collections4-4.5.0.jar">
|
||||||
<sha256 value="00f93263c267be201b8ae521b44a7137271b16688435340bf629db1bac0a5845" origin="Generated by Gradle"/>
|
<sha256 value="00f93263c267be201b8ae521b44a7137271b16688435340bf629db1bac0a5845" origin="Generated by Gradle"/>
|
||||||
@@ -995,6 +1039,11 @@
|
|||||||
<sha256 value="6f4bb954198678a528dfc8b2887a84cc3f54ae4a0b8b75c191fa28b04963e607" origin="Generated by Gradle"/>
|
<sha256 value="6f4bb954198678a528dfc8b2887a84cc3f54ae4a0b8b75c191fa28b04963e607" origin="Generated by Gradle"/>
|
||||||
</artifact>
|
</artifact>
|
||||||
</component>
|
</component>
|
||||||
|
<component group="org.apache.maven" name="maven" version="3.9.16">
|
||||||
|
<artifact name="maven-3.9.16.pom">
|
||||||
|
<sha256 value="5a761e32d3f3b5d65a70345cab4a327730c1d2000bb935bae7276dcc8fa81738" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
</component>
|
||||||
<component group="org.apache.maven" name="maven-artifact" version="3.9.14">
|
<component group="org.apache.maven" name="maven-artifact" version="3.9.14">
|
||||||
<artifact name="maven-artifact-3.9.14.jar">
|
<artifact name="maven-artifact-3.9.14.jar">
|
||||||
<sha256 value="1effa70eacbf0aa4d94ad9c7b225be031ce4317fa07da59e23b02b3e4e1231a3" origin="Generated by Gradle"/>
|
<sha256 value="1effa70eacbf0aa4d94ad9c7b225be031ce4317fa07da59e23b02b3e4e1231a3" origin="Generated by Gradle"/>
|
||||||
@@ -1003,6 +1052,14 @@
|
|||||||
<sha256 value="e668c936d22fd2c11edff52eac72c6e7fc13ba57c949096048be8debf0f9ffd2" origin="Generated by Gradle"/>
|
<sha256 value="e668c936d22fd2c11edff52eac72c6e7fc13ba57c949096048be8debf0f9ffd2" origin="Generated by Gradle"/>
|
||||||
</artifact>
|
</artifact>
|
||||||
</component>
|
</component>
|
||||||
|
<component group="org.apache.maven" name="maven-artifact" version="3.9.16">
|
||||||
|
<artifact name="maven-artifact-3.9.16.jar">
|
||||||
|
<sha256 value="54cc1c1ef932e3d4a903352111b42b4d3c3ed8e7a1d0de73b625309d9c3ad3e8" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
<artifact name="maven-artifact-3.9.16.pom">
|
||||||
|
<sha256 value="85d313bbbdbce67e199aadb4336e100c40a147881142ea4368cdebeafc02baec" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
</component>
|
||||||
<component group="org.apache.maven" name="maven-builder-support" version="3.9.14">
|
<component group="org.apache.maven" name="maven-builder-support" version="3.9.14">
|
||||||
<artifact name="maven-builder-support-3.9.14.jar">
|
<artifact name="maven-builder-support-3.9.14.jar">
|
||||||
<sha256 value="2109ff808046e4f8b356b1064060a3f224b0b0aad07ecceaf7696c3bdc0b2296" origin="Generated by Gradle"/>
|
<sha256 value="2109ff808046e4f8b356b1064060a3f224b0b0aad07ecceaf7696c3bdc0b2296" origin="Generated by Gradle"/>
|
||||||
@@ -1011,6 +1068,14 @@
|
|||||||
<sha256 value="6ed1ab2a239c5954b074dd8b75e70dbba55866840e73c53ae8b1f99a35afb7b5" origin="Generated by Gradle"/>
|
<sha256 value="6ed1ab2a239c5954b074dd8b75e70dbba55866840e73c53ae8b1f99a35afb7b5" origin="Generated by Gradle"/>
|
||||||
</artifact>
|
</artifact>
|
||||||
</component>
|
</component>
|
||||||
|
<component group="org.apache.maven" name="maven-builder-support" version="3.9.16">
|
||||||
|
<artifact name="maven-builder-support-3.9.16.jar">
|
||||||
|
<sha256 value="02972384eae3495801565fd27abb84cfd75a8e6d2cbb1ae0c556752ebc2b1cde" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
<artifact name="maven-builder-support-3.9.16.pom">
|
||||||
|
<sha256 value="c8366af883eeec0e2fd13e14cd8245969284b6e66df131f7b7c03d270f72f613" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
</component>
|
||||||
<component group="org.apache.maven" name="maven-core" version="3.9.14">
|
<component group="org.apache.maven" name="maven-core" version="3.9.14">
|
||||||
<artifact name="maven-core-3.9.14.jar">
|
<artifact name="maven-core-3.9.14.jar">
|
||||||
<sha256 value="db009d57b90a714efe86c81c7a518febb276d01ba3daf2f301e382f6560e8a58" origin="Generated by Gradle"/>
|
<sha256 value="db009d57b90a714efe86c81c7a518febb276d01ba3daf2f301e382f6560e8a58" origin="Generated by Gradle"/>
|
||||||
@@ -1019,6 +1084,14 @@
|
|||||||
<sha256 value="a7967fb392197e5fa73c7b7c3fb728f77fc50f4ad7f03679c0bbabee5c0132b3" origin="Generated by Gradle"/>
|
<sha256 value="a7967fb392197e5fa73c7b7c3fb728f77fc50f4ad7f03679c0bbabee5c0132b3" origin="Generated by Gradle"/>
|
||||||
</artifact>
|
</artifact>
|
||||||
</component>
|
</component>
|
||||||
|
<component group="org.apache.maven" name="maven-core" version="3.9.16">
|
||||||
|
<artifact name="maven-core-3.9.16.jar">
|
||||||
|
<sha256 value="5d45c72e3dbfab8b68d15ad4f12777b7d9b5fe4d4adc99c3bd51fb9641fab009" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
<artifact name="maven-core-3.9.16.pom">
|
||||||
|
<sha256 value="186f17628c5235d03e34c593122d05fdc1be9694440a54d8213b3f957d6379a4" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
</component>
|
||||||
<component group="org.apache.maven" name="maven-model" version="3.9.14">
|
<component group="org.apache.maven" name="maven-model" version="3.9.14">
|
||||||
<artifact name="maven-model-3.9.14.jar">
|
<artifact name="maven-model-3.9.14.jar">
|
||||||
<sha256 value="684f573b1b37933c5d62c1c21d5da4335b049fb8b3d9754281597a38dbbc4044" origin="Generated by Gradle"/>
|
<sha256 value="684f573b1b37933c5d62c1c21d5da4335b049fb8b3d9754281597a38dbbc4044" origin="Generated by Gradle"/>
|
||||||
@@ -1027,6 +1100,14 @@
|
|||||||
<sha256 value="e999c4ae0f12bff7585bffcaf5b0e6cb69226f27d21d5421fa4ae4e76779152a" origin="Generated by Gradle"/>
|
<sha256 value="e999c4ae0f12bff7585bffcaf5b0e6cb69226f27d21d5421fa4ae4e76779152a" origin="Generated by Gradle"/>
|
||||||
</artifact>
|
</artifact>
|
||||||
</component>
|
</component>
|
||||||
|
<component group="org.apache.maven" name="maven-model" version="3.9.16">
|
||||||
|
<artifact name="maven-model-3.9.16.jar">
|
||||||
|
<sha256 value="f59d86a507c241bf17bbf050050d1b8b9c0d34d5010a7e2386c3cab7c83b93de" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
<artifact name="maven-model-3.9.16.pom">
|
||||||
|
<sha256 value="64883ffcfd91ddaadb4181740236a823bfe0880e0c20f5beee74bc32a93914a7" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
</component>
|
||||||
<component group="org.apache.maven" name="maven-model-builder" version="3.9.14">
|
<component group="org.apache.maven" name="maven-model-builder" version="3.9.14">
|
||||||
<artifact name="maven-model-builder-3.9.14.jar">
|
<artifact name="maven-model-builder-3.9.14.jar">
|
||||||
<sha256 value="9a6f4deb11bd6fe3f8b11036ed46f34cded3b00fc638f242327537bfb53c9d3f" origin="Generated by Gradle"/>
|
<sha256 value="9a6f4deb11bd6fe3f8b11036ed46f34cded3b00fc638f242327537bfb53c9d3f" origin="Generated by Gradle"/>
|
||||||
@@ -1035,6 +1116,14 @@
|
|||||||
<sha256 value="7548856c413b3ef5f3d35c2e812beae847dd21809f8e5d3621d5da2d7bdcfe6f" origin="Generated by Gradle"/>
|
<sha256 value="7548856c413b3ef5f3d35c2e812beae847dd21809f8e5d3621d5da2d7bdcfe6f" origin="Generated by Gradle"/>
|
||||||
</artifact>
|
</artifact>
|
||||||
</component>
|
</component>
|
||||||
|
<component group="org.apache.maven" name="maven-model-builder" version="3.9.16">
|
||||||
|
<artifact name="maven-model-builder-3.9.16.jar">
|
||||||
|
<sha256 value="002be86d1f53b36f559a0786034a17598d71087f1935f205a1f9e51d4dd124b3" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
<artifact name="maven-model-builder-3.9.16.pom">
|
||||||
|
<sha256 value="6e2a59eadd77b244fe60fc60b7ed3a9b2dee704d44a4f5baca4f9d3be6e53e10" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
</component>
|
||||||
<component group="org.apache.maven" name="maven-parent" version="39">
|
<component group="org.apache.maven" name="maven-parent" version="39">
|
||||||
<artifact name="maven-parent-39.pom">
|
<artifact name="maven-parent-39.pom">
|
||||||
<sha256 value="cfe4820aa1d96ae51d1dc5b0e2a9dc582c42478c24c95ca8238f547e60bef721" origin="Generated by Gradle"/>
|
<sha256 value="cfe4820aa1d96ae51d1dc5b0e2a9dc582c42478c24c95ca8238f547e60bef721" origin="Generated by Gradle"/>
|
||||||
@@ -1045,6 +1134,11 @@
|
|||||||
<sha256 value="82d0112ba1907ff5fd13a2485829c97df66c6a81e075359a561a422f7d1582d3" origin="Generated by Gradle"/>
|
<sha256 value="82d0112ba1907ff5fd13a2485829c97df66c6a81e075359a561a422f7d1582d3" origin="Generated by Gradle"/>
|
||||||
</artifact>
|
</artifact>
|
||||||
</component>
|
</component>
|
||||||
|
<component group="org.apache.maven" name="maven-parent" version="48">
|
||||||
|
<artifact name="maven-parent-48.pom">
|
||||||
|
<sha256 value="cc9eed84b90a96cbc33aefecc93facb9a49f960ad678909162c579356cfe12c9" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
</component>
|
||||||
<component group="org.apache.maven" name="maven-plugin-api" version="3.9.14">
|
<component group="org.apache.maven" name="maven-plugin-api" version="3.9.14">
|
||||||
<artifact name="maven-plugin-api-3.9.14.jar">
|
<artifact name="maven-plugin-api-3.9.14.jar">
|
||||||
<sha256 value="062445ef3ae988e245cca68ccec915de64703ec615badf2d9d61da9ee6f1a245" origin="Generated by Gradle"/>
|
<sha256 value="062445ef3ae988e245cca68ccec915de64703ec615badf2d9d61da9ee6f1a245" origin="Generated by Gradle"/>
|
||||||
@@ -1053,6 +1147,14 @@
|
|||||||
<sha256 value="754d855dfe4c605400620b0c7ef8708c9a413ef629b07f214767ebb15ab7f99a" origin="Generated by Gradle"/>
|
<sha256 value="754d855dfe4c605400620b0c7ef8708c9a413ef629b07f214767ebb15ab7f99a" origin="Generated by Gradle"/>
|
||||||
</artifact>
|
</artifact>
|
||||||
</component>
|
</component>
|
||||||
|
<component group="org.apache.maven" name="maven-plugin-api" version="3.9.16">
|
||||||
|
<artifact name="maven-plugin-api-3.9.16.jar">
|
||||||
|
<sha256 value="37cb5e483e23327cf4ba18f920b45e000d20eec0428a086a5c1bd6bbdecf088c" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
<artifact name="maven-plugin-api-3.9.16.pom">
|
||||||
|
<sha256 value="6fe60dbb3157b9466a6b18f70a4c6eca7f68978eaf015ce164bb2471e1acbc12" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
</component>
|
||||||
<component group="org.apache.maven" name="maven-repository-metadata" version="3.9.14">
|
<component group="org.apache.maven" name="maven-repository-metadata" version="3.9.14">
|
||||||
<artifact name="maven-repository-metadata-3.9.14.jar">
|
<artifact name="maven-repository-metadata-3.9.14.jar">
|
||||||
<sha256 value="307e4920b17a8fdd764b556a7569de9bdd384d6b5f3f6a63dfb2ef186f02da2a" origin="Generated by Gradle"/>
|
<sha256 value="307e4920b17a8fdd764b556a7569de9bdd384d6b5f3f6a63dfb2ef186f02da2a" origin="Generated by Gradle"/>
|
||||||
@@ -1061,6 +1163,14 @@
|
|||||||
<sha256 value="9c399dc0a741c5023a28a2a38d4a49a80059078431be99b3d4571831c3915fdc" origin="Generated by Gradle"/>
|
<sha256 value="9c399dc0a741c5023a28a2a38d4a49a80059078431be99b3d4571831c3915fdc" origin="Generated by Gradle"/>
|
||||||
</artifact>
|
</artifact>
|
||||||
</component>
|
</component>
|
||||||
|
<component group="org.apache.maven" name="maven-repository-metadata" version="3.9.16">
|
||||||
|
<artifact name="maven-repository-metadata-3.9.16.jar">
|
||||||
|
<sha256 value="bc3dd413b89a16b695f35a5d0496b58ea2787c30b7b6062c130d24d6d764720f" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
<artifact name="maven-repository-metadata-3.9.16.pom">
|
||||||
|
<sha256 value="1c018bbd6cd513b43df5dd64c82579b14c414d35f99e06a0c3878b05ae070912" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
</component>
|
||||||
<component group="org.apache.maven" name="maven-resolver-provider" version="3.9.14">
|
<component group="org.apache.maven" name="maven-resolver-provider" version="3.9.14">
|
||||||
<artifact name="maven-resolver-provider-3.9.14.jar">
|
<artifact name="maven-resolver-provider-3.9.14.jar">
|
||||||
<sha256 value="a5bc340ffe35325c55a01762feee374abb2a9e2b14191ea39c1ef646c2754f65" origin="Generated by Gradle"/>
|
<sha256 value="a5bc340ffe35325c55a01762feee374abb2a9e2b14191ea39c1ef646c2754f65" origin="Generated by Gradle"/>
|
||||||
@@ -1069,6 +1179,14 @@
|
|||||||
<sha256 value="355db0ea3f355a1c575523a13e31f8b9fdace732474c74afae6ad90fce722e33" origin="Generated by Gradle"/>
|
<sha256 value="355db0ea3f355a1c575523a13e31f8b9fdace732474c74afae6ad90fce722e33" origin="Generated by Gradle"/>
|
||||||
</artifact>
|
</artifact>
|
||||||
</component>
|
</component>
|
||||||
|
<component group="org.apache.maven" name="maven-resolver-provider" version="3.9.16">
|
||||||
|
<artifact name="maven-resolver-provider-3.9.16.jar">
|
||||||
|
<sha256 value="72a2d6aad3708e2c708b659b292ae9587a4d170ec88f48466290318730221118" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
<artifact name="maven-resolver-provider-3.9.16.pom">
|
||||||
|
<sha256 value="33723bba45ff2a1581bc2840836b7d9c9aa262af676f4f61222e179b56ea8818" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
</component>
|
||||||
<component group="org.apache.maven" name="maven-settings" version="3.9.14">
|
<component group="org.apache.maven" name="maven-settings" version="3.9.14">
|
||||||
<artifact name="maven-settings-3.9.14.jar">
|
<artifact name="maven-settings-3.9.14.jar">
|
||||||
<sha256 value="0e5492e07136565b1ef72a981e99797999183e54f589e00aa608cbadc4cb9bda" origin="Generated by Gradle"/>
|
<sha256 value="0e5492e07136565b1ef72a981e99797999183e54f589e00aa608cbadc4cb9bda" origin="Generated by Gradle"/>
|
||||||
@@ -1077,6 +1195,14 @@
|
|||||||
<sha256 value="a49d6f6434b40c1ef8b63dccf8615aa9505d236d0e9fe6045e5f9513a42aadbe" origin="Generated by Gradle"/>
|
<sha256 value="a49d6f6434b40c1ef8b63dccf8615aa9505d236d0e9fe6045e5f9513a42aadbe" origin="Generated by Gradle"/>
|
||||||
</artifact>
|
</artifact>
|
||||||
</component>
|
</component>
|
||||||
|
<component group="org.apache.maven" name="maven-settings" version="3.9.16">
|
||||||
|
<artifact name="maven-settings-3.9.16.jar">
|
||||||
|
<sha256 value="322ae5b23f4b7b6ce7896bd33a793143dbbbbcc16c4475e2d7eebbd8f755da92" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
<artifact name="maven-settings-3.9.16.pom">
|
||||||
|
<sha256 value="d3df811fe57832933adef90c119cbf3abb941884cf840cc9777aa639cf1a145d" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
</component>
|
||||||
<component group="org.apache.maven" name="maven-settings-builder" version="3.9.14">
|
<component group="org.apache.maven" name="maven-settings-builder" version="3.9.14">
|
||||||
<artifact name="maven-settings-builder-3.9.14.jar">
|
<artifact name="maven-settings-builder-3.9.14.jar">
|
||||||
<sha256 value="1d3cf59f9dc6af77f7a1052aea598535b4f59926a9fe92cecce3ecfb5b91ff1f" origin="Generated by Gradle"/>
|
<sha256 value="1d3cf59f9dc6af77f7a1052aea598535b4f59926a9fe92cecce3ecfb5b91ff1f" origin="Generated by Gradle"/>
|
||||||
@@ -1085,6 +1211,14 @@
|
|||||||
<sha256 value="8ca00532860ab13c7b5398df095fc3820c223955c32974c5d5b6bc6e45357c61" origin="Generated by Gradle"/>
|
<sha256 value="8ca00532860ab13c7b5398df095fc3820c223955c32974c5d5b6bc6e45357c61" origin="Generated by Gradle"/>
|
||||||
</artifact>
|
</artifact>
|
||||||
</component>
|
</component>
|
||||||
|
<component group="org.apache.maven" name="maven-settings-builder" version="3.9.16">
|
||||||
|
<artifact name="maven-settings-builder-3.9.16.jar">
|
||||||
|
<sha256 value="226f1cbb0b4d414eff773fb151a8977837c98073e14d5963794e6cf464e6adbb" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
<artifact name="maven-settings-builder-3.9.16.pom">
|
||||||
|
<sha256 value="bce946240bf297982f524666674acff5c14878a24849f1f3c3489cd1d829ed48" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
</component>
|
||||||
<component group="org.apache.maven.resolver" name="maven-resolver" version="1.9.27">
|
<component group="org.apache.maven.resolver" name="maven-resolver" version="1.9.27">
|
||||||
<artifact name="maven-resolver-1.9.27.pom">
|
<artifact name="maven-resolver-1.9.27.pom">
|
||||||
<sha256 value="8924b41711cce058f83c46d79ad83d6e04edcf13ed5631ec11135656e0b87f56" origin="Generated by Gradle"/>
|
<sha256 value="8924b41711cce058f83c46d79ad83d6e04edcf13ed5631ec11135656e0b87f56" origin="Generated by Gradle"/>
|
||||||
@@ -1244,6 +1378,11 @@
|
|||||||
<sha256 value="89a1bc79e46c35ab108b7e215bb2c5c215ff8f3af1ae3cfef82d9a2b33b06c51" origin="Generated by Gradle"/>
|
<sha256 value="89a1bc79e46c35ab108b7e215bb2c5c215ff8f3af1ae3cfef82d9a2b33b06c51" origin="Generated by Gradle"/>
|
||||||
</artifact>
|
</artifact>
|
||||||
</component>
|
</component>
|
||||||
|
<component group="org.codehaus.plexus" name="plexus" version="25">
|
||||||
|
<artifact name="plexus-25.pom">
|
||||||
|
<sha256 value="faa7947c2020967ad0c92b259ee9fa361d05e90cd036d17c37098bb1edaea3a3" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
</component>
|
||||||
<component group="org.codehaus.plexus" name="plexus" version="8">
|
<component group="org.codehaus.plexus" name="plexus" version="8">
|
||||||
<artifact name="plexus-8.pom">
|
<artifact name="plexus-8.pom">
|
||||||
<sha256 value="ffa349db04e7abf65885bdc5a2062f4197c0ff9d3f1f4e2aa5720b77233f742c" origin="Generated by Gradle"/>
|
<sha256 value="ffa349db04e7abf65885bdc5a2062f4197c0ff9d3f1f4e2aa5720b77233f742c" origin="Generated by Gradle"/>
|
||||||
@@ -1257,6 +1396,14 @@
|
|||||||
<sha256 value="04842f331b0225b85a5e20439710d228ea7a6302abe6d53c9c9846fbc5bf99ff" origin="Generated by Gradle"/>
|
<sha256 value="04842f331b0225b85a5e20439710d228ea7a6302abe6d53c9c9846fbc5bf99ff" origin="Generated by Gradle"/>
|
||||||
</artifact>
|
</artifact>
|
||||||
</component>
|
</component>
|
||||||
|
<component group="org.codehaus.plexus" name="plexus-classworlds" version="2.11.0">
|
||||||
|
<artifact name="plexus-classworlds-2.11.0.jar">
|
||||||
|
<sha256 value="8971f135490070bc5fde7413fcc8db7c997fda4bebfb5c31185900d66edcbbb2" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
<artifact name="plexus-classworlds-2.11.0.pom">
|
||||||
|
<sha256 value="281d317bf8a5fe818708cdd00e377dd234ec949f498d30f4b363f6b9771e1fa2" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
</component>
|
||||||
<component group="org.codehaus.plexus" name="plexus-classworlds" version="2.9.0">
|
<component group="org.codehaus.plexus" name="plexus-classworlds" version="2.9.0">
|
||||||
<artifact name="plexus-classworlds-2.9.0.jar">
|
<artifact name="plexus-classworlds-2.9.0.jar">
|
||||||
<sha256 value="1ad3292cd563381e3fd632f3fded1988f9e9b2be7a9f3db63ff4c4cedba13fa5" origin="Generated by Gradle"/>
|
<sha256 value="1ad3292cd563381e3fd632f3fded1988f9e9b2be7a9f3db63ff4c4cedba13fa5" origin="Generated by Gradle"/>
|
||||||
@@ -1302,6 +1449,14 @@
|
|||||||
<sha256 value="6138300481471c7fe6aeb115f912961f886e1a46ee9c2bd2841b65184824da28" origin="Generated by Gradle"/>
|
<sha256 value="6138300481471c7fe6aeb115f912961f886e1a46ee9c2bd2841b65184824da28" origin="Generated by Gradle"/>
|
||||||
</artifact>
|
</artifact>
|
||||||
</component>
|
</component>
|
||||||
|
<component group="org.codehaus.plexus" name="plexus-utils" version="3.6.1">
|
||||||
|
<artifact name="plexus-utils-3.6.1.jar">
|
||||||
|
<sha256 value="05a63effd67e2d6b9d610cc82e2bd7473289d34802e57a529b28110f28af5679" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
<artifact name="plexus-utils-3.6.1.pom">
|
||||||
|
<sha256 value="c8397373781af640a76c5da88f1674293b4fc9a2391d0768ee3fc791883b040d" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
</component>
|
||||||
<component group="org.codehaus.woodstox" name="stax2-api" version="4.2.2">
|
<component group="org.codehaus.woodstox" name="stax2-api" version="4.2.2">
|
||||||
<artifact name="stax2-api-4.2.2.jar">
|
<artifact name="stax2-api-4.2.2.jar">
|
||||||
<sha256 value="a61c48d553efad78bc01fffc4ac528bebbae64cbaec170b2a5e39cf61eb51abe" origin="Generated by Gradle"/>
|
<sha256 value="a61c48d553efad78bc01fffc4ac528bebbae64cbaec170b2a5e39cf61eb51abe" origin="Generated by Gradle"/>
|
||||||
@@ -1326,11 +1481,24 @@
|
|||||||
<sha256 value="efe3734bc5b5e390b7ddd5cc7e86a5aca1a0377534e3420962f0931327c88d10" origin="Generated by Gradle"/>
|
<sha256 value="efe3734bc5b5e390b7ddd5cc7e86a5aca1a0377534e3420962f0931327c88d10" origin="Generated by Gradle"/>
|
||||||
</artifact>
|
</artifact>
|
||||||
</component>
|
</component>
|
||||||
|
<component group="org.cyclonedx" name="cyclonedx-gradle-plugin" version="3.3.0">
|
||||||
|
<artifact name="cyclonedx-gradle-plugin-3.3.0.jar">
|
||||||
|
<sha256 value="9bf283e7e451cedf536b263733cf4ddca2329b3cffe19a05ccb8cc1f90a098e8" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
<artifact name="cyclonedx-gradle-plugin-3.3.0.module">
|
||||||
|
<sha256 value="92c20482c05782eec05b9c0db1b2ed615160147e52c179c3e080538103e26239" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
</component>
|
||||||
<component group="org.cyclonedx.bom" name="org.cyclonedx.bom.gradle.plugin" version="3.2.4">
|
<component group="org.cyclonedx.bom" name="org.cyclonedx.bom.gradle.plugin" version="3.2.4">
|
||||||
<artifact name="org.cyclonedx.bom.gradle.plugin-3.2.4.pom">
|
<artifact name="org.cyclonedx.bom.gradle.plugin-3.2.4.pom">
|
||||||
<sha256 value="9a8e381d2369288b6c3198b3062e8099229abddafd0a49beb631fd999ea07b9a" origin="Generated by Gradle"/>
|
<sha256 value="9a8e381d2369288b6c3198b3062e8099229abddafd0a49beb631fd999ea07b9a" origin="Generated by Gradle"/>
|
||||||
</artifact>
|
</artifact>
|
||||||
</component>
|
</component>
|
||||||
|
<component group="org.cyclonedx.bom" name="org.cyclonedx.bom.gradle.plugin" version="3.3.0">
|
||||||
|
<artifact name="org.cyclonedx.bom.gradle.plugin-3.3.0.pom">
|
||||||
|
<sha256 value="f59df2c670269e7f5e3d9b2b539b9f435db5e88a2d545ab4811f2217d2cc5c68" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
</component>
|
||||||
<component group="org.eclipse.ee4j" name="project" version="1.0.5">
|
<component group="org.eclipse.ee4j" name="project" version="1.0.5">
|
||||||
<artifact name="project-1.0.5.pom">
|
<artifact name="project-1.0.5.pom">
|
||||||
<sha256 value="916b4794d8d8220a59a3fdf6a64dbe794aeb23395e888b81ae36a9b5a2c591a6" origin="Generated by Gradle"/>
|
<sha256 value="916b4794d8d8220a59a3fdf6a64dbe794aeb23395e888b81ae36a9b5a2c591a6" origin="Generated by Gradle"/>
|
||||||
@@ -1558,6 +1726,14 @@
|
|||||||
<sha256 value="08a02856e487c9357f9b29e38745f8ae805848111e72d15aad0352338f1632e1" origin="Generated by Gradle"/>
|
<sha256 value="08a02856e487c9357f9b29e38745f8ae805848111e72d15aad0352338f1632e1" origin="Generated by Gradle"/>
|
||||||
</artifact>
|
</artifact>
|
||||||
</component>
|
</component>
|
||||||
|
<component group="org.junit" name="junit-bom" version="5.14.4">
|
||||||
|
<artifact name="junit-bom-5.14.4.module">
|
||||||
|
<sha256 value="8a5e98d131de7d7aadb1ee88bfd86d66e62a7c2e2a4074a3b2498b03d236eb64" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
<artifact name="junit-bom-5.14.4.pom">
|
||||||
|
<sha256 value="5706e8f29a0a07f56efbbea4a0670793414194bb8d24d8143ba1e787a2f32856" origin="Generated by Gradle"/>
|
||||||
|
</artifact>
|
||||||
|
</component>
|
||||||
<component group="org.junit" name="junit-bom" version="5.9.3">
|
<component group="org.junit" name="junit-bom" version="5.9.3">
|
||||||
<artifact name="junit-bom-5.9.3.module">
|
<artifact name="junit-bom-5.9.3.module">
|
||||||
<sha256 value="b401fd25901e582a524aa5343c4b39e28bc56e24961c1069bf2b4bbfcee46b93" origin="Generated by Gradle"/>
|
<sha256 value="b401fd25901e582a524aa5343c4b39e28bc56e24961c1069bf2b4bbfcee46b93" origin="Generated by Gradle"/>
|
||||||
|
|||||||
@@ -45,6 +45,7 @@ nav:
|
|||||||
|
|
||||||
- Integration:
|
- Integration:
|
||||||
- Overview: programmatic-usage.md
|
- Overview: programmatic-usage.md
|
||||||
|
- Model Selection and Loading: model-selection-and-loading.md
|
||||||
- Loading and Building Stemmers: programmatic-loading-and-building.md
|
- Loading and Building Stemmers: programmatic-loading-and-building.md
|
||||||
- Querying and Ambiguity Handling: programmatic-querying-and-ambiguity.md
|
- Querying and Ambiguity Handling: programmatic-querying-and-ambiguity.md
|
||||||
- Extending and Persisting Compiled Tries: programmatic-extending-and-persistence.md
|
- Extending and Persisting Compiled Tries: programmatic-extending-and-persistence.md
|
||||||
@@ -52,6 +53,8 @@ nav:
|
|||||||
- CLI Compilation: cli-compilation.md
|
- CLI Compilation: cli-compilation.md
|
||||||
|
|
||||||
- Dictionaries and Languages:
|
- Dictionaries and Languages:
|
||||||
|
- Stemmer Models: stemmer-models.md
|
||||||
|
- Published Model Catalog: stemmer-model-catalog.md
|
||||||
- Built-in Languages: built-in-languages.md
|
- Built-in Languages: built-in-languages.md
|
||||||
- Dictionary Format: dictionary-format.md
|
- Dictionary Format: dictionary-format.md
|
||||||
- Contributing Dictionaries: contributing-dictionaries.md
|
- Contributing Dictionaries: contributing-dictionaries.md
|
||||||
@@ -68,6 +71,9 @@ nav:
|
|||||||
- Benchmark Results: benchmarks/index.md
|
- Benchmark Results: benchmarks/index.md
|
||||||
- Reference:
|
- Reference:
|
||||||
- Methodology: benchmarks/reference/methodology.md
|
- Methodology: benchmarks/reference/methodology.md
|
||||||
|
- Linguistic Quality Methodology: benchmarks/reference/linguistic-quality.md
|
||||||
|
- Tested Stemmers: benchmarks/reference/tested-stemmers.md
|
||||||
|
- Reproducibility and Raw Data: benchmarks/reference/reproducibility.md
|
||||||
- Corpora: benchmarks/reference/corpora.md
|
- Corpora: benchmarks/reference/corpora.md
|
||||||
- Environment and Reports: benchmarks/reference/environment.md
|
- Environment and Reports: benchmarks/reference/environment.md
|
||||||
- English Dictionary Coverage: benchmarks/reference/english-coverage.md
|
- English Dictionary Coverage: benchmarks/reference/english-coverage.md
|
||||||
@@ -96,5 +102,7 @@ nav:
|
|||||||
|
|
||||||
- Quality and Operations:
|
- Quality and Operations:
|
||||||
- Quality and Operations: quality-and-operations.md
|
- Quality and Operations: quality-and-operations.md
|
||||||
|
- Stemming Quality: stemming-quality.md
|
||||||
- Reports: reports.md
|
- Reports: reports.md
|
||||||
|
- Historical Builds: builds.md
|
||||||
- Test taxonomy and execution filtering: test-taxonomy-and-filtering.md
|
- Test taxonomy and execution filtering: test-taxonomy-and-filtering.md
|
||||||
|
|||||||
101
models/bom/build.gradle
Normal file
101
models/bom/build.gradle
Normal file
@@ -0,0 +1,101 @@
|
|||||||
|
import groovy.xml.XmlParser
|
||||||
|
|
||||||
|
plugins {
|
||||||
|
id 'java-platform'
|
||||||
|
id 'maven-publish'
|
||||||
|
id 'signing'
|
||||||
|
}
|
||||||
|
|
||||||
|
group = 'org.egothor'
|
||||||
|
version = providers.fileContents(rootProject.layout.projectDirectory.file('models/catalog-version.txt'))
|
||||||
|
.asText.map(String::trim).get()
|
||||||
|
|
||||||
|
Properties modelTopology = new Properties()
|
||||||
|
rootProject.file('models/model-projects.properties').withInputStream { InputStream input ->
|
||||||
|
modelTopology.load(input)
|
||||||
|
}
|
||||||
|
List<String> modelIds = modelTopology.stringPropertyNames().toList().sort()
|
||||||
|
|
||||||
|
dependencies {
|
||||||
|
constraints {
|
||||||
|
modelIds.each { String modelId ->
|
||||||
|
api project(":models:${modelId}")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
publishing {
|
||||||
|
publications {
|
||||||
|
bom(MavenPublication) {
|
||||||
|
from components.javaPlatform
|
||||||
|
artifactId = 'radixor-models-bom'
|
||||||
|
pom {
|
||||||
|
name = 'Radixor Stemmer Models BOM'
|
||||||
|
description = 'Maven dependency-management BOM containing recommended versions for published Radixor models.'
|
||||||
|
packaging = 'pom'
|
||||||
|
url = 'https://github.com/leogalambos/Radixor'
|
||||||
|
licenses {
|
||||||
|
license {
|
||||||
|
name = 'BSD-3-Clause'
|
||||||
|
url = 'https://spdx.org/licenses/BSD-3-Clause.html'
|
||||||
|
distribution = 'repo'
|
||||||
|
}
|
||||||
|
}
|
||||||
|
developers {
|
||||||
|
developer {
|
||||||
|
id = 'egothor'
|
||||||
|
name = 'Leo Galambos'
|
||||||
|
email = 'egothor@gmail.com'
|
||||||
|
}
|
||||||
|
}
|
||||||
|
scm {
|
||||||
|
url = 'https://github.com/leogalambos/Radixor'
|
||||||
|
connection = 'scm:git:https://github.com/leogalambos/Radixor.git'
|
||||||
|
developerConnection = 'scm:git:ssh://git@github.com/leogalambos/Radixor.git'
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
repositories {
|
||||||
|
maven {
|
||||||
|
name = 'catalogStaging'
|
||||||
|
url = rootProject.layout.buildDirectory.dir('model-catalog-staging-repository').get().asFile.toURI()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
String signingKey = providers.environmentVariable('SIGNING_KEY').orNull
|
||||||
|
String signingPassword = providers.environmentVariable('SIGNING_PASSWORD').orNull
|
||||||
|
signing {
|
||||||
|
required = { providers.environmentVariable('GITHUB_REF_TYPE').orNull == 'tag' }
|
||||||
|
if (signingKey != null && !signingKey.isBlank()) {
|
||||||
|
useInMemoryPgpKeys(signingKey, signingPassword)
|
||||||
|
sign publishing.publications.bom
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.register('verifyPomOnlyPlatform') {
|
||||||
|
group = 'verification'
|
||||||
|
description = 'Verifies the POM-only model dependency-management platform.'
|
||||||
|
dependsOn(tasks.named('generatePomFileForBomPublication'))
|
||||||
|
doLast {
|
||||||
|
File pomFile = layout.buildDirectory.file('publications/bom/pom-default.xml').get().asFile
|
||||||
|
Node pom = new XmlParser().parse(pomFile)
|
||||||
|
List<Node> constraints = pom.dependencyManagement.dependencies.dependency as List<Node>
|
||||||
|
List<String> artifactIds = constraints.collect { Node dependency -> dependency.artifactId.text() }
|
||||||
|
List<String> expected = modelIds.collect { String modelId -> "radixor-model-${modelId}" }
|
||||||
|
if (pom.packaging.text() != 'pom' || artifactIds != expected) {
|
||||||
|
throw new GradleException('radixor-models-bom must publish exactly the ordered model constraints as Maven packaging pom.')
|
||||||
|
}
|
||||||
|
if (!pom.dependencies.isEmpty()) {
|
||||||
|
throw new GradleException('radixor-models-bom must not introduce runtime model dependencies.')
|
||||||
|
}
|
||||||
|
if (!tasks.withType(Jar).isEmpty()) {
|
||||||
|
throw new GradleException('radixor-models-bom must not create binary, sources, or Javadoc JARs.')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
tasks.named('check') {
|
||||||
|
dependsOn(tasks.named('verifyPomOnlyPlatform'))
|
||||||
|
}
|
||||||
1
models/catalog-version.txt
Normal file
1
models/catalog-version.txt
Normal file
@@ -0,0 +1 @@
|
|||||||
|
2026.1
|
||||||
23
models/cs-cz-default/build.gradle
Normal file
23
models/cs-cz-default/build.gradle
Normal file
@@ -0,0 +1,23 @@
|
|||||||
|
plugins {
|
||||||
|
id 'org.egothor.radixor.model'
|
||||||
|
}
|
||||||
|
|
||||||
|
radixorModel {
|
||||||
|
modelId = 'cs-cz-default'
|
||||||
|
language = 'CS_CZ'
|
||||||
|
displayName = 'Czech default model'
|
||||||
|
defaultModel = true
|
||||||
|
sourceName = 'UniMorph'
|
||||||
|
sourceVersion = 'not-recorded-in-legacy-import'
|
||||||
|
sourceRevision = 'not-recorded-in-legacy-import'
|
||||||
|
sourceProject = 'UniMorph'
|
||||||
|
sourceRepository = 'https://github.com/unimorph/ces'
|
||||||
|
sourceDataset = 'UniMorph Czech morphological dataset (`ces`); repository also documents non-distributed MorfFlex-CZ data'
|
||||||
|
sourceRevisionStatus = 'not-recorded-in-legacy-import'
|
||||||
|
sourceLicense = 'CC-BY-SA-3.0'
|
||||||
|
sourceLicenseUri = 'https://creativecommons.org/licenses/by-sa/3.0/'
|
||||||
|
sourceAttribution = 'UniMorph; Witold Kieraś is credited for the separate MorfFlex-CZ conversion'
|
||||||
|
sourceVerificationDate = '2026-07-22'
|
||||||
|
transformationsSummary = 'Cleaning, normalization, grouping inflected forms by lemma, deduplication, filtering invalid rows, reformatting into Radixor dictionary groups, GZip packaging, and generation of runtime descriptor and checksum metadata'
|
||||||
|
noticeFileName = 'NOTICE-model-data.txt'
|
||||||
|
}
|
||||||
1
models/cs-cz-default/model-version.txt
Normal file
1
models/cs-cz-default/model-version.txt
Normal file
@@ -0,0 +1 @@
|
|||||||
|
1.0.0
|
||||||
38
models/cs-cz-default/src/modelInput/NOTICE-model-data.txt
Normal file
38
models/cs-cz-default/src/modelInput/NOTICE-model-data.txt
Normal file
@@ -0,0 +1,38 @@
|
|||||||
|
Radixor model-data notice
|
||||||
|
|
||||||
|
Radixor-derived model data
|
||||||
|
|
||||||
|
Copyright (C) 2026, Leo Galambos.
|
||||||
|
|
||||||
|
Copyright and, where applicable, database rights are claimed in the
|
||||||
|
Radixor-specific selection, verification, cleaning, normalization,
|
||||||
|
grouping, deduplication, filtering, reformatting, metadata preparation,
|
||||||
|
and packaging of this model, to the extent protected by applicable law.
|
||||||
|
|
||||||
|
The underlying morphological data remains attributed to UniMorph and
|
||||||
|
the upstream contributors identified in this notice.
|
||||||
|
|
||||||
|
This derived model data, including Radixor's protectable contributions,
|
||||||
|
is distributed under Creative Commons Attribution-ShareAlike 3.0
|
||||||
|
Unported.
|
||||||
|
|
||||||
|
Model ID: cs-cz-default
|
||||||
|
Radixor language: CS_CZ
|
||||||
|
Source project: UniMorph
|
||||||
|
Official repository: https://github.com/unimorph/ces
|
||||||
|
Upstream dataset: UniMorph Czech morphological dataset (`ces`); repository also documents non-distributed MorfFlex-CZ data
|
||||||
|
Upstream lexical source: Wiktionary; the CC BY-NC-SA MorfFlex-CZ dataset is excluded
|
||||||
|
Attribution: UniMorph; Witold Kieraś is credited for the separate MorfFlex-CZ conversion
|
||||||
|
License:
|
||||||
|
Creative Commons Attribution-ShareAlike 3.0 Unported
|
||||||
|
Canonical license URI: https://creativecommons.org/licenses/by-sa/3.0/
|
||||||
|
Source revision: not-recorded-in-legacy-import
|
||||||
|
Revision status: not-recorded-in-legacy-import
|
||||||
|
|
||||||
|
The exact UniMorph commit used for the original Radixor import was not recorded. The model remains attributed to the official UniMorph language repository and is distributed under the repository's stated data license.
|
||||||
|
|
||||||
|
Radixor modifications: Cleaning, normalization, grouping inflected forms by lemma, deduplication, filtering invalid rows, reformatting into Radixor dictionary groups, GZip packaging, and generation of runtime descriptor and checksum metadata.
|
||||||
|
|
||||||
|
The derived model data is distributed under CC BY-SA 3.0. UniMorph supplies morphological data; Radixor constructs its own patch-command trie at runtime. Neither UniMorph nor any upstream contributor endorses Radixor.
|
||||||
|
|
||||||
|
Upstream information verified: 2026-07-22
|
||||||
23
models/da-dk-default/build.gradle
Normal file
23
models/da-dk-default/build.gradle
Normal file
@@ -0,0 +1,23 @@
|
|||||||
|
plugins {
|
||||||
|
id 'org.egothor.radixor.model'
|
||||||
|
}
|
||||||
|
|
||||||
|
radixorModel {
|
||||||
|
modelId = 'da-dk-default'
|
||||||
|
language = 'DA_DK'
|
||||||
|
displayName = 'Danish default model'
|
||||||
|
defaultModel = true
|
||||||
|
sourceName = 'UniMorph'
|
||||||
|
sourceVersion = 'not-recorded-in-legacy-import'
|
||||||
|
sourceRevision = 'not-recorded-in-legacy-import'
|
||||||
|
sourceProject = 'UniMorph'
|
||||||
|
sourceRepository = 'https://github.com/unimorph/dan'
|
||||||
|
sourceDataset = 'UniMorph Danish morphological dataset (`dan`)'
|
||||||
|
sourceRevisionStatus = 'not-recorded-in-legacy-import'
|
||||||
|
sourceLicense = 'CC-BY-SA-3.0'
|
||||||
|
sourceLicenseUri = 'https://creativecommons.org/licenses/by-sa/3.0/'
|
||||||
|
sourceAttribution = 'UniMorph and Wikipedia contributors'
|
||||||
|
sourceVerificationDate = '2026-07-22'
|
||||||
|
transformationsSummary = 'Cleaning, normalization, grouping inflected forms by lemma, deduplication, filtering invalid rows, reformatting into Radixor dictionary groups, GZip packaging, and generation of runtime descriptor and checksum metadata'
|
||||||
|
noticeFileName = 'NOTICE-model-data.txt'
|
||||||
|
}
|
||||||
1
models/da-dk-default/model-version.txt
Normal file
1
models/da-dk-default/model-version.txt
Normal file
@@ -0,0 +1 @@
|
|||||||
|
1.0.0
|
||||||
38
models/da-dk-default/src/modelInput/NOTICE-model-data.txt
Normal file
38
models/da-dk-default/src/modelInput/NOTICE-model-data.txt
Normal file
@@ -0,0 +1,38 @@
|
|||||||
|
Radixor model-data notice
|
||||||
|
|
||||||
|
Radixor-derived model data
|
||||||
|
|
||||||
|
Copyright (C) 2026, Leo Galambos.
|
||||||
|
|
||||||
|
Copyright and, where applicable, database rights are claimed in the
|
||||||
|
Radixor-specific selection, verification, cleaning, normalization,
|
||||||
|
grouping, deduplication, filtering, reformatting, metadata preparation,
|
||||||
|
and packaging of this model, to the extent protected by applicable law.
|
||||||
|
|
||||||
|
The underlying morphological data remains attributed to UniMorph and
|
||||||
|
the upstream contributors identified in this notice.
|
||||||
|
|
||||||
|
This derived model data, including Radixor's protectable contributions,
|
||||||
|
is distributed under Creative Commons Attribution-ShareAlike 3.0
|
||||||
|
Unported.
|
||||||
|
|
||||||
|
Model ID: da-dk-default
|
||||||
|
Radixor language: DA_DK
|
||||||
|
Source project: UniMorph
|
||||||
|
Official repository: https://github.com/unimorph/dan
|
||||||
|
Upstream dataset: UniMorph Danish morphological dataset (`dan`)
|
||||||
|
Upstream lexical source: Wikipedia
|
||||||
|
Attribution: UniMorph and Wikipedia contributors
|
||||||
|
License:
|
||||||
|
Creative Commons Attribution-ShareAlike 3.0 Unported
|
||||||
|
Canonical license URI: https://creativecommons.org/licenses/by-sa/3.0/
|
||||||
|
Source revision: not-recorded-in-legacy-import
|
||||||
|
Revision status: not-recorded-in-legacy-import
|
||||||
|
|
||||||
|
The exact UniMorph commit used for the original Radixor import was not recorded. The model remains attributed to the official UniMorph language repository and is distributed under the repository's stated data license.
|
||||||
|
|
||||||
|
Radixor modifications: Cleaning, normalization, grouping inflected forms by lemma, deduplication, filtering invalid rows, reformatting into Radixor dictionary groups, GZip packaging, and generation of runtime descriptor and checksum metadata.
|
||||||
|
|
||||||
|
The derived model data is distributed under CC BY-SA 3.0. UniMorph supplies morphological data; Radixor constructs its own patch-command trie at runtime. Neither UniMorph nor any upstream contributor endorses Radixor.
|
||||||
|
|
||||||
|
Upstream information verified: 2026-07-22
|
||||||
23
models/de-de-default/build.gradle
Normal file
23
models/de-de-default/build.gradle
Normal file
@@ -0,0 +1,23 @@
|
|||||||
|
plugins {
|
||||||
|
id 'org.egothor.radixor.model'
|
||||||
|
}
|
||||||
|
|
||||||
|
radixorModel {
|
||||||
|
modelId = 'de-de-default'
|
||||||
|
language = 'DE_DE'
|
||||||
|
displayName = 'German default model'
|
||||||
|
defaultModel = true
|
||||||
|
sourceName = 'UniMorph'
|
||||||
|
sourceVersion = 'not-recorded-in-legacy-import'
|
||||||
|
sourceRevision = 'not-recorded-in-legacy-import'
|
||||||
|
sourceProject = 'UniMorph'
|
||||||
|
sourceRepository = 'https://github.com/unimorph/deu'
|
||||||
|
sourceDataset = 'UniMorph German morphological dataset (`deu`)'
|
||||||
|
sourceRevisionStatus = 'not-recorded-in-legacy-import'
|
||||||
|
sourceLicense = 'CC-BY-SA-3.0'
|
||||||
|
sourceLicenseUri = 'https://creativecommons.org/licenses/by-sa/3.0/'
|
||||||
|
sourceAttribution = 'UniMorph and English Wiktionary contributors'
|
||||||
|
sourceVerificationDate = '2026-07-22'
|
||||||
|
transformationsSummary = 'Cleaning, normalization, grouping inflected forms by lemma, deduplication, filtering invalid rows, reformatting into Radixor dictionary groups, GZip packaging, and generation of runtime descriptor and checksum metadata'
|
||||||
|
noticeFileName = 'NOTICE-model-data.txt'
|
||||||
|
}
|
||||||
1
models/de-de-default/model-version.txt
Normal file
1
models/de-de-default/model-version.txt
Normal file
@@ -0,0 +1 @@
|
|||||||
|
1.0.0
|
||||||
38
models/de-de-default/src/modelInput/NOTICE-model-data.txt
Normal file
38
models/de-de-default/src/modelInput/NOTICE-model-data.txt
Normal file
@@ -0,0 +1,38 @@
|
|||||||
|
Radixor model-data notice
|
||||||
|
|
||||||
|
Radixor-derived model data
|
||||||
|
|
||||||
|
Copyright (C) 2026, Leo Galambos.
|
||||||
|
|
||||||
|
Copyright and, where applicable, database rights are claimed in the
|
||||||
|
Radixor-specific selection, verification, cleaning, normalization,
|
||||||
|
grouping, deduplication, filtering, reformatting, metadata preparation,
|
||||||
|
and packaging of this model, to the extent protected by applicable law.
|
||||||
|
|
||||||
|
The underlying morphological data remains attributed to UniMorph and
|
||||||
|
the upstream contributors identified in this notice.
|
||||||
|
|
||||||
|
This derived model data, including Radixor's protectable contributions,
|
||||||
|
is distributed under Creative Commons Attribution-ShareAlike 3.0
|
||||||
|
Unported.
|
||||||
|
|
||||||
|
Model ID: de-de-default
|
||||||
|
Radixor language: DE_DE
|
||||||
|
Source project: UniMorph
|
||||||
|
Official repository: https://github.com/unimorph/deu
|
||||||
|
Upstream dataset: UniMorph German morphological dataset (`deu`)
|
||||||
|
Upstream lexical source: English Wiktionary
|
||||||
|
Attribution: UniMorph and English Wiktionary contributors
|
||||||
|
License:
|
||||||
|
Creative Commons Attribution-ShareAlike 3.0 Unported
|
||||||
|
Canonical license URI: https://creativecommons.org/licenses/by-sa/3.0/
|
||||||
|
Source revision: not-recorded-in-legacy-import
|
||||||
|
Revision status: not-recorded-in-legacy-import
|
||||||
|
|
||||||
|
The exact UniMorph commit used for the original Radixor import was not recorded. The model remains attributed to the official UniMorph language repository and is distributed under the repository's stated data license.
|
||||||
|
|
||||||
|
Radixor modifications: Cleaning, normalization, grouping inflected forms by lemma, deduplication, filtering invalid rows, reformatting into Radixor dictionary groups, GZip packaging, and generation of runtime descriptor and checksum metadata.
|
||||||
|
|
||||||
|
The derived model data is distributed under CC BY-SA 3.0. UniMorph supplies morphological data; Radixor constructs its own patch-command trie at runtime. Neither UniMorph nor any upstream contributor endorses Radixor.
|
||||||
|
|
||||||
|
Upstream information verified: 2026-07-22
|
||||||
BIN
models/de-de-default/src/modelInput/stemmer.gz
Normal file
BIN
models/de-de-default/src/modelInput/stemmer.gz
Normal file
Binary file not shown.
23
models/es-es-default/build.gradle
Normal file
23
models/es-es-default/build.gradle
Normal file
@@ -0,0 +1,23 @@
|
|||||||
|
plugins {
|
||||||
|
id 'org.egothor.radixor.model'
|
||||||
|
}
|
||||||
|
|
||||||
|
radixorModel {
|
||||||
|
modelId = 'es-es-default'
|
||||||
|
language = 'ES_ES'
|
||||||
|
displayName = 'Spanish default model'
|
||||||
|
defaultModel = true
|
||||||
|
sourceName = 'UniMorph'
|
||||||
|
sourceVersion = 'not-recorded-in-legacy-import'
|
||||||
|
sourceRevision = 'not-recorded-in-legacy-import'
|
||||||
|
sourceProject = 'UniMorph'
|
||||||
|
sourceRepository = 'https://github.com/unimorph/spa'
|
||||||
|
sourceDataset = 'UniMorph Spanish morphological dataset (`spa`)'
|
||||||
|
sourceRevisionStatus = 'not-recorded-in-legacy-import'
|
||||||
|
sourceLicense = 'CC-BY-SA-3.0'
|
||||||
|
sourceLicenseUri = 'https://creativecommons.org/licenses/by-sa/3.0/'
|
||||||
|
sourceAttribution = 'UniMorph and English Wiktionary contributors'
|
||||||
|
sourceVerificationDate = '2026-07-22'
|
||||||
|
transformationsSummary = 'Cleaning, normalization, grouping inflected forms by lemma, deduplication, filtering invalid rows, reformatting into Radixor dictionary groups, GZip packaging, and generation of runtime descriptor and checksum metadata'
|
||||||
|
noticeFileName = 'NOTICE-model-data.txt'
|
||||||
|
}
|
||||||
1
models/es-es-default/model-version.txt
Normal file
1
models/es-es-default/model-version.txt
Normal file
@@ -0,0 +1 @@
|
|||||||
|
1.0.0
|
||||||
38
models/es-es-default/src/modelInput/NOTICE-model-data.txt
Normal file
38
models/es-es-default/src/modelInput/NOTICE-model-data.txt
Normal file
@@ -0,0 +1,38 @@
|
|||||||
|
Radixor model-data notice
|
||||||
|
|
||||||
|
Radixor-derived model data
|
||||||
|
|
||||||
|
Copyright (C) 2026, Leo Galambos.
|
||||||
|
|
||||||
|
Copyright and, where applicable, database rights are claimed in the
|
||||||
|
Radixor-specific selection, verification, cleaning, normalization,
|
||||||
|
grouping, deduplication, filtering, reformatting, metadata preparation,
|
||||||
|
and packaging of this model, to the extent protected by applicable law.
|
||||||
|
|
||||||
|
The underlying morphological data remains attributed to UniMorph and
|
||||||
|
the upstream contributors identified in this notice.
|
||||||
|
|
||||||
|
This derived model data, including Radixor's protectable contributions,
|
||||||
|
is distributed under Creative Commons Attribution-ShareAlike 3.0
|
||||||
|
Unported.
|
||||||
|
|
||||||
|
Model ID: es-es-default
|
||||||
|
Radixor language: ES_ES
|
||||||
|
Source project: UniMorph
|
||||||
|
Official repository: https://github.com/unimorph/spa
|
||||||
|
Upstream dataset: UniMorph Spanish morphological dataset (`spa`)
|
||||||
|
Upstream lexical source: English Wiktionary
|
||||||
|
Attribution: UniMorph and English Wiktionary contributors
|
||||||
|
License:
|
||||||
|
Creative Commons Attribution-ShareAlike 3.0 Unported
|
||||||
|
Canonical license URI: https://creativecommons.org/licenses/by-sa/3.0/
|
||||||
|
Source revision: not-recorded-in-legacy-import
|
||||||
|
Revision status: not-recorded-in-legacy-import
|
||||||
|
|
||||||
|
The exact UniMorph commit used for the original Radixor import was not recorded. The model remains attributed to the official UniMorph language repository and is distributed under the repository's stated data license.
|
||||||
|
|
||||||
|
Radixor modifications: Cleaning, normalization, grouping inflected forms by lemma, deduplication, filtering invalid rows, reformatting into Radixor dictionary groups, GZip packaging, and generation of runtime descriptor and checksum metadata.
|
||||||
|
|
||||||
|
The derived model data is distributed under CC BY-SA 3.0. UniMorph supplies morphological data; Radixor constructs its own patch-command trie at runtime. Neither UniMorph nor any upstream contributor endorses Radixor.
|
||||||
|
|
||||||
|
Upstream information verified: 2026-07-22
|
||||||
23
models/fa-ir-default/build.gradle
Normal file
23
models/fa-ir-default/build.gradle
Normal file
@@ -0,0 +1,23 @@
|
|||||||
|
plugins {
|
||||||
|
id 'org.egothor.radixor.model'
|
||||||
|
}
|
||||||
|
|
||||||
|
radixorModel {
|
||||||
|
modelId = 'fa-ir-default'
|
||||||
|
language = 'FA_IR'
|
||||||
|
displayName = 'Persian default model'
|
||||||
|
defaultModel = true
|
||||||
|
sourceName = 'UniMorph'
|
||||||
|
sourceVersion = 'not-recorded-in-legacy-import'
|
||||||
|
sourceRevision = 'not-recorded-in-legacy-import'
|
||||||
|
sourceProject = 'UniMorph'
|
||||||
|
sourceRepository = 'https://github.com/unimorph/fas'
|
||||||
|
sourceDataset = 'UniMorph Persian morphological dataset (`fas`)'
|
||||||
|
sourceRevisionStatus = 'not-recorded-in-legacy-import'
|
||||||
|
sourceLicense = 'CC-BY-SA-3.0'
|
||||||
|
sourceLicenseUri = 'https://creativecommons.org/licenses/by-sa/3.0/'
|
||||||
|
sourceAttribution = 'UniMorph and Wikipedia contributors'
|
||||||
|
sourceVerificationDate = '2026-07-22'
|
||||||
|
transformationsSummary = 'Cleaning, normalization, grouping inflected forms by lemma, deduplication, filtering invalid rows, reformatting into Radixor dictionary groups, GZip packaging, and generation of runtime descriptor and checksum metadata'
|
||||||
|
noticeFileName = 'NOTICE-model-data.txt'
|
||||||
|
}
|
||||||
1
models/fa-ir-default/model-version.txt
Normal file
1
models/fa-ir-default/model-version.txt
Normal file
@@ -0,0 +1 @@
|
|||||||
|
1.0.0
|
||||||
38
models/fa-ir-default/src/modelInput/NOTICE-model-data.txt
Normal file
38
models/fa-ir-default/src/modelInput/NOTICE-model-data.txt
Normal file
@@ -0,0 +1,38 @@
|
|||||||
|
Radixor model-data notice
|
||||||
|
|
||||||
|
Radixor-derived model data
|
||||||
|
|
||||||
|
Copyright (C) 2026, Leo Galambos.
|
||||||
|
|
||||||
|
Copyright and, where applicable, database rights are claimed in the
|
||||||
|
Radixor-specific selection, verification, cleaning, normalization,
|
||||||
|
grouping, deduplication, filtering, reformatting, metadata preparation,
|
||||||
|
and packaging of this model, to the extent protected by applicable law.
|
||||||
|
|
||||||
|
The underlying morphological data remains attributed to UniMorph and
|
||||||
|
the upstream contributors identified in this notice.
|
||||||
|
|
||||||
|
This derived model data, including Radixor's protectable contributions,
|
||||||
|
is distributed under Creative Commons Attribution-ShareAlike 3.0
|
||||||
|
Unported.
|
||||||
|
|
||||||
|
Model ID: fa-ir-default
|
||||||
|
Radixor language: FA_IR
|
||||||
|
Source project: UniMorph
|
||||||
|
Official repository: https://github.com/unimorph/fas
|
||||||
|
Upstream dataset: UniMorph Persian morphological dataset (`fas`)
|
||||||
|
Upstream lexical source: Wikipedia
|
||||||
|
Attribution: UniMorph and Wikipedia contributors
|
||||||
|
License:
|
||||||
|
Creative Commons Attribution-ShareAlike 3.0 Unported
|
||||||
|
Canonical license URI: https://creativecommons.org/licenses/by-sa/3.0/
|
||||||
|
Source revision: not-recorded-in-legacy-import
|
||||||
|
Revision status: not-recorded-in-legacy-import
|
||||||
|
|
||||||
|
The exact UniMorph commit used for the original Radixor import was not recorded. The model remains attributed to the official UniMorph language repository and is distributed under the repository's stated data license.
|
||||||
|
|
||||||
|
Radixor modifications: Cleaning, normalization, grouping inflected forms by lemma, deduplication, filtering invalid rows, reformatting into Radixor dictionary groups, GZip packaging, and generation of runtime descriptor and checksum metadata.
|
||||||
|
|
||||||
|
The derived model data is distributed under CC BY-SA 3.0. UniMorph supplies morphological data; Radixor constructs its own patch-command trie at runtime. Neither UniMorph nor any upstream contributor endorses Radixor.
|
||||||
|
|
||||||
|
Upstream information verified: 2026-07-22
|
||||||
23
models/fi-fi-default/build.gradle
Normal file
23
models/fi-fi-default/build.gradle
Normal file
@@ -0,0 +1,23 @@
|
|||||||
|
plugins {
|
||||||
|
id 'org.egothor.radixor.model'
|
||||||
|
}
|
||||||
|
|
||||||
|
radixorModel {
|
||||||
|
modelId = 'fi-fi-default'
|
||||||
|
language = 'FI_FI'
|
||||||
|
displayName = 'Finnish default model'
|
||||||
|
defaultModel = true
|
||||||
|
sourceName = 'UniMorph'
|
||||||
|
sourceVersion = 'not-recorded-in-legacy-import'
|
||||||
|
sourceRevision = 'not-recorded-in-legacy-import'
|
||||||
|
sourceProject = 'UniMorph'
|
||||||
|
sourceRepository = 'https://github.com/unimorph/fin'
|
||||||
|
sourceDataset = 'UniMorph Finnish morphological dataset (`fin`)'
|
||||||
|
sourceRevisionStatus = 'not-recorded-in-legacy-import'
|
||||||
|
sourceLicense = 'CC-BY-SA-3.0'
|
||||||
|
sourceLicenseUri = 'https://creativecommons.org/licenses/by-sa/3.0/'
|
||||||
|
sourceAttribution = 'UniMorph and Wikipedia contributors'
|
||||||
|
sourceVerificationDate = '2026-07-22'
|
||||||
|
transformationsSummary = 'Cleaning, normalization, grouping inflected forms by lemma, deduplication, filtering invalid rows, reformatting into Radixor dictionary groups, GZip packaging, and generation of runtime descriptor and checksum metadata'
|
||||||
|
noticeFileName = 'NOTICE-model-data.txt'
|
||||||
|
}
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user