refactor: dictionary lookups from O(dictionary size) to O(word length)
isWellKnown/isKnownCity scanned the whole stem list (60, then 1111+
after the city dictionary) with startsWith on every call. Reversed the
check: try decreasing-length prefixes of the input word against a
HashSet of stems instead — same result, but bounded by word length,
not dictionary size.
Also pulls the vowel-stripping declension trick (shared verbatim
between NameDictionary and ToponymDictionary since the city dictionary
was added) into one Declension helper, and extracts the {{TYPE:value}}
benchmark-line parser — duplicated between BenchmarkTest and
LargeTextTest since the large-text test was added — into
BenchmarkFixtures.
Verified: same 125 tests pass, benchmark F1 numbers unchanged, native
image under 0.5 CPU/250MB shows no regression at 1000/1800/3000 RPS.
Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
e00062ea9f
commit
f553e79c75
@@ -11,8 +11,8 @@ import java.io.UncheckedIOException;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Comparator;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Locale;
|
||||
import java.util.Set;
|
||||
@@ -41,7 +41,7 @@ public final class NameDictionary {
|
||||
private static final long RECHECK_MILLIS = 1000;
|
||||
|
||||
private static final List<String> GIVEN_NAME_STEMS = load("/names/given-names.txt").stream()
|
||||
.map(NameDictionary::withoutInflectedEnding)
|
||||
.map(Declension::withoutInflectedEnding)
|
||||
.distinct()
|
||||
.sorted(Comparator.comparingInt(String::length).reversed())
|
||||
.toList();
|
||||
@@ -49,15 +49,14 @@ public final class NameDictionary {
|
||||
// творительный падежи образует заменой «-а» на «-ой» («Набиуллиной»), а не
|
||||
// дописыванием — без отсечения «а» их startsWith не поймает. Тот же приём,
|
||||
// что и для личных имён.
|
||||
private static final List<String> BUNDLED_WELL_KNOWN_STEMS = load("/names/well-known.txt").stream()
|
||||
.map(NameDictionary::withoutInflectedEnding)
|
||||
.distinct()
|
||||
.toList();
|
||||
private static final Set<String> BUNDLED_WELL_KNOWN_STEMS = load("/names/well-known.txt").stream()
|
||||
.map(Declension::withoutInflectedEnding)
|
||||
.collect(Collectors.toUnmodifiableSet());
|
||||
|
||||
private static volatile Path externalFile = Path.of(ConfigProvider.getConfig()
|
||||
.getOptionalValue("pdguard.well-known-file", String.class)
|
||||
.orElse("config/well-known.txt"));
|
||||
private static volatile List<String> wellKnownStems = BUNDLED_WELL_KNOWN_STEMS;
|
||||
private static volatile Set<String> wellKnownStems = BUNDLED_WELL_KNOWN_STEMS;
|
||||
private static volatile long externalTimestamp;
|
||||
private static volatile long lastCheck;
|
||||
|
||||
@@ -76,17 +75,6 @@ public final class NameDictionary {
|
||||
private NameDictionary() {
|
||||
}
|
||||
|
||||
/**
|
||||
* Отбрасывает у основы конечную гласную, которая меняется по падежам:
|
||||
* Ольга → Ольг (Ольги, Ольге, Ольгой), Николай → Никола (Николая, Николаю).
|
||||
*/
|
||||
private static String withoutInflectedEnding(String stem) {
|
||||
if (stem.length() >= 4 && "аяйь".indexOf(stem.charAt(stem.length() - 1)) >= 0) {
|
||||
return stem.substring(0, stem.length() - 1);
|
||||
}
|
||||
return stem;
|
||||
}
|
||||
|
||||
/**
|
||||
* Есть ли среди слов личное имя из словаря в любом падеже.
|
||||
*
|
||||
@@ -116,14 +104,22 @@ public final class NameDictionary {
|
||||
return false;
|
||||
}
|
||||
|
||||
/** Содержит ли текст упоминание известного человека — из сборки или дописанных сверху. */
|
||||
/**
|
||||
* Содержит ли текст упоминание известного человека — из сборки или дописанных
|
||||
* сверху.
|
||||
*
|
||||
* <p>Проверяются префиксы слова по множеству, а не каждая основа по слову:
|
||||
* при тысяче с лишним записей (столько городов в {@link ToponymDictionary},
|
||||
* тот же приём) перебор списка на каждое слово текста был бы заметен, а
|
||||
* префиксов у слова — не больше, чем в нём букв.
|
||||
*/
|
||||
public static boolean isWellKnown(String value) {
|
||||
refreshIfChanged();
|
||||
List<String> stems = wellKnownStems;
|
||||
Set<String> stems = wellKnownStems;
|
||||
for (String word : value.split("\\P{L}+")) {
|
||||
String lower = word.toLowerCase(Locale.ROOT);
|
||||
for (String stem : stems) {
|
||||
if (lower.startsWith(stem.toLowerCase(Locale.ROOT))) {
|
||||
for (int length = lower.length(); length > 0; length--) {
|
||||
if (stems.contains(lower.substring(0, length))) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
@@ -152,14 +148,14 @@ public final class NameDictionary {
|
||||
}
|
||||
try {
|
||||
externalTimestamp = Files.getLastModifiedTime(externalFile).toMillis();
|
||||
List<String> merged = new ArrayList<>(BUNDLED_WELL_KNOWN_STEMS);
|
||||
Set<String> merged = new HashSet<>(BUNDLED_WELL_KNOWN_STEMS);
|
||||
for (String line : Files.readAllLines(externalFile, StandardCharsets.UTF_8)) {
|
||||
String trimmed = withoutInflectedEnding(line.trim());
|
||||
if (!trimmed.isEmpty() && !trimmed.startsWith("#") && !merged.contains(trimmed)) {
|
||||
String trimmed = Declension.withoutInflectedEnding(line.trim());
|
||||
if (!trimmed.isEmpty() && !trimmed.startsWith("#")) {
|
||||
merged.add(trimmed);
|
||||
}
|
||||
}
|
||||
wellKnownStems = List.copyOf(merged);
|
||||
wellKnownStems = Set.copyOf(merged);
|
||||
LOG.infof("Денилист дополнен из %s: %d имён сверх встроенных",
|
||||
externalFile.toAbsolutePath(), merged.size() - BUNDLED_WELL_KNOWN_STEMS.size());
|
||||
} catch (IOException e) {
|
||||
|
||||
Reference in New Issue
Block a user