benchmark-generated.txt: a fresh dataset (not used to tune the rules)
covering every PD type from the spec plus format variations — date
order, "серия ... номер ...", register-independence. Surfaced two real
gaps worth tracking: FIO detection needs a capitalized first letter,
so "клиент иванова мария петровна" (all lowercase) isn't found; the
NER cascade flags plain capitalized nouns ("Портрет", "Рим") as names
when the base rules alone don't.
LargeTextTest: the existing large-text test just repeated one
email+card sentence 4000 times, so 27 of 28 PD types never ran at
scale. Replaces it with ~400,000 characters built from shuffled lines
of the new dataset — recall on it holds at 0.970, matching small-scale
numbers, and the NER cascade stays sub-second on the whole thing.
Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
193 lines
9.2 KiB
Java
193 lines
9.2 KiB
Java
package ru.pdguard;
|
|
|
|
import org.junit.jupiter.api.Test;
|
|
import ru.pdguard.config.SystemPolicy;
|
|
import ru.pdguard.core.PayloadStore;
|
|
import ru.pdguard.core.Pipeline;
|
|
import ru.pdguard.core.Span;
|
|
import ru.pdguard.detect.NameCascade;
|
|
import ru.pdguard.detect.RuleRegistry;
|
|
import ru.pdguard.mask.Masker;
|
|
|
|
import java.io.BufferedReader;
|
|
import java.io.IOException;
|
|
import java.io.InputStream;
|
|
import java.io.InputStreamReader;
|
|
import java.nio.charset.StandardCharsets;
|
|
import java.nio.file.Files;
|
|
import java.nio.file.Path;
|
|
import java.util.ArrayList;
|
|
import java.util.Collections;
|
|
import java.util.List;
|
|
import java.util.Objects;
|
|
import java.util.Optional;
|
|
import java.util.Random;
|
|
import java.util.regex.Matcher;
|
|
import java.util.regex.Pattern;
|
|
|
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
|
|
|
/**
|
|
* Качество и скорость на большом тексте — не повторе одного и того же
|
|
* предложения, а перемешанных строках из {@code benchmark-generated.txt}
|
|
* (все типы ПД вперемешку с чистым текстом), растянутых до объёма из ТЗ
|
|
* (около 100 000 токенов, ~400 КБ по оценке из README).
|
|
*
|
|
* <p>Раздутый повтором одной строки текст проверяет только то, что цикл не
|
|
* падает на объёме: под маской всегда один и тот же тип, а остальные правила
|
|
* не задействуются вовсе. Здесь размер и разнообразие проверяются вместе.
|
|
*/
|
|
class LargeTextTest {
|
|
|
|
private static final String MODEL_PATH = "models/ru-ner-person.bin";
|
|
private static final Pattern MARKUP = Pattern.compile("\\{\\{([A-Z_]+):([^}]*)}}");
|
|
|
|
/** Целевой объём: README оценивает 100 000 токенов как ~400 КБ текста. */
|
|
private static final int TARGET_CHARS = 400_000;
|
|
|
|
private record Line(String text, List<Span> gold) {
|
|
}
|
|
|
|
/** Строки набора без разметки — чистый текст для перемешивания. */
|
|
private static List<Line> loadLines() {
|
|
List<Line> lines = new ArrayList<>();
|
|
try (InputStream in = LargeTextTest.class.getResourceAsStream("/benchmark-generated.txt");
|
|
BufferedReader reader = new BufferedReader(
|
|
new InputStreamReader(Objects.requireNonNull(in), StandardCharsets.UTF_8))) {
|
|
String raw;
|
|
while ((raw = reader.readLine()) != null) {
|
|
String trimmed = raw.trim();
|
|
if (!trimmed.isEmpty() && !trimmed.startsWith("#")) {
|
|
lines.add(parse(trimmed));
|
|
}
|
|
}
|
|
} catch (IOException e) {
|
|
throw new IllegalStateException(e);
|
|
}
|
|
return lines;
|
|
}
|
|
|
|
private static Line parse(String line) {
|
|
StringBuilder text = new StringBuilder(line.length());
|
|
List<Span> gold = new ArrayList<>();
|
|
Matcher m = MARKUP.matcher(line);
|
|
int cursor = 0;
|
|
while (m.find()) {
|
|
text.append(line, cursor, m.start());
|
|
int start = text.length();
|
|
text.append(m.group(2));
|
|
gold.add(new Span(start, text.length(), m.group(1), 0));
|
|
cursor = m.end();
|
|
}
|
|
text.append(line, cursor, line.length());
|
|
return new Line(text.toString(), gold);
|
|
}
|
|
|
|
/**
|
|
* Перемешивает исходные строки (фиксированный seed — детерминированный
|
|
* тест) и склеивает их через перенос строки, пока не наберётся целевой
|
|
* объём. Смещения золотых фрагментов пересчитываются под общий текст.
|
|
*/
|
|
private static Line buildLargeText(int targetChars, long seed) {
|
|
List<Line> pool = new ArrayList<>(loadLines());
|
|
Random random = new Random(seed);
|
|
StringBuilder text = new StringBuilder(targetChars + 1024);
|
|
List<Span> gold = new ArrayList<>();
|
|
|
|
while (text.length() < targetChars) {
|
|
Collections.shuffle(pool, random);
|
|
for (Line line : pool) {
|
|
int offset = text.length();
|
|
text.append(line.text()).append('\n');
|
|
for (Span span : line.gold()) {
|
|
gold.add(new Span(span.start() + offset, span.end() + offset, span.type(), 0));
|
|
}
|
|
if (text.length() >= targetChars) {
|
|
break;
|
|
}
|
|
}
|
|
}
|
|
return new Line(text.toString(), gold);
|
|
}
|
|
|
|
/**
|
|
* Маскирование и обратное преобразование на большом тексте дают
|
|
* побайтово тот же результат, что и исходный текст — при объёме на
|
|
* порядок больше, чем в остальных тестах, и с разнородным содержимым,
|
|
* а не одним повторяющимся предложением.
|
|
*/
|
|
@Test
|
|
void roundTripOnLargeMixedText() {
|
|
Line large = buildLargeText(TARGET_CHARS, 1);
|
|
Pipeline pipeline = new Pipeline(new RuleRegistry(), new Masker(),
|
|
new PayloadStore(large.text().length() * 2L, 30));
|
|
|
|
long maskStarted = System.nanoTime();
|
|
String masked = pipeline.process(large.text(), "large-mixed-1", SystemPolicy.DEFAULT);
|
|
long maskMillis = (System.nanoTime() - maskStarted) / 1_000_000;
|
|
|
|
long unmaskStarted = System.nanoTime();
|
|
String restored = pipeline.process(masked, "large-mixed-1", SystemPolicy.DEFAULT);
|
|
long unmaskMillis = (System.nanoTime() - unmaskStarted) / 1_000_000;
|
|
|
|
assertEquals(large.text(), restored, "демаскирование не восстановило исходный текст");
|
|
assertTrue(maskMillis < 5000, "маскирование " + large.text().length() + " знаков заняло " + maskMillis + " мс");
|
|
assertTrue(unmaskMillis < 1000, "демаскирование заняло " + unmaskMillis + " мс");
|
|
|
|
System.out.printf("%nБольшой текст: %d знаков, маскирование %d мс, демаскирование %d мс%n",
|
|
large.text().length(), maskMillis, unmaskMillis);
|
|
}
|
|
|
|
/**
|
|
* Полнота детекции не должна проседать на объёме: каждый золотой
|
|
* фрагмент из перемешанных строк обязан быть найден в общем потоке
|
|
* текста, а не только когда он единственный в маленькой строке.
|
|
*/
|
|
@Test
|
|
void recallHoldsAtScale() {
|
|
Line large = buildLargeText(TARGET_CHARS, 2);
|
|
Pipeline pipeline = new Pipeline(new RuleRegistry(), new Masker(), new PayloadStore(1L, 30));
|
|
|
|
List<Span> found = pipeline.findPersonalData(large.text(), SystemPolicy.DEFAULT);
|
|
int hit = 0;
|
|
for (Span gold : large.gold()) {
|
|
if (found.stream().anyMatch(f -> f.type().equals(gold.type()) && f.overlaps(gold))) {
|
|
hit++;
|
|
}
|
|
}
|
|
double recall = large.gold().isEmpty() ? 1.0 : (double) hit / large.gold().size();
|
|
System.out.printf("%nПолнота на большом тексте: %d из %d (%.3f)%n", hit, large.gold().size(), recall);
|
|
|
|
assertTrue(recall >= 0.85,
|
|
String.format("полнота на большом тексте упала до %.3f (%d/%d)", recall, hit, large.gold().size()));
|
|
}
|
|
|
|
/**
|
|
* Вторая ступень ограничена числом кандидатов на запрос
|
|
* ({@code pdguard.ner.max-candidates}), поэтому объём текста не должен
|
|
* превращать её в квадратичную нагрузку — проверяем на том же большом
|
|
* тексте, что и остальные тесты, а не на маленьком образце.
|
|
*/
|
|
@Test
|
|
void nameCascadeStaysBoundedOnLargeText() throws IOException {
|
|
Path model = Path.of(MODEL_PATH);
|
|
if (!Files.isReadable(model)) {
|
|
System.out.println("Модель " + model.toAbsolutePath() + " не собрана, пропускаю");
|
|
return;
|
|
}
|
|
Line large = buildLargeText(TARGET_CHARS, 3);
|
|
Pipeline pipeline = new Pipeline(new RuleRegistry(), new Masker(),
|
|
new PayloadStore(large.text().length() * 2L, 30),
|
|
new NameCascade(Optional.of(MODEL_PATH), 16, 4));
|
|
|
|
long started = System.nanoTime();
|
|
pipeline.process(large.text(), "large-cascade-1", SystemPolicy.DEFAULT);
|
|
long millis = (System.nanoTime() - started) / 1_000_000;
|
|
|
|
System.out.printf("%nБольшой текст со второй ступенью: %d знаков за %d мс%n",
|
|
large.text().length(), millis);
|
|
assertTrue(millis < 5000, "со второй ступенью обработка заняла " + millis + " мс");
|
|
}
|
|
}
|