This commit is contained in:
dakocha3
2026-09-21 17:40:25 +03:00
commit 309188d191
52 changed files with 5124 additions and 0 deletions
+38
View File
@@ -0,0 +1,38 @@
import sys
limit = int(sys.argv[1]) if len(sys.argv) > 1 else 50000
out, sent, inside, kept = [], [], False, 0
def flush():
global sent, inside
if sent:
if inside:
sent.append('<END>')
out.append(' '.join(sent))
sent, inside = [], False
for line in sys.stdin:
line = line.rstrip('\n')
if not line:
flush()
if len(out) >= limit:
break
continue
if line.startswith('#'):
continue
cols = line.split('\t')
if len(cols) < 10 or '-' in cols[0]:
continue
word, tag = cols[1], cols[9]
person = tag.endswith('-PER')
begins = tag.startswith('Tag=B-PER')
if person and (not inside or begins):
if inside:
sent.append('<END>')
sent.append('<START:person>')
inside = True
elif not person and inside:
sent.append('<END>')
inside = False
sent.append(word)
flush()
sys.stderr.write(f"предложений: {len(out)}, с персонами: {sum(1 for l in out if 'START:person' in l)}\n")
print('\n'.join(out[:limit]))
+69
View File
@@ -0,0 +1,69 @@
import glob, os, sys
def load_doc(base):
order, text = [], {}
for line in open(base + '.tokens', encoding='utf-8'):
p = line.split()
if len(p) >= 4:
order.append(p[0]); text[p[0]] = ' '.join(p[3:])
index = {tid: i for i, tid in enumerate(order)}
spans = {}
for line in open(base + '.spans', encoding='utf-8'):
p = line.split('#')[0].split()
if len(p) >= 6:
spans[p[0]] = (p[4], int(p[5]))
person = set()
for line in open(base + '.objects', encoding='utf-8'):
p = line.split('#')[0].split()
if len(p) < 3 or p[1] != 'Person':
continue
for sid in p[2:]:
if sid in spans:
first, cnt = spans[sid]
i = index.get(first)
if i is None:
continue
for j in range(i, min(i + cnt, len(order))):
person.add(order[j])
return order, text, person
def sentences(order, text, person, limit=40):
cur = []
for tid in order:
cur.append(tid)
if text[tid] in ('.', '!', '?', '…') or len(cur) >= limit:
yield cur; cur = []
if cur:
yield cur
def emit(order, text, person):
out = []
for sent in sentences(order, text, person):
words, inside = [], False
for tid in sent:
is_person = tid in person
if is_person and not inside:
words.append('<START:person>'); inside = True
elif not is_person and inside:
words.append('<END>'); inside = False
words.append(text[tid])
if inside:
words.append('<END>')
if any(t in person for t in sent) or len(out) % 3 == 0:
out.append(' '.join(words))
return out
root = sys.argv[1]
target = sys.argv[2]
lines, persons = [], 0
for part in ('devset', 'testset'):
for tok in sorted(glob.glob(os.path.join(root, part, '*.tokens'))):
base = tok[:-len('.tokens')]
order, text, person = load_doc(base)
persons += len(person)
lines.extend(emit(order, text, person))
with open(target, 'w', encoding='utf-8') as f:
f.write('\n'.join(lines) + '\n')
print(f"предложений: {len(lines)}, размеченных токенов-персон: {persons}")
+39
View File
@@ -0,0 +1,39 @@
#!/usr/bin/env bash
# Обучение модели для второй ступени распознавания имён.
#
# Модель в репозиторий не кладётся: она весит мегабайты и собирается из открытых
# корпусов за несколько минут. Без модели сервис работает на одних правилах.
#
# ./tools/train-ner.sh # быстрый вариант, только factRuEval
# ./tools/train-ner.sh full # плюс префикс Nerus, качество заметно выше
set -euo pipefail
cd "$(dirname "$0")/.."
MODE="${1:-quick}"
WORK="$(mktemp -d)"
trap 'rm -rf "$WORK"' EXIT
echo "1. factRuEval-2016 — ручная разметка, 1965 предложений"
curl -sSL -o "$WORK/fre.tar.gz" \
https://codeload.github.com/dialogue-evaluation/factRuEval-2016/tar.gz/refs/heads/master
tar xzf "$WORK/fre.tar.gz" -C "$WORK"
python3 tools/factrueval-to-opennlp.py "$WORK/factRuEval-2016-master" "$WORK/train.txt"
if [ "$MODE" = "full" ]; then
echo "2. Nerus — автоматическая разметка, берём префикс потоком (400 тыс. предложений)"
curl -sS -r 0-400000000 \
https://storage.yandexcloud.net/natasha-nerus/data/nerus_lenta.conllu.gz \
| gunzip 2>/dev/null \
| python3 tools/conllu-to-opennlp.py 400000 >> "$WORK/train.txt"
fi
echo "3. Обучение, несколько минут"
mvn -q -B test-compile
mvn -q -B dependency:build-classpath -Dmdep.outputFile="$WORK/cp.txt"
mkdir -p models
java -Xmx6g -Dner.iterations=100 -Dner.cutoff=5 \
-cp "target/test-classes:target/classes:$(cat "$WORK/cp.txt")" \
ru.pdguard.tools.NerTrainer "$WORK/train.txt" models/ru-ner-person.bin
echo
echo "Готово. Включить: pdguard.ner.model=models/ru-ner-person.bin"