Init
This commit is contained in:
@@ -0,0 +1,38 @@
|
||||
import sys
|
||||
limit = int(sys.argv[1]) if len(sys.argv) > 1 else 50000
|
||||
out, sent, inside, kept = [], [], False, 0
|
||||
def flush():
|
||||
global sent, inside
|
||||
if sent:
|
||||
if inside:
|
||||
sent.append('<END>')
|
||||
out.append(' '.join(sent))
|
||||
sent, inside = [], False
|
||||
|
||||
for line in sys.stdin:
|
||||
line = line.rstrip('\n')
|
||||
if not line:
|
||||
flush()
|
||||
if len(out) >= limit:
|
||||
break
|
||||
continue
|
||||
if line.startswith('#'):
|
||||
continue
|
||||
cols = line.split('\t')
|
||||
if len(cols) < 10 or '-' in cols[0]:
|
||||
continue
|
||||
word, tag = cols[1], cols[9]
|
||||
person = tag.endswith('-PER')
|
||||
begins = tag.startswith('Tag=B-PER')
|
||||
if person and (not inside or begins):
|
||||
if inside:
|
||||
sent.append('<END>')
|
||||
sent.append('<START:person>')
|
||||
inside = True
|
||||
elif not person and inside:
|
||||
sent.append('<END>')
|
||||
inside = False
|
||||
sent.append(word)
|
||||
flush()
|
||||
sys.stderr.write(f"предложений: {len(out)}, с персонами: {sum(1 for l in out if 'START:person' in l)}\n")
|
||||
print('\n'.join(out[:limit]))
|
||||
@@ -0,0 +1,69 @@
|
||||
import glob, os, sys
|
||||
|
||||
def load_doc(base):
|
||||
order, text = [], {}
|
||||
for line in open(base + '.tokens', encoding='utf-8'):
|
||||
p = line.split()
|
||||
if len(p) >= 4:
|
||||
order.append(p[0]); text[p[0]] = ' '.join(p[3:])
|
||||
index = {tid: i for i, tid in enumerate(order)}
|
||||
|
||||
spans = {}
|
||||
for line in open(base + '.spans', encoding='utf-8'):
|
||||
p = line.split('#')[0].split()
|
||||
if len(p) >= 6:
|
||||
spans[p[0]] = (p[4], int(p[5]))
|
||||
|
||||
person = set()
|
||||
for line in open(base + '.objects', encoding='utf-8'):
|
||||
p = line.split('#')[0].split()
|
||||
if len(p) < 3 or p[1] != 'Person':
|
||||
continue
|
||||
for sid in p[2:]:
|
||||
if sid in spans:
|
||||
first, cnt = spans[sid]
|
||||
i = index.get(first)
|
||||
if i is None:
|
||||
continue
|
||||
for j in range(i, min(i + cnt, len(order))):
|
||||
person.add(order[j])
|
||||
return order, text, person
|
||||
|
||||
def sentences(order, text, person, limit=40):
|
||||
cur = []
|
||||
for tid in order:
|
||||
cur.append(tid)
|
||||
if text[tid] in ('.', '!', '?', '…') or len(cur) >= limit:
|
||||
yield cur; cur = []
|
||||
if cur:
|
||||
yield cur
|
||||
|
||||
def emit(order, text, person):
|
||||
out = []
|
||||
for sent in sentences(order, text, person):
|
||||
words, inside = [], False
|
||||
for tid in sent:
|
||||
is_person = tid in person
|
||||
if is_person and not inside:
|
||||
words.append('<START:person>'); inside = True
|
||||
elif not is_person and inside:
|
||||
words.append('<END>'); inside = False
|
||||
words.append(text[tid])
|
||||
if inside:
|
||||
words.append('<END>')
|
||||
if any(t in person for t in sent) or len(out) % 3 == 0:
|
||||
out.append(' '.join(words))
|
||||
return out
|
||||
|
||||
root = sys.argv[1]
|
||||
target = sys.argv[2]
|
||||
lines, persons = [], 0
|
||||
for part in ('devset', 'testset'):
|
||||
for tok in sorted(glob.glob(os.path.join(root, part, '*.tokens'))):
|
||||
base = tok[:-len('.tokens')]
|
||||
order, text, person = load_doc(base)
|
||||
persons += len(person)
|
||||
lines.extend(emit(order, text, person))
|
||||
with open(target, 'w', encoding='utf-8') as f:
|
||||
f.write('\n'.join(lines) + '\n')
|
||||
print(f"предложений: {len(lines)}, размеченных токенов-персон: {persons}")
|
||||
Executable
+39
@@ -0,0 +1,39 @@
|
||||
#!/usr/bin/env bash
|
||||
# Обучение модели для второй ступени распознавания имён.
|
||||
#
|
||||
# Модель в репозиторий не кладётся: она весит мегабайты и собирается из открытых
|
||||
# корпусов за несколько минут. Без модели сервис работает на одних правилах.
|
||||
#
|
||||
# ./tools/train-ner.sh # быстрый вариант, только factRuEval
|
||||
# ./tools/train-ner.sh full # плюс префикс Nerus, качество заметно выше
|
||||
set -euo pipefail
|
||||
cd "$(dirname "$0")/.."
|
||||
|
||||
MODE="${1:-quick}"
|
||||
WORK="$(mktemp -d)"
|
||||
trap 'rm -rf "$WORK"' EXIT
|
||||
|
||||
echo "1. factRuEval-2016 — ручная разметка, 1965 предложений"
|
||||
curl -sSL -o "$WORK/fre.tar.gz" \
|
||||
https://codeload.github.com/dialogue-evaluation/factRuEval-2016/tar.gz/refs/heads/master
|
||||
tar xzf "$WORK/fre.tar.gz" -C "$WORK"
|
||||
python3 tools/factrueval-to-opennlp.py "$WORK/factRuEval-2016-master" "$WORK/train.txt"
|
||||
|
||||
if [ "$MODE" = "full" ]; then
|
||||
echo "2. Nerus — автоматическая разметка, берём префикс потоком (400 тыс. предложений)"
|
||||
curl -sS -r 0-400000000 \
|
||||
https://storage.yandexcloud.net/natasha-nerus/data/nerus_lenta.conllu.gz \
|
||||
| gunzip 2>/dev/null \
|
||||
| python3 tools/conllu-to-opennlp.py 400000 >> "$WORK/train.txt"
|
||||
fi
|
||||
|
||||
echo "3. Обучение, несколько минут"
|
||||
mvn -q -B test-compile
|
||||
mvn -q -B dependency:build-classpath -Dmdep.outputFile="$WORK/cp.txt"
|
||||
mkdir -p models
|
||||
java -Xmx6g -Dner.iterations=100 -Dner.cutoff=5 \
|
||||
-cp "target/test-classes:target/classes:$(cat "$WORK/cp.txt")" \
|
||||
ru.pdguard.tools.NerTrainer "$WORK/train.txt" models/ru-ner-person.bin
|
||||
|
||||
echo
|
||||
echo "Готово. Включить: pdguard.ner.model=models/ru-ner-person.bin"
|
||||
Reference in New Issue
Block a user