70 lines
2.3 KiB
Python
70 lines
2.3 KiB
Python
import glob, os, sys
|
|
|
|
def load_doc(base):
|
|
order, text = [], {}
|
|
for line in open(base + '.tokens', encoding='utf-8'):
|
|
p = line.split()
|
|
if len(p) >= 4:
|
|
order.append(p[0]); text[p[0]] = ' '.join(p[3:])
|
|
index = {tid: i for i, tid in enumerate(order)}
|
|
|
|
spans = {}
|
|
for line in open(base + '.spans', encoding='utf-8'):
|
|
p = line.split('#')[0].split()
|
|
if len(p) >= 6:
|
|
spans[p[0]] = (p[4], int(p[5]))
|
|
|
|
person = set()
|
|
for line in open(base + '.objects', encoding='utf-8'):
|
|
p = line.split('#')[0].split()
|
|
if len(p) < 3 or p[1] != 'Person':
|
|
continue
|
|
for sid in p[2:]:
|
|
if sid in spans:
|
|
first, cnt = spans[sid]
|
|
i = index.get(first)
|
|
if i is None:
|
|
continue
|
|
for j in range(i, min(i + cnt, len(order))):
|
|
person.add(order[j])
|
|
return order, text, person
|
|
|
|
def sentences(order, text, person, limit=40):
|
|
cur = []
|
|
for tid in order:
|
|
cur.append(tid)
|
|
if text[tid] in ('.', '!', '?', '…') or len(cur) >= limit:
|
|
yield cur; cur = []
|
|
if cur:
|
|
yield cur
|
|
|
|
def emit(order, text, person):
|
|
out = []
|
|
for sent in sentences(order, text, person):
|
|
words, inside = [], False
|
|
for tid in sent:
|
|
is_person = tid in person
|
|
if is_person and not inside:
|
|
words.append('<START:person>'); inside = True
|
|
elif not is_person and inside:
|
|
words.append('<END>'); inside = False
|
|
words.append(text[tid])
|
|
if inside:
|
|
words.append('<END>')
|
|
if any(t in person for t in sent) or len(out) % 3 == 0:
|
|
out.append(' '.join(words))
|
|
return out
|
|
|
|
root = sys.argv[1]
|
|
target = sys.argv[2]
|
|
lines, persons = [], 0
|
|
for part in ('devset', 'testset'):
|
|
for tok in sorted(glob.glob(os.path.join(root, part, '*.tokens'))):
|
|
base = tok[:-len('.tokens')]
|
|
order, text, person = load_doc(base)
|
|
persons += len(person)
|
|
lines.extend(emit(order, text, person))
|
|
with open(target, 'w', encoding='utf-8') as f:
|
|
f.write('\n'.join(lines) + '\n')
|
|
print(f"предложений: {len(lines)}, размеченных токенов-персон: {persons}")
|