Init
This commit is contained in:
@@ -0,0 +1,69 @@
|
||||
import glob, os, sys
|
||||
|
||||
def load_doc(base):
|
||||
order, text = [], {}
|
||||
for line in open(base + '.tokens', encoding='utf-8'):
|
||||
p = line.split()
|
||||
if len(p) >= 4:
|
||||
order.append(p[0]); text[p[0]] = ' '.join(p[3:])
|
||||
index = {tid: i for i, tid in enumerate(order)}
|
||||
|
||||
spans = {}
|
||||
for line in open(base + '.spans', encoding='utf-8'):
|
||||
p = line.split('#')[0].split()
|
||||
if len(p) >= 6:
|
||||
spans[p[0]] = (p[4], int(p[5]))
|
||||
|
||||
person = set()
|
||||
for line in open(base + '.objects', encoding='utf-8'):
|
||||
p = line.split('#')[0].split()
|
||||
if len(p) < 3 or p[1] != 'Person':
|
||||
continue
|
||||
for sid in p[2:]:
|
||||
if sid in spans:
|
||||
first, cnt = spans[sid]
|
||||
i = index.get(first)
|
||||
if i is None:
|
||||
continue
|
||||
for j in range(i, min(i + cnt, len(order))):
|
||||
person.add(order[j])
|
||||
return order, text, person
|
||||
|
||||
def sentences(order, text, person, limit=40):
|
||||
cur = []
|
||||
for tid in order:
|
||||
cur.append(tid)
|
||||
if text[tid] in ('.', '!', '?', '…') or len(cur) >= limit:
|
||||
yield cur; cur = []
|
||||
if cur:
|
||||
yield cur
|
||||
|
||||
def emit(order, text, person):
|
||||
out = []
|
||||
for sent in sentences(order, text, person):
|
||||
words, inside = [], False
|
||||
for tid in sent:
|
||||
is_person = tid in person
|
||||
if is_person and not inside:
|
||||
words.append('<START:person>'); inside = True
|
||||
elif not is_person and inside:
|
||||
words.append('<END>'); inside = False
|
||||
words.append(text[tid])
|
||||
if inside:
|
||||
words.append('<END>')
|
||||
if any(t in person for t in sent) or len(out) % 3 == 0:
|
||||
out.append(' '.join(words))
|
||||
return out
|
||||
|
||||
root = sys.argv[1]
|
||||
target = sys.argv[2]
|
||||
lines, persons = [], 0
|
||||
for part in ('devset', 'testset'):
|
||||
for tok in sorted(glob.glob(os.path.join(root, part, '*.tokens'))):
|
||||
base = tok[:-len('.tokens')]
|
||||
order, text, person = load_doc(base)
|
||||
persons += len(person)
|
||||
lines.extend(emit(order, text, person))
|
||||
with open(target, 'w', encoding='utf-8') as f:
|
||||
f.write('\n'.join(lines) + '\n')
|
||||
print(f"предложений: {len(lines)}, размеченных токенов-персон: {persons}")
|
||||
Reference in New Issue
Block a user