-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathFrameNetLib.py
More file actions
157 lines (137 loc) · 4.67 KB
/
Copy pathFrameNetLib.py
File metadata and controls
157 lines (137 loc) · 4.67 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
from string import whitespace
class NoManualAnnotation(Exception):
pass
class FeeFeCollision(Exception):
pass
FEE_TYPE_DICT = {
'Core': 'core',
'Peripheral': 'non-core',
'Extra-Thematic': 'extra-thematic'
}
def extract_FEs_for_frame(frame):
result = {
'core': [],
'non-core': [],
'extra-thematic': []
}
for child in frame:
if child.tag.endswith('FE') and child.attrib['coreType'] in FEE_TYPE_DICT:
fe_type = FEE_TYPE_DICT[child.attrib['coreType']]
result[fe_type].append(child.attrib['name'])
return {'frame': frame.attrib['name'], 'frame_elements': result}
def extract_FEs(sentence):
'''
Extracts FEE and frame-element annotations from a sentence.
'''
# Extract text
for child in sentence:
if child.tag.endswith('text'):
text = child.text
break
else:
raise ValueError('Found no text in the sentence.')
# Find the manual annotation
for child in sentence:
if child.tag.endswith('annotationSet') and child.attrib['status'] == 'MANUAL':
annotation = child
break
else:
raise NoManualAnnotation(
'Found no manual annotation for the sentence.')
# Extract the FEE
for child in annotation:
if child.tag.endswith('layer') and child.attrib['name'] == 'Target':
try:
label = child[0]
except IndexError:
continue
fee_boundaries = (
int(label.attrib['start']),
int(label.attrib['end']))
break
else:
raise ValueError('Found no FEE for the sentence.')
# Extract the FEs
frame_elements = []
for child in annotation:
if child.tag.endswith('layer') and child.attrib['name'] == 'FE':
for label in child:
try:
# Check if a FE has the same extent as the FEE.
# Cf. bark.v: it is both a FEE and a core FE
# "Sound". We filter out such cases.
if int(label.attrib['start']) == fee_boundaries[0]:
raise FeeFeCollision
frame_elements.append({
'type': label.attrib['name'],
'start': int(label.attrib['start']),
'end': int(label.attrib['end'])
})
# Some frame elements may be left unexpressed
except KeyError:
continue
frame_elements.sort(key=lambda fe: fe['start'])
return {
'text': text,
'FEE_boundaries': fee_boundaries,
'FEs': frame_elements
}
def get_tokens_with_ranges(text):
tokens = []
ranges = []
inside_token = False
buffer = []
current_start = 0
for i, char in enumerate(text):
if char in whitespace:
if inside_token:
tokens.append(''.join(buffer))
buffer.clear()
ranges.append((current_start, i-1))
inside_token = False
continue
if not inside_token:
current_start = i
inside_token = True
buffer.append(char)
if buffer:
tokens.append(''.join(buffer))
ranges.append((current_start, len(text)))
return tokens, ranges
def prepend_class(label, fe_dict):
if label.startswith('FRAME:'):
return label
elif label in fe_dict['core']:
return f'CORE:{label}'
elif label in fe_dict['non-core']:
return f'NONCORE:{label}'
elif label in fe_dict['extra-thematic']:
return f'EXTRATHEMATIC:{label}'
else:
return None # "Core-Unexpressed"
def prepare_sentence(sentence_dict, frame, fe_dict):
'''
Converts a dict with text, FEE boundaries, and FE types and boundaries
into a pair of aligned lists with tokens on one hand and the name
of the frame and class (core vs. non-core) and types of FEs on the
other.
'''
text = sentence_dict['text']
range_to_label = {}
range_to_label[sentence_dict['FEE_boundaries']] = f'FRAME:{frame}'
for fe in sentence_dict['FEs']:
range_to_label[(fe['start'], fe['end'])] = fe['type']
tokens, ranges = get_tokens_with_ranges(text)
labels = []
for range in ranges:
# We hope to not see overlapping ranges
# and only check for the start of the
# interval.
lo, _ = range
for (r_lo, r_hi), label in range_to_label.items():
if lo >= r_lo and lo < r_hi:
labels.append(prepend_class(label, fe_dict))
break
else:
labels.append(None)
return tokens, labels