Writing
NLP (5) — Open Chinese Entity–Relation Extraction via Dependency Parsing
NLP-focused notes on dependency-based open entity–relation extraction in Chinese.
Preface: This article is filed under NLP because it mainly covers NLP topics. Prior reading:
-
pyltp — HIT Language Cloud Python interface: https://www.jianshu.com/p/867478f0e674
-
Open Chinese entity–relation extraction via dependency parsing, Li Mingyao (searchable on Baidu)
-
Zheng Shanshan — Open-domain entity–relation extraction based on Chinese syntactic features
Environment Setup
Install pyltp: 1) If pip fails, adjust the version. 2) Download the matching model from http://ltp.ai/download.html
Prior Work
Bootstrapping enables semi-supervised distant relation extraction: seed templates extract feature words; nearest-neighbor rules auto-generate more templates. During expansion, incorrect templates accumulate and errors compound, hurting final quality.
Extracting Triples
Pipeline: segmentation → POS tagging → NER (person, place, organization only; tag others manually) → dependency parsing. Overview:

Reference: https://www.ltp-cloud.com/intro/#ner_how
Code:
Loading models:
segmentor = Segmentor()
segmentor.load(os.path.join(MODELDIR, "cws.model"))
postagger = Postagger()
postagger.load(os.path.join(MODELDIR, "pos.model"))
parser = Parser()
parser.load(os.path.join(MODELDIR, "parser.model"))
recognizer = NamedEntityRecognizer()
recognizer.load(os.path.join(MODELDIR, "ner.model"))
Core relation extraction:
for index in range(len(postags)):
# 抽取以谓词为中心的事实三元组
if postags[index] == 'v':
child_dict = child_dict_list[index]
# 主谓宾
if child_dict.__contains__('SBV') and child_dict.__contains__('VOB'):
e1 = complete_e(words, postags, child_dict_list, child_dict['SBV'][0])
r = words[index]
e2 = complete_e(words, postags, child_dict_list, child_dict['VOB'][0])
out_file.write("主语谓语宾语关系\t(%s, %s, %s)\n" % (e1, r, e2))
print(e1,r,e2)
out_file.flush()
# 定语后置,动宾关系
if arcs[index].relation == 'ATT':
if child_dict.__contains__('VOB'):
e1 = complete_e(words, postags, child_dict_list, arcs[index].head - 1)
r = words[index]
e2 = complete_e(words, postags, child_dict_list, child_dict['VOB'][0])
temp_string = r+e2
if temp_string == e1[:len(temp_string)]:
e1 = e1[len(temp_string):]
if temp_string not in e1:
out_file.write("定语后置动宾关系\t(%s, %s, %s)\n" % (e1, r, e2))
out_file.flush()
# 含有介宾关系的主谓动补关系
if child_dict.__contains__('SBV') and child_dict.__contains__('CMP'):
#e1 = words[child_dict['SBV'][0]]
e1 = complete_e(words, postags, child_dict_list, child_dict['SBV'][0])
cmp_index = child_dict['CMP'][0]
r = words[index] + words[cmp_index]
if child_dict_list[cmp_index].__contains__('POB'):
e2 = complete_e(words, postags, child_dict_list, child_dict_list[cmp_index]['POB'][0])
out_file.write("介宾关系主谓动补\t(%s, %s, %s)\n" % (e1, r, e2))
out_file.flush()
# 尝试抽取命名实体有关的三元组
if netags[index][0] == 'S' or netags[index][0] == 'B':
ni = index
if netags[ni][0] == 'B':
while netags[ni][0] != 'E':
ni += 1
e1 = ''.join(words[index:ni+1])
else:
e1 = words[ni]
if arcs[ni].relation == 'ATT' and postags[arcs[ni].head-1] == 'n' and netags[arcs[ni].head-1] == 'O':
r = complete_e(words, postags, child_dict_list, arcs[ni].head-1)
if e1 in r:
r = r[(r.index(e1)+len(e1)):]
if arcs[arcs[ni].head-1].relation == 'ATT' and netags[arcs[arcs[ni].head-1].head-1] != 'O':
e2 = complete_e(words, postags, child_dict_list, arcs[arcs[ni].head-1].head-1)
mi = arcs[arcs[ni].head-1].head-1
li = mi
if netags[mi][0] == 'B':
while netags[mi][0] != 'E':
mi += 1
e = ''.join(words[li+1:mi+1])
e2 += e
if r in e2:
e2 = e2[(e2.index(r)+len(r)):]
if r+e2 in sentence:
out_file.write("人名//地名//机构\t(%s, %s, %s)\n" % (e1, r, e2))
out_file.flush()