# -*- coding: utf-8 -*- """NER 实体识别预处理。 按 ARCHITECTURE.md Phase 4 拆分建议,从 ``interface/Preprocessing.py`` 迁出。 类型:PREPROCESS。 原位置:``interface/Preprocessing.py`` 中以下函数: - ``get_preprocessed_entitys`` — NER 实体识别与提取 - ``union_ner`` — 连续实体合并(弃用) - ``union_result`` — 模型结果拼接 ``interface/Preprocessing.py`` 仍 re-export 以上全部名称,老 import 不受影响。 """ from __future__ import absolute_import import re import time from BiddingKG.dl.common.logging import log from BiddingKG.dl.common.Utils import get_money_entity, cut_repeat_name, changeIndexFromWordToWords, findAllIndex, clean_company, timeFormat from BiddingKG.dl.interface.Entitys import Entity from BiddingKG.dl.interface.predictor import getPredictor from BiddingKG.dl.predictors.table_prem import TableTag2List from BiddingKG.dl.common.nerUtils import * from BiddingKG.dl.money.moneySource.ruleExtra import extract_moneySource from BiddingKG.dl.time.re_servicetime import extract_servicetime from BiddingKG.dl.relation_extraction.re_email import extract_email from BiddingKG.dl.ratio.re_ratio import extract_ratio from BiddingKG.dl.entityLink.entityLink import * __all__ = [ "get_preprocessed_entitys", "union_ner", "union_result", ] def union_ner(list_ner): result_list = [] union_index = [] union_index_set = set() for i in range(len(list_ner)-1): if len(set([str(list_ner[i][2]),str(list_ner[i+1][2])])&set(["org","company"]))==2: if list_ner[i][1]-list_ner[i+1][0]==1: union_index_set.add(i) union_index_set.add(i+1) union_index.append((i,i+1)) for i in range(len(list_ner)): if i not in union_index_set: result_list.append(list_ner[i]) for item in union_index: #print(str(list_ner[item[0]][3])+str(list_ner[item[1]][3])) result_list.append((list_ner[item[0]][0],list_ner[item[1]][1],'company',str(list_ner[item[0]][3])+str(list_ner[item[1]][3]))) return result_list def get_preprocessed_entitys(list_sentences,useselffool=True,cost_time=dict()): ''' :param list_sentences:分局情况 :param cost_time: :return: list_entitys ''' list_entitys = [] not_extract_roles = ['黄埔军校', '国有资产管理处', '五金建材', '铝合金门窗', '华电XX发电有限公司', '华电XXX发电有限公司', '中标(成交)公司', '贵州茅台', '贵州茅台酒', '陕西省省级国', '纪检监察部门', '融安金桔', '海达源组织', '海达源织', '成交出版社', '中国大学', '第七条交易中心', '铭牌标明出厂', '三、广州分公司'] # 需要过滤掉的企业单位 short_full_dic = { "中交一航局": "中交第一航务工程局有限公司", "中交二公局": "中交第二公路工程局有限公司", "中交二航局": "中交第二航务工程局有限公司", "中兴": "中兴通讯股份有限公司", "中国人寿嘉兴分公司": "中国人寿保险股份有限公司嘉兴分公司", "中国有色集团": "中国有色矿业集团有限公司", "中建": "中国建筑集团有限公司", "中核": "中国核工业集团有限公司", "中船重工": "中国船舶重工集团有限公司", "中铁": "中国铁路工程集团有限公司", "北京农商银行": "北京农村商业银行股份有限公司", "华为": "华为技术有限公司", "南航物流": "南方航空物流股份有限公司", "周宁农信联社": "周宁县农村信用合作联社", "嘉秀集团": "嘉兴市嘉秀发展投资控股集团有限公司", "国电": "国电电力发展股份有限公司", "国网": "国家电网有限公司", "国网北京电力": "国网北京市电力公司", "国网江苏电力": "国网江苏省电力有限公司", "国网江西电力": "国网江西省电力有限公司", "国网浙江电力舟山供电公司": "国网浙江省电力有限公司舟山供电公司", "国网浙江电力金华供电公司": "国网浙江省电力有限公司金华供电公司", "国网湖北电力": "国网湖北省电力有限公司", "山东电建三公司": "山东电力建设第三工程有限公司", "成都市青羊区国资局": "成都市青羊区国有资产监督管理局", "新华保险": "新华人寿保险股份有限公司", "昌吉州人民医院": "昌吉回族自治州人民医院", "格力": "珠海格力电器股份有限公司", "欣创环保": "安徽欣创节能环保科技股份有限公司", "水电四局": "中国水利水电第四工程局有限公司", "江河集团": "江河创建集团股份有限公司", "江苏交控": "江苏交通控股有限公司", "泾河集团": "陕西省西咸新区泾河新城开发建设(集团)有限公司", "海尔": "海尔集团", "现代牧业集团": "现代牧业(集团)有限公司", "美的": "美的集团", "联想": "联想集团", "远东股份": "远东智慧能源股份有限公司", "首都会展集团": "首都会展(集团)有限公司", "马钢股份": "马鞍山钢铁股份有限公司" } for list_sentence in list_sentences: sentences = [] list_entitys_temp = [] for _sentence in list_sentence: sentences.append(_sentence.sentence_text) time1 = time.time() ''' tokens_all = fool.cut(sentences) #pos_all = fool.LEXICAL_ANALYSER.pos(tokens_all) #ner_tag_all = fool.LEXICAL_ANALYSER.ner_labels(sentences,tokens_all) ner_entitys_all = fool.ner(sentences) ''' #限流执行 key_nerToken = "nerToken" start_time = time.time() found_yeji = 0 # 2021/8/6 增加判断是否正文包含评标结果 及类似业绩判断用于过滤后面的金额 # found_pingbiao = False ner_entitys_all = getNers(sentences,useselffool=useselffool) if key_nerToken not in cost_time: cost_time[key_nerToken] = 0 cost_time[key_nerToken] += round(time.time()-start_time,2) doctextcon_sentence_len = sum([1 for sentence in list_sentence if not sentence.in_attachment]) company_dict = set() company_index = dict((i,set()) for i in range(len(list_sentence))) for sentence_index in range(len(list_sentence)): list_sentence_entitys = [] sentence_text = list_sentence[sentence_index].sentence_text tokens = list_sentence[sentence_index].tokens doc_id = list_sentence[sentence_index].doc_id in_attachment = list_sentence[sentence_index].in_attachment list_tokenbegin = [] begin = 0 for i in range(0,len(tokens)): list_tokenbegin.append(begin) begin += len(str(tokens[i])) list_tokenbegin.append(begin+1) #pos_tag = pos_all[sentence_index] pos_tag = "" ner_entitys = ner_entitys_all[sentence_index] # 20250320 注释掉下面代码 避免带来异常实体 # '''正则识别角色实体 经营部|经销部|电脑部|服务部|复印部|印刷部|彩印部|装饰部|修理部|汽修部|修理店|零售店|设计店|服务店|家具店|专卖店|分店|文具行|商行|印刷厂|修理厂|维修中心|修配中心|养护中心|服务中心|会馆|文化馆|超市|门市|商场|家具城|印刷社|经销处''' # for it in re.finditer( # '(?P(((单一来源|中标|中选|中价|成交)(供应商|供货商|服务商|候选人|单位|人))|(供应商|供货商|服务商|候选人))(名称)?[为::]+)(?P([()\u4e00-\u9fa5]{5,20})(厂|中心|超市|门市|商场|工作室|文印室|城|部|店|站|馆|行|社|处))[,。]', # sentence_text): # for k, v in it.groupdict().items(): # if k == 'text_key_word': # keyword = v # if k == 'text': # entity = v # b = it.start() + len(keyword) # e = it.end() - 1 # if (b, e, 'location', entity) in ner_entitys: # ner_entitys.remove((b, e, 'location', entity)) # ner_entitys.append((b, e, 'company', entity)) # elif (b, e, 'org', entity) not in ner_entitys and (b, e, 'company', entity) not in ner_entitys: # ner_entitys.append((b, e, 'company', entity)) # # for it in re.finditer( # '(?P((建设|招租|招标|采购)(单位|人)|业主)(名称)?[为::]+)(?P[\u4e00-\u9fa5]{2,4}[省市县区镇]([()\u4e00-\u9fa5]{2,20})(管理处|办公室|委员会|村委会|纪念馆|监狱|管教所|修养所|社区|农场|林场|羊场|猪场|石场|村|幼儿园|海关|殡仪馆)|海门\w{2,15}村)[,。]', # sentence_text): # for k, v in it.groupdict().items(): # if k == 'text_key_word': # keyword = v # if k == 'text': # entity = v # b = it.start() + len(keyword) # e = it.end() - 1 # if (b, e, 'location', entity) in ner_entitys: # ner_entitys.remove((b, e, 'location', entity)) # ner_entitys.append((b, e, 'org', entity)) # if (b, e, 'org', entity) not in ner_entitys and (b, e, 'company', entity) not in ner_entitys: # ner_entitys.append((b, e, 'org', entity)) for ner_entity in ner_entitys: if ner_entity[2] in ['company','org']: company_dict.add((ner_entity[2],ner_entity[3])) company_index[sentence_index].add((ner_entity[0],ner_entity[1])) #识别package ner_time_list = [] #识别实体 for ner_entity in ner_entitys: begin_index_temp = ner_entity[0] end_index_temp = ner_entity[1] entity_type = ner_entity[2] entity_text = ner_entity[3] if entity_type in ["org", "company"] and re.search('^((特殊)?普通合伙)|^(有限合伙)', sentence_text[end_index_temp:]): # 规则补充合伙关键词 partnership = re.search('^((特殊)?普通合伙)|^(有限合伙)', sentence_text[end_index_temp:]).group(0) end_index_temp += len(partnership) entity_text += partnership if entity_type == 'location' and re.search('^\w{2,4}[市县]\w{2,15}(中心|监狱|殡仪馆|水利站)$', entity_text) and \ re.search('\d[楼层号]', entity_text)==None: # 2024/06/07 修改错误地址实体为角色 entity_type = 'org' elif entity_type in ["org", "company"] and re.search('地址:$', sentence_text[:begin_index_temp]): # 20250421 修复地址识别错为角色 地址:新疆阿拉尔幸福镇十三团,2、运维公司名称:政采云有限公司 entity_type = 'location' if begin_index_temp>0 and '县' in entity_text and re.match('前郭尔罗斯蒙古族自治县|积石山县', sentence_text[begin_index_temp-1:end_index_temp]): #20240905 修复实体识别少字问题 entity_text = sentence_text[begin_index_temp-1] + entity_text begin_index_temp -= 1 ner_entity = (begin_index_temp, end_index_temp, entity_type, entity_text) elif entity_text == '中华人民共和国' and re.search('^\w{2,4}海关', sentence_text[end_index_temp: end_index_temp+6]): # 2024/04/24 修复 采购单位:中华人民共和国汕尾海关, 识别不到海关 ser = re.search('^\w{2,4}海关', sentence_text[end_index_temp: end_index_temp+6]) entity_text += ser.group(0) end_index_temp += ser.end() ner_entity = (begin_index_temp, end_index_temp, entity_type, entity_text) elif entity_text.startswith('中选人为'): # 20251224 修复 697862525 中选人为资阳厚业劳务有限公司 提取为公司 entity_text = entity_text[4:] begin_index_temp += 4 ner_entity = (begin_index_temp, end_index_temp, entity_type, entity_text) elif entity_text.startswith('方') and re.search('(出让|受让|成交|中标)方$', sentence_text[begin_index_temp-2:begin_index_temp+1]): # 20260805 修复 809205092 方平山县小觉镇郄家庄村民委员会 提取为公司 entity_text = entity_text[1:] begin_index_temp += 1 ner_entity = (begin_index_temp, end_index_temp, entity_type, entity_text) if entity_type=='time': ner_time_list.append((begin_index_temp,end_index_temp)) if entity_type in ["org","company"] and not isLegalEnterprise(entity_text): continue # 实体长度限制 if entity_type in ["org","company"] and len(entity_text)>30: continue if entity_type == "person" and len(entity_text) > 20: continue elif entity_type=="person" and len(entity_text)>10 and len(re.findall("[\u4e00-\u9fa5]",entity_text))begin_index_temp: begin_index = j-1 break begin_index_temp += len(str(entity_text)) for j in range(begin_index,len(list_tokenbegin)): if list_tokenbegin[j]>=begin_index_temp: end_index = j-1 break entity_id = "%s_%d_%d_%d"%(doc_id,sentence_index,begin_index,end_index) #去掉标点符号 if entity_type!='time': entity_text = re.sub("[,,。:!&@$\*\s;;]","",entity_text) # 215553737 entity_text = entity_text.replace("(","(").replace(")",")") if isinstance(entity_text,str) else entity_text # 组织机构实体名称补充 if entity_type in ["org", "company"]: if entity_text in not_extract_roles: # 过滤掉名称在 需要过滤企业单位列表里的 continue if not re.search("有限责任公司|有限公司",entity_text): fix_name = re.search("(有限)([责贵]?任?)(公?司?)",entity_text) if fix_name: if len(fix_name.group(2))>0: _text = fix_name.group() if '司' in _text: entity_text = entity_text.replace(_text, "有限责任公司") else: _text = re.search(_text + "[^司]{0,5}司", entity_text) if _text: _text = _text.group() entity_text = entity_text.replace(_text, "有限责任公司") else: entity_text = entity_text.replace(entity_text[fix_name.start():], "有限责任公司") elif len(fix_name.group(3))>0: _text = fix_name.group() if '司' in _text: entity_text = entity_text.replace(_text, "有限公司") else: _text = re.search(_text + "[^司]{0,3}司", entity_text) if _text: _text = _text.group() entity_text = entity_text.replace(_text, "有限公司") else: entity_text = entity_text.replace(entity_text[fix_name.start():], "有限公司") elif re.search("有限$", entity_text): entity_text = re.sub("有限$","有限公司",entity_text) entity_text = entity_text.replace("有公司","有限公司") '''下面对公司实体进行清洗''' entity_text = clean_company(entity_text) if entity_text == '': continue entity_text = cut_repeat_name(entity_text) # 20231201 重复名称去重 如:中山大学附属第一医院中山大学附属第一医院中山大学附属第一医院 entity_text = short_full_dic.get(entity_text, entity_text) # 简称映射字典 match = re.match('路桥(第[一二三四五六七八九十]+分公司)$', entity_text) # 20260311 修复 甘肃路桥集中采购管理平台 714787791 采购人只公布简称 if match: entity_text = '甘肃路桥建设集团有限公司'+match.group(1) list_sentence_entitys.append(Entity(doc_id,entity_id,entity_text,entity_type,sentence_index,begin_index,end_index,ner_entity[0],ner_entity[1],in_attachment=in_attachment)) # 标记文章末尾的"发布人”、“发布时间”实体 if sentence_index==len(list_sentence)-1 or sentence_index==doctextcon_sentence_len-1: if len(list_sentence_entitys[-2:])==2: second2last = list_sentence_entitys[-2] last = list_sentence_entitys[-1] if (second2last.entity_type in ["company",'org'] and last.entity_type=="time") or ( second2last.entity_type=="time" and last.entity_type in ["company",'org']): if last.wordOffset_begin - second2last.wordOffset_end < 6 and len(sentence_text) - last.wordOffset_end<6: last.is_tail = True second2last.is_tail = True #使用正则识别金额 money_list, found_yeji = get_money_entity(sentence_text, found_yeji, in_attachment) entity_type = "money" for money in money_list: # print('money: ', money) entity_text, begin_index, end_index, unit, notes = money end_index = end_index - 1 if entity_text.endswith(',') else end_index entity_id = "%s_%d_%d_%d" % (doc_id, sentence_index, begin_index, end_index) _exists = False for item in list_sentence_entitys: if item.entity_id==entity_id and item.entity_type==entity_type: _exists = True if (begin_index >=item.wordOffset_begin and begin_indexitem.wordOffset_begin and end_index<=item.wordOffset_end): _exists = True # print('_exists: ',begin_index, end_index, item.wordOffset_begin, item.wordOffset_end, item.entity_text, item.entity_type) if not _exists: if float(entity_text)>1: # if symbol == '-': # 负值金额保留负号 # entity_text = '-'+entity_text # 20230414 取消符号 begin_words = changeIndexFromWordToWords(tokens, begin_index) end_words = changeIndexFromWordToWords(tokens, end_index) # print('金额位置: ', begin_index, begin_words,end_index, end_words) # print('金额召回: ', entity_text, sentence_text[begin_index:end_index], tokens[begin_words:end_words]) list_sentence_entitys.append(Entity(doc_id,entity_id,entity_text,entity_type,sentence_index,begin_words,end_words,begin_index,end_index,in_attachment=in_attachment)) list_sentence_entitys[-1].notes = notes # 2021/7/20 新增金额备注 list_sentence_entitys[-1].money_unit = unit # 2021/7/20 新增金额备注 # print('预处理中的 金额:%s, 单位:%s'%(entity_text,unit)) # print(entity_text,unit,notes) # "联系人"正则补充提取 2021/11/15 新增 list_person_text = [entity.entity_text for entity in list_sentence_entitys if entity.entity_type=='person'] error_text = ['交易','机构','教育','项目','公司','中标','开标','截标','监督','政府','国家','中国','技术','投标','传真','网址','电子邮', '联系','联系电','联系地','采购代','邮政编','邮政','电话','手机','手机号','联系人','地址','地点','邮箱','邮编','联系方','招标','招标人','代理', '代理人','采购','附件','注意','登录','报名','踏勘',"测试",'交货'] list_person_text = set(list_person_text + error_text) re_person = re.compile("联系人[::]([\u4e00-\u9fa5]工)|" "联系人[::]([\u4e00-\u9fa5]{2,3})(?=,?联系)|" "联系人[::]([\u4e00-\u9fa5]{2,3})(?=[,。;、])" ) list_person = [] if not in_attachment: for match_result in re_person.finditer(sentence_text): match_text = match_result.group() entity_text = match_text[4:] wordOffset_begin = match_result.start() + 4 wordOffset_end = match_result.end() # print(text[wordOffset_begin:wordOffset_end]) # 排除一些不为人名的实体 if re.search("^[\u4e00-\u9fa5]{7,}([,。]|$)",sentence_text[wordOffset_begin:wordOffset_begin+20]): continue if entity_text not in list_person_text and entity_text[:2] not in list_person_text: _person = dict() _person['body'] = entity_text _person['begin_index'] = wordOffset_begin _person['end_index'] = wordOffset_end list_person.append(_person) entity_type = "person" for person in list_person: begin_index_temp = person['begin_index'] for j in range(len(list_tokenbegin)): if list_tokenbegin[j] == begin_index_temp: begin_index = j break elif list_tokenbegin[j] > begin_index_temp: begin_index = j - 1 break index = person['end_index'] end_index_temp = index for j in range(begin_index, len(list_tokenbegin)): if list_tokenbegin[j] >= index: end_index = j - 1 break entity_id = "%s_%d_%d_%d" % (doc_id, sentence_index, begin_index, end_index) entity_text = person['body'] list_sentence_entitys.append( Entity(doc_id, entity_id, entity_text, entity_type, sentence_index, begin_index, end_index, begin_index_temp, end_index_temp,in_attachment=in_attachment)) # 时间实体格式补充 re_time_new = re.compile("20\d{2}-\d{1,2}-\d{1,2}|" "20\d{2}-(:?0[1-9]|1[0-2]|[1-9])|" "20\d{2}/\d{1,2}/\d{1,2}|" "20\d{2}\.\d{1,2}\.\d{1,2}|" "20\d{2}(?:0[1-9]|1[0-2])(?:0[1-9]|[1-2][0-9]|3[0-1])") entity_type = "time" for _time in re.finditer(re_time_new,sentence_text): entity_text = _time.group() begin_index_temp = _time.start() end_index_temp = _time.end() is_same = False for t_index in ner_time_list: if begin_index_temp>=t_index[0] and end_index_temp<=t_index[1]: is_same = True break if is_same: continue if _time.start()!=0 and re.search("\d",sentence_text[_time.start()-1:_time.start()]): continue # 纯数字格式,例:20190509 if re.search("^\d{8}$",entity_text): if _time.end()!=len(sentence_text) and re.search("[\da-zA-z]",sentence_text[_time.end():_time.end()+1]): continue elif _time.start()!=0 and re.search("[\da-zA-z]",sentence_text[max(0,_time.start()-1):_time.start()]): continue entity_text = entity_text[:4] + "-" + entity_text[4:6] + "-" + entity_text[6:8] # 例:2025-05 if re.search("^20\d{2}-(:?0[1-9]|1[0-2]|[1-9])$",entity_text): if _time.end()!=len(sentence_text) and re.search("[\da-zA-z]",sentence_text[_time.end():_time.end()+1]): continue elif _time.start()!=0 and re.search("[\da-zA-z]",sentence_text[max(0,_time.start()-1):_time.start()]): continue if not timeFormat(entity_text): continue for j in range(len(list_tokenbegin)): if list_tokenbegin[j] == begin_index_temp: begin_index = j break elif list_tokenbegin[j] > begin_index_temp: begin_index = j - 1 break for j in range(begin_index, len(list_tokenbegin)): if list_tokenbegin[j] >= end_index_temp: end_index = j - 1 break entity_id = "%s_%d_%d_%d" % (doc_id, sentence_index, begin_index, end_index) list_sentence_entitys.append( Entity(doc_id, entity_id, entity_text, entity_type, sentence_index, begin_index, end_index, begin_index_temp, end_index_temp, in_attachment=in_attachment)) # 资金来源提取 2020/12/30 新增 list_moneySource = extract_moneySource(sentence_text) entity_type = "moneysource" for moneySource in list_moneySource: entity_text = moneySource['body'] if len(entity_text)>50: continue begin_index_temp = moneySource['begin_index'] for j in range(len(list_tokenbegin)): if list_tokenbegin[j] == begin_index_temp: begin_index = j break elif list_tokenbegin[j] > begin_index_temp: begin_index = j - 1 break index = moneySource['end_index'] end_index_temp = index for j in range(begin_index, len(list_tokenbegin)): if list_tokenbegin[j] >= index: end_index = j - 1 break entity_id = "%s_%d_%d_%d" % (doc_id, sentence_index, begin_index, end_index) list_sentence_entitys.append( Entity(doc_id, entity_id, entity_text, entity_type, sentence_index, begin_index, end_index, begin_index_temp, end_index_temp,in_attachment=in_attachment,prob=moneySource['prob'])) # 电子邮箱提取 2021/11/04 新增 list_email = extract_email(sentence_text) entity_type = "email" # 电子邮箱 for email in list_email: begin_index_temp = email['begin_index'] for j in range(len(list_tokenbegin)): if list_tokenbegin[j] == begin_index_temp: begin_index = j break elif list_tokenbegin[j] > begin_index_temp: begin_index = j - 1 break index = email['end_index'] end_index_temp = index for j in range(begin_index, len(list_tokenbegin)): if list_tokenbegin[j] >= index: end_index = j - 1 break entity_id = "%s_%d_%d_%d" % (doc_id, sentence_index, begin_index, end_index) entity_text = email['body'] list_sentence_entitys.append( Entity(doc_id, entity_id, entity_text, entity_type, sentence_index, begin_index, end_index, begin_index_temp, end_index_temp,in_attachment=in_attachment)) # 服务期限提取 2020/12/30 新增 list_servicetime = extract_servicetime(sentence_text) entity_type = "serviceTime" for servicetime in list_servicetime: entity_text = servicetime['body'] begin_index_temp = servicetime['begin_index'] for j in range(len(list_tokenbegin)): if list_tokenbegin[j] == begin_index_temp: begin_index = j break elif list_tokenbegin[j] > begin_index_temp: begin_index = j - 1 break index = servicetime['end_index'] end_index_temp = index for j in range(begin_index, len(list_tokenbegin)): if list_tokenbegin[j] >= index: end_index = j - 1 break entity_id = "%s_%d_%d_%d" % (doc_id, sentence_index, begin_index, end_index) list_sentence_entitys.append( Entity(doc_id, entity_id, entity_text, entity_type, sentence_index, begin_index, end_index, begin_index_temp, end_index_temp,in_attachment=in_attachment, prob=servicetime["prob"])) # 2021/12/29 新增比率提取 list_ratio = extract_ratio(sentence_text) entity_type = "ratio" for ratio in list_ratio: # print("ratio", ratio) begin_index_temp = ratio['begin_index'] for j in range(len(list_tokenbegin)): if list_tokenbegin[j] == begin_index_temp: begin_index = j break elif list_tokenbegin[j] > begin_index_temp: begin_index = j - 1 break index = ratio['end_index'] end_index_temp = index for j in range(begin_index, len(list_tokenbegin)): if list_tokenbegin[j] >= index: end_index = j - 1 break entity_id = "%s_%d_%d_%d" % (doc_id, sentence_index, begin_index, end_index) entity_text = ratio['body'] ratio_value = (ratio['value'],ratio['type']) _entity = Entity(doc_id, entity_id, entity_text, entity_type, sentence_index, begin_index, end_index, begin_index_temp, end_index_temp,in_attachment=in_attachment) _entity.ratio_value = ratio_value list_sentence_entitys.append(_entity) list_sentence_entitys.sort(key=lambda x:x.begin_index) list_entitys_temp = list_entitys_temp+list_sentence_entitys # 补充ner模型未识别全的company/org实体 for sentence_index in range(len(list_sentence)): sentence_text = list_sentence[sentence_index].sentence_text tokens = list_sentence[sentence_index].tokens doc_id = list_sentence[sentence_index].doc_id in_attachment = list_sentence[sentence_index].in_attachment list_tokenbegin = [] begin = 0 for i in range(0, len(tokens)): list_tokenbegin.append(begin) begin += len(str(tokens[i])) list_tokenbegin.append(begin + 1) add_sentence_entitys = [] company_dict = sorted(list(company_dict),key=lambda x:len(x[1]),reverse=True) for company_type,company_text in company_dict: begin_index_list = findAllIndex(company_text,sentence_text) for begin_index in begin_index_list: is_continue = False for t_begin,t_end in list(company_index[sentence_index]): if begin_index>=t_begin and begin_index+len(company_text)<=t_end: is_continue = True break if not is_continue: add_sentence_entitys.append((begin_index,begin_index+len(company_text),company_type,company_text)) company_index[sentence_index].add((begin_index,begin_index+len(company_text))) else: continue for ner_entity in add_sentence_entitys: begin_index_temp = ner_entity[0] end_index_temp = ner_entity[1] entity_type = ner_entity[2] entity_text = ner_entity[3] if entity_type in ["org","company"] and not isLegalEnterprise(entity_text): continue for j in range(len(list_tokenbegin)): if list_tokenbegin[j]==begin_index_temp: begin_index = j break elif list_tokenbegin[j]>begin_index_temp: begin_index = j-1 break begin_index_temp += len(str(entity_text)) for j in range(begin_index,len(list_tokenbegin)): if list_tokenbegin[j]>=begin_index_temp: end_index = j-1 break entity_id = "%s_%d_%d_%d"%(doc_id,sentence_index,begin_index,end_index) if entity_type in ["org","company"] and entity_text in not_extract_roles: # 过滤掉名称在 需要过滤企业单位列表里的 continue #去掉标点符号 entity_text = re.sub("[,,。:!&@$\*]","",entity_text) entity_text = entity_text.replace("(","(").replace(")",")") if isinstance(entity_text,str) else entity_text list_entitys_temp.append(Entity(doc_id,entity_id,entity_text,entity_type,sentence_index,begin_index,end_index,ner_entity[0],ner_entity[1],in_attachment=in_attachment)) list_entitys_temp.sort(key=lambda x:(x.sentence_index,x.begin_index)) list_entitys.append(list_entitys_temp) return list_entitys def union_result(codeName,prem): ''' @summary:模型的结果拼成字典 @param: codeName:编号名称模型的结果字典 prem:拿到属性的角色的字典 @return:拼接起来的字典 ''' result = [] assert len(codeName)==len(prem) for item_code,item_prem in zip(codeName,prem): result.append(dict(item_code,**item_prem)) return result