# -*- coding: utf-8 -*- """``PREMPredict`` / ``EPCPredict`` — 角色金额模型与联系人模型。 Phase 5 从 ``interface/predictor.py`` 迁出: - ``PREMPredict``(约 795-1229 行) - ``EPCPredict``(约 1230-1577 行) 原 ``from common.Utils import *`` / ``from interface.modelFactory import *`` 已替换为显式 import。两个类均无 ``os.path.dirname(__file__)`` 实际路径引用 (仅注释中出现,保持原样未改)。 """ from __future__ import absolute_import import re import numpy as np from BiddingKG.dl.common.context_utils import spanWindow from BiddingKG.dl.interface.Entitys import Entity from BiddingKG.dl.interface.modelFactory import ( Model_role_classify_word, Model_money_classify, Model_person_classify, ) from BiddingKG.dl.predictors.role_context import build_role_contexts from BiddingKG.dl.predictors.role_rule_engine import RoleRuleEngine from BiddingKG.dl.services.external_inference.paieas import ( USE_PAI_EAS, tf_predict_pb2, vpc_requests, role_url, role_authorization, money_url, money_authorization, person_url, person_authorization, ) __all__ = ["PREMPredict", "EPCPredict"] #角色金额模型 class PREMPredict(): def __init__(self,config=None): #self.model_role_file = os.path.abspath("../role/models/model_role.model.hdf5") # self.model_role_file = os.path.dirname(__file__)+"/../role/log/new_biLSTM-ep012-loss0.028-val_loss0.040-f10.954.h5" self.model_role = Model_role_classify_word(config=config) self.model_money = Model_money_classify(config=config) # Phase C:模型后修正规则外置到 dl/rules/patterns/role_context_fix.yaml, # 由 RoleRuleEngine 按 stage 执行(行为等价原内联 if/elif 链,Phase D 双跑验证) self.rule_engine = RoleRuleEngine() # self.role_file = open('/data/python/lsm/role_model_predict.txt', 'a', encoding='utf-8') # self.money_file = open('/data/python/lsm/money_model_predict.txt', 'a', encoding='utf-8') return def search_role_data(self,list_sentences,list_entitys,contexts=None): ''' @summary:根据句子list和实体list查询角色模型的输入数据 @param: list_sentences:文章的sentences list_entitys:文章的entitys contexts:Phase A 预计算上下文(build_role_contexts 输出), None 时内部构建(角色/金额模型共享一次实体-句子配对) @return:角色模型的输入数据 ''' if contexts is None: contexts = build_role_contexts(list_sentences, list_entitys) text_list = [] data_x = [] points_entitys = [] for ctx_list in contexts: for ctx in ctx_list: entity = ctx.entity if entity.entity_type in ['org','company']: text_list.append(ctx.role_model_text) # item_x = embedding(spanWindow(tokens=sentence.tokens,begin_index=entity.begin_index,end_index=entity.end_index,size=settings.MODEL_ROLE_INPUT_SHAPE[1]),shape=settings.MODEL_ROLE_INPUT_SHAPE) # item_x = self.model_role.encode(tokens=sentence.tokens,begin_index=entity.begin_index,end_index=entity.end_index,entity_text=entity.entity_text) item_x = self.model_role.encode_word(sentence_text=ctx.sentence_text, begin_index=entity.wordOffset_begin, end_index=entity.wordOffset_end, size=30) data_x.append(item_x) points_entitys.append(entity) if len(points_entitys)==0: return None return [data_x,points_entitys, text_list] def search_money_data(self,list_sentences,list_entitys,contexts=None): ''' @summary:根据句子list和实体list查询金额模型的输入数据 @param: list_sentences:文章的sentences list_entitys:文章的entitys contexts:Phase A 预计算上下文(build_role_contexts 输出), None 时内部构建(角色/金额模型共享一次实体-句子配对) @return:金额模型的输入数据 ''' if contexts is None: contexts = build_role_contexts(list_sentences, list_entitys) text_list = [] data_x = [] points_entitys = [] for ctx_list in contexts: for ctx in ctx_list: entity = ctx.entity if entity.entity_type=="money": text_list.append(ctx.money_model_text) #item_x = embedding(spanWindow(tokens=sentence.tokens,begin_index=entity.begin_index,end_index=entity.end_index,size=settings.MODEL_MONEY_INPUT_SHAPE[1]),shape=settings.MODEL_MONEY_INPUT_SHAPE) #item_x = embedding_word(spanWindow(tokens=sentence.tokens, begin_index=entity.begin_index, end_index=entity.end_index, size=10, center_include=True, word_flag=True),shape=settings.MODEL_MONEY_INPUT_SHAPE) item_x = self.model_money.encode(tokens=ctx.sentence.tokens,begin_index=entity.begin_index,end_index=entity.end_index) data_x.append(item_x) points_entitys.append(entity) if len(points_entitys)==0: return None return [data_x,points_entitys, text_list] def predict_role(self,list_sentences, list_entitys, contexts=None): ''' @param contexts: Phase A 预计算上下文(build_role_contexts 输出), None 时内部构建;与 predict_money 共用同一份避免重复配对。 ''' datas = self.search_role_data(list_sentences, list_entitys, contexts=contexts) if datas is None: return points_entitys = datas[1] text_list = datas[2] if USE_PAI_EAS: _data = datas[0] _data = np.transpose(np.array(_data),(1,0,2)) request = tf_predict_pb2.PredictRequest() request.inputs["input0"].dtype = tf_predict_pb2.DT_FLOAT request.inputs["input0"].array_shape.dim.extend(np.shape(_data[0])) request.inputs["input0"].float_val.extend(np.array(_data[0],dtype=np.float64).reshape(-1)) request.inputs["input1"].dtype = tf_predict_pb2.DT_FLOAT request.inputs["input1"].array_shape.dim.extend(np.shape(_data[1])) request.inputs["input1"].float_val.extend(np.array(_data[1],dtype=np.float64).reshape(-1)) request.inputs["input2"].dtype = tf_predict_pb2.DT_FLOAT request.inputs["input2"].array_shape.dim.extend(np.shape(_data[2])) request.inputs["input2"].float_val.extend(np.array(_data[2],dtype=np.float64).reshape(-1)) request_data = request.SerializeToString() list_outputs = ["outputs"] _result = vpc_requests(role_url, role_authorization, request_data, list_outputs) if _result is not None: predict_y = _result["outputs"] else: predict_y = self.model_role.predict(datas[0]) else: predict_y = self.model_role.predict(np.array(datas[0],dtype=np.float64)) for i in range(len(predict_y)): entity = points_entitys[i] label = np.argmax(predict_y[i]) values = predict_y[i] # text = text_list[i] text_tup = text_list[i] front, middle, behind = text_tup # Phase C:内联修正规则外置到 # dl/rules/patterns/role_context_fix.yaml(stage: global_pre/seq_check/ # l0/l2/l5_win_yes/notify/l1/l34/l5),由 RoleRuleEngine 执行, # 行为等价原 if/elif 链(Phase D badcase 双跑回归验证)。 label = self.rule_engine.correct_role(entity, label, values, front, middle, behind) entity.set_Role(label, values) def predict_money(self,list_sentences,list_entitys,contexts=None): ''' @param contexts: Phase A 预计算上下文(build_role_contexts 输出), None 时内部构建;与 predict_role 共用同一份避免重复配对。 ''' datas = self.search_money_data(list_sentences, list_entitys, contexts=contexts) if datas is None: return points_entitys = datas[1] _data = datas[0] text_list = datas[2] if USE_PAI_EAS: _data = np.transpose(np.array(_data),(1,0,2,3)) request = tf_predict_pb2.PredictRequest() request.inputs["input0"].dtype = tf_predict_pb2.DT_FLOAT request.inputs["input0"].array_shape.dim.extend(np.shape(_data[0])) request.inputs["input0"].float_val.extend(np.array(_data[0],dtype=np.float64).reshape(-1)) request.inputs["input1"].dtype = tf_predict_pb2.DT_FLOAT request.inputs["input1"].array_shape.dim.extend(np.shape(_data[1])) request.inputs["input1"].float_val.extend(np.array(_data[1],dtype=np.float64).reshape(-1)) request.inputs["input2"].dtype = tf_predict_pb2.DT_FLOAT request.inputs["input2"].array_shape.dim.extend(np.shape(_data[2])) request.inputs["input2"].float_val.extend(np.array(_data[2],dtype=np.float64).reshape(-1)) request_data = request.SerializeToString() list_outputs = ["outputs"] _result = vpc_requests(money_url, money_authorization, request_data, list_outputs) if _result is not None: predict_y = _result["outputs"] else: predict_y = self.model_money.predict(_data) else: predict_y = self.model_money.predict(_data) for i in range(len(predict_y)): entity = points_entitys[i] label = np.argmax(predict_y[i]) values = predict_y[i] # text = text_list[i] text_tup = text_list[i] front, middle, behind = text_tup # Phase C:内联修正规则外置到 # dl/rules/patterns/role_money_fix.yaml(stage: m1/m0/m_bid), # 由 RoleRuleEngine 执行,行为等价原 if/elif 链 # (Phase D badcase 双跑回归验证)。 label = self.rule_engine.correct_money(entity, label, values, front, middle, behind) entity.set_Money(label, values) def correct_money_by_rule(self, title, list_entitys, list_articles): # Phase C:标题/正文类别批量修正外置到 # dl/rules/patterns/role_money_fix.yaml(stage: doc_title), # 由 RoleRuleEngine 执行,行为等价原实现(Phase D 双跑回归验证)。 self.rule_engine.correct_money_by_doc(title, list_articles[0].content, list_entitys) def predict(self,list_sentences,list_entitys,contexts=None): ''' @param contexts: Phase A 预计算上下文(build_role_contexts 输出), None 时内部构建一次,role/money 两个模型共享, 消除原来各自的实体-句子配对与切片构建。 ''' if contexts is None: contexts = build_role_contexts(list_sentences, list_entitys) self.predict_role(list_sentences,list_entitys,contexts=contexts) self.predict_money(list_sentences,list_entitys,contexts=contexts) #联系人模型 class EPCPredict(): def __init__(self,config=None): self.model_person = Model_person_classify(config=config) def search_person_data(self,list_sentences,list_entitys): ''' @summary:根据句子list和实体list查询联系人模型的输入数据 @param: list_sentences:文章的sentences list_entitys:文章的entitys @return:联系人模型的输入数据 ''' data_x = [] points_entitys = [] pre_texts = [] for list_entity,list_sentence in zip(list_entitys,list_sentences): p_entitys = 0 dict_index_sentence = {} for _sentence in list_sentence: dict_index_sentence[_sentence.sentence_index] = _sentence _list_entity = [entity for entity in list_entity if entity.entity_type=="person"] while(p_entitys 1: # _dianhua = phoneFromList(have_phone[1:]) # else: # _dianhua = phoneFromList(have_phone) # elif have_key: # if entity.entity_text != last_person and s0.find(last_person) != -1 and s1.find( # last_person_phone) != -1: # if len(have_key) > 1: # _dianhua = phoneFromList(have_key[1:]) # else: # _dianhua = phoneFromList(have_key) # elif have_phone2: # if entity.entity_text != last_person and s0.find(last_person) != -1 and s0.find( # last_person_phone) != -1: # if len(have_phone2) > 1: # _dianhua = phoneFromList(have_phone2[1:]) # else: # _dianhua = phoneFromList(have_phone2) # elif have_key2: # if entity.entity_text != last_person and s0.find(last_person) != -1 and s0.find( # last_person_phone) != -1: # if len(have_key2) > 1: # _dianhua = phoneFromList(have_key2[1:]) # else: # _dianhua = phoneFromList(have_key2) # elif have_phone3: # if entity.entity_text != last_person and s4.find(last_person) != -1 and s3.find( # last_person_phone) != -1: # if len(have_phone3) > 1: # _dianhua = phoneFromList(have_phone3[1:]) # else: # _dianhua = phoneFromList(have_phone3) # elif have_key3: # if entity.entity_text != last_person and s4.find(last_person) != -1 and s3.find( # last_person_phone) != -1: # if len(have_key3) > 1: # _dianhua = phoneFromList(have_key3[1:]) # else: # _dianhua = phoneFromList(have_key3) # elif have_phone4: # if entity.entity_text != last_person and s4.find(last_person) != -1 and s4.find( # last_person_phone) != -1: # if len(have_phone4) > 1: # _dianhua = phoneFromList(have_phone4) # else: # _dianhua = phoneFromList(have_phone4) # elif have_key4: # if entity.entity_text != last_person and s4.find(last_person) != -1 and s4.find( # last_person_phone) != -1: # if len(have_key4) > 1: # _dianhua = phoneFromList(have_key4) # else: # _dianhua = phoneFromList(have_key4) # else: # _dianhua = "" # # dict_context_itemx[_key] = [item_x, _dianhua] # dict_context_itemx[_key] = [_dianhua] # # points_entitys.append(entity) # # dianhua.append(_dianhua) # last_person = entity.entity_text # if _dianhua: # # 更新联系人entity联系方式(person_phone) # entity.person_phone = _dianhua # last_person_phone = _dianhua # else: # last_person_phone = "####****++++$^" # p_entitys += 1 from scipy.optimize import linear_sum_assignment from BiddingKG.dl.interface.Entitys import Match def dispatch(match_list): main_roles = list(set([match.main_role for match in match_list])) attributes = list(set([match.attribute for match in match_list])) label = np.zeros(shape=(len(main_roles), len(attributes))) for match in match_list: main_role = match.main_role attribute = match.attribute value = match.value label[main_roles.index(main_role), attributes.index(attribute)] = value + 10000 # print(label) gragh = -label # km算法 row, col = linear_sum_assignment(gragh) max_dispatch = [(i, j) for i, j, value in zip(row, col, gragh[row, col]) if value] return [Match(main_roles[row], attributes[col]) for row, col in max_dispatch] # km算法 key_word = re.compile('((?:电话|联系方式|联系人).{0,4}?)(\d{7,12})') phone = re.compile('1[3|4|5|6|7|8|9][0-9][-—-―]?\d{4}[-—-―]?\d{4}|' '\+86.?1[3|4|5|6|7|8|9]\d{9}|' '0\d{2,3}[-—-―][1-9]\d{6,7}/[1-9]\d{6,10}|' '0\d{2,3}[-—-―]\d{7,8}转\d{1,4}|' '0\d{2,3}[-—-―]?[1-9]\d{6,7}|' '[\(|\(]0\d{2,3}[\)|\)]-?\d{7,8}-?\d{,4}|' '[1-9]\d{6,7}') phone_entitys = [] for _sentence in list_sentence: sentence_text = _sentence.sentence_text res_set = set() for i in re.finditer(phone,sentence_text): res_set.add((i.group(),i.start(),i.end())) for i in re.finditer(key_word,sentence_text): res_set.add((i.group(2),i.start()+len(i.group(1)),i.end())) for item in list(res_set): phone_left = sentence_text[max(0,item[1]-10):item[1]] phone_right = sentence_text[item[2]:item[2]+8] # 排除传真号 和 其它错误项 if re.search("传,?真|信,?箱|邮,?箱",phone_left): if not re.search("电,?话",phone_left): continue if re.search("帐,?号|编,?号|报,?价|证,?号|价,?格|[\((]万?元[\))]",phone_left): continue if re.search("[.,]\d{2,}",phone_right): continue _entity = Entity(_sentence.doc_id, None, item[0], "phone", _sentence.sentence_index, None, None,item[1], item[2],in_attachment=_sentence.in_attachment) phone_entitys.append(_entity) person_entitys = [] for entity in list_entity: if entity.entity_type == "person": entity.person_phone = "" person_entitys.append(entity) _list_entity = phone_entitys + person_entitys _list_entity = sorted(_list_entity,key=lambda x:(x.sentence_index,x.wordOffset_begin)) words_num_dict = dict() last_words_num = 0 list_sentence = sorted(list_sentence, key=lambda x: x.sentence_index) for sentence in list_sentence: _index = sentence.sentence_index if _index == 0: words_num_dict[_index] = 0 else: words_num_dict[_index] = words_num_dict[_index - 1] + last_words_num last_words_num = len(sentence.sentence_text) match_list = [] for index in range(len(_list_entity)): entity = _list_entity[index] if entity.entity_type=="person" and entity.label in [1,2,3]: match_nums = 0 for after_index in range(index + 1, min(len(_list_entity), index + 5)): after_entity = _list_entity[after_index] if after_entity.entity_type=="phone": sentence_distance = after_entity.sentence_index - entity.sentence_index distance = (words_num_dict[after_entity.sentence_index] + after_entity.wordOffset_begin) - ( words_num_dict[entity.sentence_index] + entity.wordOffset_end) if sentence_distance < 2 and distance < 50: value = (-1 / 2 * (distance ** 2)) / 10000 match_list.append(Match(entity, after_entity, value)) match_nums += 1 else: break if after_entity.entity_type=="person": if after_entity.label not in [1,2,3]: break if not match_nums: for previous_index in range(index-1, max(0,index-5), -1): previous_entity = _list_entity[previous_index] if previous_entity.entity_type == "phone": sentence_distance = entity.sentence_index - previous_entity.sentence_index distance = (words_num_dict[entity.sentence_index] + entity.wordOffset_begin) - ( words_num_dict[previous_entity.sentence_index] + previous_entity.wordOffset_end) if sentence_distance < 1 and distance<30: # 前向 没有 /10000 value = (-1 / 2 * (distance ** 2)) match_list.append(Match(entity, previous_entity, value)) else: break result = dispatch(match_list) for match in result: entity = match.main_role # 更新 list_entity entity_index = list_entity.index(entity) list_entity[entity_index].person_phone = match.attribute.entity_text def predict(self,list_sentences,list_entitys): self.predict_person(list_sentences,list_entitys)