| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581 |
- # -*- coding: utf-8 -*-
- """``PREMPredict`` / ``EPCPredict`` — 角色金额模型与联系人模型。
- Phase 5 从 ``interface/predictor.py`` 迁出:
- - ``PREMPredict``(约 795-1229 行)
- - ``EPCPredict``(约 1230-1577 行)
- 原 ``from common.Utils import *`` / ``from interface.modelFactory import *``
- 已替换为显式 import。两个类均无 ``os.path.dirname(__file__)`` 实际路径引用
- (仅注释中出现,保持原样未改)。
- """
- from __future__ import absolute_import
- import re
- import numpy as np
- from BiddingKG.dl.common.context_utils import spanWindow
- from BiddingKG.dl.interface.Entitys import Entity
- from BiddingKG.dl.interface.modelFactory import (
- Model_role_classify_word,
- Model_money_classify,
- Model_person_classify,
- )
- from BiddingKG.dl.predictors.role_context import build_role_contexts
- from BiddingKG.dl.predictors.role_rule_engine import RoleRuleEngine
- from BiddingKG.dl.services.external_inference.paieas import (
- USE_PAI_EAS,
- tf_predict_pb2,
- vpc_requests,
- role_url,
- role_authorization,
- money_url,
- money_authorization,
- person_url,
- person_authorization,
- )
- __all__ = ["PREMPredict", "EPCPredict"]
- #角色金额模型
- class PREMPredict():
-
- def __init__(self,config=None):
- #self.model_role_file = os.path.abspath("../role/models/model_role.model.hdf5")
- # self.model_role_file = os.path.dirname(__file__)+"/../role/log/new_biLSTM-ep012-loss0.028-val_loss0.040-f10.954.h5"
- self.model_role = Model_role_classify_word(config=config)
- self.model_money = Model_money_classify(config=config)
- # Phase C:模型后修正规则外置到 dl/rules/patterns/role_context_fix.yaml,
- # 由 RoleRuleEngine 按 stage 执行(行为等价原内联 if/elif 链,Phase D 双跑验证)
- self.rule_engine = RoleRuleEngine()
- # self.role_file = open('/data/python/lsm/role_model_predict.txt', 'a', encoding='utf-8')
- # self.money_file = open('/data/python/lsm/money_model_predict.txt', 'a', encoding='utf-8')
- return
-
- def search_role_data(self,list_sentences,list_entitys,contexts=None):
- '''
- @summary:根据句子list和实体list查询角色模型的输入数据
- @param:
- list_sentences:文章的sentences
- list_entitys:文章的entitys
- contexts:Phase A 预计算上下文(build_role_contexts 输出),
- None 时内部构建(角色/金额模型共享一次实体-句子配对)
- @return:角色模型的输入数据
- '''
- if contexts is None:
- contexts = build_role_contexts(list_sentences, list_entitys)
- text_list = []
- data_x = []
- points_entitys = []
- for ctx_list in contexts:
- for ctx in ctx_list:
- entity = ctx.entity
- if entity.entity_type in ['org','company']:
- text_list.append(ctx.role_model_text)
- # item_x = embedding(spanWindow(tokens=sentence.tokens,begin_index=entity.begin_index,end_index=entity.end_index,size=settings.MODEL_ROLE_INPUT_SHAPE[1]),shape=settings.MODEL_ROLE_INPUT_SHAPE)
- # item_x = self.model_role.encode(tokens=sentence.tokens,begin_index=entity.begin_index,end_index=entity.end_index,entity_text=entity.entity_text)
- item_x = self.model_role.encode_word(sentence_text=ctx.sentence_text, begin_index=entity.wordOffset_begin, end_index=entity.wordOffset_end, size=30)
- data_x.append(item_x)
- points_entitys.append(entity)
- if len(points_entitys)==0:
- return None
- return [data_x,points_entitys, text_list]
-
-
- def search_money_data(self,list_sentences,list_entitys,contexts=None):
- '''
- @summary:根据句子list和实体list查询金额模型的输入数据
- @param:
- list_sentences:文章的sentences
- list_entitys:文章的entitys
- contexts:Phase A 预计算上下文(build_role_contexts 输出),
- None 时内部构建(角色/金额模型共享一次实体-句子配对)
- @return:金额模型的输入数据
- '''
- if contexts is None:
- contexts = build_role_contexts(list_sentences, list_entitys)
- text_list = []
- data_x = []
- points_entitys = []
- for ctx_list in contexts:
- for ctx in ctx_list:
- entity = ctx.entity
- if entity.entity_type=="money":
- text_list.append(ctx.money_model_text)
- #item_x = embedding(spanWindow(tokens=sentence.tokens,begin_index=entity.begin_index,end_index=entity.end_index,size=settings.MODEL_MONEY_INPUT_SHAPE[1]),shape=settings.MODEL_MONEY_INPUT_SHAPE)
- #item_x = embedding_word(spanWindow(tokens=sentence.tokens, begin_index=entity.begin_index, end_index=entity.end_index, size=10, center_include=True, word_flag=True),shape=settings.MODEL_MONEY_INPUT_SHAPE)
- item_x = self.model_money.encode(tokens=ctx.sentence.tokens,begin_index=entity.begin_index,end_index=entity.end_index)
- data_x.append(item_x)
- points_entitys.append(entity)
- if len(points_entitys)==0:
- return None
- return [data_x,points_entitys, text_list]
-
- def predict_role(self,list_sentences, list_entitys, contexts=None):
- '''
- @param contexts: Phase A 预计算上下文(build_role_contexts 输出),
- None 时内部构建;与 predict_money 共用同一份避免重复配对。
- '''
- datas = self.search_role_data(list_sentences, list_entitys, contexts=contexts)
- if datas is None:
- return
- points_entitys = datas[1]
- text_list = datas[2]
- if USE_PAI_EAS:
- _data = datas[0]
- _data = np.transpose(np.array(_data),(1,0,2))
- request = tf_predict_pb2.PredictRequest()
- request.inputs["input0"].dtype = tf_predict_pb2.DT_FLOAT
- request.inputs["input0"].array_shape.dim.extend(np.shape(_data[0]))
- request.inputs["input0"].float_val.extend(np.array(_data[0],dtype=np.float64).reshape(-1))
- request.inputs["input1"].dtype = tf_predict_pb2.DT_FLOAT
- request.inputs["input1"].array_shape.dim.extend(np.shape(_data[1]))
- request.inputs["input1"].float_val.extend(np.array(_data[1],dtype=np.float64).reshape(-1))
- request.inputs["input2"].dtype = tf_predict_pb2.DT_FLOAT
- request.inputs["input2"].array_shape.dim.extend(np.shape(_data[2]))
- request.inputs["input2"].float_val.extend(np.array(_data[2],dtype=np.float64).reshape(-1))
- request_data = request.SerializeToString()
- list_outputs = ["outputs"]
- _result = vpc_requests(role_url, role_authorization, request_data, list_outputs)
- if _result is not None:
- predict_y = _result["outputs"]
- else:
- predict_y = self.model_role.predict(datas[0])
- else:
- predict_y = self.model_role.predict(np.array(datas[0],dtype=np.float64))
- for i in range(len(predict_y)):
- entity = points_entitys[i]
- label = np.argmax(predict_y[i])
- values = predict_y[i]
- # text = text_list[i]
- text_tup = text_list[i]
- front, middle, behind = text_tup
- # Phase C:内联修正规则外置到
- # dl/rules/patterns/role_context_fix.yaml(stage: global_pre/seq_check/
- # l0/l2/l5_win_yes/notify/l1/l34/l5),由 RoleRuleEngine 执行,
- # 行为等价原 if/elif 链(Phase D badcase 双跑回归验证)。
- label = self.rule_engine.correct_role(entity, label, values, front, middle, behind)
- entity.set_Role(label, values)
- def predict_money(self,list_sentences,list_entitys,contexts=None):
- '''
- @param contexts: Phase A 预计算上下文(build_role_contexts 输出),
- None 时内部构建;与 predict_role 共用同一份避免重复配对。
- '''
- datas = self.search_money_data(list_sentences, list_entitys, contexts=contexts)
- if datas is None:
- return
- points_entitys = datas[1]
- _data = datas[0]
- text_list = datas[2]
- if USE_PAI_EAS:
- _data = np.transpose(np.array(_data),(1,0,2,3))
- request = tf_predict_pb2.PredictRequest()
- request.inputs["input0"].dtype = tf_predict_pb2.DT_FLOAT
- request.inputs["input0"].array_shape.dim.extend(np.shape(_data[0]))
- request.inputs["input0"].float_val.extend(np.array(_data[0],dtype=np.float64).reshape(-1))
- request.inputs["input1"].dtype = tf_predict_pb2.DT_FLOAT
- request.inputs["input1"].array_shape.dim.extend(np.shape(_data[1]))
- request.inputs["input1"].float_val.extend(np.array(_data[1],dtype=np.float64).reshape(-1))
- request.inputs["input2"].dtype = tf_predict_pb2.DT_FLOAT
- request.inputs["input2"].array_shape.dim.extend(np.shape(_data[2]))
- request.inputs["input2"].float_val.extend(np.array(_data[2],dtype=np.float64).reshape(-1))
- request_data = request.SerializeToString()
- list_outputs = ["outputs"]
- _result = vpc_requests(money_url, money_authorization, request_data, list_outputs)
- if _result is not None:
- predict_y = _result["outputs"]
- else:
- predict_y = self.model_money.predict(_data)
- else:
- predict_y = self.model_money.predict(_data)
- for i in range(len(predict_y)):
- entity = points_entitys[i]
- label = np.argmax(predict_y[i])
- values = predict_y[i]
- # text = text_list[i]
- text_tup = text_list[i]
- front, middle, behind = text_tup
- # Phase C:内联修正规则外置到
- # dl/rules/patterns/role_money_fix.yaml(stage: m1/m0/m_bid),
- # 由 RoleRuleEngine 执行,行为等价原 if/elif 链
- # (Phase D badcase 双跑回归验证)。
- label = self.rule_engine.correct_money(entity, label, values, front, middle, behind)
- entity.set_Money(label, values)
- def correct_money_by_rule(self, title, list_entitys, list_articles):
- # Phase C:标题/正文类别批量修正外置到
- # dl/rules/patterns/role_money_fix.yaml(stage: doc_title),
- # 由 RoleRuleEngine 执行,行为等价原实现(Phase D 双跑回归验证)。
- self.rule_engine.correct_money_by_doc(title, list_articles[0].content, list_entitys)
- def predict(self,list_sentences,list_entitys,contexts=None):
- '''
- @param contexts: Phase A 预计算上下文(build_role_contexts 输出),
- None 时内部构建一次,role/money 两个模型共享,
- 消除原来各自的实体-句子配对与切片构建。
- '''
- if contexts is None:
- contexts = build_role_contexts(list_sentences, list_entitys)
- self.predict_role(list_sentences,list_entitys,contexts=contexts)
- self.predict_money(list_sentences,list_entitys,contexts=contexts)
-
- #联系人模型
- class EPCPredict():
-
- def __init__(self,config=None):
- self.model_person = Model_person_classify(config=config)
-
- def search_person_data(self,list_sentences,list_entitys):
- '''
- @summary:根据句子list和实体list查询联系人模型的输入数据
- @param:
- list_sentences:文章的sentences
- list_entitys:文章的entitys
- @return:联系人模型的输入数据
- '''
- data_x = []
- points_entitys = []
- pre_texts = []
- for list_entity,list_sentence in zip(list_entitys,list_sentences):
-
- p_entitys = 0
- dict_index_sentence = {}
- for _sentence in list_sentence:
- dict_index_sentence[_sentence.sentence_index] = _sentence
- _list_entity = [entity for entity in list_entity if entity.entity_type=="person"]
- while(p_entitys<len(_list_entity)):
- entity = _list_entity[p_entitys]
- if entity.entity_type=="person":
- sentence = dict_index_sentence[entity.sentence_index]
- item_x = self.model_person.encode(tokens=sentence.tokens,begin_index=entity.begin_index,end_index=entity.end_index)
- data_x.append(item_x)
- points_entitys.append(entity)
- pre_texts.append(spanWindow(tokens=sentence.tokens,begin_index=entity.begin_index,end_index=entity.end_index,size=20))
- p_entitys += 1
- if len(points_entitys)==0:
- return None
-
- # return [data_x,points_entitys,dianhua]
- return [data_x,points_entitys, pre_texts]
- def predict_person(self,list_sentences, list_entitys):
- datas = self.search_person_data(list_sentences, list_entitys)
- if datas is None:
- return
- points_entitys = datas[1]
- pre_texts = datas[2]
- # phone = datas[2]
- if USE_PAI_EAS:
- _data = datas[0]
- _data = np.transpose(np.array(_data),(1,0,2,3))
- request = tf_predict_pb2.PredictRequest()
- request.inputs["input0"].dtype = tf_predict_pb2.DT_FLOAT
- request.inputs["input0"].array_shape.dim.extend(np.shape(_data[0]))
- request.inputs["input0"].float_val.extend(np.array(_data[0],dtype=np.float64).reshape(-1))
- request.inputs["input1"].dtype = tf_predict_pb2.DT_FLOAT
- request.inputs["input1"].array_shape.dim.extend(np.shape(_data[1]))
- request.inputs["input1"].float_val.extend(np.array(_data[1],dtype=np.float64).reshape(-1))
- request_data = request.SerializeToString()
- list_outputs = ["outputs"]
- _result = vpc_requests(person_url, person_authorization, request_data, list_outputs)
- if _result is not None:
- predict_y = _result["outputs"]
- else:
- predict_y = self.model_person.predict(datas[0])
- else:
- predict_y = self.model_person.predict(datas[0])
- # assert len(predict_y)==len(points_entitys)==len(phone)
- assert len(predict_y)==len(points_entitys)
- for i in range(len(predict_y)):
- entity = points_entitys[i]
- label = np.argmax(predict_y[i])
- pre_text = ''.join(pre_texts[i][0])
- # print('pre_text', pre_text)
- if label==0 and re.search('(谈判|磋商|询价|资格审查|评审专家|(评选|议标|评标|评审)委员会?|专家|评委)(小?组|小?组成员)?(成员|名单)[:,](\w{2,4}((组长)|(成员))?[、,,])*$', pre_text):
- # print(entity.entity_text, re.search('(谈判|磋商|询价|资格审查|评审专家|(评选|议标|评标|评审)委员会?|专家|评委)(小?组|小?组成员)?(成员|名单)[:,](\w{2,4}((组长)|(成员))?[、,,])*$', pre_text).group(0))
- label = 4
- values = []
- for item in predict_y[i]:
- values.append(item)
- # phone_number = phone[i]
- # entity.set_Person(label,values,phone_number)
- entity.set_Person(label,values,[])
- # 为联系人匹配电话
- # self.person_search_phone(list_sentences, list_entitys)
- def person_search_phone(self,list_sentences, list_entitys):
- def phoneFromList(phones):
- # for phone in phones:
- # if len(phone)==11:
- # return re.sub('电话[:|:]|联系方式[:|:]','',phone)
- return re.sub('电话[:|:]|联系方式[:|:]', '', phones[0])
- for list_entity, list_sentence in zip(list_entitys, list_sentences):
- # p_entitys = 0
- # p_sentences = 0
- #
- # key_word = re.compile('电话[:|:].{0,4}\d{7,12}|联系方式[:|:].{0,4}\d{7,12}')
- # # phone = re.compile('1[3|4|5|7|8][0-9][-—-]?\d{4}[-—-]?\d{4}|\d{3,4}[-—]\d{7,8}/\d{3,8}|\d{3,4}[-—]\d{7,8}转\d{1,4}|\d{3,4}[-—]\d{7,8}|[\(|\(]0\d{2,3}[\)|\)]-?\d{7,8}-?\d{,4}') # 联系电话
- # # 2020/11/25 增加发现的号码段
- # phone = re.compile('1[3|4|5|6|7|8|9][0-9][-—-]?\d{4}[-—-]?\d{4}|'
- # '\d{3,4}[-—][1-9]\d{6,7}/\d{3,8}|'
- # '\d{3,4}[-—]\d{7,8}转\d{1,4}|'
- # '\d{3,4}[-—]?[1-9]\d{6,7}|'
- # '[\(|\(]0\d{2,3}[\)|\)]-?\d{7,8}-?\d{,4}|'
- # '[1-9]\d{6,7}') # 联系电话
- # dict_index_sentence = {}
- # for _sentence in list_sentence:
- # dict_index_sentence[_sentence.sentence_index] = _sentence
- #
- # dict_context_itemx = {}
- # last_person = "####****++++$$^"
- # last_person_phone = "####****++++$^"
- # _list_entity = [entity for entity in list_entity if entity.entity_type == "person"]
- # while (p_entitys < len(_list_entity)):
- # entity = _list_entity[p_entitys]
- # if entity.entity_type == "person" and entity.label in [1,2,3]:
- # sentence = dict_index_sentence[entity.sentence_index]
- # # item_x = embedding(spanWindow(tokens=sentence.tokens,begin_index=entity.begin_index,end_index=entity.end_index,size=settings.MODEL_PERSON_INPUT_SHAPE[1]),shape=settings.MODEL_PERSON_INPUT_SHAPE)
- #
- # # s = spanWindow(tokens=sentence.tokens,begin_index=entity.begin_index,end_index=entity.end_index,size=20)
- #
- # # 2021/5/8 取上下文的句子,解决表格处理的分句问题
- # left_sentence = dict_index_sentence.get(entity.sentence_index - 1)
- # left_sentence_tokens = left_sentence.tokens if left_sentence else []
- # right_sentence = dict_index_sentence.get(entity.sentence_index + 1)
- # right_sentence_tokens = right_sentence.tokens if right_sentence else []
- # entity_beginIndex = entity.begin_index + len(left_sentence_tokens)
- # entity_endIndex = entity.end_index + len(left_sentence_tokens)
- # context_sentences_tokens = left_sentence_tokens + sentence.tokens + right_sentence_tokens
- # s = spanWindow(tokens=context_sentences_tokens, begin_index=entity_beginIndex,
- # end_index=entity_endIndex, size=20)
- #
- # _key = "".join(["".join(x) for x in s])
- # if _key in dict_context_itemx:
- # _dianhua = dict_context_itemx[_key][0]
- # else:
- # s1 = ''.join(s[1])
- # # s1 = re.sub(',)', '-', s1)
- # s1 = re.sub('\s', '', s1)
- # have_key = re.findall(key_word, s1)
- # have_phone = re.findall(phone, s1)
- # s0 = ''.join(s[0])
- # # s0 = re.sub(',)', '-', s0)
- # s0 = re.sub('\s', '', s0)
- # have_key2 = re.findall(key_word, s0)
- # have_phone2 = re.findall(phone, s0)
- #
- # s3 = ''.join(s[1])
- # # s0 = re.sub(',)', '-', s0)
- # s3 = re.sub(',|,|\s', '', s3)
- # have_key3 = re.findall(key_word, s3)
- # have_phone3 = re.findall(phone, s3)
- #
- # s4 = ''.join(s[0])
- # # s0 = re.sub(',)', '-', s0)
- # s4 = re.sub(',|,|\s', '', s0)
- # have_key4 = re.findall(key_word, s4)
- # have_phone4 = re.findall(phone, s4)
- #
- # _dianhua = ""
- # if have_phone:
- # if entity.entity_text != last_person and s0.find(last_person) != -1 and s1.find(
- # last_person_phone) != -1:
- # if len(have_phone) > 1:
- # _dianhua = phoneFromList(have_phone[1:])
- # else:
- # _dianhua = phoneFromList(have_phone)
- # elif have_key:
- # if entity.entity_text != last_person and s0.find(last_person) != -1 and s1.find(
- # last_person_phone) != -1:
- # if len(have_key) > 1:
- # _dianhua = phoneFromList(have_key[1:])
- # else:
- # _dianhua = phoneFromList(have_key)
- # elif have_phone2:
- # if entity.entity_text != last_person and s0.find(last_person) != -1 and s0.find(
- # last_person_phone) != -1:
- # if len(have_phone2) > 1:
- # _dianhua = phoneFromList(have_phone2[1:])
- # else:
- # _dianhua = phoneFromList(have_phone2)
- # elif have_key2:
- # if entity.entity_text != last_person and s0.find(last_person) != -1 and s0.find(
- # last_person_phone) != -1:
- # if len(have_key2) > 1:
- # _dianhua = phoneFromList(have_key2[1:])
- # else:
- # _dianhua = phoneFromList(have_key2)
- # elif have_phone3:
- # if entity.entity_text != last_person and s4.find(last_person) != -1 and s3.find(
- # last_person_phone) != -1:
- # if len(have_phone3) > 1:
- # _dianhua = phoneFromList(have_phone3[1:])
- # else:
- # _dianhua = phoneFromList(have_phone3)
- # elif have_key3:
- # if entity.entity_text != last_person and s4.find(last_person) != -1 and s3.find(
- # last_person_phone) != -1:
- # if len(have_key3) > 1:
- # _dianhua = phoneFromList(have_key3[1:])
- # else:
- # _dianhua = phoneFromList(have_key3)
- # elif have_phone4:
- # if entity.entity_text != last_person and s4.find(last_person) != -1 and s4.find(
- # last_person_phone) != -1:
- # if len(have_phone4) > 1:
- # _dianhua = phoneFromList(have_phone4)
- # else:
- # _dianhua = phoneFromList(have_phone4)
- # elif have_key4:
- # if entity.entity_text != last_person and s4.find(last_person) != -1 and s4.find(
- # last_person_phone) != -1:
- # if len(have_key4) > 1:
- # _dianhua = phoneFromList(have_key4)
- # else:
- # _dianhua = phoneFromList(have_key4)
- # else:
- # _dianhua = ""
- # # dict_context_itemx[_key] = [item_x, _dianhua]
- # dict_context_itemx[_key] = [_dianhua]
- # # points_entitys.append(entity)
- # # dianhua.append(_dianhua)
- # last_person = entity.entity_text
- # if _dianhua:
- # # 更新联系人entity联系方式(person_phone)
- # entity.person_phone = _dianhua
- # last_person_phone = _dianhua
- # else:
- # last_person_phone = "####****++++$^"
- # p_entitys += 1
- from scipy.optimize import linear_sum_assignment
- from BiddingKG.dl.interface.Entitys import Match
- def dispatch(match_list):
- main_roles = list(set([match.main_role for match in match_list]))
- attributes = list(set([match.attribute for match in match_list]))
- label = np.zeros(shape=(len(main_roles), len(attributes)))
- for match in match_list:
- main_role = match.main_role
- attribute = match.attribute
- value = match.value
- label[main_roles.index(main_role), attributes.index(attribute)] = value + 10000
- # print(label)
- gragh = -label
- # km算法
- row, col = linear_sum_assignment(gragh)
- max_dispatch = [(i, j) for i, j, value in zip(row, col, gragh[row, col]) if value]
- return [Match(main_roles[row], attributes[col]) for row, col in max_dispatch]
- # km算法
- key_word = re.compile('((?:电话|联系方式|联系人).{0,4}?)(\d{7,12})')
- phone = re.compile('1[3|4|5|6|7|8|9][0-9][-—-―]?\d{4}[-—-―]?\d{4}|'
- '\+86.?1[3|4|5|6|7|8|9]\d{9}|'
- '0\d{2,3}[-—-―][1-9]\d{6,7}/[1-9]\d{6,10}|'
- '0\d{2,3}[-—-―]\d{7,8}转\d{1,4}|'
- '0\d{2,3}[-—-―]?[1-9]\d{6,7}|'
- '[\(|\(]0\d{2,3}[\)|\)]-?\d{7,8}-?\d{,4}|'
- '[1-9]\d{6,7}')
- phone_entitys = []
- for _sentence in list_sentence:
- sentence_text = _sentence.sentence_text
- res_set = set()
- for i in re.finditer(phone,sentence_text):
- res_set.add((i.group(),i.start(),i.end()))
- for i in re.finditer(key_word,sentence_text):
- res_set.add((i.group(2),i.start()+len(i.group(1)),i.end()))
- for item in list(res_set):
- phone_left = sentence_text[max(0,item[1]-10):item[1]]
- phone_right = sentence_text[item[2]:item[2]+8]
- # 排除传真号 和 其它错误项
- if re.search("传,?真|信,?箱|邮,?箱",phone_left):
- if not re.search("电,?话",phone_left):
- continue
- if re.search("帐,?号|编,?号|报,?价|证,?号|价,?格|[\((]万?元[\))]",phone_left):
- continue
- if re.search("[.,]\d{2,}",phone_right):
- continue
- _entity = Entity(_sentence.doc_id, None, item[0], "phone", _sentence.sentence_index, None, None,item[1], item[2],in_attachment=_sentence.in_attachment)
- phone_entitys.append(_entity)
- person_entitys = []
- for entity in list_entity:
- if entity.entity_type == "person":
- entity.person_phone = ""
- person_entitys.append(entity)
- _list_entity = phone_entitys + person_entitys
- _list_entity = sorted(_list_entity,key=lambda x:(x.sentence_index,x.wordOffset_begin))
- words_num_dict = dict()
- last_words_num = 0
- list_sentence = sorted(list_sentence, key=lambda x: x.sentence_index)
- for sentence in list_sentence:
- _index = sentence.sentence_index
- if _index == 0:
- words_num_dict[_index] = 0
- else:
- words_num_dict[_index] = words_num_dict[_index - 1] + last_words_num
- last_words_num = len(sentence.sentence_text)
- match_list = []
- for index in range(len(_list_entity)):
- entity = _list_entity[index]
- if entity.entity_type=="person" and entity.label in [1,2,3]:
- match_nums = 0
- for after_index in range(index + 1, min(len(_list_entity), index + 5)):
- after_entity = _list_entity[after_index]
- if after_entity.entity_type=="phone":
- sentence_distance = after_entity.sentence_index - entity.sentence_index
- distance = (words_num_dict[after_entity.sentence_index] + after_entity.wordOffset_begin) - (
- words_num_dict[entity.sentence_index] + entity.wordOffset_end)
- if sentence_distance < 2 and distance < 50:
- value = (-1 / 2 * (distance ** 2)) / 10000
- match_list.append(Match(entity, after_entity, value))
- match_nums += 1
- else:
- break
- if after_entity.entity_type=="person":
- if after_entity.label not in [1,2,3]:
- break
- if not match_nums:
- for previous_index in range(index-1, max(0,index-5), -1):
- previous_entity = _list_entity[previous_index]
- if previous_entity.entity_type == "phone":
- sentence_distance = entity.sentence_index - previous_entity.sentence_index
- distance = (words_num_dict[entity.sentence_index] + entity.wordOffset_begin) - (
- words_num_dict[previous_entity.sentence_index] + previous_entity.wordOffset_end)
- if sentence_distance < 1 and distance<30:
- # 前向 没有 /10000
- value = (-1 / 2 * (distance ** 2))
- match_list.append(Match(entity, previous_entity, value))
- else:
- break
- result = dispatch(match_list)
- for match in result:
- entity = match.main_role
- # 更新 list_entity
- entity_index = list_entity.index(entity)
- list_entity[entity_index].person_phone = match.attribute.entity_text
- def predict(self,list_sentences,list_entitys):
- self.predict_person(list_sentences,list_entitys)
|