prem.py 31 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581
  1. # -*- coding: utf-8 -*-
  2. """``PREMPredict`` / ``EPCPredict`` — 角色金额模型与联系人模型。
  3. Phase 5 从 ``interface/predictor.py`` 迁出:
  4. - ``PREMPredict``(约 795-1229 行)
  5. - ``EPCPredict``(约 1230-1577 行)
  6. 原 ``from common.Utils import *`` / ``from interface.modelFactory import *``
  7. 已替换为显式 import。两个类均无 ``os.path.dirname(__file__)`` 实际路径引用
  8. (仅注释中出现,保持原样未改)。
  9. """
  10. from __future__ import absolute_import
  11. import re
  12. import numpy as np
  13. from BiddingKG.dl.common.context_utils import spanWindow
  14. from BiddingKG.dl.interface.Entitys import Entity
  15. from BiddingKG.dl.interface.modelFactory import (
  16. Model_role_classify_word,
  17. Model_money_classify,
  18. Model_person_classify,
  19. )
  20. from BiddingKG.dl.predictors.role_context import build_role_contexts
  21. from BiddingKG.dl.predictors.role_rule_engine import RoleRuleEngine
  22. from BiddingKG.dl.services.external_inference.paieas import (
  23. USE_PAI_EAS,
  24. tf_predict_pb2,
  25. vpc_requests,
  26. role_url,
  27. role_authorization,
  28. money_url,
  29. money_authorization,
  30. person_url,
  31. person_authorization,
  32. )
  33. __all__ = ["PREMPredict", "EPCPredict"]
  34. #角色金额模型
  35. class PREMPredict():
  36. def __init__(self,config=None):
  37. #self.model_role_file = os.path.abspath("../role/models/model_role.model.hdf5")
  38. # self.model_role_file = os.path.dirname(__file__)+"/../role/log/new_biLSTM-ep012-loss0.028-val_loss0.040-f10.954.h5"
  39. self.model_role = Model_role_classify_word(config=config)
  40. self.model_money = Model_money_classify(config=config)
  41. # Phase C:模型后修正规则外置到 dl/rules/patterns/role_context_fix.yaml,
  42. # 由 RoleRuleEngine 按 stage 执行(行为等价原内联 if/elif 链,Phase D 双跑验证)
  43. self.rule_engine = RoleRuleEngine()
  44. # self.role_file = open('/data/python/lsm/role_model_predict.txt', 'a', encoding='utf-8')
  45. # self.money_file = open('/data/python/lsm/money_model_predict.txt', 'a', encoding='utf-8')
  46. return
  47. def search_role_data(self,list_sentences,list_entitys,contexts=None):
  48. '''
  49. @summary:根据句子list和实体list查询角色模型的输入数据
  50. @param:
  51. list_sentences:文章的sentences
  52. list_entitys:文章的entitys
  53. contexts:Phase A 预计算上下文(build_role_contexts 输出),
  54. None 时内部构建(角色/金额模型共享一次实体-句子配对)
  55. @return:角色模型的输入数据
  56. '''
  57. if contexts is None:
  58. contexts = build_role_contexts(list_sentences, list_entitys)
  59. text_list = []
  60. data_x = []
  61. points_entitys = []
  62. for ctx_list in contexts:
  63. for ctx in ctx_list:
  64. entity = ctx.entity
  65. if entity.entity_type in ['org','company']:
  66. text_list.append(ctx.role_model_text)
  67. # item_x = embedding(spanWindow(tokens=sentence.tokens,begin_index=entity.begin_index,end_index=entity.end_index,size=settings.MODEL_ROLE_INPUT_SHAPE[1]),shape=settings.MODEL_ROLE_INPUT_SHAPE)
  68. # item_x = self.model_role.encode(tokens=sentence.tokens,begin_index=entity.begin_index,end_index=entity.end_index,entity_text=entity.entity_text)
  69. item_x = self.model_role.encode_word(sentence_text=ctx.sentence_text, begin_index=entity.wordOffset_begin, end_index=entity.wordOffset_end, size=30)
  70. data_x.append(item_x)
  71. points_entitys.append(entity)
  72. if len(points_entitys)==0:
  73. return None
  74. return [data_x,points_entitys, text_list]
  75. def search_money_data(self,list_sentences,list_entitys,contexts=None):
  76. '''
  77. @summary:根据句子list和实体list查询金额模型的输入数据
  78. @param:
  79. list_sentences:文章的sentences
  80. list_entitys:文章的entitys
  81. contexts:Phase A 预计算上下文(build_role_contexts 输出),
  82. None 时内部构建(角色/金额模型共享一次实体-句子配对)
  83. @return:金额模型的输入数据
  84. '''
  85. if contexts is None:
  86. contexts = build_role_contexts(list_sentences, list_entitys)
  87. text_list = []
  88. data_x = []
  89. points_entitys = []
  90. for ctx_list in contexts:
  91. for ctx in ctx_list:
  92. entity = ctx.entity
  93. if entity.entity_type=="money":
  94. text_list.append(ctx.money_model_text)
  95. #item_x = embedding(spanWindow(tokens=sentence.tokens,begin_index=entity.begin_index,end_index=entity.end_index,size=settings.MODEL_MONEY_INPUT_SHAPE[1]),shape=settings.MODEL_MONEY_INPUT_SHAPE)
  96. #item_x = embedding_word(spanWindow(tokens=sentence.tokens, begin_index=entity.begin_index, end_index=entity.end_index, size=10, center_include=True, word_flag=True),shape=settings.MODEL_MONEY_INPUT_SHAPE)
  97. item_x = self.model_money.encode(tokens=ctx.sentence.tokens,begin_index=entity.begin_index,end_index=entity.end_index)
  98. data_x.append(item_x)
  99. points_entitys.append(entity)
  100. if len(points_entitys)==0:
  101. return None
  102. return [data_x,points_entitys, text_list]
  103. def predict_role(self,list_sentences, list_entitys, contexts=None):
  104. '''
  105. @param contexts: Phase A 预计算上下文(build_role_contexts 输出),
  106. None 时内部构建;与 predict_money 共用同一份避免重复配对。
  107. '''
  108. datas = self.search_role_data(list_sentences, list_entitys, contexts=contexts)
  109. if datas is None:
  110. return
  111. points_entitys = datas[1]
  112. text_list = datas[2]
  113. if USE_PAI_EAS:
  114. _data = datas[0]
  115. _data = np.transpose(np.array(_data),(1,0,2))
  116. request = tf_predict_pb2.PredictRequest()
  117. request.inputs["input0"].dtype = tf_predict_pb2.DT_FLOAT
  118. request.inputs["input0"].array_shape.dim.extend(np.shape(_data[0]))
  119. request.inputs["input0"].float_val.extend(np.array(_data[0],dtype=np.float64).reshape(-1))
  120. request.inputs["input1"].dtype = tf_predict_pb2.DT_FLOAT
  121. request.inputs["input1"].array_shape.dim.extend(np.shape(_data[1]))
  122. request.inputs["input1"].float_val.extend(np.array(_data[1],dtype=np.float64).reshape(-1))
  123. request.inputs["input2"].dtype = tf_predict_pb2.DT_FLOAT
  124. request.inputs["input2"].array_shape.dim.extend(np.shape(_data[2]))
  125. request.inputs["input2"].float_val.extend(np.array(_data[2],dtype=np.float64).reshape(-1))
  126. request_data = request.SerializeToString()
  127. list_outputs = ["outputs"]
  128. _result = vpc_requests(role_url, role_authorization, request_data, list_outputs)
  129. if _result is not None:
  130. predict_y = _result["outputs"]
  131. else:
  132. predict_y = self.model_role.predict(datas[0])
  133. else:
  134. predict_y = self.model_role.predict(np.array(datas[0],dtype=np.float64))
  135. for i in range(len(predict_y)):
  136. entity = points_entitys[i]
  137. label = np.argmax(predict_y[i])
  138. values = predict_y[i]
  139. # text = text_list[i]
  140. text_tup = text_list[i]
  141. front, middle, behind = text_tup
  142. # Phase C:内联修正规则外置到
  143. # dl/rules/patterns/role_context_fix.yaml(stage: global_pre/seq_check/
  144. # l0/l2/l5_win_yes/notify/l1/l34/l5),由 RoleRuleEngine 执行,
  145. # 行为等价原 if/elif 链(Phase D badcase 双跑回归验证)。
  146. label = self.rule_engine.correct_role(entity, label, values, front, middle, behind)
  147. entity.set_Role(label, values)
  148. def predict_money(self,list_sentences,list_entitys,contexts=None):
  149. '''
  150. @param contexts: Phase A 预计算上下文(build_role_contexts 输出),
  151. None 时内部构建;与 predict_role 共用同一份避免重复配对。
  152. '''
  153. datas = self.search_money_data(list_sentences, list_entitys, contexts=contexts)
  154. if datas is None:
  155. return
  156. points_entitys = datas[1]
  157. _data = datas[0]
  158. text_list = datas[2]
  159. if USE_PAI_EAS:
  160. _data = np.transpose(np.array(_data),(1,0,2,3))
  161. request = tf_predict_pb2.PredictRequest()
  162. request.inputs["input0"].dtype = tf_predict_pb2.DT_FLOAT
  163. request.inputs["input0"].array_shape.dim.extend(np.shape(_data[0]))
  164. request.inputs["input0"].float_val.extend(np.array(_data[0],dtype=np.float64).reshape(-1))
  165. request.inputs["input1"].dtype = tf_predict_pb2.DT_FLOAT
  166. request.inputs["input1"].array_shape.dim.extend(np.shape(_data[1]))
  167. request.inputs["input1"].float_val.extend(np.array(_data[1],dtype=np.float64).reshape(-1))
  168. request.inputs["input2"].dtype = tf_predict_pb2.DT_FLOAT
  169. request.inputs["input2"].array_shape.dim.extend(np.shape(_data[2]))
  170. request.inputs["input2"].float_val.extend(np.array(_data[2],dtype=np.float64).reshape(-1))
  171. request_data = request.SerializeToString()
  172. list_outputs = ["outputs"]
  173. _result = vpc_requests(money_url, money_authorization, request_data, list_outputs)
  174. if _result is not None:
  175. predict_y = _result["outputs"]
  176. else:
  177. predict_y = self.model_money.predict(_data)
  178. else:
  179. predict_y = self.model_money.predict(_data)
  180. for i in range(len(predict_y)):
  181. entity = points_entitys[i]
  182. label = np.argmax(predict_y[i])
  183. values = predict_y[i]
  184. # text = text_list[i]
  185. text_tup = text_list[i]
  186. front, middle, behind = text_tup
  187. # Phase C:内联修正规则外置到
  188. # dl/rules/patterns/role_money_fix.yaml(stage: m1/m0/m_bid),
  189. # 由 RoleRuleEngine 执行,行为等价原 if/elif 链
  190. # (Phase D badcase 双跑回归验证)。
  191. label = self.rule_engine.correct_money(entity, label, values, front, middle, behind)
  192. entity.set_Money(label, values)
  193. def correct_money_by_rule(self, title, list_entitys, list_articles):
  194. # Phase C:标题/正文类别批量修正外置到
  195. # dl/rules/patterns/role_money_fix.yaml(stage: doc_title),
  196. # 由 RoleRuleEngine 执行,行为等价原实现(Phase D 双跑回归验证)。
  197. self.rule_engine.correct_money_by_doc(title, list_articles[0].content, list_entitys)
  198. def predict(self,list_sentences,list_entitys,contexts=None):
  199. '''
  200. @param contexts: Phase A 预计算上下文(build_role_contexts 输出),
  201. None 时内部构建一次,role/money 两个模型共享,
  202. 消除原来各自的实体-句子配对与切片构建。
  203. '''
  204. if contexts is None:
  205. contexts = build_role_contexts(list_sentences, list_entitys)
  206. self.predict_role(list_sentences,list_entitys,contexts=contexts)
  207. self.predict_money(list_sentences,list_entitys,contexts=contexts)
  208. #联系人模型
  209. class EPCPredict():
  210. def __init__(self,config=None):
  211. self.model_person = Model_person_classify(config=config)
  212. def search_person_data(self,list_sentences,list_entitys):
  213. '''
  214. @summary:根据句子list和实体list查询联系人模型的输入数据
  215. @param:
  216. list_sentences:文章的sentences
  217. list_entitys:文章的entitys
  218. @return:联系人模型的输入数据
  219. '''
  220. data_x = []
  221. points_entitys = []
  222. pre_texts = []
  223. for list_entity,list_sentence in zip(list_entitys,list_sentences):
  224. p_entitys = 0
  225. dict_index_sentence = {}
  226. for _sentence in list_sentence:
  227. dict_index_sentence[_sentence.sentence_index] = _sentence
  228. _list_entity = [entity for entity in list_entity if entity.entity_type=="person"]
  229. while(p_entitys<len(_list_entity)):
  230. entity = _list_entity[p_entitys]
  231. if entity.entity_type=="person":
  232. sentence = dict_index_sentence[entity.sentence_index]
  233. item_x = self.model_person.encode(tokens=sentence.tokens,begin_index=entity.begin_index,end_index=entity.end_index)
  234. data_x.append(item_x)
  235. points_entitys.append(entity)
  236. pre_texts.append(spanWindow(tokens=sentence.tokens,begin_index=entity.begin_index,end_index=entity.end_index,size=20))
  237. p_entitys += 1
  238. if len(points_entitys)==0:
  239. return None
  240. # return [data_x,points_entitys,dianhua]
  241. return [data_x,points_entitys, pre_texts]
  242. def predict_person(self,list_sentences, list_entitys):
  243. datas = self.search_person_data(list_sentences, list_entitys)
  244. if datas is None:
  245. return
  246. points_entitys = datas[1]
  247. pre_texts = datas[2]
  248. # phone = datas[2]
  249. if USE_PAI_EAS:
  250. _data = datas[0]
  251. _data = np.transpose(np.array(_data),(1,0,2,3))
  252. request = tf_predict_pb2.PredictRequest()
  253. request.inputs["input0"].dtype = tf_predict_pb2.DT_FLOAT
  254. request.inputs["input0"].array_shape.dim.extend(np.shape(_data[0]))
  255. request.inputs["input0"].float_val.extend(np.array(_data[0],dtype=np.float64).reshape(-1))
  256. request.inputs["input1"].dtype = tf_predict_pb2.DT_FLOAT
  257. request.inputs["input1"].array_shape.dim.extend(np.shape(_data[1]))
  258. request.inputs["input1"].float_val.extend(np.array(_data[1],dtype=np.float64).reshape(-1))
  259. request_data = request.SerializeToString()
  260. list_outputs = ["outputs"]
  261. _result = vpc_requests(person_url, person_authorization, request_data, list_outputs)
  262. if _result is not None:
  263. predict_y = _result["outputs"]
  264. else:
  265. predict_y = self.model_person.predict(datas[0])
  266. else:
  267. predict_y = self.model_person.predict(datas[0])
  268. # assert len(predict_y)==len(points_entitys)==len(phone)
  269. assert len(predict_y)==len(points_entitys)
  270. for i in range(len(predict_y)):
  271. entity = points_entitys[i]
  272. label = np.argmax(predict_y[i])
  273. pre_text = ''.join(pre_texts[i][0])
  274. # print('pre_text', pre_text)
  275. if label==0 and re.search('(谈判|磋商|询价|资格审查|评审专家|(评选|议标|评标|评审)委员会?|专家|评委)(小?组|小?组成员)?(成员|名单)[:,](\w{2,4}((组长)|(成员))?[、,,])*$', pre_text):
  276. # print(entity.entity_text, re.search('(谈判|磋商|询价|资格审查|评审专家|(评选|议标|评标|评审)委员会?|专家|评委)(小?组|小?组成员)?(成员|名单)[:,](\w{2,4}((组长)|(成员))?[、,,])*$', pre_text).group(0))
  277. label = 4
  278. values = []
  279. for item in predict_y[i]:
  280. values.append(item)
  281. # phone_number = phone[i]
  282. # entity.set_Person(label,values,phone_number)
  283. entity.set_Person(label,values,[])
  284. # 为联系人匹配电话
  285. # self.person_search_phone(list_sentences, list_entitys)
  286. def person_search_phone(self,list_sentences, list_entitys):
  287. def phoneFromList(phones):
  288. # for phone in phones:
  289. # if len(phone)==11:
  290. # return re.sub('电话[:|:]|联系方式[:|:]','',phone)
  291. return re.sub('电话[:|:]|联系方式[:|:]', '', phones[0])
  292. for list_entity, list_sentence in zip(list_entitys, list_sentences):
  293. # p_entitys = 0
  294. # p_sentences = 0
  295. #
  296. # key_word = re.compile('电话[:|:].{0,4}\d{7,12}|联系方式[:|:].{0,4}\d{7,12}')
  297. # # phone = re.compile('1[3|4|5|7|8][0-9][-—-]?\d{4}[-—-]?\d{4}|\d{3,4}[-—]\d{7,8}/\d{3,8}|\d{3,4}[-—]\d{7,8}转\d{1,4}|\d{3,4}[-—]\d{7,8}|[\(|\(]0\d{2,3}[\)|\)]-?\d{7,8}-?\d{,4}') # 联系电话
  298. # # 2020/11/25 增加发现的号码段
  299. # phone = re.compile('1[3|4|5|6|7|8|9][0-9][-—-]?\d{4}[-—-]?\d{4}|'
  300. # '\d{3,4}[-—][1-9]\d{6,7}/\d{3,8}|'
  301. # '\d{3,4}[-—]\d{7,8}转\d{1,4}|'
  302. # '\d{3,4}[-—]?[1-9]\d{6,7}|'
  303. # '[\(|\(]0\d{2,3}[\)|\)]-?\d{7,8}-?\d{,4}|'
  304. # '[1-9]\d{6,7}') # 联系电话
  305. # dict_index_sentence = {}
  306. # for _sentence in list_sentence:
  307. # dict_index_sentence[_sentence.sentence_index] = _sentence
  308. #
  309. # dict_context_itemx = {}
  310. # last_person = "####****++++$$^"
  311. # last_person_phone = "####****++++$^"
  312. # _list_entity = [entity for entity in list_entity if entity.entity_type == "person"]
  313. # while (p_entitys < len(_list_entity)):
  314. # entity = _list_entity[p_entitys]
  315. # if entity.entity_type == "person" and entity.label in [1,2,3]:
  316. # sentence = dict_index_sentence[entity.sentence_index]
  317. # # item_x = embedding(spanWindow(tokens=sentence.tokens,begin_index=entity.begin_index,end_index=entity.end_index,size=settings.MODEL_PERSON_INPUT_SHAPE[1]),shape=settings.MODEL_PERSON_INPUT_SHAPE)
  318. #
  319. # # s = spanWindow(tokens=sentence.tokens,begin_index=entity.begin_index,end_index=entity.end_index,size=20)
  320. #
  321. # # 2021/5/8 取上下文的句子,解决表格处理的分句问题
  322. # left_sentence = dict_index_sentence.get(entity.sentence_index - 1)
  323. # left_sentence_tokens = left_sentence.tokens if left_sentence else []
  324. # right_sentence = dict_index_sentence.get(entity.sentence_index + 1)
  325. # right_sentence_tokens = right_sentence.tokens if right_sentence else []
  326. # entity_beginIndex = entity.begin_index + len(left_sentence_tokens)
  327. # entity_endIndex = entity.end_index + len(left_sentence_tokens)
  328. # context_sentences_tokens = left_sentence_tokens + sentence.tokens + right_sentence_tokens
  329. # s = spanWindow(tokens=context_sentences_tokens, begin_index=entity_beginIndex,
  330. # end_index=entity_endIndex, size=20)
  331. #
  332. # _key = "".join(["".join(x) for x in s])
  333. # if _key in dict_context_itemx:
  334. # _dianhua = dict_context_itemx[_key][0]
  335. # else:
  336. # s1 = ''.join(s[1])
  337. # # s1 = re.sub(',)', '-', s1)
  338. # s1 = re.sub('\s', '', s1)
  339. # have_key = re.findall(key_word, s1)
  340. # have_phone = re.findall(phone, s1)
  341. # s0 = ''.join(s[0])
  342. # # s0 = re.sub(',)', '-', s0)
  343. # s0 = re.sub('\s', '', s0)
  344. # have_key2 = re.findall(key_word, s0)
  345. # have_phone2 = re.findall(phone, s0)
  346. #
  347. # s3 = ''.join(s[1])
  348. # # s0 = re.sub(',)', '-', s0)
  349. # s3 = re.sub(',|,|\s', '', s3)
  350. # have_key3 = re.findall(key_word, s3)
  351. # have_phone3 = re.findall(phone, s3)
  352. #
  353. # s4 = ''.join(s[0])
  354. # # s0 = re.sub(',)', '-', s0)
  355. # s4 = re.sub(',|,|\s', '', s0)
  356. # have_key4 = re.findall(key_word, s4)
  357. # have_phone4 = re.findall(phone, s4)
  358. #
  359. # _dianhua = ""
  360. # if have_phone:
  361. # if entity.entity_text != last_person and s0.find(last_person) != -1 and s1.find(
  362. # last_person_phone) != -1:
  363. # if len(have_phone) > 1:
  364. # _dianhua = phoneFromList(have_phone[1:])
  365. # else:
  366. # _dianhua = phoneFromList(have_phone)
  367. # elif have_key:
  368. # if entity.entity_text != last_person and s0.find(last_person) != -1 and s1.find(
  369. # last_person_phone) != -1:
  370. # if len(have_key) > 1:
  371. # _dianhua = phoneFromList(have_key[1:])
  372. # else:
  373. # _dianhua = phoneFromList(have_key)
  374. # elif have_phone2:
  375. # if entity.entity_text != last_person and s0.find(last_person) != -1 and s0.find(
  376. # last_person_phone) != -1:
  377. # if len(have_phone2) > 1:
  378. # _dianhua = phoneFromList(have_phone2[1:])
  379. # else:
  380. # _dianhua = phoneFromList(have_phone2)
  381. # elif have_key2:
  382. # if entity.entity_text != last_person and s0.find(last_person) != -1 and s0.find(
  383. # last_person_phone) != -1:
  384. # if len(have_key2) > 1:
  385. # _dianhua = phoneFromList(have_key2[1:])
  386. # else:
  387. # _dianhua = phoneFromList(have_key2)
  388. # elif have_phone3:
  389. # if entity.entity_text != last_person and s4.find(last_person) != -1 and s3.find(
  390. # last_person_phone) != -1:
  391. # if len(have_phone3) > 1:
  392. # _dianhua = phoneFromList(have_phone3[1:])
  393. # else:
  394. # _dianhua = phoneFromList(have_phone3)
  395. # elif have_key3:
  396. # if entity.entity_text != last_person and s4.find(last_person) != -1 and s3.find(
  397. # last_person_phone) != -1:
  398. # if len(have_key3) > 1:
  399. # _dianhua = phoneFromList(have_key3[1:])
  400. # else:
  401. # _dianhua = phoneFromList(have_key3)
  402. # elif have_phone4:
  403. # if entity.entity_text != last_person and s4.find(last_person) != -1 and s4.find(
  404. # last_person_phone) != -1:
  405. # if len(have_phone4) > 1:
  406. # _dianhua = phoneFromList(have_phone4)
  407. # else:
  408. # _dianhua = phoneFromList(have_phone4)
  409. # elif have_key4:
  410. # if entity.entity_text != last_person and s4.find(last_person) != -1 and s4.find(
  411. # last_person_phone) != -1:
  412. # if len(have_key4) > 1:
  413. # _dianhua = phoneFromList(have_key4)
  414. # else:
  415. # _dianhua = phoneFromList(have_key4)
  416. # else:
  417. # _dianhua = ""
  418. # # dict_context_itemx[_key] = [item_x, _dianhua]
  419. # dict_context_itemx[_key] = [_dianhua]
  420. # # points_entitys.append(entity)
  421. # # dianhua.append(_dianhua)
  422. # last_person = entity.entity_text
  423. # if _dianhua:
  424. # # 更新联系人entity联系方式(person_phone)
  425. # entity.person_phone = _dianhua
  426. # last_person_phone = _dianhua
  427. # else:
  428. # last_person_phone = "####****++++$^"
  429. # p_entitys += 1
  430. from scipy.optimize import linear_sum_assignment
  431. from BiddingKG.dl.interface.Entitys import Match
  432. def dispatch(match_list):
  433. main_roles = list(set([match.main_role for match in match_list]))
  434. attributes = list(set([match.attribute for match in match_list]))
  435. label = np.zeros(shape=(len(main_roles), len(attributes)))
  436. for match in match_list:
  437. main_role = match.main_role
  438. attribute = match.attribute
  439. value = match.value
  440. label[main_roles.index(main_role), attributes.index(attribute)] = value + 10000
  441. # print(label)
  442. gragh = -label
  443. # km算法
  444. row, col = linear_sum_assignment(gragh)
  445. max_dispatch = [(i, j) for i, j, value in zip(row, col, gragh[row, col]) if value]
  446. return [Match(main_roles[row], attributes[col]) for row, col in max_dispatch]
  447. # km算法
  448. key_word = re.compile('((?:电话|联系方式|联系人).{0,4}?)(\d{7,12})')
  449. phone = re.compile('1[3|4|5|6|7|8|9][0-9][-—-―]?\d{4}[-—-―]?\d{4}|'
  450. '\+86.?1[3|4|5|6|7|8|9]\d{9}|'
  451. '0\d{2,3}[-—-―][1-9]\d{6,7}/[1-9]\d{6,10}|'
  452. '0\d{2,3}[-—-―]\d{7,8}转\d{1,4}|'
  453. '0\d{2,3}[-—-―]?[1-9]\d{6,7}|'
  454. '[\(|\(]0\d{2,3}[\)|\)]-?\d{7,8}-?\d{,4}|'
  455. '[1-9]\d{6,7}')
  456. phone_entitys = []
  457. for _sentence in list_sentence:
  458. sentence_text = _sentence.sentence_text
  459. res_set = set()
  460. for i in re.finditer(phone,sentence_text):
  461. res_set.add((i.group(),i.start(),i.end()))
  462. for i in re.finditer(key_word,sentence_text):
  463. res_set.add((i.group(2),i.start()+len(i.group(1)),i.end()))
  464. for item in list(res_set):
  465. phone_left = sentence_text[max(0,item[1]-10):item[1]]
  466. phone_right = sentence_text[item[2]:item[2]+8]
  467. # 排除传真号 和 其它错误项
  468. if re.search("传,?真|信,?箱|邮,?箱",phone_left):
  469. if not re.search("电,?话",phone_left):
  470. continue
  471. if re.search("帐,?号|编,?号|报,?价|证,?号|价,?格|[\((]万?元[\))]",phone_left):
  472. continue
  473. if re.search("[.,]\d{2,}",phone_right):
  474. continue
  475. _entity = Entity(_sentence.doc_id, None, item[0], "phone", _sentence.sentence_index, None, None,item[1], item[2],in_attachment=_sentence.in_attachment)
  476. phone_entitys.append(_entity)
  477. person_entitys = []
  478. for entity in list_entity:
  479. if entity.entity_type == "person":
  480. entity.person_phone = ""
  481. person_entitys.append(entity)
  482. _list_entity = phone_entitys + person_entitys
  483. _list_entity = sorted(_list_entity,key=lambda x:(x.sentence_index,x.wordOffset_begin))
  484. words_num_dict = dict()
  485. last_words_num = 0
  486. list_sentence = sorted(list_sentence, key=lambda x: x.sentence_index)
  487. for sentence in list_sentence:
  488. _index = sentence.sentence_index
  489. if _index == 0:
  490. words_num_dict[_index] = 0
  491. else:
  492. words_num_dict[_index] = words_num_dict[_index - 1] + last_words_num
  493. last_words_num = len(sentence.sentence_text)
  494. match_list = []
  495. for index in range(len(_list_entity)):
  496. entity = _list_entity[index]
  497. if entity.entity_type=="person" and entity.label in [1,2,3]:
  498. match_nums = 0
  499. for after_index in range(index + 1, min(len(_list_entity), index + 5)):
  500. after_entity = _list_entity[after_index]
  501. if after_entity.entity_type=="phone":
  502. sentence_distance = after_entity.sentence_index - entity.sentence_index
  503. distance = (words_num_dict[after_entity.sentence_index] + after_entity.wordOffset_begin) - (
  504. words_num_dict[entity.sentence_index] + entity.wordOffset_end)
  505. if sentence_distance < 2 and distance < 50:
  506. value = (-1 / 2 * (distance ** 2)) / 10000
  507. match_list.append(Match(entity, after_entity, value))
  508. match_nums += 1
  509. else:
  510. break
  511. if after_entity.entity_type=="person":
  512. if after_entity.label not in [1,2,3]:
  513. break
  514. if not match_nums:
  515. for previous_index in range(index-1, max(0,index-5), -1):
  516. previous_entity = _list_entity[previous_index]
  517. if previous_entity.entity_type == "phone":
  518. sentence_distance = entity.sentence_index - previous_entity.sentence_index
  519. distance = (words_num_dict[entity.sentence_index] + entity.wordOffset_begin) - (
  520. words_num_dict[previous_entity.sentence_index] + previous_entity.wordOffset_end)
  521. if sentence_distance < 1 and distance<30:
  522. # 前向 没有 /10000
  523. value = (-1 / 2 * (distance ** 2))
  524. match_list.append(Match(entity, previous_entity, value))
  525. else:
  526. break
  527. result = dispatch(match_list)
  528. for match in result:
  529. entity = match.main_role
  530. # 更新 list_entity
  531. entity_index = list_entity.index(entity)
  532. list_entity[entity_index].person_phone = match.attribute.entity_text
  533. def predict(self,list_sentences,list_entitys):
  534. self.predict_person(list_sentences,list_entitys)