ner.py 36 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632
  1. # -*- coding: utf-8 -*-
  2. """NER 实体识别预处理。
  3. 按 ARCHITECTURE.md Phase 4 拆分建议,从 ``interface/Preprocessing.py`` 迁出。
  4. 类型:PREPROCESS。
  5. 原位置:``interface/Preprocessing.py`` 中以下函数:
  6. - ``get_preprocessed_entitys`` — NER 实体识别与提取
  7. - ``union_ner`` — 连续实体合并(弃用)
  8. - ``union_result`` — 模型结果拼接
  9. ``interface/Preprocessing.py`` 仍 re-export 以上全部名称,老 import 不受影响。
  10. """
  11. from __future__ import absolute_import
  12. import re
  13. import time
  14. from BiddingKG.dl.common.logging import log
  15. from BiddingKG.dl.common.Utils import get_money_entity, cut_repeat_name, changeIndexFromWordToWords, findAllIndex, clean_company, timeFormat
  16. from BiddingKG.dl.interface.Entitys import Entity
  17. from BiddingKG.dl.interface.predictor import getPredictor
  18. from BiddingKG.dl.predictors.table_prem import TableTag2List
  19. from BiddingKG.dl.common.nerUtils import *
  20. from BiddingKG.dl.money.moneySource.ruleExtra import extract_moneySource
  21. from BiddingKG.dl.time.re_servicetime import extract_servicetime
  22. from BiddingKG.dl.relation_extraction.re_email import extract_email
  23. from BiddingKG.dl.ratio.re_ratio import extract_ratio
  24. from BiddingKG.dl.entityLink.entityLink import *
  25. __all__ = [
  26. "get_preprocessed_entitys",
  27. "union_ner",
  28. "union_result",
  29. ]
  30. def union_ner(list_ner):
  31. result_list = []
  32. union_index = []
  33. union_index_set = set()
  34. for i in range(len(list_ner)-1):
  35. if len(set([str(list_ner[i][2]),str(list_ner[i+1][2])])&set(["org","company"]))==2:
  36. if list_ner[i][1]-list_ner[i+1][0]==1:
  37. union_index_set.add(i)
  38. union_index_set.add(i+1)
  39. union_index.append((i,i+1))
  40. for i in range(len(list_ner)):
  41. if i not in union_index_set:
  42. result_list.append(list_ner[i])
  43. for item in union_index:
  44. #print(str(list_ner[item[0]][3])+str(list_ner[item[1]][3]))
  45. result_list.append((list_ner[item[0]][0],list_ner[item[1]][1],'company',str(list_ner[item[0]][3])+str(list_ner[item[1]][3])))
  46. return result_list
  47. def get_preprocessed_entitys(list_sentences,useselffool=True,cost_time=dict()):
  48. '''
  49. :param list_sentences:分局情况
  50. :param cost_time:
  51. :return: list_entitys
  52. '''
  53. list_entitys = []
  54. not_extract_roles = ['黄埔军校', '国有资产管理处', '五金建材', '铝合金门窗', '华电XX发电有限公司', '华电XXX发电有限公司',
  55. '中标(成交)公司', '贵州茅台', '贵州茅台酒', '陕西省省级国', '纪检监察部门', '融安金桔', '海达源组织',
  56. '海达源织', '成交出版社', '中国大学', '第七条交易中心', '铭牌标明出厂', '三、广州分公司'] # 需要过滤掉的企业单位
  57. short_full_dic = {
  58. "中交一航局": "中交第一航务工程局有限公司",
  59. "中交二公局": "中交第二公路工程局有限公司",
  60. "中交二航局": "中交第二航务工程局有限公司",
  61. "中兴": "中兴通讯股份有限公司",
  62. "中国人寿嘉兴分公司": "中国人寿保险股份有限公司嘉兴分公司",
  63. "中国有色集团": "中国有色矿业集团有限公司",
  64. "中建": "中国建筑集团有限公司",
  65. "中核": "中国核工业集团有限公司",
  66. "中船重工": "中国船舶重工集团有限公司",
  67. "中铁": "中国铁路工程集团有限公司",
  68. "北京农商银行": "北京农村商业银行股份有限公司",
  69. "华为": "华为技术有限公司",
  70. "南航物流": "南方航空物流股份有限公司",
  71. "周宁农信联社": "周宁县农村信用合作联社",
  72. "嘉秀集团": "嘉兴市嘉秀发展投资控股集团有限公司",
  73. "国电": "国电电力发展股份有限公司",
  74. "国网": "国家电网有限公司",
  75. "国网北京电力": "国网北京市电力公司",
  76. "国网江苏电力": "国网江苏省电力有限公司",
  77. "国网江西电力": "国网江西省电力有限公司",
  78. "国网浙江电力舟山供电公司": "国网浙江省电力有限公司舟山供电公司",
  79. "国网浙江电力金华供电公司": "国网浙江省电力有限公司金华供电公司",
  80. "国网湖北电力": "国网湖北省电力有限公司",
  81. "山东电建三公司": "山东电力建设第三工程有限公司",
  82. "成都市青羊区国资局": "成都市青羊区国有资产监督管理局",
  83. "新华保险": "新华人寿保险股份有限公司",
  84. "昌吉州人民医院": "昌吉回族自治州人民医院",
  85. "格力": "珠海格力电器股份有限公司",
  86. "欣创环保": "安徽欣创节能环保科技股份有限公司",
  87. "水电四局": "中国水利水电第四工程局有限公司",
  88. "江河集团": "江河创建集团股份有限公司",
  89. "江苏交控": "江苏交通控股有限公司",
  90. "泾河集团": "陕西省西咸新区泾河新城开发建设(集团)有限公司",
  91. "海尔": "海尔集团",
  92. "现代牧业集团": "现代牧业(集团)有限公司",
  93. "美的": "美的集团",
  94. "联想": "联想集团",
  95. "远东股份": "远东智慧能源股份有限公司",
  96. "首都会展集团": "首都会展(集团)有限公司",
  97. "马钢股份": "马鞍山钢铁股份有限公司"
  98. }
  99. for list_sentence in list_sentences:
  100. sentences = []
  101. list_entitys_temp = []
  102. for _sentence in list_sentence:
  103. sentences.append(_sentence.sentence_text)
  104. time1 = time.time()
  105. '''
  106. tokens_all = fool.cut(sentences)
  107. #pos_all = fool.LEXICAL_ANALYSER.pos(tokens_all)
  108. #ner_tag_all = fool.LEXICAL_ANALYSER.ner_labels(sentences,tokens_all)
  109. ner_entitys_all = fool.ner(sentences)
  110. '''
  111. #限流执行
  112. key_nerToken = "nerToken"
  113. start_time = time.time()
  114. found_yeji = 0 # 2021/8/6 增加判断是否正文包含评标结果 及类似业绩判断用于过滤后面的金额
  115. # found_pingbiao = False
  116. ner_entitys_all = getNers(sentences,useselffool=useselffool)
  117. if key_nerToken not in cost_time:
  118. cost_time[key_nerToken] = 0
  119. cost_time[key_nerToken] += round(time.time()-start_time,2)
  120. doctextcon_sentence_len = sum([1 for sentence in list_sentence if not sentence.in_attachment])
  121. company_dict = set()
  122. company_index = dict((i,set()) for i in range(len(list_sentence)))
  123. for sentence_index in range(len(list_sentence)):
  124. list_sentence_entitys = []
  125. sentence_text = list_sentence[sentence_index].sentence_text
  126. tokens = list_sentence[sentence_index].tokens
  127. doc_id = list_sentence[sentence_index].doc_id
  128. in_attachment = list_sentence[sentence_index].in_attachment
  129. list_tokenbegin = []
  130. begin = 0
  131. for i in range(0,len(tokens)):
  132. list_tokenbegin.append(begin)
  133. begin += len(str(tokens[i]))
  134. list_tokenbegin.append(begin+1)
  135. #pos_tag = pos_all[sentence_index]
  136. pos_tag = ""
  137. ner_entitys = ner_entitys_all[sentence_index]
  138. # 20250320 注释掉下面代码 避免带来异常实体
  139. # '''正则识别角色实体 经营部|经销部|电脑部|服务部|复印部|印刷部|彩印部|装饰部|修理部|汽修部|修理店|零售店|设计店|服务店|家具店|专卖店|分店|文具行|商行|印刷厂|修理厂|维修中心|修配中心|养护中心|服务中心|会馆|文化馆|超市|门市|商场|家具城|印刷社|经销处'''
  140. # for it in re.finditer(
  141. # '(?P<text_key_word>(((单一来源|中标|中选|中价|成交)(供应商|供货商|服务商|候选人|单位|人))|(供应商|供货商|服务商|候选人))(名称)?[为::]+)(?P<text>([()\u4e00-\u9fa5]{5,20})(厂|中心|超市|门市|商场|工作室|文印室|城|部|店|站|馆|行|社|处))[,。]',
  142. # sentence_text):
  143. # for k, v in it.groupdict().items():
  144. # if k == 'text_key_word':
  145. # keyword = v
  146. # if k == 'text':
  147. # entity = v
  148. # b = it.start() + len(keyword)
  149. # e = it.end() - 1
  150. # if (b, e, 'location', entity) in ner_entitys:
  151. # ner_entitys.remove((b, e, 'location', entity))
  152. # ner_entitys.append((b, e, 'company', entity))
  153. # elif (b, e, 'org', entity) not in ner_entitys and (b, e, 'company', entity) not in ner_entitys:
  154. # ner_entitys.append((b, e, 'company', entity))
  155. #
  156. # for it in re.finditer(
  157. # '(?P<text_key_word>((建设|招租|招标|采购)(单位|人)|业主)(名称)?[为::]+)(?P<text>[\u4e00-\u9fa5]{2,4}[省市县区镇]([()\u4e00-\u9fa5]{2,20})(管理处|办公室|委员会|村委会|纪念馆|监狱|管教所|修养所|社区|农场|林场|羊场|猪场|石场|村|幼儿园|海关|殡仪馆)|海门\w{2,15}村)[,。]',
  158. # sentence_text):
  159. # for k, v in it.groupdict().items():
  160. # if k == 'text_key_word':
  161. # keyword = v
  162. # if k == 'text':
  163. # entity = v
  164. # b = it.start() + len(keyword)
  165. # e = it.end() - 1
  166. # if (b, e, 'location', entity) in ner_entitys:
  167. # ner_entitys.remove((b, e, 'location', entity))
  168. # ner_entitys.append((b, e, 'org', entity))
  169. # if (b, e, 'org', entity) not in ner_entitys and (b, e, 'company', entity) not in ner_entitys:
  170. # ner_entitys.append((b, e, 'org', entity))
  171. for ner_entity in ner_entitys:
  172. if ner_entity[2] in ['company','org']:
  173. company_dict.add((ner_entity[2],ner_entity[3]))
  174. company_index[sentence_index].add((ner_entity[0],ner_entity[1]))
  175. #识别package
  176. ner_time_list = []
  177. #识别实体
  178. for ner_entity in ner_entitys:
  179. begin_index_temp = ner_entity[0]
  180. end_index_temp = ner_entity[1]
  181. entity_type = ner_entity[2]
  182. entity_text = ner_entity[3]
  183. if entity_type in ["org", "company"] and re.search('^((特殊)?普通合伙)|^(有限合伙)', sentence_text[end_index_temp:]): # 规则补充合伙关键词
  184. partnership = re.search('^((特殊)?普通合伙)|^(有限合伙)', sentence_text[end_index_temp:]).group(0)
  185. end_index_temp += len(partnership)
  186. entity_text += partnership
  187. if entity_type == 'location' and re.search('^\w{2,4}[市县]\w{2,15}(中心|监狱|殡仪馆|水利站)$', entity_text) and \
  188. re.search('\d[楼层号]', entity_text)==None: # 2024/06/07 修改错误地址实体为角色
  189. entity_type = 'org'
  190. elif entity_type in ["org", "company"] and re.search('地址:$', sentence_text[:begin_index_temp]): # 20250421 修复地址识别错为角色 地址:新疆阿拉尔幸福镇十三团,2、运维公司名称:政采云有限公司
  191. entity_type = 'location'
  192. if begin_index_temp>0 and '县' in entity_text and re.match('前郭尔罗斯蒙古族自治县|积石山县', sentence_text[begin_index_temp-1:end_index_temp]): #20240905 修复实体识别少字问题
  193. entity_text = sentence_text[begin_index_temp-1] + entity_text
  194. begin_index_temp -= 1
  195. ner_entity = (begin_index_temp, end_index_temp, entity_type, entity_text)
  196. elif entity_text == '中华人民共和国' and re.search('^\w{2,4}海关', sentence_text[end_index_temp: end_index_temp+6]): # 2024/04/24 修复 采购单位:中华人民共和国汕尾海关, 识别不到海关
  197. ser = re.search('^\w{2,4}海关', sentence_text[end_index_temp: end_index_temp+6])
  198. entity_text += ser.group(0)
  199. end_index_temp += ser.end()
  200. ner_entity = (begin_index_temp, end_index_temp, entity_type, entity_text)
  201. elif entity_text.startswith('中选人为'): # 20251224 修复 697862525 中选人为资阳厚业劳务有限公司 提取为公司
  202. entity_text = entity_text[4:]
  203. begin_index_temp += 4
  204. ner_entity = (begin_index_temp, end_index_temp, entity_type, entity_text)
  205. elif entity_text.startswith('方') and re.search('(出让|受让|成交|中标)方$', sentence_text[begin_index_temp-2:begin_index_temp+1]): # 20260805 修复 809205092 方平山县小觉镇郄家庄村民委员会 提取为公司
  206. entity_text = entity_text[1:]
  207. begin_index_temp += 1
  208. ner_entity = (begin_index_temp, end_index_temp, entity_type, entity_text)
  209. if entity_type=='time':
  210. ner_time_list.append((begin_index_temp,end_index_temp))
  211. if entity_type in ["org","company"] and not isLegalEnterprise(entity_text):
  212. continue
  213. # 实体长度限制
  214. if entity_type in ["org","company"] and len(entity_text)>30:
  215. continue
  216. if entity_type == "person" and len(entity_text) > 20:
  217. continue
  218. elif entity_type=="person" and len(entity_text)>10 and len(re.findall("[\u4e00-\u9fa5]",entity_text))<len(entity_text)/2:
  219. continue
  220. # 识别不完整的组织机构补充
  221. # if entity_type in ["org"]:
  222. # end_words = re.search("^[\u4e00-\u9fa5]{,5}(?:办公室|部|中心|处|会)",sentence_text[end_index_temp:end_index_temp+10]) # 2024/4/7 注释掉 273356356 江门市新会区大鳌镇农村集体资产资源交易中心受新会
  223. # if end_words:
  224. # entity_text = entity_text + end_words.group()
  225. for j in range(len(list_tokenbegin)):
  226. if list_tokenbegin[j]==begin_index_temp:
  227. begin_index = j
  228. break
  229. elif list_tokenbegin[j]>begin_index_temp:
  230. begin_index = j-1
  231. break
  232. begin_index_temp += len(str(entity_text))
  233. for j in range(begin_index,len(list_tokenbegin)):
  234. if list_tokenbegin[j]>=begin_index_temp:
  235. end_index = j-1
  236. break
  237. entity_id = "%s_%d_%d_%d"%(doc_id,sentence_index,begin_index,end_index)
  238. #去掉标点符号
  239. if entity_type!='time':
  240. entity_text = re.sub("[,,。:!&@$\*\s;;]","",entity_text) # 215553737
  241. entity_text = entity_text.replace("(","(").replace(")",")") if isinstance(entity_text,str) else entity_text
  242. # 组织机构实体名称补充
  243. if entity_type in ["org", "company"]:
  244. if entity_text in not_extract_roles: # 过滤掉名称在 需要过滤企业单位列表里的
  245. continue
  246. if not re.search("有限责任公司|有限公司",entity_text):
  247. fix_name = re.search("(有限)([责贵]?任?)(公?司?)",entity_text)
  248. if fix_name:
  249. if len(fix_name.group(2))>0:
  250. _text = fix_name.group()
  251. if '司' in _text:
  252. entity_text = entity_text.replace(_text, "有限责任公司")
  253. else:
  254. _text = re.search(_text + "[^司]{0,5}司", entity_text)
  255. if _text:
  256. _text = _text.group()
  257. entity_text = entity_text.replace(_text, "有限责任公司")
  258. else:
  259. entity_text = entity_text.replace(entity_text[fix_name.start():], "有限责任公司")
  260. elif len(fix_name.group(3))>0:
  261. _text = fix_name.group()
  262. if '司' in _text:
  263. entity_text = entity_text.replace(_text, "有限公司")
  264. else:
  265. _text = re.search(_text + "[^司]{0,3}司", entity_text)
  266. if _text:
  267. _text = _text.group()
  268. entity_text = entity_text.replace(_text, "有限公司")
  269. else:
  270. entity_text = entity_text.replace(entity_text[fix_name.start():], "有限公司")
  271. elif re.search("有限$", entity_text):
  272. entity_text = re.sub("有限$","有限公司",entity_text)
  273. entity_text = entity_text.replace("有公司","有限公司")
  274. '''下面对公司实体进行清洗'''
  275. entity_text = clean_company(entity_text)
  276. if entity_text == '':
  277. continue
  278. entity_text = cut_repeat_name(entity_text) # 20231201 重复名称去重 如:中山大学附属第一医院中山大学附属第一医院中山大学附属第一医院
  279. entity_text = short_full_dic.get(entity_text, entity_text) # 简称映射字典
  280. match = re.match('路桥(第[一二三四五六七八九十]+分公司)$', entity_text) # 20260311 修复 甘肃路桥集中采购管理平台 714787791 采购人只公布简称
  281. if match:
  282. entity_text = '甘肃路桥建设集团有限公司'+match.group(1)
  283. list_sentence_entitys.append(Entity(doc_id,entity_id,entity_text,entity_type,sentence_index,begin_index,end_index,ner_entity[0],ner_entity[1],in_attachment=in_attachment))
  284. # 标记文章末尾的"发布人”、“发布时间”实体
  285. if sentence_index==len(list_sentence)-1 or sentence_index==doctextcon_sentence_len-1:
  286. if len(list_sentence_entitys[-2:])==2:
  287. second2last = list_sentence_entitys[-2]
  288. last = list_sentence_entitys[-1]
  289. if (second2last.entity_type in ["company",'org'] and last.entity_type=="time") or (
  290. second2last.entity_type=="time" and last.entity_type in ["company",'org']):
  291. if last.wordOffset_begin - second2last.wordOffset_end < 6 and len(sentence_text) - last.wordOffset_end<6:
  292. last.is_tail = True
  293. second2last.is_tail = True
  294. #使用正则识别金额
  295. money_list, found_yeji = get_money_entity(sentence_text, found_yeji, in_attachment)
  296. entity_type = "money"
  297. for money in money_list:
  298. # print('money: ', money)
  299. entity_text, begin_index, end_index, unit, notes = money
  300. end_index = end_index - 1 if entity_text.endswith(',') else end_index
  301. entity_id = "%s_%d_%d_%d" % (doc_id, sentence_index, begin_index, end_index)
  302. _exists = False
  303. for item in list_sentence_entitys:
  304. if item.entity_id==entity_id and item.entity_type==entity_type:
  305. _exists = True
  306. if (begin_index >=item.wordOffset_begin and begin_index<item.wordOffset_end) or (end_index>item.wordOffset_begin and end_index<=item.wordOffset_end):
  307. _exists = True
  308. # print('_exists: ',begin_index, end_index, item.wordOffset_begin, item.wordOffset_end, item.entity_text, item.entity_type)
  309. if not _exists:
  310. if float(entity_text)>1:
  311. # if symbol == '-': # 负值金额保留负号
  312. # entity_text = '-'+entity_text # 20230414 取消符号
  313. begin_words = changeIndexFromWordToWords(tokens, begin_index)
  314. end_words = changeIndexFromWordToWords(tokens, end_index)
  315. # print('金额位置: ', begin_index, begin_words,end_index, end_words)
  316. # print('金额召回: ', entity_text, sentence_text[begin_index:end_index], tokens[begin_words:end_words])
  317. list_sentence_entitys.append(Entity(doc_id,entity_id,entity_text,entity_type,sentence_index,begin_words,end_words,begin_index,end_index,in_attachment=in_attachment))
  318. list_sentence_entitys[-1].notes = notes # 2021/7/20 新增金额备注
  319. list_sentence_entitys[-1].money_unit = unit # 2021/7/20 新增金额备注
  320. # print('预处理中的 金额:%s, 单位:%s'%(entity_text,unit))
  321. # print(entity_text,unit,notes)
  322. # "联系人"正则补充提取 2021/11/15 新增
  323. list_person_text = [entity.entity_text for entity in list_sentence_entitys if entity.entity_type=='person']
  324. error_text = ['交易','机构','教育','项目','公司','中标','开标','截标','监督','政府','国家','中国','技术','投标','传真','网址','电子邮',
  325. '联系','联系电','联系地','采购代','邮政编','邮政','电话','手机','手机号','联系人','地址','地点','邮箱','邮编','联系方','招标','招标人','代理',
  326. '代理人','采购','附件','注意','登录','报名','踏勘',"测试",'交货']
  327. list_person_text = set(list_person_text + error_text)
  328. re_person = re.compile("联系人[::]([\u4e00-\u9fa5]工)|"
  329. "联系人[::]([\u4e00-\u9fa5]{2,3})(?=,?联系)|"
  330. "联系人[::]([\u4e00-\u9fa5]{2,3})(?=[,。;、])"
  331. )
  332. list_person = []
  333. if not in_attachment:
  334. for match_result in re_person.finditer(sentence_text):
  335. match_text = match_result.group()
  336. entity_text = match_text[4:]
  337. wordOffset_begin = match_result.start() + 4
  338. wordOffset_end = match_result.end()
  339. # print(text[wordOffset_begin:wordOffset_end])
  340. # 排除一些不为人名的实体
  341. if re.search("^[\u4e00-\u9fa5]{7,}([,。]|$)",sentence_text[wordOffset_begin:wordOffset_begin+20]):
  342. continue
  343. if entity_text not in list_person_text and entity_text[:2] not in list_person_text:
  344. _person = dict()
  345. _person['body'] = entity_text
  346. _person['begin_index'] = wordOffset_begin
  347. _person['end_index'] = wordOffset_end
  348. list_person.append(_person)
  349. entity_type = "person"
  350. for person in list_person:
  351. begin_index_temp = person['begin_index']
  352. for j in range(len(list_tokenbegin)):
  353. if list_tokenbegin[j] == begin_index_temp:
  354. begin_index = j
  355. break
  356. elif list_tokenbegin[j] > begin_index_temp:
  357. begin_index = j - 1
  358. break
  359. index = person['end_index']
  360. end_index_temp = index
  361. for j in range(begin_index, len(list_tokenbegin)):
  362. if list_tokenbegin[j] >= index:
  363. end_index = j - 1
  364. break
  365. entity_id = "%s_%d_%d_%d" % (doc_id, sentence_index, begin_index, end_index)
  366. entity_text = person['body']
  367. list_sentence_entitys.append(
  368. Entity(doc_id, entity_id, entity_text, entity_type, sentence_index, begin_index, end_index,
  369. begin_index_temp, end_index_temp,in_attachment=in_attachment))
  370. # 时间实体格式补充
  371. re_time_new = re.compile("20\d{2}-\d{1,2}-\d{1,2}|"
  372. "20\d{2}-(:?0[1-9]|1[0-2]|[1-9])|"
  373. "20\d{2}/\d{1,2}/\d{1,2}|"
  374. "20\d{2}\.\d{1,2}\.\d{1,2}|"
  375. "20\d{2}(?:0[1-9]|1[0-2])(?:0[1-9]|[1-2][0-9]|3[0-1])")
  376. entity_type = "time"
  377. for _time in re.finditer(re_time_new,sentence_text):
  378. entity_text = _time.group()
  379. begin_index_temp = _time.start()
  380. end_index_temp = _time.end()
  381. is_same = False
  382. for t_index in ner_time_list:
  383. if begin_index_temp>=t_index[0] and end_index_temp<=t_index[1]:
  384. is_same = True
  385. break
  386. if is_same:
  387. continue
  388. if _time.start()!=0 and re.search("\d",sentence_text[_time.start()-1:_time.start()]):
  389. continue
  390. # 纯数字格式,例:20190509
  391. if re.search("^\d{8}$",entity_text):
  392. if _time.end()!=len(sentence_text) and re.search("[\da-zA-z]",sentence_text[_time.end():_time.end()+1]):
  393. continue
  394. elif _time.start()!=0 and re.search("[\da-zA-z]",sentence_text[max(0,_time.start()-1):_time.start()]):
  395. continue
  396. entity_text = entity_text[:4] + "-" + entity_text[4:6] + "-" + entity_text[6:8]
  397. # 例:2025-05
  398. if re.search("^20\d{2}-(:?0[1-9]|1[0-2]|[1-9])$",entity_text):
  399. if _time.end()!=len(sentence_text) and re.search("[\da-zA-z]",sentence_text[_time.end():_time.end()+1]):
  400. continue
  401. elif _time.start()!=0 and re.search("[\da-zA-z]",sentence_text[max(0,_time.start()-1):_time.start()]):
  402. continue
  403. if not timeFormat(entity_text):
  404. continue
  405. for j in range(len(list_tokenbegin)):
  406. if list_tokenbegin[j] == begin_index_temp:
  407. begin_index = j
  408. break
  409. elif list_tokenbegin[j] > begin_index_temp:
  410. begin_index = j - 1
  411. break
  412. for j in range(begin_index, len(list_tokenbegin)):
  413. if list_tokenbegin[j] >= end_index_temp:
  414. end_index = j - 1
  415. break
  416. entity_id = "%s_%d_%d_%d" % (doc_id, sentence_index, begin_index, end_index)
  417. list_sentence_entitys.append(
  418. Entity(doc_id, entity_id, entity_text, entity_type, sentence_index, begin_index, end_index,
  419. begin_index_temp, end_index_temp, in_attachment=in_attachment))
  420. # 资金来源提取 2020/12/30 新增
  421. list_moneySource = extract_moneySource(sentence_text)
  422. entity_type = "moneysource"
  423. for moneySource in list_moneySource:
  424. entity_text = moneySource['body']
  425. if len(entity_text)>50:
  426. continue
  427. begin_index_temp = moneySource['begin_index']
  428. for j in range(len(list_tokenbegin)):
  429. if list_tokenbegin[j] == begin_index_temp:
  430. begin_index = j
  431. break
  432. elif list_tokenbegin[j] > begin_index_temp:
  433. begin_index = j - 1
  434. break
  435. index = moneySource['end_index']
  436. end_index_temp = index
  437. for j in range(begin_index, len(list_tokenbegin)):
  438. if list_tokenbegin[j] >= index:
  439. end_index = j - 1
  440. break
  441. entity_id = "%s_%d_%d_%d" % (doc_id, sentence_index, begin_index, end_index)
  442. list_sentence_entitys.append(
  443. Entity(doc_id, entity_id, entity_text, entity_type, sentence_index, begin_index, end_index,
  444. begin_index_temp, end_index_temp,in_attachment=in_attachment,prob=moneySource['prob']))
  445. # 电子邮箱提取 2021/11/04 新增
  446. list_email = extract_email(sentence_text)
  447. entity_type = "email" # 电子邮箱
  448. for email in list_email:
  449. begin_index_temp = email['begin_index']
  450. for j in range(len(list_tokenbegin)):
  451. if list_tokenbegin[j] == begin_index_temp:
  452. begin_index = j
  453. break
  454. elif list_tokenbegin[j] > begin_index_temp:
  455. begin_index = j - 1
  456. break
  457. index = email['end_index']
  458. end_index_temp = index
  459. for j in range(begin_index, len(list_tokenbegin)):
  460. if list_tokenbegin[j] >= index:
  461. end_index = j - 1
  462. break
  463. entity_id = "%s_%d_%d_%d" % (doc_id, sentence_index, begin_index, end_index)
  464. entity_text = email['body']
  465. list_sentence_entitys.append(
  466. Entity(doc_id, entity_id, entity_text, entity_type, sentence_index, begin_index, end_index,
  467. begin_index_temp, end_index_temp,in_attachment=in_attachment))
  468. # 服务期限提取 2020/12/30 新增
  469. list_servicetime = extract_servicetime(sentence_text)
  470. entity_type = "serviceTime"
  471. for servicetime in list_servicetime:
  472. entity_text = servicetime['body']
  473. begin_index_temp = servicetime['begin_index']
  474. for j in range(len(list_tokenbegin)):
  475. if list_tokenbegin[j] == begin_index_temp:
  476. begin_index = j
  477. break
  478. elif list_tokenbegin[j] > begin_index_temp:
  479. begin_index = j - 1
  480. break
  481. index = servicetime['end_index']
  482. end_index_temp = index
  483. for j in range(begin_index, len(list_tokenbegin)):
  484. if list_tokenbegin[j] >= index:
  485. end_index = j - 1
  486. break
  487. entity_id = "%s_%d_%d_%d" % (doc_id, sentence_index, begin_index, end_index)
  488. list_sentence_entitys.append(
  489. Entity(doc_id, entity_id, entity_text, entity_type, sentence_index, begin_index, end_index,
  490. begin_index_temp, end_index_temp,in_attachment=in_attachment, prob=servicetime["prob"]))
  491. # 2021/12/29 新增比率提取
  492. list_ratio = extract_ratio(sentence_text)
  493. entity_type = "ratio"
  494. for ratio in list_ratio:
  495. # print("ratio", ratio)
  496. begin_index_temp = ratio['begin_index']
  497. for j in range(len(list_tokenbegin)):
  498. if list_tokenbegin[j] == begin_index_temp:
  499. begin_index = j
  500. break
  501. elif list_tokenbegin[j] > begin_index_temp:
  502. begin_index = j - 1
  503. break
  504. index = ratio['end_index']
  505. end_index_temp = index
  506. for j in range(begin_index, len(list_tokenbegin)):
  507. if list_tokenbegin[j] >= index:
  508. end_index = j - 1
  509. break
  510. entity_id = "%s_%d_%d_%d" % (doc_id, sentence_index, begin_index, end_index)
  511. entity_text = ratio['body']
  512. ratio_value = (ratio['value'],ratio['type'])
  513. _entity = Entity(doc_id, entity_id, entity_text, entity_type, sentence_index, begin_index, end_index,
  514. begin_index_temp, end_index_temp,in_attachment=in_attachment)
  515. _entity.ratio_value = ratio_value
  516. list_sentence_entitys.append(_entity)
  517. list_sentence_entitys.sort(key=lambda x:x.begin_index)
  518. list_entitys_temp = list_entitys_temp+list_sentence_entitys
  519. # 补充ner模型未识别全的company/org实体
  520. for sentence_index in range(len(list_sentence)):
  521. sentence_text = list_sentence[sentence_index].sentence_text
  522. tokens = list_sentence[sentence_index].tokens
  523. doc_id = list_sentence[sentence_index].doc_id
  524. in_attachment = list_sentence[sentence_index].in_attachment
  525. list_tokenbegin = []
  526. begin = 0
  527. for i in range(0, len(tokens)):
  528. list_tokenbegin.append(begin)
  529. begin += len(str(tokens[i]))
  530. list_tokenbegin.append(begin + 1)
  531. add_sentence_entitys = []
  532. company_dict = sorted(list(company_dict),key=lambda x:len(x[1]),reverse=True)
  533. for company_type,company_text in company_dict:
  534. begin_index_list = findAllIndex(company_text,sentence_text)
  535. for begin_index in begin_index_list:
  536. is_continue = False
  537. for t_begin,t_end in list(company_index[sentence_index]):
  538. if begin_index>=t_begin and begin_index+len(company_text)<=t_end:
  539. is_continue = True
  540. break
  541. if not is_continue:
  542. add_sentence_entitys.append((begin_index,begin_index+len(company_text),company_type,company_text))
  543. company_index[sentence_index].add((begin_index,begin_index+len(company_text)))
  544. else:
  545. continue
  546. for ner_entity in add_sentence_entitys:
  547. begin_index_temp = ner_entity[0]
  548. end_index_temp = ner_entity[1]
  549. entity_type = ner_entity[2]
  550. entity_text = ner_entity[3]
  551. if entity_type in ["org","company"] and not isLegalEnterprise(entity_text):
  552. continue
  553. for j in range(len(list_tokenbegin)):
  554. if list_tokenbegin[j]==begin_index_temp:
  555. begin_index = j
  556. break
  557. elif list_tokenbegin[j]>begin_index_temp:
  558. begin_index = j-1
  559. break
  560. begin_index_temp += len(str(entity_text))
  561. for j in range(begin_index,len(list_tokenbegin)):
  562. if list_tokenbegin[j]>=begin_index_temp:
  563. end_index = j-1
  564. break
  565. entity_id = "%s_%d_%d_%d"%(doc_id,sentence_index,begin_index,end_index)
  566. if entity_type in ["org","company"] and entity_text in not_extract_roles: # 过滤掉名称在 需要过滤企业单位列表里的
  567. continue
  568. #去掉标点符号
  569. entity_text = re.sub("[,,。:!&@$\*]","",entity_text)
  570. entity_text = entity_text.replace("(","(").replace(")",")") if isinstance(entity_text,str) else entity_text
  571. list_entitys_temp.append(Entity(doc_id,entity_id,entity_text,entity_type,sentence_index,begin_index,end_index,ner_entity[0],ner_entity[1],in_attachment=in_attachment))
  572. list_entitys_temp.sort(key=lambda x:(x.sentence_index,x.begin_index))
  573. list_entitys.append(list_entitys_temp)
  574. return list_entitys
  575. def union_result(codeName,prem):
  576. '''
  577. @summary:模型的结果拼成字典
  578. @param:
  579. codeName:编号名称模型的结果字典
  580. prem:拿到属性的角色的字典
  581. @return:拼接起来的字典
  582. '''
  583. result = []
  584. assert len(codeName)==len(prem)
  585. for item_code,item_prem in zip(codeName,prem):
  586. result.append(dict(item_code,**item_prem))
  587. return result