district.py 78 KB

12345678910111213141516171819202122232425262728293031323334353637383940414243444546474849505152535455565758596061626364656667686970717273747576777879808182838485868788899091929394959697989910010110210310410510610710810911011111211311411511611711811912012112212312412512612712812913013113213313413513613713813914014114214314414514614714814915015115215315415515615715815916016116216316416516616716816917017117217317417517617717817918018118218318418518618718818919019119219319419519619719819920020120220320420520620720820921021121221321421521621721821922022122222322422522622722822923023123223323423523623723823924024124224324424524624724824925025125225325425525625725825926026126226326426526626726826927027127227327427527627727827928028128228328428528628728828929029129229329429529629729829930030130230330430530630730830931031131231331431531631731831932032132232332432532632732832933033133233333433533633733833934034134234334434534634734834935035135235335435535635735835936036136236336436536636736836937037137237337437537637737837938038138238338438538638738838939039139239339439539639739839940040140240340440540640740840941041141241341441541641741841942042142242342442542642742842943043143243343443543643743843944044144244344444544644744844945045145245345445545645745845946046146246346446546646746846947047147247347447547647747847948048148248348448548648748848949049149249349449549649749849950050150250350450550650750850951051151251351451551651751851952052152252352452552652752852953053153253353453553653753853954054154254354454554654754854955055155255355455555655755855956056156256356456556656756856957057157257357457557657757857958058158258358458558658758858959059159259359459559659759859960060160260360460560660760860961061161261361461561661761861962062162262362462562662762862963063163263363463563663763863964064164264364464564664764864965065165265365465565665765865966066166266366466566666766866967067167267367467567667767867968068168268368468568668768868969069169269369469569669769869970070170270370470570670770870971071171271371471571671771871972072172272372472572672772872973073173273373473573673773873974074174274374474574674774874975075175275375475575675775875976076176276376476576676776876977077177277377477577677777877978078178278378478578678778878979079179279379479579679779879980080180280380480580680780880981081181281381481581681781881982082182282382482582682782882983083183283383483583683783883984084184284384484584684784884985085185285385485585685785885986086186286386486586686786886987087187287387487587687787887988088188288388488588688788888989089189289389489589689789889990090190290390490590690790890991091191291391491591691791891992092192292392492592692792892993093193293393493593693793893994094194294394494594694794894995095195295395495595695795895996096196296396496596696796896997097197297397497597697797897998098198298398498598698798898999099199299399499599699799899910001001100210031004100510061007100810091010101110121013101410151016101710181019102010211022102310241025102610271028102910301031103210331034103510361037103810391040104110421043104410451046104710481049105010511052105310541055105610571058105910601061106210631064106510661067106810691070107110721073107410751076107710781079108010811082108310841085108610871088108910901091109210931094109510961097109810991100110111021103110411051106110711081109111011111112111311141115111611171118111911201121112211231124112511261127112811291130113111321133113411351136113711381139114011411142114311441145114611471148114911501151115211531154115511561157115811591160116111621163116411651166116711681169117011711172117311741175117611771178117911801181118211831184118511861187118811891190119111921193119411951196119711981199120012011202120312041205120612071208120912101211121212131214121512161217121812191220122112221223122412251226122712281229123012311232123312341235123612371238123912401241124212431244124512461247124812491250125112521253125412551256125712581259126012611262126312641265126612671268126912701271127212731274127512761277127812791280128112821283128412851286
  1. # -*- coding: utf-8 -*-
  2. """``DistrictPredictor`` — 地区匹配规则预测。
  3. Phase 5 从 ``interface/predictor.py``(约 6770-8032 行)迁出。
  4. 原 ``from common.Utils import *`` / ``from common.nerUtils import *``
  5. 已替换为显式 import;``os.path.dirname(__file__)`` 路径引用替换为
  6. ``predictors._common.INTERFACE_DIR``。使用 district_tuple.pkl、
  7. area_variance_dic.pkl 等地区字典。
  8. """
  9. from __future__ import absolute_import
  10. import os
  11. import re
  12. import pickle
  13. from collections import Counter
  14. from BiddingKG.dl.common.logging import log
  15. from BiddingKG.dl.common.nerUtils import getTokens, getNers
  16. from BiddingKG.dl.entityLink.entityLink import get_business_data
  17. from BiddingKG.dl.predictors._common import INTERFACE_DIR
  18. __all__ = ["DistrictPredictor"]
  19. class DistrictPredictor():
  20. def __init__(self):
  21. # with open(os.path.dirname(__file__)+'/district_dic.pkl', 'rb') as f:
  22. # dist_dic = pickle.load(f)
  23. # short_name = '|'.join(sorted(set([v['简称'] for v in dist_dic.values()]), key=lambda x: len(x), reverse=True))
  24. # full_name = '|'.join(sorted(set([v['全称'] for v in dist_dic.values()]), key=lambda x: len(x), reverse=True))
  25. # short2id = {}
  26. # full2id = {}
  27. # for k, v in dist_dic.items():
  28. # if v['简称'] not in short2id:
  29. # short2id[v['简称']] = [k]
  30. # else:
  31. # short2id[v['简称']].append(k)
  32. # if v['全称'] not in full2id:
  33. # full2id[v['全称']] = [k]
  34. # else:
  35. # full2id[v['全称']].append(k)
  36. # self.dist_dic = dist_dic
  37. # self.short_name = short_name
  38. # self.full_name = full_name
  39. # self.short2id = short2id
  40. # self.full2id = full2id
  41. # # self.f = open(os.path.dirname(__file__)+'/../test/data/district_predict.txt', 'w', encoding='utf-8')
  42. with open(os.path.join(INTERFACE_DIR, 'district_tuple.pkl'), 'rb') as f:
  43. district_tuple = pickle.load(f)
  44. self.p_pro, self.p_city, self.p_dis, self.idx_dic, self.full_dic, self.short_dic = district_tuple
  45. # self.pettern = "((?P<prov>%s)(?P<city>%s)?(?P<dist>%s)?)|((?P<city1>%s)(?P<dist1>%s)?)|(?P<dist2>%s)" % (
  46. # self.p_pro, self.p_city, self.p_dis, self.p_city, self.p_dis, self.p_dis)
  47. short_pro = '黑龙江|内蒙古|青海|陕西|辽宁|贵州|西藏|福建|甘肃|湖南|湖北|海南|浙江|河南|河北|江西|江苏|新疆|广西|广东|山西|山东|安徽|宁夏|四川|吉林|云南'
  48. self.pettern = "(?P<prov>%s)##(?P<city>%s)##(?P<dist>%s)" % (
  49. self.p_pro, self.p_city, self.p_dis)
  50. self.pettern_pro = re.compile("(?P<prov>%s)"%self.p_pro)
  51. self.pettern_city = re.compile("(%s)?(?P<city>%s)"%(short_pro, self.p_city)) # 20250925补充省简称,解决 海南昌江 匹配为南昌问题
  52. self.pettern_dist = re.compile("(%s)?(?P<dist>%s)"%(short_pro, self.p_dis))
  53. with open(os.path.join(INTERFACE_DIR, "area_variance_dic.pkl"), 'rb') as f: # 20241113 地区变更新旧名称对照字典
  54. self.area_variance_dic = pickle.load(f)
  55. self.multi_dist = ['向阳区', '宝山区', '南沙区', '和平区', '新城区', '鼓楼区', '南山区', '白云区', '朝阳区',
  56. '江北区', '城关区', '永定区', '普陀区', '长安区', '市中区', '西安区', '通州区', '西湖区',
  57. '龙华区', '城中区', '河东区', '桥西区', '青山区', '新华区', '铁西区', '铁东区', '海州区', '滨海新区']
  58. def find_whole_areas(self, text, pettern, area_variance_dic, full_dic, weight=1):
  59. '''
  60. 通过正则匹配字符串返回地址
  61. :param pettern: 地址正则 广东省|广西省|...
  62. :param text: 待匹配文本
  63. :return:
  64. '''
  65. province_l, city_l, district_l = [], [], []
  66. citys = []
  67. text = str(text).replace('(', '(').replace(')', ')')
  68. text = re.sub('\d{2,4}年度?|[\d/-]{1,5}[月日]|\d+|[a-zA-Z0-9]+', ' ', text)
  69. text = re.sub(
  70. '复合肥|海南岛|兴业银行|双河口|阳光|杭州湾|新城区|中粮屯河|老城(区|改造|更新|升级|翻新)|沙县小吃|北京时间|福田汽车|中山(大学|公园|纪念堂)|孙中山|海天水泥|阳光采购|示范县|珠江城?|西九龙站|广州路北|安阳山村|电信|联通|北京现代|祁连山|锡铁山|大黄山(?!市)|红旗汽车', # 570445994 广州路北侧 预测为 广州 路北
  71. ' ', text) # 544151395 赤壁市老城区燃气管道老化更新改造
  72. text = re.sub('珠海城市', '珠海', text) # 修复 426624023 珠海城市 预测为海城市
  73. text = re.sub('怒江州', '怒江傈僳族自治州', text) # 修复 423589589 所属地域:怒江州 识别为广西 - 崇左 - 江州
  74. text = re.sub('茂名滨海新区', '茂名市', text)
  75. text = re.sub('中山([东南西][部区环]|黄圃|南头|东凤|小榄|石岐|翠亨|南朗)', '中山市', text)
  76. text = re.sub('横州市', '横县', text) # 例:547363890 修复广西南宁横州 不在地区表问题
  77. text = re.sub('广东中山', '广东中山市', text)
  78. text = re.sub('朝阳柳城经济开发区', '朝阳市', text)
  79. text = re.sub('安徽徽运城市', '安徽', text) # 653054198 安徽徽运城市运营管理有限公司
  80. text = re.sub('西安丰镇', '宝应', text) # 修复 681946122 宝应县西安丰镇
  81. text = re.sub('西城区', '西城', text) # 修复 653130377 标题 董家口电厂至新区西城区长输热力管线工程目 预测错北京西城
  82. # 处理地区+大学不在该地区问题
  83. text = re.sub('(河北工业大学)', '天津市', text)
  84. text = re.sub('(西藏民族大学)', '咸阳市', text)
  85. text = re.sub('(四川外国语大学|四川美术学院)', '重庆市', text)
  86. text = re.sub('(中山大学)', '广州市', text)
  87. text = re.sub('(滨州医学院)', '烟台市', text)
  88. ser = re.search('海南(昌江|白沙|乐东|陵水|保亭|琼中)(黎族)?', text)
  89. if ser and '黎族' not in ser.group(0):
  90. text = text.replace(ser.group(0), ser.group(0) + '黎族')
  91. for k, v in area_variance_dic.items(): # 20241113 根据地区变更信息替换文本
  92. text = text.replace(k, v)
  93. text = re.sub('\s+', ' ', text)
  94. if re.search('[\u4e00-\u9fa5]', text) == None:
  95. return province_l, city_l, district_l
  96. name_set = set() # 提取到的所有地址集合
  97. tokens = getTokens([text], useselffool=True)[0]
  98. # print('句子:', text)
  99. # print('分词:', tokens)
  100. for pettern in [self.pettern_pro, self.pettern_city, self.pettern_dist]: # pettern.split('##')
  101. for it in re.finditer(pettern, text):
  102. if it.group(0) == '站前': # 20240314 修复类似 中铁二局新建沪苏湖铁路工程站前VI标项目 错识别为 省份:辽宁, 城市:营口,区县:站前
  103. continue
  104. for k, v in it.groupdict().items():
  105. if v != None:
  106. if it.end() == it.end(k) and re.search('[省市区县州旗盟]$', v) == None and re.search(
  107. '^([东南西北中一二三四五六七八九十大小]?(村|镇|街|路|道|社区|巷|坊)|酒店|宾馆|经济开发区|开发区|新区|公园|广场|公馆|小区|幼儿园)', # |医院|[大中小]学 # 20250917取消地区+医院过滤,大部分是在该地区
  108. # 城市不匹配为区的地址 修复 滨州北海经济开发区 北海新区 等提取为北海
  109. text[it.end(k):]) != None and not re.match('路桥', text[it.end(k):]): # 修复 746181930 甘肃路桥 被路去掉
  110. continue
  111. if k in ['prov']:
  112. if v in full_dic['province']:
  113. score = 2
  114. else:
  115. score = 1
  116. if re.search('^(\w{,2}[分支](公司|局|行|院|干?线)|(机务|车辆|车务|工务|电务|供电|动车)?段|地铁|(火车|高铁)?站|港|地区|区域|基地)'
  117. , text[it.end(k):]) or re.search('^((%s)|\-%s)' % (v, v),
  118. text[max(0, it.start(k) - 1):]):
  119. score += 1
  120. elif re.search('大学|学院', text[:it.start(k)]) and re.search('分校|校区', text[it.end(k):]): # 长春市第八十七中学南阳校区 不在南阳市
  121. score += 1
  122. if len(v) < 3 and v not in tokens: # 不在分词里面概率降低
  123. score /= 2
  124. else:
  125. score += it.end(k) / len(text) / 10
  126. province_l.append((v, score * weight))
  127. elif k in ['city', 'city1']:
  128. if v in full_dic['city']:
  129. score = 2
  130. else:
  131. score = 1
  132. if re.search('^(\w{,2}[分支](公司|局|行|院|干?线)|(机务|车辆|车务|工务|电务|供电|动车)?段|地铁|(火车|高铁)?站|港|地区|区域|基地)'
  133. , text[it.end(k):]) or re.search('^((%s)|\-%s)' % (v, v),
  134. text[max(0, it.start(k) - 1):]):
  135. score += 1
  136. elif re.search('大学|学院', text[:it.start(k)]) and re.search('分校|校区', text[it.end(k):]):
  137. score += 1
  138. if len(v) < 3 and v not in tokens: # 不在分词里面概率降低 优化 653058471 山东恒通化工股份有限公司 错分 通化
  139. score /= 2
  140. else:
  141. score += it.end(k) / len(text) / 10 # 优化 572840045 上海铁路公安局合肥公安处 这种表达
  142. city_l.append((v, score * weight))
  143. citys.append(v)
  144. elif k in ['dist', 'dist1', 'dist2']:
  145. if v in ['东区', '西区', '城区', '郊区', '矿区', '东至']:
  146. continue
  147. elif v.endswith('城市') and text[it.start(k)-1:it.start(k)+1] in citys: # 修复 上海城市 珠海城市 等 预测为 海城市
  148. continue
  149. if v in self.multi_dist or re.search('\w城区$', v): # 多个城市有的区概率降低
  150. score = 0.5
  151. elif v in full_dic['district'] and (len(v) > 2 or v.endswith('县')): # 20250709 修复 萧县 等概率过低
  152. score = 2
  153. else:
  154. score = 0.5
  155. if re.search('^(\w{,2}[分支](公司|局|行|院|干?线)|(机务|车辆|车务|工务|电务|供电|动车)?段|地铁|(火车|高铁)?站|港|地区|区域|基地)'
  156. , text[it.end(k):]) or (
  157. re.match('\s*%s' % v, text) and it.start(k) < 2) or re.search(
  158. '^((%s)|\-%s)' % (v, v), text[max(0, it.start(k) - 1):]):
  159. score += 0.5
  160. elif re.search('大学|学院', text[:it.start(k)]) and re.search('分校|校区', text[it.end(k):]):
  161. score += 0.5
  162. if len(v) < 3 and v not in tokens: # 不在分词里面概率降低,三字以上不受限制 ,避免类似 德令哈工务段 分词不对
  163. score /= 2
  164. # score += it.end(k) / len(text) / 10
  165. district_l.append((v, score * weight))
  166. name_set.add(v)
  167. if len(name_set) > 1: # 解决 类似 山西宁武 匹配出 山西 西宁 宁武 问题 修复 653057648 把峨眉山市作为眉山市
  168. names = re.findall('|'.join(sorted(name_set, key=lambda x: len(x), reverse=True)), text)
  169. province_l = [it for it in province_l if it[0] in names]
  170. city_l = [it for it in city_l if it[0] in names]
  171. district_l = [it for it in district_l if it[0] in names]
  172. return province_l, city_l, district_l
  173. def merge_score(self, province_l, city_l, district_l, full_dic, short_dic, idx_dic, filter_short_dist=True):
  174. '''
  175. 合并分数,下级地区分数加到上级
  176. :param province_l: 提取到的省份列表 [(name, score)]
  177. :param city_l: 提取到的城市列表 [(name, score)]
  178. :param district_l: 提取到的区县列表 [(name, score)]
  179. :param filter_short_dist: 是否过滤不在省份下的区县简称权重
  180. :return:
  181. '''
  182. pro_ids = dict()
  183. city_ids = dict()
  184. dis_ids = dict()
  185. for pro in province_l:
  186. name, score = pro
  187. idx = full_dic['province'][name] if name in full_dic['province'] else short_dic['province'][name]
  188. if idx not in pro_ids:
  189. pro_ids[idx] = 0
  190. pro_ids[idx] += score
  191. tmp_pro = {}
  192. for city in city_l:
  193. name, score = city
  194. if name in full_dic['city']:
  195. for idx in full_dic['city'][name]:
  196. if idx not in city_ids:
  197. city_ids[idx] = 0
  198. city_ids[idx] += score
  199. pro_idx = idx_dic[idx]['省']
  200. if pro_idx in tmp_pro:
  201. tmp_pro[pro_idx] += score
  202. else:
  203. tmp_pro[pro_idx] = score
  204. elif name in short_dic['city']:
  205. for idx in short_dic['city'][name]:
  206. if idx not in city_ids:
  207. city_ids[idx] = 0
  208. city_ids[idx] += score
  209. pro_idx = idx_dic[idx]['省']
  210. if pro_ids != {} and pro_idx not in pro_ids: # 如果省份不为空且简称不在省份分值降低 优化 653150317 海南农垦阳江农场有限公司 错分 阳江
  211. score -= 0.1
  212. if pro_idx in tmp_pro:
  213. tmp_pro[pro_idx] += score
  214. else:
  215. tmp_pro[pro_idx] = score
  216. if set(tmp_pro) & set(pro_ids) != set():
  217. for k, v in tmp_pro.items():
  218. if k in pro_ids:
  219. pro_ids[k] += v
  220. else:
  221. pro_ids[k] = v
  222. else:
  223. pro_ids.update(tmp_pro)
  224. tmp_pro = {}
  225. tmp_city = {}
  226. for dis in district_l:
  227. name, score = dis
  228. if name in full_dic['district']:
  229. for idx in full_dic['district'][name]:
  230. if idx not in dis_ids:
  231. dis_ids[idx] = 0
  232. dis_ids[idx] += score
  233. pro_idx = idx_dic[idx]['省']
  234. if pro_idx in tmp_pro:
  235. tmp_pro[pro_idx] += score
  236. else:
  237. if name in self.multi_dist: # 多个城市重复名称,需过滤
  238. continue
  239. tmp_pro[pro_idx] = score
  240. city_idx = idx_dic[idx]['市']
  241. if city_idx in tmp_city:
  242. tmp_city[city_idx] += score
  243. else:
  244. if name in self.multi_dist: # 多个城市重复名称,需过滤
  245. continue
  246. tmp_city[city_idx] = score
  247. elif name in short_dic['district']:
  248. for idx in short_dic['district'][name]:
  249. if idx not in dis_ids:
  250. dis_ids[idx] = 0
  251. dis_ids[idx] += score
  252. pro_idx = idx_dic[idx]['省']
  253. if pro_ids != {} and pro_idx not in pro_ids: # 如果省份不为空且简称不在省份分值降低
  254. score -= 0.1
  255. if filter_short_dist and score < 1: # pro_idx not in pro_ids
  256. continue
  257. if pro_idx in tmp_pro:
  258. tmp_pro[pro_idx] += score
  259. else:
  260. tmp_pro[pro_idx] = score
  261. city_idx = idx_dic[idx]['市']
  262. if city_idx in tmp_city:
  263. tmp_city[city_idx] += score
  264. else:
  265. tmp_city[city_idx] = score
  266. if set(tmp_pro) & set(pro_ids) != set():
  267. for k, v in tmp_pro.items():
  268. if k in pro_ids:
  269. pro_ids[k] += v
  270. else:
  271. pro_ids.update(tmp_pro)
  272. if set(tmp_city) & set(city_ids) != set():
  273. for k, v in tmp_city.items():
  274. if k in city_ids:
  275. city_ids[k] += v
  276. else:
  277. city_ids.update(tmp_city)
  278. return pro_ids, city_ids, dis_ids
  279. @staticmethod
  280. def get_final_addr(pro_ids, city_ids, dis_ids, idx_dic):
  281. '''
  282. 先把所有匹配的全称、简称转为id,如果省份不为空,城市不为空且有城市属于省份的取该城市
  283. :param province_l: 匹配到的所有省份
  284. :param city_l: 匹配到的所有城市
  285. :param district_l: 匹配到的所有区县
  286. :return:
  287. '''
  288. big_area = ""
  289. pred_pro = ""
  290. pred_city = ""
  291. pred_dis = ""
  292. final_pro = ""
  293. final_city = ""
  294. prob = 0
  295. max_score = 0
  296. code_dic = {
  297. 'province_code': '',
  298. 'city_code': '',
  299. 'district_code': ''
  300. }
  301. if len(pro_ids) >= 1:
  302. pro_l = sorted([(k, v) for k, v in pro_ids.items()], key=lambda x: x[1], reverse=True)
  303. scores = [it[1] for it in pro_l]
  304. prob = max(scores) / sum(scores)
  305. max_score = max(scores)
  306. final_pro, score = pro_l[0]
  307. if score >= 0.01:
  308. pred_pro = idx_dic[final_pro]['返回名称']
  309. big_area = idx_dic[final_pro]['大区']
  310. code_dic['province_code'] = idx_dic[final_pro]['编码']
  311. if pred_pro != "" and len(city_ids) >= 1:
  312. city_l = sorted([(k, v) for k, v in city_ids.items()], key=lambda x: x[1], reverse=True)
  313. for it in city_l:
  314. if idx_dic[it[0]]['省'] == final_pro:
  315. final_city = it[0]
  316. pred_city = idx_dic[final_city]['返回名称']
  317. code_dic['city_code'] = idx_dic[final_city]['编码']
  318. break
  319. if final_city != "" and len(set(dis_ids)) >= 1:
  320. dis_l = sorted([(k, v) for k, v in dis_ids.items()], key=lambda x: x[1], reverse=True)
  321. for it in dis_l:
  322. if idx_dic[it[0]]['市'] == final_city:
  323. pred_dis = idx_dic[it[0]]['返回名称']
  324. code_dic['district_code'] = idx_dic[it[0]]['编码']
  325. elif pred_pro != "" and pred_city == "" and len(set(dis_ids)) >= 1: # 20241111 省份不为空,市为空,如果区县在省份下,补充对应的市县
  326. dis_l = sorted([(k, v) for k, v in dis_ids.items()], key=lambda x: x[1], reverse=True)
  327. for it in dis_l:
  328. if idx_dic[it[0]]['省'] == final_pro:
  329. pred_city = idx_dic[idx_dic[it[0]]['市']]['返回名称']
  330. pred_dis = idx_dic[it[0]]['返回名称']
  331. code_dic['city_code'] = idx_dic[idx_dic[it[0]]['市']]['编码']
  332. code_dic['district_code'] = idx_dic[it[0]]['编码']
  333. return big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic
  334. @staticmethod
  335. def get_ree_addr(prem):
  336. tenderee = ""
  337. tenderee_address = ""
  338. try:
  339. for v in prem.values():
  340. for link in v['roleList']:
  341. if link['role_name'] == 'tenderee' and tenderee == "":
  342. tenderee = link['role_text']
  343. tenderee_address = link['address']
  344. except Exception as e:
  345. print('解析prem 获取招标人、及地址出错')
  346. return tenderee, tenderee_address
  347. @staticmethod
  348. def get_role_address(text):
  349. '''正则匹配获取招标人地址
  350. 3:地址直接在招标人后面 招标人:xxx,地址:xxx
  351. 4:招标、代理一起,两个地址一起 招标人:xxx, 代理人:xxx, 地址:xxx, 地址:xxx.
  352. '''
  353. p3 = '(招标|采购|甲)(人|方|单位)(信息:|(甲方))?(名称)?:[\w()]{4,15},(联系)?地址:(?P<addr>(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+|\w{2,15}[,。])'
  354. p4 = '(招标|采购|甲)(人|方|单位)(信息:|(甲方))?(名称)?:[\w()]{4,15},(招标|采购)?代理(人|机构)(名称)?:[\w()]{4,15},(联系)?地址:(?P<addr>(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+|\w{2,35}),(联系)?地址:'
  355. p5 = '(采购|招标)(人|单位)(联系)?地址:(?P<addr>(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+|\w{2,15}[,。])'
  356. if re.search(p3, text):
  357. return re.search(p3, text).group('addr')
  358. elif re.search(p4, text):
  359. return re.search(p4, text).group('addr')
  360. elif re.search(p5, text):
  361. return re.search(p5, text).group('addr')
  362. else:
  363. return ''
  364. @staticmethod
  365. def get_all_addr(list_entity, filter_attach=False):
  366. '''
  367. 获取所有招标或代理人名称及所有地址
  368. :param list_entity: 实体列表
  369. :param filter_attach: 是否过滤附件实体
  370. :return:
  371. '''
  372. tenderee_l = []
  373. addr_l = []
  374. for ent in list_entity:
  375. if ent.entity_type not in ['org', 'company', 'location'] or (filter_attach and ent.in_attachment):
  376. continue
  377. if ent.entity_type == 'location' and len(ent.entity_text) > 2:
  378. addr_l.append(ent.entity_text)
  379. elif ent.entity_type in ['org', 'company']:
  380. if ent.label in [0, 1]: # 加招标或代理
  381. tenderee_l.append(ent.entity_text)
  382. elif re.search('[局委]$', ent.entity_text):
  383. tenderee_l.append(ent.entity_text)
  384. if len(addr_l) > 10: # 只取前10个地址
  385. break
  386. return ' '.join(set(addr_l)), ' '.join(set(tenderee_l))
  387. def addr_process(self, addr_text):
  388. addr_text = addr_text.replace('(', '(').replace(')', ')')
  389. if re.search('[省市县]', addr_text) == None:
  390. ser = re.search('\w{2,}区', addr_text)
  391. ser2 = re.match('\w{2,5}镇', addr_text)
  392. if ser:
  393. addr_text = addr_text[:ser.end()]
  394. elif ser2:
  395. addr_text = addr_text[:ser2.end()]
  396. return addr_text
  397. def predict_area(self,docid, title, content, web_source_name, prem={}, addr_dic={}, list_entity=[]):
  398. if re.match('([^,]{3,50}网络竞价会),', content): # 修复 701397187 标题只有拍卖的东西,第一句有地址
  399. title_auction = re.match('([^,]{3,50}网络竞价会),', content).group(1)
  400. if title.find(title_auction[:3]) == -1:
  401. title += ' ' + title_auction
  402. filter_attach = False # 是否过滤附件内容
  403. if '##attachment##' in content:
  404. main, att = content.split('##attachment##')
  405. if 2000 < len(main) < len(att): # 正文超过500字且附件比正文长过滤附件内容
  406. filter_attach = True
  407. # print('正文超过500字过滤附件内容')
  408. ree, addr_ree = self.get_ree_addr(prem)
  409. addr_ree = self.addr_process(addr_ree)
  410. addr_bus = ''
  411. if len(addr_ree) < 3 and ree != '':
  412. have_bus, bus_dic = get_business_data(ree)
  413. if have_bus:
  414. addr_bus = '%s %s %s' % (bus_dic.get('province', ''), bus_dic.get('city', ''), bus_dic.get('district', ''))
  415. all_addr, tenderees = self.get_all_addr(list_entity, filter_attach)
  416. return self.predict_distrist(docid, title, web_source_name, ree, addr_ree, addr_bus, addr_dic, tenderees, all_addr)
  417. def predict_distrist(self,docid, title, web_source_name, ree, addr_ree, addr_bus,location_dic={}, tenderees='', all_addr=''):
  418. '''
  419. 优先顺序项目地址、收货地址、招标人地址 / 工商地址、开标地址 / 联系地址、站源名称
  420. '''
  421. area_dic = {'area': '全国', 'province': '全国', 'city': '未知', 'district': '未知', "is_in_text": False}
  422. in_content = False
  423. not_sure = True # 是否不确定地区
  424. msc = "" # 日志信息
  425. addr_dic = {}
  426. prov_name_l = [] # 保存预测到的省份名称
  427. city_name_l = [] # 保存预测到的城市名称
  428. final_key = ''
  429. company_title = ''
  430. title_raw = title
  431. if len(title.strip()) > 4:
  432. ner_title = getNers([title], True)[0]
  433. company_title = []
  434. for ner in ner_title:
  435. if ner[2] in ['org', 'company']:
  436. company_title.append(ner[3])
  437. title = title.replace(ner[3], '#')
  438. company_title = "#".join(company_title)
  439. if re.search('拍卖|法院', ree):
  440. company_title += ' ' + ree
  441. ree = ''
  442. if web_source_name in ('中原云商', '中原云商电子招投标平台'): # 修复某些公告没地区
  443. web_source_name = '河南中原云商'
  444. for key in ['addr_project', 'addr_delivery', 'addr_bidopen', 'addr_bidsend', 'addr_contact']:
  445. addr = location_dic.get(key, '')
  446. addr = self.addr_process(addr)
  447. if len(addr) < 2:
  448. continue
  449. province_l, city_l, district_l = self.find_whole_areas('%s'%addr, self.pettern, self.area_variance_dic, self.full_dic)
  450. if len(province_l+city_l+district_l)>0:
  451. pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic,self.short_dic, self.idx_dic)
  452. big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = self.get_final_addr(pro_ids, city_ids, dis_ids, self.idx_dic)
  453. if max_score < 2 and len(addr) > len(''.join([it[0] for it in province_l+city_l+district_l]))*2: # 修复 712151540 北京银行大厦6楼厨房及22楼 提取为北京
  454. log('地区匹配非正常地址:%s, 预测为:%s %s %s, docid:%s'%(addr, pred_pro, pred_city, pred_dis, docid))
  455. continue
  456. if pred_pro != '':
  457. addr_dic[key] = {
  458. 'keyword': (province_l, city_l, district_l),
  459. 'result': (big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic)
  460. }
  461. msc += "使用%s信息:%s, 预测为:%s %s %s;" % (key, addr, pred_pro, pred_city, pred_dis)
  462. prov_name_l.append(pred_pro)
  463. if pred_city != '':
  464. city_name_l.append(pred_city)
  465. if key == 'addr_project' and addr_dic['addr_project']['result'][2] != '' and addr_dic['addr_project']['result'][4] > 0.6 and (addr_dic['addr_project']['result'][5] >= 2 or addr_dic['addr_project']['keyword'][2]==[]):
  466. big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic['addr_project']['result']
  467. not_sure = False
  468. final_key = key
  469. msc += "最终使用:%s,预测为:%s %s %s;"%(final_key, pred_pro, pred_city, pred_dis)
  470. if addr.endswith('#在附件'):
  471. log('地区匹配使用附件中的项目地址:%s;预测为:%s %s;docid:%s'%(addr, pred_pro, pred_city, docid))
  472. break
  473. if not_sure:
  474. for key, text in zip(['tenderee', 'company_title', 'web_source_name'], [ree, company_title, web_source_name]):
  475. if len(text) < 4:
  476. continue
  477. weight = 0.48 if key == 'web_source_name' else 1
  478. province_l, city_l, district_l = self.find_whole_areas('%s'%text, self.pettern, self.area_variance_dic, self.full_dic, weight=weight)
  479. if len(province_l+city_l+district_l)>0:
  480. pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic,self.short_dic, self.idx_dic)
  481. big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = self.get_final_addr(pro_ids, city_ids, dis_ids, self.idx_dic)
  482. if pred_pro != '':
  483. addr_dic[key] = {
  484. 'keyword': (province_l, city_l, district_l),
  485. 'result': (big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic)
  486. }
  487. msc += "使用%s信息:%s, 预测为:%s %s %s;" % (key, text, pred_pro, pred_city, pred_dis)
  488. prov_name_l.append(pred_pro)
  489. if pred_city != '':
  490. city_name_l.append(pred_city)
  491. elif key == 'company_title' and re.search('\w{1,}[省市县]', text): # 修复 2025年可克达拉市政府部门独立办公楼聘用保安保洁采购服务 实体提取为 克达拉市政府 造成市漏提
  492. title = title_raw
  493. # 提取招标地址
  494. if len(addr_ree) >= 3:
  495. province_l, city_l, district_l = self.find_whole_areas('%s' % addr_ree, self.pettern, self.area_variance_dic,self.full_dic)
  496. if len(province_l + city_l + district_l) > 0:
  497. pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic,self.short_dic, self.idx_dic)
  498. big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = self.get_final_addr(pro_ids,city_ids,dis_ids,self.idx_dic)
  499. if pred_pro != '':
  500. addr_dic['addr_tenderee'] = {
  501. 'keyword': (province_l, city_l, district_l),
  502. 'result': (big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic)
  503. }
  504. msc += "使用招标人地址:%s, 预测为:%s %s %s;" % (addr_ree, pred_pro, pred_city, pred_dis)
  505. prov_name_l.append(pred_pro)
  506. if pred_city != '':
  507. city_name_l.append(pred_city)
  508. # 提取招标人工商登记地址
  509. if len(addr_bus) >= 3:
  510. province_l, city_l, district_l = self.find_whole_areas('%s' % addr_bus, self.pettern, self.area_variance_dic,self.full_dic)
  511. if len(province_l + city_l + district_l) > 0:
  512. pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic,self.short_dic, self.idx_dic)
  513. big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = self.get_final_addr(pro_ids,city_ids,dis_ids,self.idx_dic)
  514. if pred_pro != '':
  515. addr_dic['addr_bus'] = {
  516. 'keyword': (province_l, city_l, district_l),
  517. 'result': (big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic)
  518. }
  519. msc += "使用招标人工商登记地址:%s, 预测为:%s %s %s;" % (addr_bus, pred_pro, pred_city, pred_dis)
  520. prov_name_l.append(pred_pro)
  521. if pred_city != '':
  522. city_name_l.append(pred_city)
  523. # 提取标题地址
  524. if len(title) > 3:
  525. province_l, city_l, district_l = self.find_whole_areas('%s' % title, self.pettern, self.area_variance_dic,self.full_dic)
  526. if len(province_l + city_l + district_l) > 0:
  527. pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic,self.short_dic, self.idx_dic)
  528. big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = self.get_final_addr(pro_ids,city_ids,dis_ids,self.idx_dic)
  529. if pred_pro != '':
  530. addr_dic['addr_title'] = {
  531. 'keyword': (province_l, city_l, district_l),
  532. 'result': (big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic)
  533. }
  534. msc += "使用标题地址:%s, 预测为:%s %s %s;" % (title, pred_pro, pred_city, pred_dis)
  535. prov_name_l.append(pred_pro)
  536. if pred_city != '':
  537. city_name_l.append(pred_city)
  538. if len(addr_dic) == 1:
  539. for key in addr_dic:
  540. big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic[key]['result']
  541. final_key = key
  542. msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis)
  543. elif len(addr_dic) > 1:
  544. prov_count = Counter(prov_name_l)
  545. city_count = Counter(city_name_l)
  546. if len(prov_count) == 1 and len(city_count) == 1:
  547. score_max = 0
  548. for key in ['addr_project', 'addr_delivery', 'addr_tenderee', 'addr_title', 'tenderee', 'company_title',
  549. 'addr_bidopen', 'addr_bidsend', 'addr_contact', 'addr_bus', 'web_source_name']:
  550. if key in addr_dic and addr_dic[key]['result'][1] == prov_name_l[0] and addr_dic[key]['result'][2] == city_name_l[0] and addr_dic[key]['result'][5] > score_max:
  551. score_max = addr_dic[key]['result'][5]
  552. final_key = key
  553. if addr_dic[final_key]['result'][3] != '': # 2026/2/2 省市匹配,且有区县停止循环
  554. break
  555. big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic[final_key]['result']
  556. msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis)
  557. # if key in addr_dic and addr_dic[key]['result'][1] == prov_name_l[0] and addr_dic[key]['result'][2] == city_name_l[0]:
  558. # # big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic[key]['result']
  559. # final_key = key
  560. # msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis)
  561. # break
  562. elif 'addr_project' in addr_dic and addr_dic['addr_project']['result'][2] == '': # 只有省份
  563. big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic['addr_project']['result']
  564. final_key = 'addr_project'
  565. score_max = 0
  566. tmp_key = ''
  567. for key in ['addr_delivery', 'addr_tenderee', 'addr_title', 'tenderee', 'company_title',
  568. 'addr_bidopen', 'addr_bidsend', 'addr_contact', 'addr_bus', 'web_source_name']:
  569. if key in addr_dic and addr_dic[key]['result'][1] == pred_pro and addr_dic[key]['result'][2] != '' and addr_dic[key]['result'][5] > score_max:
  570. score_max = addr_dic[key]['result'][5]
  571. tmp_key = key
  572. if addr_dic[tmp_key]['result'][3] != '': # 2026/2/2 省市匹配,且有区县停止循环
  573. break
  574. if tmp_key != '':
  575. final_key += ',%s' % tmp_key
  576. big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic[tmp_key]['result']
  577. msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis)
  578. elif 'addr_delivery' in addr_dic and addr_dic['addr_delivery']['result'][2] != '' and addr_dic['addr_delivery']['result'][4] > 0.7 and addr_dic['addr_delivery']['result'][5] >= 2:
  579. big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic['addr_delivery']['result']
  580. final_key = 'addr_delivery'
  581. msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis)
  582. elif 'addr_title' in addr_dic and addr_dic['addr_title']['result'][2] != '' and addr_dic['addr_title']['result'][4] > 0.7 and addr_dic['addr_title']['result'][5] >= 2:
  583. big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic['addr_title']['result']
  584. final_key = 'addr_title'
  585. msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis)
  586. elif 'tenderee' in addr_dic and addr_dic['tenderee']['result'][2] != '' and addr_dic['tenderee']['result'][4] > 0.7 and addr_dic['tenderee']['result'][5] >= 2: # 补概率 653492303 广州市妇女儿童医疗中心柳州医院 预测错广州
  587. big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic['tenderee']['result']
  588. final_key = 'tenderee'
  589. msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis)
  590. elif 'addr_tenderee' in addr_dic and addr_dic['addr_tenderee']['result'][2] != '' and addr_dic['addr_tenderee']['result'][5] >= 2:
  591. big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic['addr_tenderee']['result']
  592. final_key = 'addr_tenderee'
  593. msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis)
  594. else:
  595. province_l, city_l, district_l = [], [], []
  596. for key in addr_dic:
  597. if key in ['addr_delivery', 'addr_bidopen', 'addr_bidsend', 'addr_contact', 'addr_bus']: # 地址类型只加一项,因为一般都有三项造成单个地址总分过大
  598. if addr_dic[key]['keyword'][2]:
  599. district_l += addr_dic[key]['keyword'][2]
  600. elif addr_dic[key]['keyword'][1]:
  601. city_l += addr_dic[key]['keyword'][1]
  602. else:
  603. province_l += addr_dic[key]['keyword'][0]
  604. else:
  605. province_l += addr_dic[key]['keyword'][0]
  606. city_l += addr_dic[key]['keyword'][1]
  607. district_l += addr_dic[key]['keyword'][2]
  608. if key in ['addr_project', 'addr_title']: # 加重项目、标题地址权重
  609. province_l += addr_dic[key]['keyword'][0]
  610. city_l += addr_dic[key]['keyword'][1]
  611. district_l += addr_dic[key]['keyword'][2]
  612. pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic, self.short_dic, self.idx_dic)
  613. big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = self.get_final_addr(pro_ids, city_ids, dis_ids, self.idx_dic)
  614. if pred_city != '':
  615. score_max = 0
  616. for key in ['addr_delivery', 'addr_tenderee', 'addr_title', 'tenderee', 'company_title',
  617. 'addr_bidopen', 'addr_bidsend', 'addr_contact', 'addr_bus', 'web_source_name']:
  618. if key in addr_dic and addr_dic[key]['result'][2] == pred_city and addr_dic[key]['result'][5] > score_max:
  619. final_key = key
  620. score_max = addr_dic[key]['result'][5]
  621. if addr_dic[final_key]['result'][3] == pred_dis: # 2026/2/2 省市匹配,且有区县停止循环
  622. break
  623. if pred_pro != '' and final_key == "":
  624. for key in ['addr_delivery', 'addr_tenderee', 'addr_title', 'tenderee', 'company_title',
  625. 'addr_bidopen', 'addr_bidsend', 'addr_contact', 'addr_bus', 'web_source_name']:
  626. if key in addr_dic and addr_dic[key]['result'][1] == pred_pro:
  627. final_key = key
  628. break
  629. msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis)
  630. else:
  631. # 取所有的地址
  632. province_l, city_l, district_l = self.find_whole_areas('%s %s %s' % (title_raw, tenderees, all_addr), self.pettern, self.area_variance_dic,self.full_dic, weight=0.5)
  633. pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic, self.short_dic, self.idx_dic)
  634. big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = self.get_final_addr(pro_ids, city_ids, dis_ids, self.idx_dic)
  635. final_key = 'all_addr'
  636. in_content = True
  637. msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis)
  638. if len(addr_dic) > 0 and pred_city == '': # 如果非正文提取缺少城市,补充
  639. province_l, city_l, district_l = self.find_whole_areas('%s %s %s' % (title_raw, tenderees, all_addr), self.pettern, self.area_variance_dic,self.full_dic, weight=0.5)
  640. pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic, self.short_dic, self.idx_dic)
  641. rs_tmp = self.get_final_addr(pro_ids, city_ids, dis_ids, self.idx_dic)
  642. if rs_tmp[1] == pred_pro and rs_tmp[4] > 0.5 and rs_tmp[2] != '':
  643. big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = rs_tmp
  644. msc += "只有省份,所有地址补充市县:%s %s ;"%(pred_city, pred_dis)
  645. final_key += ',all_addr'
  646. dist_source = {
  647. 'addr_project': '项目地址',
  648. 'addr_delivery': '收货地址',
  649. 'addr_title': '标题',
  650. 'tenderee': "招标人",
  651. 'company_title': '标题公司',
  652. 'addr_tenderee': '招标人地址',
  653. 'addr_bus': '招标人工商地址',
  654. 'web_source_name': '站源',
  655. 'addr_bidopen': '开标地址',
  656. 'addr_bidsend': '邮件地址',
  657. 'addr_contact': '联系地址',
  658. 'all_addr': '文中所有地址'
  659. }
  660. if big_area != "":
  661. area_dic['area'] = big_area
  662. if pred_pro != "":
  663. area_dic['province'] = pred_pro
  664. source_l = [dist_source.get(k, '') for k in final_key.split(',')]
  665. area_dic['dist_source'] = ';'.join(source_l)
  666. if pred_city != "":
  667. area_dic['city'] = pred_city
  668. if pred_dis != "":
  669. area_dic['district'] = pred_dis
  670. for k, v in code_dic.items():
  671. if v != '':
  672. area_dic[k] = v
  673. if pred_city in ['北京', '天津', '上海', '重庆']: # 直辖市调整县级
  674. area_dic['city'] = area_dic['district']
  675. area_dic['district'] = '未知'
  676. if 'district_code' in area_dic:
  677. area_dic['city_code'] = area_dic['district_code']
  678. area_dic.pop('district_code')
  679. if area_dic['city'] == '未知' and 'city_code' in area_dic:
  680. area_dic.pop('city_code')
  681. area_dic['is_in_text'] = in_content
  682. # area_dic['prob'] = prob
  683. # area_dic['max_score'] = max_score
  684. # print('最终地址:', pred_pro, pred_city, pred_dis)
  685. if prob < 0.6 or max_score < 2:
  686. log('地区匹配预测,最终结果:%s %s %s, 预测:%s, docid:%s' % (pred_pro, pred_city, pred_dis, msc, docid))
  687. return {'district': area_dic}
  688. def predict_area_backup(self,docid, title, content, web_source_name, prem={}, addr_dic={}, list_entity=[]):
  689. area_dic = {'area': '全国', 'province': '全国', 'city': '未知', 'district': '未知', "is_in_text": False}
  690. addr_project = addr_dic.get('addr_project', '').replace('(', '(').replace(')', ')')
  691. addr_delivery = addr_dic.get('addr_delivery', '').replace('(', '(').replace(')', ')')
  692. addr_bidopen = addr_dic.get('addr_bidopen', '').replace('(', '(').replace(')', ')')
  693. addr_bidsend = addr_dic.get('addr_bidsend', '').replace('(', '(').replace(')', ')')
  694. addr_contact = addr_dic.get('addr_contact', '').replace('(', '(').replace(')', ')')
  695. in_content = False
  696. not_sure = True # 是否不确定地区
  697. filter_attach = False # 是否过滤附件内容
  698. msc = "" # 日志信息
  699. if '##attachment##' in content:
  700. main, att = content.split('##attachment##')
  701. if 500 < len(main) < len(att): # 正文超过500字且附件比正文长过滤附件内容
  702. filter_attach = True
  703. # print('正文超过500字过滤附件内容')
  704. if web_source_name in ('中原云商', '中原云商电子招投标平台'): # 修复某些公告没地区
  705. web_source_name = '河南中原云商'
  706. province_l, city_l, district_l = self.find_whole_areas('%s %s'%(title, addr_project), self.pettern, self.area_variance_dic, self.full_dic)
  707. pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic, self.short_dic, self.idx_dic)
  708. big_area_1, pred_pro_1, pred_city_1, pred_dis_1, prob_1, max_score, code_dic_1 = self.get_final_addr(pro_ids, city_ids, dis_ids, self.idx_dic)
  709. big_area, pred_pro, pred_city, pred_dis, prob, code_dic = big_area_1, pred_pro_1, pred_city_1, pred_dis_1, prob_1, code_dic_1
  710. # print('关键词1:', province_l, city_l, district_l)
  711. # print('输入:', '标题:%s; 项目地址:%s'%(title, addr_project))
  712. # print('分数:', pro_ids, city_ids, dis_ids, prob, max_score)
  713. msc += '第一次预测,标题:%s; 项目地址:%s; 预测为:%s %s ##'%(title, addr_project, pred_pro, pred_city)
  714. if pred_city_1 == "" or prob < 0.7 or max_score<2:
  715. ree, addr = self.get_ree_addr(prem)
  716. have_bus, bus_dic = get_business_data(ree) if ree != '' else False, {}
  717. if ree in title:
  718. ree = '##'
  719. rule_ree_addr = self.get_role_address(content)
  720. if rule_ree_addr:
  721. addr = rule_ree_addr
  722. # addr = content
  723. # ree = ''
  724. province_l2, city_l2, district_l2 = self.find_whole_areas('%s %s %s' % (ree, addr, addr_delivery), self.pettern, self.area_variance_dic, self.full_dic, weight=1)
  725. province_l.extend(province_l2)
  726. city_l.extend(city_l2)
  727. district_l.extend(district_l2)
  728. pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic, self.short_dic, self.idx_dic)
  729. big_area_2, pred_pro_2, pred_city_2, pred_dis_2, prob_2, max_score, code_dic_2 = self.get_final_addr(pro_ids, city_ids, dis_ids, self.idx_dic)
  730. big_area, pred_pro, pred_city, pred_dis, prob, code_dic = big_area_2, pred_pro_2, pred_city_2, pred_dis_2, prob_2, code_dic_2
  731. # print('关键词2:', province_l, city_l, district_l)
  732. # print('输入:', '招标人:%s; 招标人地址:%s; 收货地址:%s' % (ree, addr, addr_delivery))
  733. # print('分数:', pro_ids, city_ids, dis_ids, prob, max_score)
  734. msc += '第二次预测,招标人:%s; 招标人地址:%s; 收货地址:%s; 预测为:%s %s ##' % (ree, addr, addr_delivery, pred_pro, pred_city)
  735. if re.search('省|市|县|自治', addr_project) and prob_1 !=0.5 and pred_pro_1 != '' and pred_pro_1 != pred_pro_2: # 如果有项目地址使用项目地址 要有省市县等 275127622 工程地点为狮山镇颜峰综合区岐山至人和段道路, 提错 岐山
  736. not_sure = False
  737. big_area, pred_pro, pred_city, pred_dis, code_dic = big_area_1, pred_pro_1, pred_city_1, pred_dis_1, code_dic_1
  738. msc += "有项目地址,一二次预测省份不同,改为第一次结果。"
  739. if not_sure and (pred_city_2 == "" or prob < 0.7 or max_score<2):
  740. addr_bus = '%s %s'%(bus_dic.get('province', ''), bus_dic.get('city', ''))
  741. province_l3, city_l3, district_l3 = self.find_whole_areas('%s %s; %s; %s'%(addr_bus, addr_contact, addr_bidopen, addr_bidsend), self.pettern, self.area_variance_dic, self.full_dic, weight=0.6)
  742. province_l.extend(province_l3)
  743. city_l.extend(city_l3)
  744. district_l.extend(district_l3)
  745. pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic, self.short_dic, self.idx_dic)
  746. big_area_3, pred_pro_3, pred_city_3, pred_dis_3, prob_3, max_score, code_dic_3 = self.get_final_addr(pro_ids, city_ids, dis_ids, self.idx_dic)
  747. big_area, pred_pro, pred_city, pred_dis, prob, code_dic = big_area_3, pred_pro_3, pred_city_3, pred_dis_3, prob_3, code_dic_3
  748. # print('关键词3:', province_l, city_l, district_l)
  749. # print('输入:', '联系:%s, 开标:%s, 邮寄:%s'%(addr_contact, addr_bidopen, addr_bidsend))
  750. # print('分数:', pro_ids, city_ids, dis_ids, prob, max_score)
  751. msc += '第三次预测,工商地址:%s;联系:%s; 开标:%s; 邮寄:%s; 预测为:%s %s ##' % (addr_bus, addr_contact, addr_bidopen, addr_bidsend, pred_pro, pred_city)
  752. if pred_city_2 != "" and prob_2 !=0.5 and pred_city_2 != pred_city_3:
  753. not_sure = False
  754. big_area, pred_pro, pred_city, pred_dis, code_dic = big_area_2, pred_pro_2, pred_city_2, pred_dis_2, code_dic_2 # 如果招标人、招标人地址、收货地址与开标地址、联系地址等不一致,取招标人地址
  755. msc += "二三次预测城市不一致,改为第二次结果。"
  756. if not_sure and (pred_city_3 == "" or prob < 0.6 or max_score < 2):
  757. all_addr, tenderees = self.get_all_addr(list_entity, filter_attach)
  758. province_l4, city_l4, district_l4 = self.find_whole_areas('%s %s %s' % (web_source_name, tenderees, all_addr), self.pettern, self.area_variance_dic, self.full_dic, weight=0.3)
  759. province_l.extend(province_l4)
  760. city_l.extend(city_l4)
  761. district_l.extend(district_l4)
  762. pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic, self.short_dic, self.idx_dic)
  763. big_area_4, pred_pro_4, pred_city_4, pred_dis_4, prob_4, max_score, code_dic_4 = self.get_final_addr(pro_ids, city_ids,dis_ids, self.idx_dic)
  764. big_area, pred_pro, pred_city, pred_dis, prob, code_dic = big_area_4, pred_pro_4, pred_city_4, pred_dis_4, prob_4, code_dic_4
  765. if pred_city_3 != "" and prob_3 !=0.5 and pred_city_3 != pred_city_4:
  766. not_sure = False
  767. big_area, pred_pro, pred_city, pred_dis, code_dic = big_area_3, pred_pro_3, pred_city_3, pred_dis_3, code_dic_3 # 如果开标地址等提取的城市与所有地址提取的城市不一致,取开标地址等
  768. msc += "三四次预测城市不一致,改为第三次结果。"
  769. if pred_pro_3 != pred_pro_4 and (prob < 0.6 or max_score < 2) or max_score < 1:
  770. in_content = True
  771. # print('关键词4:', province_l, city_l, district_l)
  772. # print('输入:', '站源:%s, 角色:%s, 地址:%s' % (web_source_name, tenderees, all_addr))
  773. # print('分数:', pro_ids, city_ids, dis_ids, prob, max_score)
  774. msc += '第四次预测,站源:%s; 招标、代理角色:%s; 所有地址:%s; 预测为:%s %s ##' % (web_source_name, tenderees, all_addr, pred_pro, pred_city)
  775. if big_area != "":
  776. area_dic['area'] = big_area
  777. if pred_pro != "":
  778. area_dic['province'] = pred_pro
  779. if pred_city != "":
  780. area_dic['city'] = pred_city
  781. if pred_dis != "":
  782. area_dic['district'] = pred_dis
  783. for k, v in code_dic.items():
  784. if v != '':
  785. area_dic[k] = v
  786. if pred_city in ['北京', '天津', '上海', '重庆']: # 直辖市调整县级
  787. area_dic['city'] = area_dic['district']
  788. area_dic['district'] = '未知'
  789. if 'district_code' in area_dic:
  790. area_dic['city_code'] = area_dic['district_code']
  791. area_dic.pop('district_code')
  792. if area_dic['city'] == '未知' and 'city_code' in area_dic:
  793. area_dic.pop('city_code')
  794. area_dic['is_in_text'] = in_content
  795. # area_dic['prob'] = prob
  796. # area_dic['max_score'] = max_score
  797. # print('最终地址:', pred_pro, pred_city, pred_dis)
  798. if prob < 0.6 or max_score < 2:
  799. log('地区匹配预测,最终结果:%s %s %s, 预测:%s, docid:%s'%(pred_pro, pred_city, pred_dis, msc, docid))
  800. return {'district': area_dic}
  801. def get_area(self, text, web_name, in_content=False):
  802. p_pro, p_city, p_dis, idx_dic, full_dic, short_dic = self.p_pro, self.p_city, self.p_dis, self.idx_dic, self.full_dic, self.short_dic
  803. def get_final_addr(pro_ids, city_ids, dis_ids):
  804. '''
  805. 先把所有匹配的全称、简称转为id,如果省份不为空,城市不为空且有城市属于省份的取该城市
  806. :param province_l: 匹配到的所有省份
  807. :param city_l: 匹配到的所有城市
  808. :param district_l: 匹配到的所有区县
  809. :return:
  810. '''
  811. big_area = ""
  812. pred_pro = ""
  813. pred_city = ""
  814. pred_dis = ""
  815. final_pro = ""
  816. final_city = ""
  817. pro_prob = 0
  818. city_prob = 0
  819. if len(pro_ids) >= 1:
  820. pro_l = sorted([(k, v) for k, v in pro_ids.items()], key=lambda x: x[1], reverse=True)
  821. scores = [it[1] for it in pro_l]
  822. pro_prob = max(scores)/sum(scores)
  823. final_pro, score = pro_l[0]
  824. if score >= 0.01:
  825. pred_pro = idx_dic[final_pro]['返回名称']
  826. big_area = idx_dic[final_pro]['大区']
  827. # else:
  828. # print("得分过低,过滤掉", idx_dic[final_pro]['返回名称'], score)
  829. if pred_pro != "" and len(city_ids) >= 1:
  830. city_l = sorted([(k, v) for k, v in city_ids.items()], key=lambda x: x[1], reverse=True)
  831. scores = [it[1] for it in city_l]
  832. city_prob = max(scores) / sum(scores)
  833. for it in city_l:
  834. if idx_dic[it[0]]['省'] == final_pro:
  835. final_city = it[0]
  836. pred_city = idx_dic[final_city]['返回名称']
  837. break
  838. if final_city != "" and len(set(dis_ids)) >= 1:
  839. dis_l = sorted([(k, v) for k, v in dis_ids.items()], key=lambda x: x[1], reverse=True)
  840. for it in dis_l:
  841. if idx_dic[it[0]]['市'] == final_city:
  842. pred_dis = idx_dic[it[0]]['返回名称']
  843. elif pred_pro != "" and pred_city == "" and len(set(dis_ids)) >= 1: # 20241111 省份不为空,市为空,如果区县在省份下,补充对应的市县
  844. dis_l = sorted([(k, v) for k, v in dis_ids.items()], key=lambda x: x[1], reverse=True)
  845. for it in dis_l:
  846. if idx_dic[it[0]]['省'] == final_pro:
  847. pred_city = idx_dic[idx_dic[it[0]]['市']]['返回名称']
  848. pred_dis = idx_dic[it[0]]['返回名称']
  849. # print('20241111 省份不为空,市为空,如果区县在省份下,补充对应的市县: ', pred_city, pred_dis)
  850. if pred_city in ['北京', '天津', '上海', '重庆']:
  851. pred_city = pred_dis
  852. pred_dis = ""
  853. return big_area, pred_pro, pred_city, pred_dis
  854. def find_areas(pettern, text):
  855. '''
  856. 通过正则匹配字符串返回地址
  857. :param pettern: 地址正则 广东省|广西省|...
  858. :param text: 待匹配文本
  859. :return:
  860. '''
  861. addr = []
  862. for it in re.finditer(pettern, text):
  863. if re.search('[省市区县旗盟]$', it.group(0)) == None and re.search(
  864. '^([东南西北中一二三四五六七八九十大小]?(村|镇|街|路|道|社区)|酒店|宾馆)', text[it.end():]):
  865. continue
  866. if it.group(0) == '站前': # 20240314 修复类似 中铁二局新建沪苏湖铁路工程站前VI标项目 错识别为 省份:辽宁, 城市:营口,区县:站前
  867. continue
  868. if re.search('^(经济开发区|开发区|新区)', text[it.end():]) and re.search('广州市', pettern): # 城市不匹配为区的地址 修复 滨州北海经济开发区 北海新区 等提取为北海
  869. continue
  870. addr.append((it.group(0), it.start(), it.end()))
  871. if re.search('^([分支](公司|局|行|校|院|干?线)|\w{,3}段|地铁|(火车|高铁)?站|\w{,3}项目)', text[it.end():]):
  872. addr.append((it.group(0), it.start(), it.end()))
  873. return addr
  874. def chage_area2score(group_list, max_len):
  875. '''
  876. 把匹配的的地址转为分数
  877. :param group_list: [('name', b, e)]
  878. :return:
  879. '''
  880. area_list = []
  881. if group_list != []:
  882. for it in group_list:
  883. name, b, e = it
  884. area_list.append((name, (e - b + e) / max_len / 2))
  885. return area_list
  886. def find_whole_areas(text):
  887. '''
  888. 通过正则匹配字符串返回地址
  889. :param pettern: 地址正则 广东省|广西省|...
  890. :param text: 待匹配文本
  891. :return:
  892. '''
  893. pettern = "((?P<prov>%s)(?P<city>%s)?(?P<dist>%s)?)|((?P<city1>%s)(?P<dist1>%s)?)|(?P<dist2>%s)" % (
  894. p_pro, p_city, p_dis, p_city, p_dis, p_dis)
  895. province_l, city_l, district_l = [], [], []
  896. for it in re.finditer(pettern, text):
  897. if re.search('[省市区县旗盟]', it.group(0)) == None and re.search(
  898. '^([东南西北中一二三四五六七八九十大小]?(村|镇|街|路|道|社区)|酒店|宾馆)', text[it.end():]):
  899. continue
  900. if it.group(0) == '站前': # 20240314 修复类似 中铁二局新建沪苏湖铁路工程站前VI标项目 错识别为 省份:辽宁, 城市:营口,区县:站前
  901. continue
  902. for k, v in it.groupdict().items():
  903. if v != None:
  904. if k in ['prov']:
  905. province_l.append((it.group(k), it.start(k), it.end(k)))
  906. elif k in ['city', 'city1']:
  907. if re.search('^(经济开发区|开发区|新区)', text[it.end(k):]): # 城市不匹配为区的地址 修复 滨州北海经济开发区 北海新区 等提取为北海
  908. continue
  909. city_l.append((it.group(k), it.start(k), it.end(k)))
  910. if re.search('^([分支](公司|局|行|校|院|干?线)|\w{,3}段|地铁|(火车|高铁)?站|\w{,3}项目)', text[it.end(k):]):
  911. city_l.append((it.group(k), it.start(k), it.end(k)))
  912. elif k in ['dist', 'dist1', 'dist2']:
  913. if it.group(k)=='昌江' and '景德镇' not in it.group(0):
  914. district_l.append(('昌江黎族', it.start(k), it.end(k)))
  915. else:
  916. district_l.append((it.group(k), it.start(k), it.end(k)))
  917. return province_l, city_l, district_l
  918. def get_pro_city_dis_score(text, text_weight=1):
  919. text = re.sub('复合肥|海南岛|兴业银行|双河口|阳光|杭州湾|新城区|中粮屯河|老城(区|改造|更新|升级|翻新)|沙县小吃|北京时间', ' ', text) # 544151395 赤壁市老城区燃气管道老化更新改造
  920. text = re.sub('珠海城市', '珠海', text) # 修复 426624023 珠海城市 预测为海城市
  921. text = re.sub('怒江州', '怒江傈僳族自治州', text) # 修复 423589589 所属地域:怒江州 识别为广西 - 崇左 - 江州
  922. text = re.sub('茂名滨海新区', '茂名市', text)
  923. text = re.sub('中山([东南西][部区环]|黄圃|南头|东凤|小榄|石岐|翠亨|南朗)', '中山市', text)
  924. text = re.sub('横州市', '横县', text) # 例:547363890 修复广西南宁横州 不在地区表问题
  925. ser = re.search('海南(昌江|白沙|乐东|陵水|保亭|琼中)(黎族)?', text)
  926. if ser and '黎族' not in ser.group(0):
  927. text = text.replace(ser.group(0), ser.group(0)+'黎族')
  928. for k, v in self.area_variance_dic.items(): # 20241113 根据地区变更信息替换文本
  929. text = text.replace(k, v)
  930. # province_l = find_areas(p_pro, text)
  931. # city_l = find_areas(p_city, text)
  932. # district_l = find_areas(p_dis, text)
  933. province_l, city_l, district_l = find_whole_areas(text) # 20240703 优化地址提取,解决类似 海南昌江 得到 海南 南昌 结果
  934. # if len(province_l) == len(city_l) == 0:
  935. # district_l = [it for it in district_l if
  936. # re.search('[市县旗区]$', it[0])] # 20240428去掉只有区县地址且不是全称的匹配,避免错误 例 凌云工业股份有限公司 提取地区为广西白色凌云
  937. province_l = chage_area2score(province_l, max_len=len(text))
  938. city_l = chage_area2score(city_l, max_len=len(text))
  939. district_l = chage_area2score(district_l, max_len=len(text))
  940. pro_ids = dict()
  941. city_ids = dict()
  942. dis_ids = dict()
  943. for pro in province_l:
  944. name, score = pro
  945. assert (name in full_dic['province'] or name in short_dic['province'])
  946. if name in full_dic['province']:
  947. idx = full_dic['province'][name]
  948. if idx not in pro_ids:
  949. pro_ids[idx] = 0
  950. pro_ids[idx] += (score + 1)
  951. else:
  952. idx = short_dic['province'][name]
  953. if idx not in pro_ids:
  954. pro_ids[idx] = 0
  955. pro_ids[idx] += (score + 0)
  956. for city in city_l:
  957. name, score = city
  958. if name in full_dic['city']:
  959. w = 0.1 if len(full_dic['city'][name]) > 1 else 1
  960. for idx in full_dic['city'][name]:
  961. if idx not in city_ids:
  962. city_ids[idx] = 0
  963. # weight = idx_dic[idx]['权重']
  964. city_ids[idx] += (score + 2) * w
  965. pro_idx = idx_dic[idx]['省']
  966. if pro_idx in pro_ids:
  967. pro_ids[pro_idx] += (score + 2) * w
  968. else:
  969. pro_ids[pro_idx] = (score + 2) * w * 0.5
  970. elif name in short_dic['city']:
  971. w = 0.1 if len(short_dic['city'][name]) > 1 else 1
  972. for idx in short_dic['city'][name]:
  973. if idx not in city_ids:
  974. city_ids[idx] = 0
  975. weight = idx_dic[idx]['权重']
  976. city_ids[idx] += (score + 1) * w * weight
  977. pro_idx = idx_dic[idx]['省']
  978. if pro_idx in pro_ids:
  979. pro_ids[pro_idx] += (score + 1) * w * weight
  980. else:
  981. pro_ids[pro_idx] = (score + 1) * w * weight * 0.5
  982. for dis in district_l:
  983. name, score = dis
  984. if name in full_dic['district']:
  985. w = 0.1 if len(full_dic['district'][name]) > 1 else 1
  986. for idx in full_dic['district'][name]:
  987. if idx not in dis_ids:
  988. dis_ids[idx] = 0
  989. # weight = idx_dic[idx]['权重']
  990. dis_ids[idx] += (score + 1) * w
  991. pro_idx = idx_dic[idx]['省']
  992. if pro_idx in pro_ids:
  993. pro_ids[pro_idx] += (score + 1) * w
  994. else:
  995. pro_ids[pro_idx] = (score + 1) * w * 0.5
  996. city_idx = idx_dic[idx]['市']
  997. if city_idx in city_ids:
  998. city_ids[city_idx] += (score + 1) * w
  999. else:
  1000. city_ids[city_idx] = (score + 1) * w * 0.5
  1001. elif name in short_dic['district']:
  1002. w = 0.1 if len(short_dic['district'][name]) > 1 else 1
  1003. for idx in short_dic['district'][name]:
  1004. if idx not in dis_ids:
  1005. dis_ids[idx] = 0
  1006. weight = idx_dic[idx]['权重']
  1007. dis_ids[idx] += (score + 0) * w
  1008. if idx_dic[idx]['市'] not in city_ids and idx_dic[idx]['省'] not in pro_ids: # 20241111 区县简称不在获取到的省、市范围内的过滤掉
  1009. continue
  1010. pro_idx = idx_dic[idx]['省']
  1011. if pro_idx in pro_ids:
  1012. pro_ids[pro_idx] += (score + 0) * w * weight
  1013. # else: # 20241015 注销 区县简称且不在提取的省市下面,不加分,避免提取错误 例:536550843
  1014. # pro_ids[pro_idx] = (score + 0) * w * weight * 0.5
  1015. city_idx = idx_dic[idx]['市']
  1016. if city_idx in city_ids:
  1017. city_ids[city_idx] += (score + 0) * w * weight
  1018. # else: # 20241015 注销 区县简称且不在提取的省市下面,不加分,避免提取错误 例:536550843
  1019. # city_ids[city_idx] = (score + 0) * w * weight * 0.1
  1020. elif pro_idx in pro_ids:
  1021. city_ids[city_idx] = (score + 0) * w * weight * 0.1
  1022. for k, v in pro_ids.items():
  1023. pro_ids[k] = v * text_weight
  1024. for k, v in city_ids.items():
  1025. city_ids[k] = v * text_weight
  1026. for k, v in dis_ids.items():
  1027. dis_ids[k] = v * text_weight
  1028. return pro_ids, city_ids, dis_ids
  1029. area_dic = {'area': '全国', 'province': '全国', 'city': '未知', 'district': '未知', "is_in_text": False}
  1030. pro_ids, city_ids, dis_ids = get_pro_city_dis_score(text)
  1031. pro_ids1, city_ids1, dis_ids1 = get_pro_city_dis_score(web_name, text_weight=0.01) # 20240422 修改为站源名称只取前三字,避免类似 459056219 中金岭南阳光采购平台 错提取阳光
  1032. for k in pro_ids1:
  1033. if k in pro_ids:
  1034. pro_ids[k] += pro_ids1[k]
  1035. else:
  1036. pro_ids[k] = pro_ids1[k]
  1037. for k in city_ids1:
  1038. if k in city_ids:
  1039. city_ids[k] += city_ids1[k]
  1040. else:
  1041. city_ids[k] = city_ids1[k]
  1042. for k in dis_ids1:
  1043. if k in dis_ids:
  1044. dis_ids[k] += dis_ids1[k]
  1045. else:
  1046. dis_ids[k] = dis_ids1[k]
  1047. big_area, pred_pro, pred_city, pred_dis = get_final_addr(pro_ids, city_ids, dis_ids)
  1048. if big_area != "":
  1049. area_dic['area'] = big_area
  1050. if pred_pro != "":
  1051. area_dic['province'] = pred_pro
  1052. if pred_city != "":
  1053. area_dic['city'] = pred_city
  1054. if pred_dis != "":
  1055. area_dic['district'] = pred_dis
  1056. if in_content:
  1057. area_dic['is_in_text'] = True
  1058. return {'district': area_dic}
  1059. def predict(self, project_name, prem, title, list_articles, web_source_name = "", list_entitys=""):
  1060. '''
  1061. 先匹配 project_name+tenderee+tenderee_address, 如果缺少省或市 再匹配 title+content
  1062. :param project_name:
  1063. :param prem:
  1064. :param title:
  1065. :param list_articles:
  1066. :param web_source_name:
  1067. :return:
  1068. '''
  1069. def get_ree_addr(prem):
  1070. tenderee = ""
  1071. tenderee_address = ""
  1072. try:
  1073. for v in prem[0]['prem'].values():
  1074. for link in v['roleList']:
  1075. if link['role_name'] == 'tenderee' and tenderee == "":
  1076. tenderee = link['role_text']
  1077. tenderee_address = link['address']
  1078. except Exception as e:
  1079. print('解析prem 获取招标人、及地址出错')
  1080. return tenderee, tenderee_address
  1081. def get_role_address(text):
  1082. '''正则匹配获取招标人地址
  1083. 3:地址直接在招标人后面 招标人:xxx,地址:xxx
  1084. 4:招标、代理一起,两个地址一起 招标人:xxx, 代理人:xxx, 地址:xxx, 地址:xxx.
  1085. '''
  1086. p3 = '(招标|采购|甲)(人|方|单位)(信息:|(甲方))?(名称)?:[\w()]{4,15},(联系)?地址:(?P<addr>(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+|\w{2,15}[,。])'
  1087. p4 = '(招标|采购|甲)(人|方|单位)(信息:|(甲方))?(名称)?:[\w()]{4,15},(招标|采购)?代理(人|机构)(名称)?:[\w()]{4,15},(联系)?地址:(?P<addr>(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+|\w{2,15}[,。])'
  1088. p5 = '(采购|招标)(人|单位)(联系)?地址:(?P<addr>(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+|\w{2,15}[,。])'
  1089. if re.search(p3, text):
  1090. return re.search(p3, text).group('addr')
  1091. elif re.search(p4, text):
  1092. return re.search(p4, text).group('addr')
  1093. elif re.search(p5, text):
  1094. return re.search(p5, text).group('addr')
  1095. else:
  1096. return ''
  1097. def get_project_addr(text):
  1098. p1 = '(项目|施工|实施|建设|工程|服务|交货|送货|收货|展示|看样|拍卖)(地址|地点|位置|所在地区?)(位于)?:(?P<addr>(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+([\w()]{,20}[,。])?|\w{2,15}[,。])'
  1099. p2 = '项目位于(?P<addr>\w{2}市\w{2,4}区)'
  1100. if re.search(p1, text):
  1101. return re.search(p1, text).group('addr')
  1102. elif re.search(p2, text):
  1103. return re.search(p2, text).group('addr')
  1104. else:
  1105. return ''
  1106. def get_bid_addr(text):
  1107. p2 = '(磋商|谈判|开标|投标|评标|报名|递交|评审|发售|所属)(地址|地点|所在地区?|地域):(?P<addr>(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+|\w{2,15}[,。])'
  1108. if re.search(p2, text):
  1109. return re.search(p2, text).group('addr')
  1110. else:
  1111. return ''
  1112. def get_all_addr(list_entitys):
  1113. tenderee_l = []
  1114. addr_l = []
  1115. for ent in list_entitys[0]:
  1116. if ent.entity_type == 'location' and len(ent.entity_text) > 2:
  1117. addr_l.append(ent.entity_text)
  1118. elif ent.entity_type in ['org', 'company']:
  1119. if ent.label in [0, 1]: # 加招标或代理
  1120. tenderee_l.append(ent.entity_text)
  1121. return ' '.join(addr_l), ' '.join(tenderee_l)
  1122. def get_title_addr(text):
  1123. p1 = '(?P<addr>(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+|\w{2,15}[,。])'
  1124. if re.search(p1, text):
  1125. return re.search(p1, text).group('addr')
  1126. else:
  1127. return ''
  1128. if '##attachment##' in list_articles[0].content:
  1129. content, attachment = list_articles[0].content.split('##attachment##')
  1130. if len(content) < 200:
  1131. content += attachment
  1132. else:
  1133. content = list_articles[0].content
  1134. tenderee, tenderee_address = get_ree_addr(prem)
  1135. msc = ""
  1136. pro_addr = get_project_addr(content)
  1137. if pro_addr != "" and re.search('(采购人|招标人)?指定地点', pro_addr)==None: # 排除错误项目地址 例:554024168 1.5服务地点:采购人指定地点。
  1138. msc += '使用规则提取的项目地址;'
  1139. tenderee_address = pro_addr
  1140. else:
  1141. role_addr = get_role_address(content)
  1142. if role_addr != "" and re.search('(采购人|招标人)?指定地点', role_addr)==None:
  1143. msc += '使用规则提取的联系人地址;'
  1144. tenderee_address = role_addr
  1145. if tenderee_address == "":
  1146. title_addr = get_title_addr(title)
  1147. if title_addr != "":
  1148. msc += '使用规则提取的标题地址;'
  1149. tenderee_address = title_addr
  1150. else:
  1151. bid_addr = get_bid_addr(content)
  1152. if bid_addr != "":
  1153. msc += '使用规则提取的开标地址;'
  1154. tenderee_address = bid_addr
  1155. project_name = str(project_name)
  1156. tenderee = str(tenderee)
  1157. # print('招标人地址',role_addr, tenderee_address)
  1158. project_name = project_name + title if project_name not in title else title
  1159. # project_name = project_name.replace(tenderee, '')
  1160. if len(project_name)>3:
  1161. entity_list = getNers([project_name],useselffool=False) # 2024/4/26 修改为去重项目名称中所有公司名称
  1162. for tup in entity_list[0]:
  1163. if tup[2] in ['org', 'company']:
  1164. project_name = project_name.replace(tup[3], '')
  1165. text1 = "{0} {1} {2}".format(tenderee, tenderee_address, project_name)
  1166. web_source_name = str(web_source_name) # 修复某些不是字符串类型造成报错
  1167. text1 = re.sub('复合肥|铁路|公路|新会计', ' ', text1) # 预防提取错 合肥 路南 新会 等地区
  1168. if pro_addr and re.search('\w{2,}([市县旗盟]|自治[区州县旗])', pro_addr):
  1169. if re.search('[市县旗盟]', pro_addr)==None: # 修复 486623506 项目地址不完整
  1170. pro_addr = text1 + ' '+ pro_addr
  1171. msc += '## 使用项目地址输入:%s ##;' % pro_addr
  1172. rs = self.get_area(pro_addr, '')
  1173. msc += '预测结果:省份:%s, 城市:%s,区县:%s;' % (
  1174. rs['district']['province'], rs['district']['city'], rs['district']['district'])
  1175. if rs['district']['province'] != '全国' and rs['district']['city'] != '未知':
  1176. # print('地区匹配:', msc)
  1177. return rs
  1178. # print('text1:', text1)
  1179. msc += '## 第一次预测输入:%s ##;' % text1
  1180. rs = self.get_area(text1, '') # 2024/4/22 调整第一次输入不带站源名称,避免出错
  1181. msc += '预测结果:省份:%s, 城市:%s,区县:%s;' % (
  1182. rs['district']['province'], rs['district']['city'], rs['district']['district'])
  1183. # self.f.write('%s %s \n' % (list_articles[0].id, msc))
  1184. # print('地区匹配:', msc)
  1185. if rs['district']['province'] == '全国' or rs['district']['city'] == '未知':
  1186. # msc = ""
  1187. all_addr, tenderees = get_all_addr(list_entitys)
  1188. text2 = tenderees + " " + all_addr + ' ' + title
  1189. msc += '使用实体列表所有招标人+所有地址;'
  1190. # text2 += title + content if len(content)<2000 else title + content[:1000] + content[-1000:]
  1191. text2 = re.sub('复合肥|铁路|公路|新会计', ' ', text2)
  1192. # print('text2:', text2)
  1193. msc += '## 第二次预测输入:%s %s##' % (text2,web_source_name)
  1194. rs2 = self.get_area(text2, web_source_name, in_content=True)
  1195. # rs2['district']['is_in_text'] = True
  1196. if rs['district']['province'] == '全国' and rs2['district']['province'] != '全国':
  1197. rs = rs2
  1198. elif rs['district']['province'] == rs2['district']['province'] and rs2['district']['city'] != '未知':
  1199. rs = rs2
  1200. msc += '预测结果:省份:%s, 城市:%s,区县:%s' % (
  1201. rs['district']['province'], rs['district']['city'], rs['district']['district'])
  1202. # self.f.write('%s %s \n'%(list_articles[0].id, msc))
  1203. # print('地区匹配:', msc)
  1204. return rs