| 12345678910111213141516171819202122232425262728293031323334353637383940414243444546474849505152535455565758596061626364656667686970717273747576777879808182838485868788899091929394959697989910010110210310410510610710810911011111211311411511611711811912012112212312412512612712812913013113213313413513613713813914014114214314414514614714814915015115215315415515615715815916016116216316416516616716816917017117217317417517617717817918018118218318418518618718818919019119219319419519619719819920020120220320420520620720820921021121221321421521621721821922022122222322422522622722822923023123223323423523623723823924024124224324424524624724824925025125225325425525625725825926026126226326426526626726826927027127227327427527627727827928028128228328428528628728828929029129229329429529629729829930030130230330430530630730830931031131231331431531631731831932032132232332432532632732832933033133233333433533633733833934034134234334434534634734834935035135235335435535635735835936036136236336436536636736836937037137237337437537637737837938038138238338438538638738838939039139239339439539639739839940040140240340440540640740840941041141241341441541641741841942042142242342442542642742842943043143243343443543643743843944044144244344444544644744844945045145245345445545645745845946046146246346446546646746846947047147247347447547647747847948048148248348448548648748848949049149249349449549649749849950050150250350450550650750850951051151251351451551651751851952052152252352452552652752852953053153253353453553653753853954054154254354454554654754854955055155255355455555655755855956056156256356456556656756856957057157257357457557657757857958058158258358458558658758858959059159259359459559659759859960060160260360460560660760860961061161261361461561661761861962062162262362462562662762862963063163263363463563663763863964064164264364464564664764864965065165265365465565665765865966066166266366466566666766866967067167267367467567667767867968068168268368468568668768868969069169269369469569669769869970070170270370470570670770870971071171271371471571671771871972072172272372472572672772872973073173273373473573673773873974074174274374474574674774874975075175275375475575675775875976076176276376476576676776876977077177277377477577677777877978078178278378478578678778878979079179279379479579679779879980080180280380480580680780880981081181281381481581681781881982082182282382482582682782882983083183283383483583683783883984084184284384484584684784884985085185285385485585685785885986086186286386486586686786886987087187287387487587687787887988088188288388488588688788888989089189289389489589689789889990090190290390490590690790890991091191291391491591691791891992092192292392492592692792892993093193293393493593693793893994094194294394494594694794894995095195295395495595695795895996096196296396496596696796896997097197297397497597697797897998098198298398498598698798898999099199299399499599699799899910001001100210031004100510061007100810091010101110121013101410151016101710181019102010211022102310241025102610271028102910301031103210331034103510361037103810391040104110421043104410451046104710481049105010511052105310541055105610571058105910601061106210631064106510661067106810691070107110721073107410751076107710781079108010811082108310841085108610871088108910901091109210931094109510961097109810991100110111021103110411051106110711081109111011111112111311141115111611171118111911201121112211231124112511261127112811291130113111321133113411351136113711381139114011411142114311441145114611471148114911501151115211531154115511561157115811591160116111621163116411651166116711681169117011711172117311741175117611771178117911801181118211831184118511861187118811891190119111921193119411951196119711981199120012011202120312041205120612071208120912101211121212131214121512161217121812191220122112221223122412251226122712281229123012311232123312341235123612371238123912401241124212431244124512461247124812491250125112521253125412551256125712581259126012611262126312641265126612671268126912701271127212731274127512761277127812791280128112821283128412851286 |
- # -*- coding: utf-8 -*-
- """``DistrictPredictor`` — 地区匹配规则预测。
- Phase 5 从 ``interface/predictor.py``(约 6770-8032 行)迁出。
- 原 ``from common.Utils import *`` / ``from common.nerUtils import *``
- 已替换为显式 import;``os.path.dirname(__file__)`` 路径引用替换为
- ``predictors._common.INTERFACE_DIR``。使用 district_tuple.pkl、
- area_variance_dic.pkl 等地区字典。
- """
- from __future__ import absolute_import
- import os
- import re
- import pickle
- from collections import Counter
- from BiddingKG.dl.common.logging import log
- from BiddingKG.dl.common.nerUtils import getTokens, getNers
- from BiddingKG.dl.entityLink.entityLink import get_business_data
- from BiddingKG.dl.predictors._common import INTERFACE_DIR
- __all__ = ["DistrictPredictor"]
- class DistrictPredictor():
- def __init__(self):
- # with open(os.path.dirname(__file__)+'/district_dic.pkl', 'rb') as f:
- # dist_dic = pickle.load(f)
- # short_name = '|'.join(sorted(set([v['简称'] for v in dist_dic.values()]), key=lambda x: len(x), reverse=True))
- # full_name = '|'.join(sorted(set([v['全称'] for v in dist_dic.values()]), key=lambda x: len(x), reverse=True))
- # short2id = {}
- # full2id = {}
- # for k, v in dist_dic.items():
- # if v['简称'] not in short2id:
- # short2id[v['简称']] = [k]
- # else:
- # short2id[v['简称']].append(k)
- # if v['全称'] not in full2id:
- # full2id[v['全称']] = [k]
- # else:
- # full2id[v['全称']].append(k)
- # self.dist_dic = dist_dic
- # self.short_name = short_name
- # self.full_name = full_name
- # self.short2id = short2id
- # self.full2id = full2id
- # # self.f = open(os.path.dirname(__file__)+'/../test/data/district_predict.txt', 'w', encoding='utf-8')
- with open(os.path.join(INTERFACE_DIR, 'district_tuple.pkl'), 'rb') as f:
- district_tuple = pickle.load(f)
- self.p_pro, self.p_city, self.p_dis, self.idx_dic, self.full_dic, self.short_dic = district_tuple
- # self.pettern = "((?P<prov>%s)(?P<city>%s)?(?P<dist>%s)?)|((?P<city1>%s)(?P<dist1>%s)?)|(?P<dist2>%s)" % (
- # self.p_pro, self.p_city, self.p_dis, self.p_city, self.p_dis, self.p_dis)
- short_pro = '黑龙江|内蒙古|青海|陕西|辽宁|贵州|西藏|福建|甘肃|湖南|湖北|海南|浙江|河南|河北|江西|江苏|新疆|广西|广东|山西|山东|安徽|宁夏|四川|吉林|云南'
- self.pettern = "(?P<prov>%s)##(?P<city>%s)##(?P<dist>%s)" % (
- self.p_pro, self.p_city, self.p_dis)
- self.pettern_pro = re.compile("(?P<prov>%s)"%self.p_pro)
- self.pettern_city = re.compile("(%s)?(?P<city>%s)"%(short_pro, self.p_city)) # 20250925补充省简称,解决 海南昌江 匹配为南昌问题
- self.pettern_dist = re.compile("(%s)?(?P<dist>%s)"%(short_pro, self.p_dis))
- with open(os.path.join(INTERFACE_DIR, "area_variance_dic.pkl"), 'rb') as f: # 20241113 地区变更新旧名称对照字典
- self.area_variance_dic = pickle.load(f)
- self.multi_dist = ['向阳区', '宝山区', '南沙区', '和平区', '新城区', '鼓楼区', '南山区', '白云区', '朝阳区',
- '江北区', '城关区', '永定区', '普陀区', '长安区', '市中区', '西安区', '通州区', '西湖区',
- '龙华区', '城中区', '河东区', '桥西区', '青山区', '新华区', '铁西区', '铁东区', '海州区', '滨海新区']
- def find_whole_areas(self, text, pettern, area_variance_dic, full_dic, weight=1):
- '''
- 通过正则匹配字符串返回地址
- :param pettern: 地址正则 广东省|广西省|...
- :param text: 待匹配文本
- :return:
- '''
- province_l, city_l, district_l = [], [], []
- citys = []
- text = str(text).replace('(', '(').replace(')', ')')
- text = re.sub('\d{2,4}年度?|[\d/-]{1,5}[月日]|\d+|[a-zA-Z0-9]+', ' ', text)
- text = re.sub(
- '复合肥|海南岛|兴业银行|双河口|阳光|杭州湾|新城区|中粮屯河|老城(区|改造|更新|升级|翻新)|沙县小吃|北京时间|福田汽车|中山(大学|公园|纪念堂)|孙中山|海天水泥|阳光采购|示范县|珠江城?|西九龙站|广州路北|安阳山村|电信|联通|北京现代|祁连山|锡铁山|大黄山(?!市)|红旗汽车', # 570445994 广州路北侧 预测为 广州 路北
- ' ', text) # 544151395 赤壁市老城区燃气管道老化更新改造
- text = re.sub('珠海城市', '珠海', text) # 修复 426624023 珠海城市 预测为海城市
- text = re.sub('怒江州', '怒江傈僳族自治州', text) # 修复 423589589 所属地域:怒江州 识别为广西 - 崇左 - 江州
- text = re.sub('茂名滨海新区', '茂名市', text)
- text = re.sub('中山([东南西][部区环]|黄圃|南头|东凤|小榄|石岐|翠亨|南朗)', '中山市', text)
- text = re.sub('横州市', '横县', text) # 例:547363890 修复广西南宁横州 不在地区表问题
- text = re.sub('广东中山', '广东中山市', text)
- text = re.sub('朝阳柳城经济开发区', '朝阳市', text)
- text = re.sub('安徽徽运城市', '安徽', text) # 653054198 安徽徽运城市运营管理有限公司
- text = re.sub('西安丰镇', '宝应', text) # 修复 681946122 宝应县西安丰镇
- text = re.sub('西城区', '西城', text) # 修复 653130377 标题 董家口电厂至新区西城区长输热力管线工程目 预测错北京西城
- # 处理地区+大学不在该地区问题
- text = re.sub('(河北工业大学)', '天津市', text)
- text = re.sub('(西藏民族大学)', '咸阳市', text)
- text = re.sub('(四川外国语大学|四川美术学院)', '重庆市', text)
- text = re.sub('(中山大学)', '广州市', text)
- text = re.sub('(滨州医学院)', '烟台市', text)
- ser = re.search('海南(昌江|白沙|乐东|陵水|保亭|琼中)(黎族)?', text)
- if ser and '黎族' not in ser.group(0):
- text = text.replace(ser.group(0), ser.group(0) + '黎族')
- for k, v in area_variance_dic.items(): # 20241113 根据地区变更信息替换文本
- text = text.replace(k, v)
- text = re.sub('\s+', ' ', text)
- if re.search('[\u4e00-\u9fa5]', text) == None:
- return province_l, city_l, district_l
- name_set = set() # 提取到的所有地址集合
- tokens = getTokens([text], useselffool=True)[0]
- # print('句子:', text)
- # print('分词:', tokens)
- for pettern in [self.pettern_pro, self.pettern_city, self.pettern_dist]: # pettern.split('##')
- for it in re.finditer(pettern, text):
- if it.group(0) == '站前': # 20240314 修复类似 中铁二局新建沪苏湖铁路工程站前VI标项目 错识别为 省份:辽宁, 城市:营口,区县:站前
- continue
- for k, v in it.groupdict().items():
- if v != None:
- if it.end() == it.end(k) and re.search('[省市区县州旗盟]$', v) == None and re.search(
- '^([东南西北中一二三四五六七八九十大小]?(村|镇|街|路|道|社区|巷|坊)|酒店|宾馆|经济开发区|开发区|新区|公园|广场|公馆|小区|幼儿园)', # |医院|[大中小]学 # 20250917取消地区+医院过滤,大部分是在该地区
- # 城市不匹配为区的地址 修复 滨州北海经济开发区 北海新区 等提取为北海
- text[it.end(k):]) != None and not re.match('路桥', text[it.end(k):]): # 修复 746181930 甘肃路桥 被路去掉
- continue
- if k in ['prov']:
- if v in full_dic['province']:
- score = 2
- else:
- score = 1
- if re.search('^(\w{,2}[分支](公司|局|行|院|干?线)|(机务|车辆|车务|工务|电务|供电|动车)?段|地铁|(火车|高铁)?站|港|地区|区域|基地)'
- , text[it.end(k):]) or re.search('^((%s)|\-%s)' % (v, v),
- text[max(0, it.start(k) - 1):]):
- score += 1
- elif re.search('大学|学院', text[:it.start(k)]) and re.search('分校|校区', text[it.end(k):]): # 长春市第八十七中学南阳校区 不在南阳市
- score += 1
- if len(v) < 3 and v not in tokens: # 不在分词里面概率降低
- score /= 2
- else:
- score += it.end(k) / len(text) / 10
- province_l.append((v, score * weight))
- elif k in ['city', 'city1']:
- if v in full_dic['city']:
- score = 2
- else:
- score = 1
- if re.search('^(\w{,2}[分支](公司|局|行|院|干?线)|(机务|车辆|车务|工务|电务|供电|动车)?段|地铁|(火车|高铁)?站|港|地区|区域|基地)'
- , text[it.end(k):]) or re.search('^((%s)|\-%s)' % (v, v),
- text[max(0, it.start(k) - 1):]):
- score += 1
- elif re.search('大学|学院', text[:it.start(k)]) and re.search('分校|校区', text[it.end(k):]):
- score += 1
- if len(v) < 3 and v not in tokens: # 不在分词里面概率降低 优化 653058471 山东恒通化工股份有限公司 错分 通化
- score /= 2
- else:
- score += it.end(k) / len(text) / 10 # 优化 572840045 上海铁路公安局合肥公安处 这种表达
- city_l.append((v, score * weight))
- citys.append(v)
- elif k in ['dist', 'dist1', 'dist2']:
- if v in ['东区', '西区', '城区', '郊区', '矿区', '东至']:
- continue
- elif v.endswith('城市') and text[it.start(k)-1:it.start(k)+1] in citys: # 修复 上海城市 珠海城市 等 预测为 海城市
- continue
- if v in self.multi_dist or re.search('\w城区$', v): # 多个城市有的区概率降低
- score = 0.5
- elif v in full_dic['district'] and (len(v) > 2 or v.endswith('县')): # 20250709 修复 萧县 等概率过低
- score = 2
- else:
- score = 0.5
- if re.search('^(\w{,2}[分支](公司|局|行|院|干?线)|(机务|车辆|车务|工务|电务|供电|动车)?段|地铁|(火车|高铁)?站|港|地区|区域|基地)'
- , text[it.end(k):]) or (
- re.match('\s*%s' % v, text) and it.start(k) < 2) or re.search(
- '^((%s)|\-%s)' % (v, v), text[max(0, it.start(k) - 1):]):
- score += 0.5
- elif re.search('大学|学院', text[:it.start(k)]) and re.search('分校|校区', text[it.end(k):]):
- score += 0.5
- if len(v) < 3 and v not in tokens: # 不在分词里面概率降低,三字以上不受限制 ,避免类似 德令哈工务段 分词不对
- score /= 2
- # score += it.end(k) / len(text) / 10
- district_l.append((v, score * weight))
- name_set.add(v)
- if len(name_set) > 1: # 解决 类似 山西宁武 匹配出 山西 西宁 宁武 问题 修复 653057648 把峨眉山市作为眉山市
- names = re.findall('|'.join(sorted(name_set, key=lambda x: len(x), reverse=True)), text)
- province_l = [it for it in province_l if it[0] in names]
- city_l = [it for it in city_l if it[0] in names]
- district_l = [it for it in district_l if it[0] in names]
- return province_l, city_l, district_l
- def merge_score(self, province_l, city_l, district_l, full_dic, short_dic, idx_dic, filter_short_dist=True):
- '''
- 合并分数,下级地区分数加到上级
- :param province_l: 提取到的省份列表 [(name, score)]
- :param city_l: 提取到的城市列表 [(name, score)]
- :param district_l: 提取到的区县列表 [(name, score)]
- :param filter_short_dist: 是否过滤不在省份下的区县简称权重
- :return:
- '''
- pro_ids = dict()
- city_ids = dict()
- dis_ids = dict()
- for pro in province_l:
- name, score = pro
- idx = full_dic['province'][name] if name in full_dic['province'] else short_dic['province'][name]
- if idx not in pro_ids:
- pro_ids[idx] = 0
- pro_ids[idx] += score
- tmp_pro = {}
- for city in city_l:
- name, score = city
- if name in full_dic['city']:
- for idx in full_dic['city'][name]:
- if idx not in city_ids:
- city_ids[idx] = 0
- city_ids[idx] += score
- pro_idx = idx_dic[idx]['省']
- if pro_idx in tmp_pro:
- tmp_pro[pro_idx] += score
- else:
- tmp_pro[pro_idx] = score
- elif name in short_dic['city']:
- for idx in short_dic['city'][name]:
- if idx not in city_ids:
- city_ids[idx] = 0
- city_ids[idx] += score
- pro_idx = idx_dic[idx]['省']
- if pro_ids != {} and pro_idx not in pro_ids: # 如果省份不为空且简称不在省份分值降低 优化 653150317 海南农垦阳江农场有限公司 错分 阳江
- score -= 0.1
- if pro_idx in tmp_pro:
- tmp_pro[pro_idx] += score
- else:
- tmp_pro[pro_idx] = score
- if set(tmp_pro) & set(pro_ids) != set():
- for k, v in tmp_pro.items():
- if k in pro_ids:
- pro_ids[k] += v
- else:
- pro_ids[k] = v
- else:
- pro_ids.update(tmp_pro)
- tmp_pro = {}
- tmp_city = {}
- for dis in district_l:
- name, score = dis
- if name in full_dic['district']:
- for idx in full_dic['district'][name]:
- if idx not in dis_ids:
- dis_ids[idx] = 0
- dis_ids[idx] += score
- pro_idx = idx_dic[idx]['省']
- if pro_idx in tmp_pro:
- tmp_pro[pro_idx] += score
- else:
- if name in self.multi_dist: # 多个城市重复名称,需过滤
- continue
- tmp_pro[pro_idx] = score
- city_idx = idx_dic[idx]['市']
- if city_idx in tmp_city:
- tmp_city[city_idx] += score
- else:
- if name in self.multi_dist: # 多个城市重复名称,需过滤
- continue
- tmp_city[city_idx] = score
- elif name in short_dic['district']:
- for idx in short_dic['district'][name]:
- if idx not in dis_ids:
- dis_ids[idx] = 0
- dis_ids[idx] += score
- pro_idx = idx_dic[idx]['省']
- if pro_ids != {} and pro_idx not in pro_ids: # 如果省份不为空且简称不在省份分值降低
- score -= 0.1
- if filter_short_dist and score < 1: # pro_idx not in pro_ids
- continue
- if pro_idx in tmp_pro:
- tmp_pro[pro_idx] += score
- else:
- tmp_pro[pro_idx] = score
- city_idx = idx_dic[idx]['市']
- if city_idx in tmp_city:
- tmp_city[city_idx] += score
- else:
- tmp_city[city_idx] = score
- if set(tmp_pro) & set(pro_ids) != set():
- for k, v in tmp_pro.items():
- if k in pro_ids:
- pro_ids[k] += v
- else:
- pro_ids.update(tmp_pro)
- if set(tmp_city) & set(city_ids) != set():
- for k, v in tmp_city.items():
- if k in city_ids:
- city_ids[k] += v
- else:
- city_ids.update(tmp_city)
- return pro_ids, city_ids, dis_ids
- @staticmethod
- def get_final_addr(pro_ids, city_ids, dis_ids, idx_dic):
- '''
- 先把所有匹配的全称、简称转为id,如果省份不为空,城市不为空且有城市属于省份的取该城市
- :param province_l: 匹配到的所有省份
- :param city_l: 匹配到的所有城市
- :param district_l: 匹配到的所有区县
- :return:
- '''
- big_area = ""
- pred_pro = ""
- pred_city = ""
- pred_dis = ""
- final_pro = ""
- final_city = ""
- prob = 0
- max_score = 0
- code_dic = {
- 'province_code': '',
- 'city_code': '',
- 'district_code': ''
- }
- if len(pro_ids) >= 1:
- pro_l = sorted([(k, v) for k, v in pro_ids.items()], key=lambda x: x[1], reverse=True)
- scores = [it[1] for it in pro_l]
- prob = max(scores) / sum(scores)
- max_score = max(scores)
- final_pro, score = pro_l[0]
- if score >= 0.01:
- pred_pro = idx_dic[final_pro]['返回名称']
- big_area = idx_dic[final_pro]['大区']
- code_dic['province_code'] = idx_dic[final_pro]['编码']
- if pred_pro != "" and len(city_ids) >= 1:
- city_l = sorted([(k, v) for k, v in city_ids.items()], key=lambda x: x[1], reverse=True)
- for it in city_l:
- if idx_dic[it[0]]['省'] == final_pro:
- final_city = it[0]
- pred_city = idx_dic[final_city]['返回名称']
- code_dic['city_code'] = idx_dic[final_city]['编码']
- break
- if final_city != "" and len(set(dis_ids)) >= 1:
- dis_l = sorted([(k, v) for k, v in dis_ids.items()], key=lambda x: x[1], reverse=True)
- for it in dis_l:
- if idx_dic[it[0]]['市'] == final_city:
- pred_dis = idx_dic[it[0]]['返回名称']
- code_dic['district_code'] = idx_dic[it[0]]['编码']
- elif pred_pro != "" and pred_city == "" and len(set(dis_ids)) >= 1: # 20241111 省份不为空,市为空,如果区县在省份下,补充对应的市县
- dis_l = sorted([(k, v) for k, v in dis_ids.items()], key=lambda x: x[1], reverse=True)
- for it in dis_l:
- if idx_dic[it[0]]['省'] == final_pro:
- pred_city = idx_dic[idx_dic[it[0]]['市']]['返回名称']
- pred_dis = idx_dic[it[0]]['返回名称']
- code_dic['city_code'] = idx_dic[idx_dic[it[0]]['市']]['编码']
- code_dic['district_code'] = idx_dic[it[0]]['编码']
- return big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic
- @staticmethod
- def get_ree_addr(prem):
- tenderee = ""
- tenderee_address = ""
- try:
- for v in prem.values():
- for link in v['roleList']:
- if link['role_name'] == 'tenderee' and tenderee == "":
- tenderee = link['role_text']
- tenderee_address = link['address']
- except Exception as e:
- print('解析prem 获取招标人、及地址出错')
- return tenderee, tenderee_address
- @staticmethod
- def get_role_address(text):
- '''正则匹配获取招标人地址
- 3:地址直接在招标人后面 招标人:xxx,地址:xxx
- 4:招标、代理一起,两个地址一起 招标人:xxx, 代理人:xxx, 地址:xxx, 地址:xxx.
- '''
- p3 = '(招标|采购|甲)(人|方|单位)(信息:|(甲方))?(名称)?:[\w()]{4,15},(联系)?地址:(?P<addr>(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+|\w{2,15}[,。])'
- p4 = '(招标|采购|甲)(人|方|单位)(信息:|(甲方))?(名称)?:[\w()]{4,15},(招标|采购)?代理(人|机构)(名称)?:[\w()]{4,15},(联系)?地址:(?P<addr>(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+|\w{2,35}),(联系)?地址:'
- p5 = '(采购|招标)(人|单位)(联系)?地址:(?P<addr>(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+|\w{2,15}[,。])'
- if re.search(p3, text):
- return re.search(p3, text).group('addr')
- elif re.search(p4, text):
- return re.search(p4, text).group('addr')
- elif re.search(p5, text):
- return re.search(p5, text).group('addr')
- else:
- return ''
- @staticmethod
- def get_all_addr(list_entity, filter_attach=False):
- '''
- 获取所有招标或代理人名称及所有地址
- :param list_entity: 实体列表
- :param filter_attach: 是否过滤附件实体
- :return:
- '''
- tenderee_l = []
- addr_l = []
- for ent in list_entity:
- if ent.entity_type not in ['org', 'company', 'location'] or (filter_attach and ent.in_attachment):
- continue
- if ent.entity_type == 'location' and len(ent.entity_text) > 2:
- addr_l.append(ent.entity_text)
- elif ent.entity_type in ['org', 'company']:
- if ent.label in [0, 1]: # 加招标或代理
- tenderee_l.append(ent.entity_text)
- elif re.search('[局委]$', ent.entity_text):
- tenderee_l.append(ent.entity_text)
- if len(addr_l) > 10: # 只取前10个地址
- break
- return ' '.join(set(addr_l)), ' '.join(set(tenderee_l))
- def addr_process(self, addr_text):
- addr_text = addr_text.replace('(', '(').replace(')', ')')
- if re.search('[省市县]', addr_text) == None:
- ser = re.search('\w{2,}区', addr_text)
- ser2 = re.match('\w{2,5}镇', addr_text)
- if ser:
- addr_text = addr_text[:ser.end()]
- elif ser2:
- addr_text = addr_text[:ser2.end()]
- return addr_text
- def predict_area(self,docid, title, content, web_source_name, prem={}, addr_dic={}, list_entity=[]):
- if re.match('([^,]{3,50}网络竞价会),', content): # 修复 701397187 标题只有拍卖的东西,第一句有地址
- title_auction = re.match('([^,]{3,50}网络竞价会),', content).group(1)
- if title.find(title_auction[:3]) == -1:
- title += ' ' + title_auction
- filter_attach = False # 是否过滤附件内容
- if '##attachment##' in content:
- main, att = content.split('##attachment##')
- if 2000 < len(main) < len(att): # 正文超过500字且附件比正文长过滤附件内容
- filter_attach = True
- # print('正文超过500字过滤附件内容')
- ree, addr_ree = self.get_ree_addr(prem)
- addr_ree = self.addr_process(addr_ree)
- addr_bus = ''
- if len(addr_ree) < 3 and ree != '':
- have_bus, bus_dic = get_business_data(ree)
- if have_bus:
- addr_bus = '%s %s %s' % (bus_dic.get('province', ''), bus_dic.get('city', ''), bus_dic.get('district', ''))
- all_addr, tenderees = self.get_all_addr(list_entity, filter_attach)
- return self.predict_distrist(docid, title, web_source_name, ree, addr_ree, addr_bus, addr_dic, tenderees, all_addr)
- def predict_distrist(self,docid, title, web_source_name, ree, addr_ree, addr_bus,location_dic={}, tenderees='', all_addr=''):
- '''
- 优先顺序项目地址、收货地址、招标人地址 / 工商地址、开标地址 / 联系地址、站源名称
- '''
- area_dic = {'area': '全国', 'province': '全国', 'city': '未知', 'district': '未知', "is_in_text": False}
- in_content = False
- not_sure = True # 是否不确定地区
- msc = "" # 日志信息
- addr_dic = {}
- prov_name_l = [] # 保存预测到的省份名称
- city_name_l = [] # 保存预测到的城市名称
- final_key = ''
- company_title = ''
- title_raw = title
- if len(title.strip()) > 4:
- ner_title = getNers([title], True)[0]
- company_title = []
- for ner in ner_title:
- if ner[2] in ['org', 'company']:
- company_title.append(ner[3])
- title = title.replace(ner[3], '#')
- company_title = "#".join(company_title)
- if re.search('拍卖|法院', ree):
- company_title += ' ' + ree
- ree = ''
- if web_source_name in ('中原云商', '中原云商电子招投标平台'): # 修复某些公告没地区
- web_source_name = '河南中原云商'
- for key in ['addr_project', 'addr_delivery', 'addr_bidopen', 'addr_bidsend', 'addr_contact']:
- addr = location_dic.get(key, '')
- addr = self.addr_process(addr)
- if len(addr) < 2:
- continue
- province_l, city_l, district_l = self.find_whole_areas('%s'%addr, self.pettern, self.area_variance_dic, self.full_dic)
- if len(province_l+city_l+district_l)>0:
- pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic,self.short_dic, self.idx_dic)
- big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = self.get_final_addr(pro_ids, city_ids, dis_ids, self.idx_dic)
- if max_score < 2 and len(addr) > len(''.join([it[0] for it in province_l+city_l+district_l]))*2: # 修复 712151540 北京银行大厦6楼厨房及22楼 提取为北京
- log('地区匹配非正常地址:%s, 预测为:%s %s %s, docid:%s'%(addr, pred_pro, pred_city, pred_dis, docid))
- continue
- if pred_pro != '':
- addr_dic[key] = {
- 'keyword': (province_l, city_l, district_l),
- 'result': (big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic)
- }
- msc += "使用%s信息:%s, 预测为:%s %s %s;" % (key, addr, pred_pro, pred_city, pred_dis)
- prov_name_l.append(pred_pro)
- if pred_city != '':
- city_name_l.append(pred_city)
- if key == 'addr_project' and addr_dic['addr_project']['result'][2] != '' and addr_dic['addr_project']['result'][4] > 0.6 and (addr_dic['addr_project']['result'][5] >= 2 or addr_dic['addr_project']['keyword'][2]==[]):
- big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic['addr_project']['result']
- not_sure = False
- final_key = key
- msc += "最终使用:%s,预测为:%s %s %s;"%(final_key, pred_pro, pred_city, pred_dis)
- if addr.endswith('#在附件'):
- log('地区匹配使用附件中的项目地址:%s;预测为:%s %s;docid:%s'%(addr, pred_pro, pred_city, docid))
- break
- if not_sure:
- for key, text in zip(['tenderee', 'company_title', 'web_source_name'], [ree, company_title, web_source_name]):
- if len(text) < 4:
- continue
- weight = 0.48 if key == 'web_source_name' else 1
- province_l, city_l, district_l = self.find_whole_areas('%s'%text, self.pettern, self.area_variance_dic, self.full_dic, weight=weight)
- if len(province_l+city_l+district_l)>0:
- pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic,self.short_dic, self.idx_dic)
- big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = self.get_final_addr(pro_ids, city_ids, dis_ids, self.idx_dic)
- if pred_pro != '':
- addr_dic[key] = {
- 'keyword': (province_l, city_l, district_l),
- 'result': (big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic)
- }
- msc += "使用%s信息:%s, 预测为:%s %s %s;" % (key, text, pred_pro, pred_city, pred_dis)
- prov_name_l.append(pred_pro)
- if pred_city != '':
- city_name_l.append(pred_city)
- elif key == 'company_title' and re.search('\w{1,}[省市县]', text): # 修复 2025年可克达拉市政府部门独立办公楼聘用保安保洁采购服务 实体提取为 克达拉市政府 造成市漏提
- title = title_raw
- # 提取招标地址
- if len(addr_ree) >= 3:
- province_l, city_l, district_l = self.find_whole_areas('%s' % addr_ree, self.pettern, self.area_variance_dic,self.full_dic)
- if len(province_l + city_l + district_l) > 0:
- pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic,self.short_dic, self.idx_dic)
- big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = self.get_final_addr(pro_ids,city_ids,dis_ids,self.idx_dic)
- if pred_pro != '':
- addr_dic['addr_tenderee'] = {
- 'keyword': (province_l, city_l, district_l),
- 'result': (big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic)
- }
- msc += "使用招标人地址:%s, 预测为:%s %s %s;" % (addr_ree, pred_pro, pred_city, pred_dis)
- prov_name_l.append(pred_pro)
- if pred_city != '':
- city_name_l.append(pred_city)
- # 提取招标人工商登记地址
- if len(addr_bus) >= 3:
- province_l, city_l, district_l = self.find_whole_areas('%s' % addr_bus, self.pettern, self.area_variance_dic,self.full_dic)
- if len(province_l + city_l + district_l) > 0:
- pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic,self.short_dic, self.idx_dic)
- big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = self.get_final_addr(pro_ids,city_ids,dis_ids,self.idx_dic)
- if pred_pro != '':
- addr_dic['addr_bus'] = {
- 'keyword': (province_l, city_l, district_l),
- 'result': (big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic)
- }
- msc += "使用招标人工商登记地址:%s, 预测为:%s %s %s;" % (addr_bus, pred_pro, pred_city, pred_dis)
- prov_name_l.append(pred_pro)
- if pred_city != '':
- city_name_l.append(pred_city)
- # 提取标题地址
- if len(title) > 3:
- province_l, city_l, district_l = self.find_whole_areas('%s' % title, self.pettern, self.area_variance_dic,self.full_dic)
- if len(province_l + city_l + district_l) > 0:
- pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic,self.short_dic, self.idx_dic)
- big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = self.get_final_addr(pro_ids,city_ids,dis_ids,self.idx_dic)
- if pred_pro != '':
- addr_dic['addr_title'] = {
- 'keyword': (province_l, city_l, district_l),
- 'result': (big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic)
- }
- msc += "使用标题地址:%s, 预测为:%s %s %s;" % (title, pred_pro, pred_city, pred_dis)
- prov_name_l.append(pred_pro)
- if pred_city != '':
- city_name_l.append(pred_city)
- if len(addr_dic) == 1:
- for key in addr_dic:
- big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic[key]['result']
- final_key = key
- msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis)
- elif len(addr_dic) > 1:
- prov_count = Counter(prov_name_l)
- city_count = Counter(city_name_l)
- if len(prov_count) == 1 and len(city_count) == 1:
- score_max = 0
- for key in ['addr_project', 'addr_delivery', 'addr_tenderee', 'addr_title', 'tenderee', 'company_title',
- 'addr_bidopen', 'addr_bidsend', 'addr_contact', 'addr_bus', 'web_source_name']:
- if key in addr_dic and addr_dic[key]['result'][1] == prov_name_l[0] and addr_dic[key]['result'][2] == city_name_l[0] and addr_dic[key]['result'][5] > score_max:
- score_max = addr_dic[key]['result'][5]
- final_key = key
- if addr_dic[final_key]['result'][3] != '': # 2026/2/2 省市匹配,且有区县停止循环
- break
- big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic[final_key]['result']
- msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis)
- # if key in addr_dic and addr_dic[key]['result'][1] == prov_name_l[0] and addr_dic[key]['result'][2] == city_name_l[0]:
- # # big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic[key]['result']
- # final_key = key
- # msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis)
- # break
- elif 'addr_project' in addr_dic and addr_dic['addr_project']['result'][2] == '': # 只有省份
- big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic['addr_project']['result']
- final_key = 'addr_project'
- score_max = 0
- tmp_key = ''
- for key in ['addr_delivery', 'addr_tenderee', 'addr_title', 'tenderee', 'company_title',
- 'addr_bidopen', 'addr_bidsend', 'addr_contact', 'addr_bus', 'web_source_name']:
- if key in addr_dic and addr_dic[key]['result'][1] == pred_pro and addr_dic[key]['result'][2] != '' and addr_dic[key]['result'][5] > score_max:
- score_max = addr_dic[key]['result'][5]
- tmp_key = key
- if addr_dic[tmp_key]['result'][3] != '': # 2026/2/2 省市匹配,且有区县停止循环
- break
- if tmp_key != '':
- final_key += ',%s' % tmp_key
- big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic[tmp_key]['result']
- msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis)
- elif 'addr_delivery' in addr_dic and addr_dic['addr_delivery']['result'][2] != '' and addr_dic['addr_delivery']['result'][4] > 0.7 and addr_dic['addr_delivery']['result'][5] >= 2:
- big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic['addr_delivery']['result']
- final_key = 'addr_delivery'
- msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis)
- elif 'addr_title' in addr_dic and addr_dic['addr_title']['result'][2] != '' and addr_dic['addr_title']['result'][4] > 0.7 and addr_dic['addr_title']['result'][5] >= 2:
- big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic['addr_title']['result']
- final_key = 'addr_title'
- msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis)
- elif 'tenderee' in addr_dic and addr_dic['tenderee']['result'][2] != '' and addr_dic['tenderee']['result'][4] > 0.7 and addr_dic['tenderee']['result'][5] >= 2: # 补概率 653492303 广州市妇女儿童医疗中心柳州医院 预测错广州
- big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic['tenderee']['result']
- final_key = 'tenderee'
- msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis)
- elif 'addr_tenderee' in addr_dic and addr_dic['addr_tenderee']['result'][2] != '' and addr_dic['addr_tenderee']['result'][5] >= 2:
- big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic['addr_tenderee']['result']
- final_key = 'addr_tenderee'
- msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis)
- else:
- province_l, city_l, district_l = [], [], []
- for key in addr_dic:
- if key in ['addr_delivery', 'addr_bidopen', 'addr_bidsend', 'addr_contact', 'addr_bus']: # 地址类型只加一项,因为一般都有三项造成单个地址总分过大
- if addr_dic[key]['keyword'][2]:
- district_l += addr_dic[key]['keyword'][2]
- elif addr_dic[key]['keyword'][1]:
- city_l += addr_dic[key]['keyword'][1]
- else:
- province_l += addr_dic[key]['keyword'][0]
- else:
- province_l += addr_dic[key]['keyword'][0]
- city_l += addr_dic[key]['keyword'][1]
- district_l += addr_dic[key]['keyword'][2]
- if key in ['addr_project', 'addr_title']: # 加重项目、标题地址权重
- province_l += addr_dic[key]['keyword'][0]
- city_l += addr_dic[key]['keyword'][1]
- district_l += addr_dic[key]['keyword'][2]
- pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic, self.short_dic, self.idx_dic)
- big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = self.get_final_addr(pro_ids, city_ids, dis_ids, self.idx_dic)
- if pred_city != '':
- score_max = 0
- for key in ['addr_delivery', 'addr_tenderee', 'addr_title', 'tenderee', 'company_title',
- 'addr_bidopen', 'addr_bidsend', 'addr_contact', 'addr_bus', 'web_source_name']:
- if key in addr_dic and addr_dic[key]['result'][2] == pred_city and addr_dic[key]['result'][5] > score_max:
- final_key = key
- score_max = addr_dic[key]['result'][5]
- if addr_dic[final_key]['result'][3] == pred_dis: # 2026/2/2 省市匹配,且有区县停止循环
- break
- if pred_pro != '' and final_key == "":
- for key in ['addr_delivery', 'addr_tenderee', 'addr_title', 'tenderee', 'company_title',
- 'addr_bidopen', 'addr_bidsend', 'addr_contact', 'addr_bus', 'web_source_name']:
- if key in addr_dic and addr_dic[key]['result'][1] == pred_pro:
- final_key = key
- break
- msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis)
- else:
- # 取所有的地址
- province_l, city_l, district_l = self.find_whole_areas('%s %s %s' % (title_raw, tenderees, all_addr), self.pettern, self.area_variance_dic,self.full_dic, weight=0.5)
- pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic, self.short_dic, self.idx_dic)
- big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = self.get_final_addr(pro_ids, city_ids, dis_ids, self.idx_dic)
- final_key = 'all_addr'
- in_content = True
- msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis)
- if len(addr_dic) > 0 and pred_city == '': # 如果非正文提取缺少城市,补充
- province_l, city_l, district_l = self.find_whole_areas('%s %s %s' % (title_raw, tenderees, all_addr), self.pettern, self.area_variance_dic,self.full_dic, weight=0.5)
- pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic, self.short_dic, self.idx_dic)
- rs_tmp = self.get_final_addr(pro_ids, city_ids, dis_ids, self.idx_dic)
- if rs_tmp[1] == pred_pro and rs_tmp[4] > 0.5 and rs_tmp[2] != '':
- big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = rs_tmp
- msc += "只有省份,所有地址补充市县:%s %s ;"%(pred_city, pred_dis)
- final_key += ',all_addr'
- dist_source = {
- 'addr_project': '项目地址',
- 'addr_delivery': '收货地址',
- 'addr_title': '标题',
- 'tenderee': "招标人",
- 'company_title': '标题公司',
- 'addr_tenderee': '招标人地址',
- 'addr_bus': '招标人工商地址',
- 'web_source_name': '站源',
- 'addr_bidopen': '开标地址',
- 'addr_bidsend': '邮件地址',
- 'addr_contact': '联系地址',
- 'all_addr': '文中所有地址'
- }
- if big_area != "":
- area_dic['area'] = big_area
- if pred_pro != "":
- area_dic['province'] = pred_pro
- source_l = [dist_source.get(k, '') for k in final_key.split(',')]
- area_dic['dist_source'] = ';'.join(source_l)
- if pred_city != "":
- area_dic['city'] = pred_city
- if pred_dis != "":
- area_dic['district'] = pred_dis
- for k, v in code_dic.items():
- if v != '':
- area_dic[k] = v
- if pred_city in ['北京', '天津', '上海', '重庆']: # 直辖市调整县级
- area_dic['city'] = area_dic['district']
- area_dic['district'] = '未知'
- if 'district_code' in area_dic:
- area_dic['city_code'] = area_dic['district_code']
- area_dic.pop('district_code')
- if area_dic['city'] == '未知' and 'city_code' in area_dic:
- area_dic.pop('city_code')
- area_dic['is_in_text'] = in_content
- # area_dic['prob'] = prob
- # area_dic['max_score'] = max_score
- # print('最终地址:', pred_pro, pred_city, pred_dis)
- if prob < 0.6 or max_score < 2:
- log('地区匹配预测,最终结果:%s %s %s, 预测:%s, docid:%s' % (pred_pro, pred_city, pred_dis, msc, docid))
- return {'district': area_dic}
- def predict_area_backup(self,docid, title, content, web_source_name, prem={}, addr_dic={}, list_entity=[]):
- area_dic = {'area': '全国', 'province': '全国', 'city': '未知', 'district': '未知', "is_in_text": False}
- addr_project = addr_dic.get('addr_project', '').replace('(', '(').replace(')', ')')
- addr_delivery = addr_dic.get('addr_delivery', '').replace('(', '(').replace(')', ')')
- addr_bidopen = addr_dic.get('addr_bidopen', '').replace('(', '(').replace(')', ')')
- addr_bidsend = addr_dic.get('addr_bidsend', '').replace('(', '(').replace(')', ')')
- addr_contact = addr_dic.get('addr_contact', '').replace('(', '(').replace(')', ')')
- in_content = False
- not_sure = True # 是否不确定地区
- filter_attach = False # 是否过滤附件内容
- msc = "" # 日志信息
- if '##attachment##' in content:
- main, att = content.split('##attachment##')
- if 500 < len(main) < len(att): # 正文超过500字且附件比正文长过滤附件内容
- filter_attach = True
- # print('正文超过500字过滤附件内容')
- if web_source_name in ('中原云商', '中原云商电子招投标平台'): # 修复某些公告没地区
- web_source_name = '河南中原云商'
- province_l, city_l, district_l = self.find_whole_areas('%s %s'%(title, addr_project), self.pettern, self.area_variance_dic, self.full_dic)
- pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic, self.short_dic, self.idx_dic)
- big_area_1, pred_pro_1, pred_city_1, pred_dis_1, prob_1, max_score, code_dic_1 = self.get_final_addr(pro_ids, city_ids, dis_ids, self.idx_dic)
- big_area, pred_pro, pred_city, pred_dis, prob, code_dic = big_area_1, pred_pro_1, pred_city_1, pred_dis_1, prob_1, code_dic_1
- # print('关键词1:', province_l, city_l, district_l)
- # print('输入:', '标题:%s; 项目地址:%s'%(title, addr_project))
- # print('分数:', pro_ids, city_ids, dis_ids, prob, max_score)
- msc += '第一次预测,标题:%s; 项目地址:%s; 预测为:%s %s ##'%(title, addr_project, pred_pro, pred_city)
- if pred_city_1 == "" or prob < 0.7 or max_score<2:
- ree, addr = self.get_ree_addr(prem)
- have_bus, bus_dic = get_business_data(ree) if ree != '' else False, {}
- if ree in title:
- ree = '##'
- rule_ree_addr = self.get_role_address(content)
- if rule_ree_addr:
- addr = rule_ree_addr
- # addr = content
- # ree = ''
- province_l2, city_l2, district_l2 = self.find_whole_areas('%s %s %s' % (ree, addr, addr_delivery), self.pettern, self.area_variance_dic, self.full_dic, weight=1)
- province_l.extend(province_l2)
- city_l.extend(city_l2)
- district_l.extend(district_l2)
- pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic, self.short_dic, self.idx_dic)
- big_area_2, pred_pro_2, pred_city_2, pred_dis_2, prob_2, max_score, code_dic_2 = self.get_final_addr(pro_ids, city_ids, dis_ids, self.idx_dic)
- big_area, pred_pro, pred_city, pred_dis, prob, code_dic = big_area_2, pred_pro_2, pred_city_2, pred_dis_2, prob_2, code_dic_2
- # print('关键词2:', province_l, city_l, district_l)
- # print('输入:', '招标人:%s; 招标人地址:%s; 收货地址:%s' % (ree, addr, addr_delivery))
- # print('分数:', pro_ids, city_ids, dis_ids, prob, max_score)
- msc += '第二次预测,招标人:%s; 招标人地址:%s; 收货地址:%s; 预测为:%s %s ##' % (ree, addr, addr_delivery, pred_pro, pred_city)
- if re.search('省|市|县|自治', addr_project) and prob_1 !=0.5 and pred_pro_1 != '' and pred_pro_1 != pred_pro_2: # 如果有项目地址使用项目地址 要有省市县等 275127622 工程地点为狮山镇颜峰综合区岐山至人和段道路, 提错 岐山
- not_sure = False
- big_area, pred_pro, pred_city, pred_dis, code_dic = big_area_1, pred_pro_1, pred_city_1, pred_dis_1, code_dic_1
- msc += "有项目地址,一二次预测省份不同,改为第一次结果。"
- if not_sure and (pred_city_2 == "" or prob < 0.7 or max_score<2):
- addr_bus = '%s %s'%(bus_dic.get('province', ''), bus_dic.get('city', ''))
- province_l3, city_l3, district_l3 = self.find_whole_areas('%s %s; %s; %s'%(addr_bus, addr_contact, addr_bidopen, addr_bidsend), self.pettern, self.area_variance_dic, self.full_dic, weight=0.6)
- province_l.extend(province_l3)
- city_l.extend(city_l3)
- district_l.extend(district_l3)
- pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic, self.short_dic, self.idx_dic)
- big_area_3, pred_pro_3, pred_city_3, pred_dis_3, prob_3, max_score, code_dic_3 = self.get_final_addr(pro_ids, city_ids, dis_ids, self.idx_dic)
- big_area, pred_pro, pred_city, pred_dis, prob, code_dic = big_area_3, pred_pro_3, pred_city_3, pred_dis_3, prob_3, code_dic_3
- # print('关键词3:', province_l, city_l, district_l)
- # print('输入:', '联系:%s, 开标:%s, 邮寄:%s'%(addr_contact, addr_bidopen, addr_bidsend))
- # print('分数:', pro_ids, city_ids, dis_ids, prob, max_score)
- msc += '第三次预测,工商地址:%s;联系:%s; 开标:%s; 邮寄:%s; 预测为:%s %s ##' % (addr_bus, addr_contact, addr_bidopen, addr_bidsend, pred_pro, pred_city)
- if pred_city_2 != "" and prob_2 !=0.5 and pred_city_2 != pred_city_3:
- not_sure = False
- big_area, pred_pro, pred_city, pred_dis, code_dic = big_area_2, pred_pro_2, pred_city_2, pred_dis_2, code_dic_2 # 如果招标人、招标人地址、收货地址与开标地址、联系地址等不一致,取招标人地址
- msc += "二三次预测城市不一致,改为第二次结果。"
- if not_sure and (pred_city_3 == "" or prob < 0.6 or max_score < 2):
- all_addr, tenderees = self.get_all_addr(list_entity, filter_attach)
- province_l4, city_l4, district_l4 = self.find_whole_areas('%s %s %s' % (web_source_name, tenderees, all_addr), self.pettern, self.area_variance_dic, self.full_dic, weight=0.3)
- province_l.extend(province_l4)
- city_l.extend(city_l4)
- district_l.extend(district_l4)
- pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic, self.short_dic, self.idx_dic)
- big_area_4, pred_pro_4, pred_city_4, pred_dis_4, prob_4, max_score, code_dic_4 = self.get_final_addr(pro_ids, city_ids,dis_ids, self.idx_dic)
- big_area, pred_pro, pred_city, pred_dis, prob, code_dic = big_area_4, pred_pro_4, pred_city_4, pred_dis_4, prob_4, code_dic_4
- if pred_city_3 != "" and prob_3 !=0.5 and pred_city_3 != pred_city_4:
- not_sure = False
- big_area, pred_pro, pred_city, pred_dis, code_dic = big_area_3, pred_pro_3, pred_city_3, pred_dis_3, code_dic_3 # 如果开标地址等提取的城市与所有地址提取的城市不一致,取开标地址等
- msc += "三四次预测城市不一致,改为第三次结果。"
- if pred_pro_3 != pred_pro_4 and (prob < 0.6 or max_score < 2) or max_score < 1:
- in_content = True
- # print('关键词4:', province_l, city_l, district_l)
- # print('输入:', '站源:%s, 角色:%s, 地址:%s' % (web_source_name, tenderees, all_addr))
- # print('分数:', pro_ids, city_ids, dis_ids, prob, max_score)
- msc += '第四次预测,站源:%s; 招标、代理角色:%s; 所有地址:%s; 预测为:%s %s ##' % (web_source_name, tenderees, all_addr, pred_pro, pred_city)
- if big_area != "":
- area_dic['area'] = big_area
- if pred_pro != "":
- area_dic['province'] = pred_pro
- if pred_city != "":
- area_dic['city'] = pred_city
- if pred_dis != "":
- area_dic['district'] = pred_dis
- for k, v in code_dic.items():
- if v != '':
- area_dic[k] = v
- if pred_city in ['北京', '天津', '上海', '重庆']: # 直辖市调整县级
- area_dic['city'] = area_dic['district']
- area_dic['district'] = '未知'
- if 'district_code' in area_dic:
- area_dic['city_code'] = area_dic['district_code']
- area_dic.pop('district_code')
- if area_dic['city'] == '未知' and 'city_code' in area_dic:
- area_dic.pop('city_code')
- area_dic['is_in_text'] = in_content
- # area_dic['prob'] = prob
- # area_dic['max_score'] = max_score
- # print('最终地址:', pred_pro, pred_city, pred_dis)
- if prob < 0.6 or max_score < 2:
- log('地区匹配预测,最终结果:%s %s %s, 预测:%s, docid:%s'%(pred_pro, pred_city, pred_dis, msc, docid))
- return {'district': area_dic}
- def get_area(self, text, web_name, in_content=False):
- p_pro, p_city, p_dis, idx_dic, full_dic, short_dic = self.p_pro, self.p_city, self.p_dis, self.idx_dic, self.full_dic, self.short_dic
- def get_final_addr(pro_ids, city_ids, dis_ids):
- '''
- 先把所有匹配的全称、简称转为id,如果省份不为空,城市不为空且有城市属于省份的取该城市
- :param province_l: 匹配到的所有省份
- :param city_l: 匹配到的所有城市
- :param district_l: 匹配到的所有区县
- :return:
- '''
- big_area = ""
- pred_pro = ""
- pred_city = ""
- pred_dis = ""
- final_pro = ""
- final_city = ""
- pro_prob = 0
- city_prob = 0
- if len(pro_ids) >= 1:
- pro_l = sorted([(k, v) for k, v in pro_ids.items()], key=lambda x: x[1], reverse=True)
- scores = [it[1] for it in pro_l]
- pro_prob = max(scores)/sum(scores)
- final_pro, score = pro_l[0]
- if score >= 0.01:
- pred_pro = idx_dic[final_pro]['返回名称']
- big_area = idx_dic[final_pro]['大区']
- # else:
- # print("得分过低,过滤掉", idx_dic[final_pro]['返回名称'], score)
- if pred_pro != "" and len(city_ids) >= 1:
- city_l = sorted([(k, v) for k, v in city_ids.items()], key=lambda x: x[1], reverse=True)
- scores = [it[1] for it in city_l]
- city_prob = max(scores) / sum(scores)
- for it in city_l:
- if idx_dic[it[0]]['省'] == final_pro:
- final_city = it[0]
- pred_city = idx_dic[final_city]['返回名称']
- break
- if final_city != "" and len(set(dis_ids)) >= 1:
- dis_l = sorted([(k, v) for k, v in dis_ids.items()], key=lambda x: x[1], reverse=True)
- for it in dis_l:
- if idx_dic[it[0]]['市'] == final_city:
- pred_dis = idx_dic[it[0]]['返回名称']
- elif pred_pro != "" and pred_city == "" and len(set(dis_ids)) >= 1: # 20241111 省份不为空,市为空,如果区县在省份下,补充对应的市县
- dis_l = sorted([(k, v) for k, v in dis_ids.items()], key=lambda x: x[1], reverse=True)
- for it in dis_l:
- if idx_dic[it[0]]['省'] == final_pro:
- pred_city = idx_dic[idx_dic[it[0]]['市']]['返回名称']
- pred_dis = idx_dic[it[0]]['返回名称']
- # print('20241111 省份不为空,市为空,如果区县在省份下,补充对应的市县: ', pred_city, pred_dis)
- if pred_city in ['北京', '天津', '上海', '重庆']:
- pred_city = pred_dis
- pred_dis = ""
- return big_area, pred_pro, pred_city, pred_dis
- def find_areas(pettern, text):
- '''
- 通过正则匹配字符串返回地址
- :param pettern: 地址正则 广东省|广西省|...
- :param text: 待匹配文本
- :return:
- '''
- addr = []
- for it in re.finditer(pettern, text):
- if re.search('[省市区县旗盟]$', it.group(0)) == None and re.search(
- '^([东南西北中一二三四五六七八九十大小]?(村|镇|街|路|道|社区)|酒店|宾馆)', text[it.end():]):
- continue
- if it.group(0) == '站前': # 20240314 修复类似 中铁二局新建沪苏湖铁路工程站前VI标项目 错识别为 省份:辽宁, 城市:营口,区县:站前
- continue
- if re.search('^(经济开发区|开发区|新区)', text[it.end():]) and re.search('广州市', pettern): # 城市不匹配为区的地址 修复 滨州北海经济开发区 北海新区 等提取为北海
- continue
- addr.append((it.group(0), it.start(), it.end()))
- if re.search('^([分支](公司|局|行|校|院|干?线)|\w{,3}段|地铁|(火车|高铁)?站|\w{,3}项目)', text[it.end():]):
- addr.append((it.group(0), it.start(), it.end()))
- return addr
- def chage_area2score(group_list, max_len):
- '''
- 把匹配的的地址转为分数
- :param group_list: [('name', b, e)]
- :return:
- '''
- area_list = []
- if group_list != []:
- for it in group_list:
- name, b, e = it
- area_list.append((name, (e - b + e) / max_len / 2))
- return area_list
- def find_whole_areas(text):
- '''
- 通过正则匹配字符串返回地址
- :param pettern: 地址正则 广东省|广西省|...
- :param text: 待匹配文本
- :return:
- '''
- pettern = "((?P<prov>%s)(?P<city>%s)?(?P<dist>%s)?)|((?P<city1>%s)(?P<dist1>%s)?)|(?P<dist2>%s)" % (
- p_pro, p_city, p_dis, p_city, p_dis, p_dis)
- province_l, city_l, district_l = [], [], []
- for it in re.finditer(pettern, text):
- if re.search('[省市区县旗盟]', it.group(0)) == None and re.search(
- '^([东南西北中一二三四五六七八九十大小]?(村|镇|街|路|道|社区)|酒店|宾馆)', text[it.end():]):
- continue
- if it.group(0) == '站前': # 20240314 修复类似 中铁二局新建沪苏湖铁路工程站前VI标项目 错识别为 省份:辽宁, 城市:营口,区县:站前
- continue
- for k, v in it.groupdict().items():
- if v != None:
- if k in ['prov']:
- province_l.append((it.group(k), it.start(k), it.end(k)))
- elif k in ['city', 'city1']:
- if re.search('^(经济开发区|开发区|新区)', text[it.end(k):]): # 城市不匹配为区的地址 修复 滨州北海经济开发区 北海新区 等提取为北海
- continue
- city_l.append((it.group(k), it.start(k), it.end(k)))
- if re.search('^([分支](公司|局|行|校|院|干?线)|\w{,3}段|地铁|(火车|高铁)?站|\w{,3}项目)', text[it.end(k):]):
- city_l.append((it.group(k), it.start(k), it.end(k)))
- elif k in ['dist', 'dist1', 'dist2']:
- if it.group(k)=='昌江' and '景德镇' not in it.group(0):
- district_l.append(('昌江黎族', it.start(k), it.end(k)))
- else:
- district_l.append((it.group(k), it.start(k), it.end(k)))
- return province_l, city_l, district_l
- def get_pro_city_dis_score(text, text_weight=1):
- text = re.sub('复合肥|海南岛|兴业银行|双河口|阳光|杭州湾|新城区|中粮屯河|老城(区|改造|更新|升级|翻新)|沙县小吃|北京时间', ' ', text) # 544151395 赤壁市老城区燃气管道老化更新改造
- text = re.sub('珠海城市', '珠海', text) # 修复 426624023 珠海城市 预测为海城市
- text = re.sub('怒江州', '怒江傈僳族自治州', text) # 修复 423589589 所属地域:怒江州 识别为广西 - 崇左 - 江州
- text = re.sub('茂名滨海新区', '茂名市', text)
- text = re.sub('中山([东南西][部区环]|黄圃|南头|东凤|小榄|石岐|翠亨|南朗)', '中山市', text)
- text = re.sub('横州市', '横县', text) # 例:547363890 修复广西南宁横州 不在地区表问题
- ser = re.search('海南(昌江|白沙|乐东|陵水|保亭|琼中)(黎族)?', text)
- if ser and '黎族' not in ser.group(0):
- text = text.replace(ser.group(0), ser.group(0)+'黎族')
- for k, v in self.area_variance_dic.items(): # 20241113 根据地区变更信息替换文本
- text = text.replace(k, v)
- # province_l = find_areas(p_pro, text)
- # city_l = find_areas(p_city, text)
- # district_l = find_areas(p_dis, text)
- province_l, city_l, district_l = find_whole_areas(text) # 20240703 优化地址提取,解决类似 海南昌江 得到 海南 南昌 结果
- # if len(province_l) == len(city_l) == 0:
- # district_l = [it for it in district_l if
- # re.search('[市县旗区]$', it[0])] # 20240428去掉只有区县地址且不是全称的匹配,避免错误 例 凌云工业股份有限公司 提取地区为广西白色凌云
- province_l = chage_area2score(province_l, max_len=len(text))
- city_l = chage_area2score(city_l, max_len=len(text))
- district_l = chage_area2score(district_l, max_len=len(text))
- pro_ids = dict()
- city_ids = dict()
- dis_ids = dict()
- for pro in province_l:
- name, score = pro
- assert (name in full_dic['province'] or name in short_dic['province'])
- if name in full_dic['province']:
- idx = full_dic['province'][name]
- if idx not in pro_ids:
- pro_ids[idx] = 0
- pro_ids[idx] += (score + 1)
- else:
- idx = short_dic['province'][name]
- if idx not in pro_ids:
- pro_ids[idx] = 0
- pro_ids[idx] += (score + 0)
- for city in city_l:
- name, score = city
- if name in full_dic['city']:
- w = 0.1 if len(full_dic['city'][name]) > 1 else 1
- for idx in full_dic['city'][name]:
- if idx not in city_ids:
- city_ids[idx] = 0
- # weight = idx_dic[idx]['权重']
- city_ids[idx] += (score + 2) * w
- pro_idx = idx_dic[idx]['省']
- if pro_idx in pro_ids:
- pro_ids[pro_idx] += (score + 2) * w
- else:
- pro_ids[pro_idx] = (score + 2) * w * 0.5
- elif name in short_dic['city']:
- w = 0.1 if len(short_dic['city'][name]) > 1 else 1
- for idx in short_dic['city'][name]:
- if idx not in city_ids:
- city_ids[idx] = 0
- weight = idx_dic[idx]['权重']
- city_ids[idx] += (score + 1) * w * weight
- pro_idx = idx_dic[idx]['省']
- if pro_idx in pro_ids:
- pro_ids[pro_idx] += (score + 1) * w * weight
- else:
- pro_ids[pro_idx] = (score + 1) * w * weight * 0.5
- for dis in district_l:
- name, score = dis
- if name in full_dic['district']:
- w = 0.1 if len(full_dic['district'][name]) > 1 else 1
- for idx in full_dic['district'][name]:
- if idx not in dis_ids:
- dis_ids[idx] = 0
- # weight = idx_dic[idx]['权重']
- dis_ids[idx] += (score + 1) * w
- pro_idx = idx_dic[idx]['省']
- if pro_idx in pro_ids:
- pro_ids[pro_idx] += (score + 1) * w
- else:
- pro_ids[pro_idx] = (score + 1) * w * 0.5
- city_idx = idx_dic[idx]['市']
- if city_idx in city_ids:
- city_ids[city_idx] += (score + 1) * w
- else:
- city_ids[city_idx] = (score + 1) * w * 0.5
- elif name in short_dic['district']:
- w = 0.1 if len(short_dic['district'][name]) > 1 else 1
- for idx in short_dic['district'][name]:
- if idx not in dis_ids:
- dis_ids[idx] = 0
- weight = idx_dic[idx]['权重']
- dis_ids[idx] += (score + 0) * w
- if idx_dic[idx]['市'] not in city_ids and idx_dic[idx]['省'] not in pro_ids: # 20241111 区县简称不在获取到的省、市范围内的过滤掉
- continue
- pro_idx = idx_dic[idx]['省']
- if pro_idx in pro_ids:
- pro_ids[pro_idx] += (score + 0) * w * weight
- # else: # 20241015 注销 区县简称且不在提取的省市下面,不加分,避免提取错误 例:536550843
- # pro_ids[pro_idx] = (score + 0) * w * weight * 0.5
- city_idx = idx_dic[idx]['市']
- if city_idx in city_ids:
- city_ids[city_idx] += (score + 0) * w * weight
- # else: # 20241015 注销 区县简称且不在提取的省市下面,不加分,避免提取错误 例:536550843
- # city_ids[city_idx] = (score + 0) * w * weight * 0.1
- elif pro_idx in pro_ids:
- city_ids[city_idx] = (score + 0) * w * weight * 0.1
- for k, v in pro_ids.items():
- pro_ids[k] = v * text_weight
- for k, v in city_ids.items():
- city_ids[k] = v * text_weight
- for k, v in dis_ids.items():
- dis_ids[k] = v * text_weight
- return pro_ids, city_ids, dis_ids
- area_dic = {'area': '全国', 'province': '全国', 'city': '未知', 'district': '未知', "is_in_text": False}
- pro_ids, city_ids, dis_ids = get_pro_city_dis_score(text)
- pro_ids1, city_ids1, dis_ids1 = get_pro_city_dis_score(web_name, text_weight=0.01) # 20240422 修改为站源名称只取前三字,避免类似 459056219 中金岭南阳光采购平台 错提取阳光
- for k in pro_ids1:
- if k in pro_ids:
- pro_ids[k] += pro_ids1[k]
- else:
- pro_ids[k] = pro_ids1[k]
- for k in city_ids1:
- if k in city_ids:
- city_ids[k] += city_ids1[k]
- else:
- city_ids[k] = city_ids1[k]
- for k in dis_ids1:
- if k in dis_ids:
- dis_ids[k] += dis_ids1[k]
- else:
- dis_ids[k] = dis_ids1[k]
- big_area, pred_pro, pred_city, pred_dis = get_final_addr(pro_ids, city_ids, dis_ids)
- if big_area != "":
- area_dic['area'] = big_area
- if pred_pro != "":
- area_dic['province'] = pred_pro
- if pred_city != "":
- area_dic['city'] = pred_city
- if pred_dis != "":
- area_dic['district'] = pred_dis
- if in_content:
- area_dic['is_in_text'] = True
- return {'district': area_dic}
- def predict(self, project_name, prem, title, list_articles, web_source_name = "", list_entitys=""):
- '''
- 先匹配 project_name+tenderee+tenderee_address, 如果缺少省或市 再匹配 title+content
- :param project_name:
- :param prem:
- :param title:
- :param list_articles:
- :param web_source_name:
- :return:
- '''
- def get_ree_addr(prem):
- tenderee = ""
- tenderee_address = ""
- try:
- for v in prem[0]['prem'].values():
- for link in v['roleList']:
- if link['role_name'] == 'tenderee' and tenderee == "":
- tenderee = link['role_text']
- tenderee_address = link['address']
- except Exception as e:
- print('解析prem 获取招标人、及地址出错')
- return tenderee, tenderee_address
- def get_role_address(text):
- '''正则匹配获取招标人地址
- 3:地址直接在招标人后面 招标人:xxx,地址:xxx
- 4:招标、代理一起,两个地址一起 招标人:xxx, 代理人:xxx, 地址:xxx, 地址:xxx.
- '''
- p3 = '(招标|采购|甲)(人|方|单位)(信息:|(甲方))?(名称)?:[\w()]{4,15},(联系)?地址:(?P<addr>(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+|\w{2,15}[,。])'
- p4 = '(招标|采购|甲)(人|方|单位)(信息:|(甲方))?(名称)?:[\w()]{4,15},(招标|采购)?代理(人|机构)(名称)?:[\w()]{4,15},(联系)?地址:(?P<addr>(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+|\w{2,15}[,。])'
- p5 = '(采购|招标)(人|单位)(联系)?地址:(?P<addr>(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+|\w{2,15}[,。])'
- if re.search(p3, text):
- return re.search(p3, text).group('addr')
- elif re.search(p4, text):
- return re.search(p4, text).group('addr')
- elif re.search(p5, text):
- return re.search(p5, text).group('addr')
- else:
- return ''
- def get_project_addr(text):
- p1 = '(项目|施工|实施|建设|工程|服务|交货|送货|收货|展示|看样|拍卖)(地址|地点|位置|所在地区?)(位于)?:(?P<addr>(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+([\w()]{,20}[,。])?|\w{2,15}[,。])'
- p2 = '项目位于(?P<addr>\w{2}市\w{2,4}区)'
- if re.search(p1, text):
- return re.search(p1, text).group('addr')
- elif re.search(p2, text):
- return re.search(p2, text).group('addr')
- else:
- return ''
- def get_bid_addr(text):
- p2 = '(磋商|谈判|开标|投标|评标|报名|递交|评审|发售|所属)(地址|地点|所在地区?|地域):(?P<addr>(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+|\w{2,15}[,。])'
- if re.search(p2, text):
- return re.search(p2, text).group('addr')
- else:
- return ''
- def get_all_addr(list_entitys):
- tenderee_l = []
- addr_l = []
- for ent in list_entitys[0]:
- if ent.entity_type == 'location' and len(ent.entity_text) > 2:
- addr_l.append(ent.entity_text)
- elif ent.entity_type in ['org', 'company']:
- if ent.label in [0, 1]: # 加招标或代理
- tenderee_l.append(ent.entity_text)
- return ' '.join(addr_l), ' '.join(tenderee_l)
- def get_title_addr(text):
- p1 = '(?P<addr>(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+|\w{2,15}[,。])'
- if re.search(p1, text):
- return re.search(p1, text).group('addr')
- else:
- return ''
- if '##attachment##' in list_articles[0].content:
- content, attachment = list_articles[0].content.split('##attachment##')
- if len(content) < 200:
- content += attachment
- else:
- content = list_articles[0].content
- tenderee, tenderee_address = get_ree_addr(prem)
- msc = ""
- pro_addr = get_project_addr(content)
- if pro_addr != "" and re.search('(采购人|招标人)?指定地点', pro_addr)==None: # 排除错误项目地址 例:554024168 1.5服务地点:采购人指定地点。
- msc += '使用规则提取的项目地址;'
- tenderee_address = pro_addr
- else:
- role_addr = get_role_address(content)
- if role_addr != "" and re.search('(采购人|招标人)?指定地点', role_addr)==None:
- msc += '使用规则提取的联系人地址;'
- tenderee_address = role_addr
- if tenderee_address == "":
- title_addr = get_title_addr(title)
- if title_addr != "":
- msc += '使用规则提取的标题地址;'
- tenderee_address = title_addr
- else:
- bid_addr = get_bid_addr(content)
- if bid_addr != "":
- msc += '使用规则提取的开标地址;'
- tenderee_address = bid_addr
- project_name = str(project_name)
- tenderee = str(tenderee)
- # print('招标人地址',role_addr, tenderee_address)
- project_name = project_name + title if project_name not in title else title
- # project_name = project_name.replace(tenderee, '')
- if len(project_name)>3:
- entity_list = getNers([project_name],useselffool=False) # 2024/4/26 修改为去重项目名称中所有公司名称
- for tup in entity_list[0]:
- if tup[2] in ['org', 'company']:
- project_name = project_name.replace(tup[3], '')
- text1 = "{0} {1} {2}".format(tenderee, tenderee_address, project_name)
- web_source_name = str(web_source_name) # 修复某些不是字符串类型造成报错
- text1 = re.sub('复合肥|铁路|公路|新会计', ' ', text1) # 预防提取错 合肥 路南 新会 等地区
- if pro_addr and re.search('\w{2,}([市县旗盟]|自治[区州县旗])', pro_addr):
- if re.search('[市县旗盟]', pro_addr)==None: # 修复 486623506 项目地址不完整
- pro_addr = text1 + ' '+ pro_addr
- msc += '## 使用项目地址输入:%s ##;' % pro_addr
- rs = self.get_area(pro_addr, '')
- msc += '预测结果:省份:%s, 城市:%s,区县:%s;' % (
- rs['district']['province'], rs['district']['city'], rs['district']['district'])
- if rs['district']['province'] != '全国' and rs['district']['city'] != '未知':
- # print('地区匹配:', msc)
- return rs
- # print('text1:', text1)
- msc += '## 第一次预测输入:%s ##;' % text1
- rs = self.get_area(text1, '') # 2024/4/22 调整第一次输入不带站源名称,避免出错
- msc += '预测结果:省份:%s, 城市:%s,区县:%s;' % (
- rs['district']['province'], rs['district']['city'], rs['district']['district'])
- # self.f.write('%s %s \n' % (list_articles[0].id, msc))
- # print('地区匹配:', msc)
- if rs['district']['province'] == '全国' or rs['district']['city'] == '未知':
- # msc = ""
- all_addr, tenderees = get_all_addr(list_entitys)
- text2 = tenderees + " " + all_addr + ' ' + title
- msc += '使用实体列表所有招标人+所有地址;'
- # text2 += title + content if len(content)<2000 else title + content[:1000] + content[-1000:]
- text2 = re.sub('复合肥|铁路|公路|新会计', ' ', text2)
- # print('text2:', text2)
- msc += '## 第二次预测输入:%s %s##' % (text2,web_source_name)
- rs2 = self.get_area(text2, web_source_name, in_content=True)
- # rs2['district']['is_in_text'] = True
- if rs['district']['province'] == '全国' and rs2['district']['province'] != '全国':
- rs = rs2
- elif rs['district']['province'] == rs2['district']['province'] and rs2['district']['city'] != '未知':
- rs = rs2
- msc += '预测结果:省份:%s, 城市:%s,区县:%s' % (
- rs['district']['province'], rs['district']['city'], rs['district']['district'])
- # self.f.write('%s %s \n'%(list_articles[0].id, msc))
- # print('地区匹配:', msc)
- return rs
|