# -*- coding: utf-8 -*- """``DistrictPredictor`` — 地区匹配规则预测。 Phase 5 从 ``interface/predictor.py``(约 6770-8032 行)迁出。 原 ``from common.Utils import *`` / ``from common.nerUtils import *`` 已替换为显式 import;``os.path.dirname(__file__)`` 路径引用替换为 ``predictors._common.INTERFACE_DIR``。使用 district_tuple.pkl、 area_variance_dic.pkl 等地区字典。 """ from __future__ import absolute_import import os import re import pickle from collections import Counter from BiddingKG.dl.common.logging import log from BiddingKG.dl.common.nerUtils import getTokens, getNers from BiddingKG.dl.entityLink.entityLink import get_business_data from BiddingKG.dl.predictors._common import INTERFACE_DIR __all__ = ["DistrictPredictor"] class DistrictPredictor(): def __init__(self): # with open(os.path.dirname(__file__)+'/district_dic.pkl', 'rb') as f: # dist_dic = pickle.load(f) # short_name = '|'.join(sorted(set([v['简称'] for v in dist_dic.values()]), key=lambda x: len(x), reverse=True)) # full_name = '|'.join(sorted(set([v['全称'] for v in dist_dic.values()]), key=lambda x: len(x), reverse=True)) # short2id = {} # full2id = {} # for k, v in dist_dic.items(): # if v['简称'] not in short2id: # short2id[v['简称']] = [k] # else: # short2id[v['简称']].append(k) # if v['全称'] not in full2id: # full2id[v['全称']] = [k] # else: # full2id[v['全称']].append(k) # self.dist_dic = dist_dic # self.short_name = short_name # self.full_name = full_name # self.short2id = short2id # self.full2id = full2id # # self.f = open(os.path.dirname(__file__)+'/../test/data/district_predict.txt', 'w', encoding='utf-8') with open(os.path.join(INTERFACE_DIR, 'district_tuple.pkl'), 'rb') as f: district_tuple = pickle.load(f) self.p_pro, self.p_city, self.p_dis, self.idx_dic, self.full_dic, self.short_dic = district_tuple # self.pettern = "((?P%s)(?P%s)?(?P%s)?)|((?P%s)(?P%s)?)|(?P%s)" % ( # self.p_pro, self.p_city, self.p_dis, self.p_city, self.p_dis, self.p_dis) short_pro = '黑龙江|内蒙古|青海|陕西|辽宁|贵州|西藏|福建|甘肃|湖南|湖北|海南|浙江|河南|河北|江西|江苏|新疆|广西|广东|山西|山东|安徽|宁夏|四川|吉林|云南' self.pettern = "(?P%s)##(?P%s)##(?P%s)" % ( self.p_pro, self.p_city, self.p_dis) self.pettern_pro = re.compile("(?P%s)"%self.p_pro) self.pettern_city = re.compile("(%s)?(?P%s)"%(short_pro, self.p_city)) # 20250925补充省简称,解决 海南昌江 匹配为南昌问题 self.pettern_dist = re.compile("(%s)?(?P%s)"%(short_pro, self.p_dis)) with open(os.path.join(INTERFACE_DIR, "area_variance_dic.pkl"), 'rb') as f: # 20241113 地区变更新旧名称对照字典 self.area_variance_dic = pickle.load(f) self.multi_dist = ['向阳区', '宝山区', '南沙区', '和平区', '新城区', '鼓楼区', '南山区', '白云区', '朝阳区', '江北区', '城关区', '永定区', '普陀区', '长安区', '市中区', '西安区', '通州区', '西湖区', '龙华区', '城中区', '河东区', '桥西区', '青山区', '新华区', '铁西区', '铁东区', '海州区', '滨海新区'] def find_whole_areas(self, text, pettern, area_variance_dic, full_dic, weight=1): ''' 通过正则匹配字符串返回地址 :param pettern: 地址正则 广东省|广西省|... :param text: 待匹配文本 :return: ''' province_l, city_l, district_l = [], [], [] citys = [] text = str(text).replace('(', '(').replace(')', ')') text = re.sub('\d{2,4}年度?|[\d/-]{1,5}[月日]|\d+|[a-zA-Z0-9]+', ' ', text) text = re.sub( '复合肥|海南岛|兴业银行|双河口|阳光|杭州湾|新城区|中粮屯河|老城(区|改造|更新|升级|翻新)|沙县小吃|北京时间|福田汽车|中山(大学|公园|纪念堂)|孙中山|海天水泥|阳光采购|示范县|珠江城?|西九龙站|广州路北|安阳山村|电信|联通|北京现代|祁连山|锡铁山|大黄山(?!市)|红旗汽车', # 570445994 广州路北侧 预测为 广州 路北 ' ', text) # 544151395 赤壁市老城区燃气管道老化更新改造 text = re.sub('珠海城市', '珠海', text) # 修复 426624023 珠海城市 预测为海城市 text = re.sub('怒江州', '怒江傈僳族自治州', text) # 修复 423589589 所属地域:怒江州 识别为广西 - 崇左 - 江州 text = re.sub('茂名滨海新区', '茂名市', text) text = re.sub('中山([东南西][部区环]|黄圃|南头|东凤|小榄|石岐|翠亨|南朗)', '中山市', text) text = re.sub('横州市', '横县', text) # 例:547363890 修复广西南宁横州 不在地区表问题 text = re.sub('广东中山', '广东中山市', text) text = re.sub('朝阳柳城经济开发区', '朝阳市', text) text = re.sub('安徽徽运城市', '安徽', text) # 653054198 安徽徽运城市运营管理有限公司 text = re.sub('西安丰镇', '宝应', text) # 修复 681946122 宝应县西安丰镇 text = re.sub('西城区', '西城', text) # 修复 653130377 标题 董家口电厂至新区西城区长输热力管线工程目 预测错北京西城 # 处理地区+大学不在该地区问题 text = re.sub('(河北工业大学)', '天津市', text) text = re.sub('(西藏民族大学)', '咸阳市', text) text = re.sub('(四川外国语大学|四川美术学院)', '重庆市', text) text = re.sub('(中山大学)', '广州市', text) text = re.sub('(滨州医学院)', '烟台市', text) ser = re.search('海南(昌江|白沙|乐东|陵水|保亭|琼中)(黎族)?', text) if ser and '黎族' not in ser.group(0): text = text.replace(ser.group(0), ser.group(0) + '黎族') for k, v in area_variance_dic.items(): # 20241113 根据地区变更信息替换文本 text = text.replace(k, v) text = re.sub('\s+', ' ', text) if re.search('[\u4e00-\u9fa5]', text) == None: return province_l, city_l, district_l name_set = set() # 提取到的所有地址集合 tokens = getTokens([text], useselffool=True)[0] # print('句子:', text) # print('分词:', tokens) for pettern in [self.pettern_pro, self.pettern_city, self.pettern_dist]: # pettern.split('##') for it in re.finditer(pettern, text): if it.group(0) == '站前': # 20240314 修复类似 中铁二局新建沪苏湖铁路工程站前VI标项目 错识别为 省份:辽宁, 城市:营口,区县:站前 continue for k, v in it.groupdict().items(): if v != None: if it.end() == it.end(k) and re.search('[省市区县州旗盟]$', v) == None and re.search( '^([东南西北中一二三四五六七八九十大小]?(村|镇|街|路|道|社区|巷|坊)|酒店|宾馆|经济开发区|开发区|新区|公园|广场|公馆|小区|幼儿园)', # |医院|[大中小]学 # 20250917取消地区+医院过滤,大部分是在该地区 # 城市不匹配为区的地址 修复 滨州北海经济开发区 北海新区 等提取为北海 text[it.end(k):]) != None and not re.match('路桥', text[it.end(k):]): # 修复 746181930 甘肃路桥 被路去掉 continue if k in ['prov']: if v in full_dic['province']: score = 2 else: score = 1 if re.search('^(\w{,2}[分支](公司|局|行|院|干?线)|(机务|车辆|车务|工务|电务|供电|动车)?段|地铁|(火车|高铁)?站|港|地区|区域|基地)' , text[it.end(k):]) or re.search('^((%s)|\-%s)' % (v, v), text[max(0, it.start(k) - 1):]): score += 1 elif re.search('大学|学院', text[:it.start(k)]) and re.search('分校|校区', text[it.end(k):]): # 长春市第八十七中学南阳校区 不在南阳市 score += 1 if len(v) < 3 and v not in tokens: # 不在分词里面概率降低 score /= 2 else: score += it.end(k) / len(text) / 10 province_l.append((v, score * weight)) elif k in ['city', 'city1']: if v in full_dic['city']: score = 2 else: score = 1 if re.search('^(\w{,2}[分支](公司|局|行|院|干?线)|(机务|车辆|车务|工务|电务|供电|动车)?段|地铁|(火车|高铁)?站|港|地区|区域|基地)' , text[it.end(k):]) or re.search('^((%s)|\-%s)' % (v, v), text[max(0, it.start(k) - 1):]): score += 1 elif re.search('大学|学院', text[:it.start(k)]) and re.search('分校|校区', text[it.end(k):]): score += 1 if len(v) < 3 and v not in tokens: # 不在分词里面概率降低 优化 653058471 山东恒通化工股份有限公司 错分 通化 score /= 2 else: score += it.end(k) / len(text) / 10 # 优化 572840045 上海铁路公安局合肥公安处 这种表达 city_l.append((v, score * weight)) citys.append(v) elif k in ['dist', 'dist1', 'dist2']: if v in ['东区', '西区', '城区', '郊区', '矿区', '东至']: continue elif v.endswith('城市') and text[it.start(k)-1:it.start(k)+1] in citys: # 修复 上海城市 珠海城市 等 预测为 海城市 continue if v in self.multi_dist or re.search('\w城区$', v): # 多个城市有的区概率降低 score = 0.5 elif v in full_dic['district'] and (len(v) > 2 or v.endswith('县')): # 20250709 修复 萧县 等概率过低 score = 2 else: score = 0.5 if re.search('^(\w{,2}[分支](公司|局|行|院|干?线)|(机务|车辆|车务|工务|电务|供电|动车)?段|地铁|(火车|高铁)?站|港|地区|区域|基地)' , text[it.end(k):]) or ( re.match('\s*%s' % v, text) and it.start(k) < 2) or re.search( '^((%s)|\-%s)' % (v, v), text[max(0, it.start(k) - 1):]): score += 0.5 elif re.search('大学|学院', text[:it.start(k)]) and re.search('分校|校区', text[it.end(k):]): score += 0.5 if len(v) < 3 and v not in tokens: # 不在分词里面概率降低,三字以上不受限制 ,避免类似 德令哈工务段 分词不对 score /= 2 # score += it.end(k) / len(text) / 10 district_l.append((v, score * weight)) name_set.add(v) if len(name_set) > 1: # 解决 类似 山西宁武 匹配出 山西 西宁 宁武 问题 修复 653057648 把峨眉山市作为眉山市 names = re.findall('|'.join(sorted(name_set, key=lambda x: len(x), reverse=True)), text) province_l = [it for it in province_l if it[0] in names] city_l = [it for it in city_l if it[0] in names] district_l = [it for it in district_l if it[0] in names] return province_l, city_l, district_l def merge_score(self, province_l, city_l, district_l, full_dic, short_dic, idx_dic, filter_short_dist=True): ''' 合并分数,下级地区分数加到上级 :param province_l: 提取到的省份列表 [(name, score)] :param city_l: 提取到的城市列表 [(name, score)] :param district_l: 提取到的区县列表 [(name, score)] :param filter_short_dist: 是否过滤不在省份下的区县简称权重 :return: ''' pro_ids = dict() city_ids = dict() dis_ids = dict() for pro in province_l: name, score = pro idx = full_dic['province'][name] if name in full_dic['province'] else short_dic['province'][name] if idx not in pro_ids: pro_ids[idx] = 0 pro_ids[idx] += score tmp_pro = {} for city in city_l: name, score = city if name in full_dic['city']: for idx in full_dic['city'][name]: if idx not in city_ids: city_ids[idx] = 0 city_ids[idx] += score pro_idx = idx_dic[idx]['省'] if pro_idx in tmp_pro: tmp_pro[pro_idx] += score else: tmp_pro[pro_idx] = score elif name in short_dic['city']: for idx in short_dic['city'][name]: if idx not in city_ids: city_ids[idx] = 0 city_ids[idx] += score pro_idx = idx_dic[idx]['省'] if pro_ids != {} and pro_idx not in pro_ids: # 如果省份不为空且简称不在省份分值降低 优化 653150317 海南农垦阳江农场有限公司 错分 阳江 score -= 0.1 if pro_idx in tmp_pro: tmp_pro[pro_idx] += score else: tmp_pro[pro_idx] = score if set(tmp_pro) & set(pro_ids) != set(): for k, v in tmp_pro.items(): if k in pro_ids: pro_ids[k] += v else: pro_ids[k] = v else: pro_ids.update(tmp_pro) tmp_pro = {} tmp_city = {} for dis in district_l: name, score = dis if name in full_dic['district']: for idx in full_dic['district'][name]: if idx not in dis_ids: dis_ids[idx] = 0 dis_ids[idx] += score pro_idx = idx_dic[idx]['省'] if pro_idx in tmp_pro: tmp_pro[pro_idx] += score else: if name in self.multi_dist: # 多个城市重复名称,需过滤 continue tmp_pro[pro_idx] = score city_idx = idx_dic[idx]['市'] if city_idx in tmp_city: tmp_city[city_idx] += score else: if name in self.multi_dist: # 多个城市重复名称,需过滤 continue tmp_city[city_idx] = score elif name in short_dic['district']: for idx in short_dic['district'][name]: if idx not in dis_ids: dis_ids[idx] = 0 dis_ids[idx] += score pro_idx = idx_dic[idx]['省'] if pro_ids != {} and pro_idx not in pro_ids: # 如果省份不为空且简称不在省份分值降低 score -= 0.1 if filter_short_dist and score < 1: # pro_idx not in pro_ids continue if pro_idx in tmp_pro: tmp_pro[pro_idx] += score else: tmp_pro[pro_idx] = score city_idx = idx_dic[idx]['市'] if city_idx in tmp_city: tmp_city[city_idx] += score else: tmp_city[city_idx] = score if set(tmp_pro) & set(pro_ids) != set(): for k, v in tmp_pro.items(): if k in pro_ids: pro_ids[k] += v else: pro_ids.update(tmp_pro) if set(tmp_city) & set(city_ids) != set(): for k, v in tmp_city.items(): if k in city_ids: city_ids[k] += v else: city_ids.update(tmp_city) return pro_ids, city_ids, dis_ids @staticmethod def get_final_addr(pro_ids, city_ids, dis_ids, idx_dic): ''' 先把所有匹配的全称、简称转为id,如果省份不为空,城市不为空且有城市属于省份的取该城市 :param province_l: 匹配到的所有省份 :param city_l: 匹配到的所有城市 :param district_l: 匹配到的所有区县 :return: ''' big_area = "" pred_pro = "" pred_city = "" pred_dis = "" final_pro = "" final_city = "" prob = 0 max_score = 0 code_dic = { 'province_code': '', 'city_code': '', 'district_code': '' } if len(pro_ids) >= 1: pro_l = sorted([(k, v) for k, v in pro_ids.items()], key=lambda x: x[1], reverse=True) scores = [it[1] for it in pro_l] prob = max(scores) / sum(scores) max_score = max(scores) final_pro, score = pro_l[0] if score >= 0.01: pred_pro = idx_dic[final_pro]['返回名称'] big_area = idx_dic[final_pro]['大区'] code_dic['province_code'] = idx_dic[final_pro]['编码'] if pred_pro != "" and len(city_ids) >= 1: city_l = sorted([(k, v) for k, v in city_ids.items()], key=lambda x: x[1], reverse=True) for it in city_l: if idx_dic[it[0]]['省'] == final_pro: final_city = it[0] pred_city = idx_dic[final_city]['返回名称'] code_dic['city_code'] = idx_dic[final_city]['编码'] break if final_city != "" and len(set(dis_ids)) >= 1: dis_l = sorted([(k, v) for k, v in dis_ids.items()], key=lambda x: x[1], reverse=True) for it in dis_l: if idx_dic[it[0]]['市'] == final_city: pred_dis = idx_dic[it[0]]['返回名称'] code_dic['district_code'] = idx_dic[it[0]]['编码'] elif pred_pro != "" and pred_city == "" and len(set(dis_ids)) >= 1: # 20241111 省份不为空,市为空,如果区县在省份下,补充对应的市县 dis_l = sorted([(k, v) for k, v in dis_ids.items()], key=lambda x: x[1], reverse=True) for it in dis_l: if idx_dic[it[0]]['省'] == final_pro: pred_city = idx_dic[idx_dic[it[0]]['市']]['返回名称'] pred_dis = idx_dic[it[0]]['返回名称'] code_dic['city_code'] = idx_dic[idx_dic[it[0]]['市']]['编码'] code_dic['district_code'] = idx_dic[it[0]]['编码'] return big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic @staticmethod def get_ree_addr(prem): tenderee = "" tenderee_address = "" try: for v in prem.values(): for link in v['roleList']: if link['role_name'] == 'tenderee' and tenderee == "": tenderee = link['role_text'] tenderee_address = link['address'] except Exception as e: print('解析prem 获取招标人、及地址出错') return tenderee, tenderee_address @staticmethod def get_role_address(text): '''正则匹配获取招标人地址 3:地址直接在招标人后面 招标人:xxx,地址:xxx 4:招标、代理一起,两个地址一起 招标人:xxx, 代理人:xxx, 地址:xxx, 地址:xxx. ''' p3 = '(招标|采购|甲)(人|方|单位)(信息:|(甲方))?(名称)?:[\w()]{4,15},(联系)?地址:(?P(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+|\w{2,15}[,。])' p4 = '(招标|采购|甲)(人|方|单位)(信息:|(甲方))?(名称)?:[\w()]{4,15},(招标|采购)?代理(人|机构)(名称)?:[\w()]{4,15},(联系)?地址:(?P(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+|\w{2,35}),(联系)?地址:' p5 = '(采购|招标)(人|单位)(联系)?地址:(?P(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+|\w{2,15}[,。])' if re.search(p3, text): return re.search(p3, text).group('addr') elif re.search(p4, text): return re.search(p4, text).group('addr') elif re.search(p5, text): return re.search(p5, text).group('addr') else: return '' @staticmethod def get_all_addr(list_entity, filter_attach=False): ''' 获取所有招标或代理人名称及所有地址 :param list_entity: 实体列表 :param filter_attach: 是否过滤附件实体 :return: ''' tenderee_l = [] addr_l = [] for ent in list_entity: if ent.entity_type not in ['org', 'company', 'location'] or (filter_attach and ent.in_attachment): continue if ent.entity_type == 'location' and len(ent.entity_text) > 2: addr_l.append(ent.entity_text) elif ent.entity_type in ['org', 'company']: if ent.label in [0, 1]: # 加招标或代理 tenderee_l.append(ent.entity_text) elif re.search('[局委]$', ent.entity_text): tenderee_l.append(ent.entity_text) if len(addr_l) > 10: # 只取前10个地址 break return ' '.join(set(addr_l)), ' '.join(set(tenderee_l)) def addr_process(self, addr_text): addr_text = addr_text.replace('(', '(').replace(')', ')') if re.search('[省市县]', addr_text) == None: ser = re.search('\w{2,}区', addr_text) ser2 = re.match('\w{2,5}镇', addr_text) if ser: addr_text = addr_text[:ser.end()] elif ser2: addr_text = addr_text[:ser2.end()] return addr_text def predict_area(self,docid, title, content, web_source_name, prem={}, addr_dic={}, list_entity=[]): if re.match('([^,]{3,50}网络竞价会),', content): # 修复 701397187 标题只有拍卖的东西,第一句有地址 title_auction = re.match('([^,]{3,50}网络竞价会),', content).group(1) if title.find(title_auction[:3]) == -1: title += ' ' + title_auction filter_attach = False # 是否过滤附件内容 if '##attachment##' in content: main, att = content.split('##attachment##') if 2000 < len(main) < len(att): # 正文超过500字且附件比正文长过滤附件内容 filter_attach = True # print('正文超过500字过滤附件内容') ree, addr_ree = self.get_ree_addr(prem) addr_ree = self.addr_process(addr_ree) addr_bus = '' if len(addr_ree) < 3 and ree != '': have_bus, bus_dic = get_business_data(ree) if have_bus: addr_bus = '%s %s %s' % (bus_dic.get('province', ''), bus_dic.get('city', ''), bus_dic.get('district', '')) all_addr, tenderees = self.get_all_addr(list_entity, filter_attach) return self.predict_distrist(docid, title, web_source_name, ree, addr_ree, addr_bus, addr_dic, tenderees, all_addr) def predict_distrist(self,docid, title, web_source_name, ree, addr_ree, addr_bus,location_dic={}, tenderees='', all_addr=''): ''' 优先顺序项目地址、收货地址、招标人地址 / 工商地址、开标地址 / 联系地址、站源名称 ''' area_dic = {'area': '全国', 'province': '全国', 'city': '未知', 'district': '未知', "is_in_text": False} in_content = False not_sure = True # 是否不确定地区 msc = "" # 日志信息 addr_dic = {} prov_name_l = [] # 保存预测到的省份名称 city_name_l = [] # 保存预测到的城市名称 final_key = '' company_title = '' title_raw = title if len(title.strip()) > 4: ner_title = getNers([title], True)[0] company_title = [] for ner in ner_title: if ner[2] in ['org', 'company']: company_title.append(ner[3]) title = title.replace(ner[3], '#') company_title = "#".join(company_title) if re.search('拍卖|法院', ree): company_title += ' ' + ree ree = '' if web_source_name in ('中原云商', '中原云商电子招投标平台'): # 修复某些公告没地区 web_source_name = '河南中原云商' for key in ['addr_project', 'addr_delivery', 'addr_bidopen', 'addr_bidsend', 'addr_contact']: addr = location_dic.get(key, '') addr = self.addr_process(addr) if len(addr) < 2: continue province_l, city_l, district_l = self.find_whole_areas('%s'%addr, self.pettern, self.area_variance_dic, self.full_dic) if len(province_l+city_l+district_l)>0: pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic,self.short_dic, self.idx_dic) big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = self.get_final_addr(pro_ids, city_ids, dis_ids, self.idx_dic) if max_score < 2 and len(addr) > len(''.join([it[0] for it in province_l+city_l+district_l]))*2: # 修复 712151540 北京银行大厦6楼厨房及22楼 提取为北京 log('地区匹配非正常地址:%s, 预测为:%s %s %s, docid:%s'%(addr, pred_pro, pred_city, pred_dis, docid)) continue if pred_pro != '': addr_dic[key] = { 'keyword': (province_l, city_l, district_l), 'result': (big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic) } msc += "使用%s信息:%s, 预测为:%s %s %s;" % (key, addr, pred_pro, pred_city, pred_dis) prov_name_l.append(pred_pro) if pred_city != '': city_name_l.append(pred_city) if key == 'addr_project' and addr_dic['addr_project']['result'][2] != '' and addr_dic['addr_project']['result'][4] > 0.6 and (addr_dic['addr_project']['result'][5] >= 2 or addr_dic['addr_project']['keyword'][2]==[]): big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic['addr_project']['result'] not_sure = False final_key = key msc += "最终使用:%s,预测为:%s %s %s;"%(final_key, pred_pro, pred_city, pred_dis) if addr.endswith('#在附件'): log('地区匹配使用附件中的项目地址:%s;预测为:%s %s;docid:%s'%(addr, pred_pro, pred_city, docid)) break if not_sure: for key, text in zip(['tenderee', 'company_title', 'web_source_name'], [ree, company_title, web_source_name]): if len(text) < 4: continue weight = 0.48 if key == 'web_source_name' else 1 province_l, city_l, district_l = self.find_whole_areas('%s'%text, self.pettern, self.area_variance_dic, self.full_dic, weight=weight) if len(province_l+city_l+district_l)>0: pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic,self.short_dic, self.idx_dic) big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = self.get_final_addr(pro_ids, city_ids, dis_ids, self.idx_dic) if pred_pro != '': addr_dic[key] = { 'keyword': (province_l, city_l, district_l), 'result': (big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic) } msc += "使用%s信息:%s, 预测为:%s %s %s;" % (key, text, pred_pro, pred_city, pred_dis) prov_name_l.append(pred_pro) if pred_city != '': city_name_l.append(pred_city) elif key == 'company_title' and re.search('\w{1,}[省市县]', text): # 修复 2025年可克达拉市政府部门独立办公楼聘用保安保洁采购服务 实体提取为 克达拉市政府 造成市漏提 title = title_raw # 提取招标地址 if len(addr_ree) >= 3: province_l, city_l, district_l = self.find_whole_areas('%s' % addr_ree, self.pettern, self.area_variance_dic,self.full_dic) if len(province_l + city_l + district_l) > 0: pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic,self.short_dic, self.idx_dic) big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = self.get_final_addr(pro_ids,city_ids,dis_ids,self.idx_dic) if pred_pro != '': addr_dic['addr_tenderee'] = { 'keyword': (province_l, city_l, district_l), 'result': (big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic) } msc += "使用招标人地址:%s, 预测为:%s %s %s;" % (addr_ree, pred_pro, pred_city, pred_dis) prov_name_l.append(pred_pro) if pred_city != '': city_name_l.append(pred_city) # 提取招标人工商登记地址 if len(addr_bus) >= 3: province_l, city_l, district_l = self.find_whole_areas('%s' % addr_bus, self.pettern, self.area_variance_dic,self.full_dic) if len(province_l + city_l + district_l) > 0: pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic,self.short_dic, self.idx_dic) big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = self.get_final_addr(pro_ids,city_ids,dis_ids,self.idx_dic) if pred_pro != '': addr_dic['addr_bus'] = { 'keyword': (province_l, city_l, district_l), 'result': (big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic) } msc += "使用招标人工商登记地址:%s, 预测为:%s %s %s;" % (addr_bus, pred_pro, pred_city, pred_dis) prov_name_l.append(pred_pro) if pred_city != '': city_name_l.append(pred_city) # 提取标题地址 if len(title) > 3: province_l, city_l, district_l = self.find_whole_areas('%s' % title, self.pettern, self.area_variance_dic,self.full_dic) if len(province_l + city_l + district_l) > 0: pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic,self.short_dic, self.idx_dic) big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = self.get_final_addr(pro_ids,city_ids,dis_ids,self.idx_dic) if pred_pro != '': addr_dic['addr_title'] = { 'keyword': (province_l, city_l, district_l), 'result': (big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic) } msc += "使用标题地址:%s, 预测为:%s %s %s;" % (title, pred_pro, pred_city, pred_dis) prov_name_l.append(pred_pro) if pred_city != '': city_name_l.append(pred_city) if len(addr_dic) == 1: for key in addr_dic: big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic[key]['result'] final_key = key msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis) elif len(addr_dic) > 1: prov_count = Counter(prov_name_l) city_count = Counter(city_name_l) if len(prov_count) == 1 and len(city_count) == 1: score_max = 0 for key in ['addr_project', 'addr_delivery', 'addr_tenderee', 'addr_title', 'tenderee', 'company_title', 'addr_bidopen', 'addr_bidsend', 'addr_contact', 'addr_bus', 'web_source_name']: if key in addr_dic and addr_dic[key]['result'][1] == prov_name_l[0] and addr_dic[key]['result'][2] == city_name_l[0] and addr_dic[key]['result'][5] > score_max: score_max = addr_dic[key]['result'][5] final_key = key if addr_dic[final_key]['result'][3] != '': # 2026/2/2 省市匹配,且有区县停止循环 break big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic[final_key]['result'] msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis) # if key in addr_dic and addr_dic[key]['result'][1] == prov_name_l[0] and addr_dic[key]['result'][2] == city_name_l[0]: # # big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic[key]['result'] # final_key = key # msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis) # break elif 'addr_project' in addr_dic and addr_dic['addr_project']['result'][2] == '': # 只有省份 big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic['addr_project']['result'] final_key = 'addr_project' score_max = 0 tmp_key = '' for key in ['addr_delivery', 'addr_tenderee', 'addr_title', 'tenderee', 'company_title', 'addr_bidopen', 'addr_bidsend', 'addr_contact', 'addr_bus', 'web_source_name']: if key in addr_dic and addr_dic[key]['result'][1] == pred_pro and addr_dic[key]['result'][2] != '' and addr_dic[key]['result'][5] > score_max: score_max = addr_dic[key]['result'][5] tmp_key = key if addr_dic[tmp_key]['result'][3] != '': # 2026/2/2 省市匹配,且有区县停止循环 break if tmp_key != '': final_key += ',%s' % tmp_key big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic[tmp_key]['result'] msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis) elif 'addr_delivery' in addr_dic and addr_dic['addr_delivery']['result'][2] != '' and addr_dic['addr_delivery']['result'][4] > 0.7 and addr_dic['addr_delivery']['result'][5] >= 2: big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic['addr_delivery']['result'] final_key = 'addr_delivery' msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis) elif 'addr_title' in addr_dic and addr_dic['addr_title']['result'][2] != '' and addr_dic['addr_title']['result'][4] > 0.7 and addr_dic['addr_title']['result'][5] >= 2: big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic['addr_title']['result'] final_key = 'addr_title' msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis) elif 'tenderee' in addr_dic and addr_dic['tenderee']['result'][2] != '' and addr_dic['tenderee']['result'][4] > 0.7 and addr_dic['tenderee']['result'][5] >= 2: # 补概率 653492303 广州市妇女儿童医疗中心柳州医院 预测错广州 big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic['tenderee']['result'] final_key = 'tenderee' msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis) elif 'addr_tenderee' in addr_dic and addr_dic['addr_tenderee']['result'][2] != '' and addr_dic['addr_tenderee']['result'][5] >= 2: big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = addr_dic['addr_tenderee']['result'] final_key = 'addr_tenderee' msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis) else: province_l, city_l, district_l = [], [], [] for key in addr_dic: if key in ['addr_delivery', 'addr_bidopen', 'addr_bidsend', 'addr_contact', 'addr_bus']: # 地址类型只加一项,因为一般都有三项造成单个地址总分过大 if addr_dic[key]['keyword'][2]: district_l += addr_dic[key]['keyword'][2] elif addr_dic[key]['keyword'][1]: city_l += addr_dic[key]['keyword'][1] else: province_l += addr_dic[key]['keyword'][0] else: province_l += addr_dic[key]['keyword'][0] city_l += addr_dic[key]['keyword'][1] district_l += addr_dic[key]['keyword'][2] if key in ['addr_project', 'addr_title']: # 加重项目、标题地址权重 province_l += addr_dic[key]['keyword'][0] city_l += addr_dic[key]['keyword'][1] district_l += addr_dic[key]['keyword'][2] pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic, self.short_dic, self.idx_dic) big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = self.get_final_addr(pro_ids, city_ids, dis_ids, self.idx_dic) if pred_city != '': score_max = 0 for key in ['addr_delivery', 'addr_tenderee', 'addr_title', 'tenderee', 'company_title', 'addr_bidopen', 'addr_bidsend', 'addr_contact', 'addr_bus', 'web_source_name']: if key in addr_dic and addr_dic[key]['result'][2] == pred_city and addr_dic[key]['result'][5] > score_max: final_key = key score_max = addr_dic[key]['result'][5] if addr_dic[final_key]['result'][3] == pred_dis: # 2026/2/2 省市匹配,且有区县停止循环 break if pred_pro != '' and final_key == "": for key in ['addr_delivery', 'addr_tenderee', 'addr_title', 'tenderee', 'company_title', 'addr_bidopen', 'addr_bidsend', 'addr_contact', 'addr_bus', 'web_source_name']: if key in addr_dic and addr_dic[key]['result'][1] == pred_pro: final_key = key break msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis) else: # 取所有的地址 province_l, city_l, district_l = self.find_whole_areas('%s %s %s' % (title_raw, tenderees, all_addr), self.pettern, self.area_variance_dic,self.full_dic, weight=0.5) pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic, self.short_dic, self.idx_dic) big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = self.get_final_addr(pro_ids, city_ids, dis_ids, self.idx_dic) final_key = 'all_addr' in_content = True msc += "最终使用:%s,预测为:%s %s %s;" % (final_key, pred_pro, pred_city, pred_dis) if len(addr_dic) > 0 and pred_city == '': # 如果非正文提取缺少城市,补充 province_l, city_l, district_l = self.find_whole_areas('%s %s %s' % (title_raw, tenderees, all_addr), self.pettern, self.area_variance_dic,self.full_dic, weight=0.5) pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic, self.short_dic, self.idx_dic) rs_tmp = self.get_final_addr(pro_ids, city_ids, dis_ids, self.idx_dic) if rs_tmp[1] == pred_pro and rs_tmp[4] > 0.5 and rs_tmp[2] != '': big_area, pred_pro, pred_city, pred_dis, prob, max_score, code_dic = rs_tmp msc += "只有省份,所有地址补充市县:%s %s ;"%(pred_city, pred_dis) final_key += ',all_addr' dist_source = { 'addr_project': '项目地址', 'addr_delivery': '收货地址', 'addr_title': '标题', 'tenderee': "招标人", 'company_title': '标题公司', 'addr_tenderee': '招标人地址', 'addr_bus': '招标人工商地址', 'web_source_name': '站源', 'addr_bidopen': '开标地址', 'addr_bidsend': '邮件地址', 'addr_contact': '联系地址', 'all_addr': '文中所有地址' } if big_area != "": area_dic['area'] = big_area if pred_pro != "": area_dic['province'] = pred_pro source_l = [dist_source.get(k, '') for k in final_key.split(',')] area_dic['dist_source'] = ';'.join(source_l) if pred_city != "": area_dic['city'] = pred_city if pred_dis != "": area_dic['district'] = pred_dis for k, v in code_dic.items(): if v != '': area_dic[k] = v if pred_city in ['北京', '天津', '上海', '重庆']: # 直辖市调整县级 area_dic['city'] = area_dic['district'] area_dic['district'] = '未知' if 'district_code' in area_dic: area_dic['city_code'] = area_dic['district_code'] area_dic.pop('district_code') if area_dic['city'] == '未知' and 'city_code' in area_dic: area_dic.pop('city_code') area_dic['is_in_text'] = in_content # area_dic['prob'] = prob # area_dic['max_score'] = max_score # print('最终地址:', pred_pro, pred_city, pred_dis) if prob < 0.6 or max_score < 2: log('地区匹配预测,最终结果:%s %s %s, 预测:%s, docid:%s' % (pred_pro, pred_city, pred_dis, msc, docid)) return {'district': area_dic} def predict_area_backup(self,docid, title, content, web_source_name, prem={}, addr_dic={}, list_entity=[]): area_dic = {'area': '全国', 'province': '全国', 'city': '未知', 'district': '未知', "is_in_text": False} addr_project = addr_dic.get('addr_project', '').replace('(', '(').replace(')', ')') addr_delivery = addr_dic.get('addr_delivery', '').replace('(', '(').replace(')', ')') addr_bidopen = addr_dic.get('addr_bidopen', '').replace('(', '(').replace(')', ')') addr_bidsend = addr_dic.get('addr_bidsend', '').replace('(', '(').replace(')', ')') addr_contact = addr_dic.get('addr_contact', '').replace('(', '(').replace(')', ')') in_content = False not_sure = True # 是否不确定地区 filter_attach = False # 是否过滤附件内容 msc = "" # 日志信息 if '##attachment##' in content: main, att = content.split('##attachment##') if 500 < len(main) < len(att): # 正文超过500字且附件比正文长过滤附件内容 filter_attach = True # print('正文超过500字过滤附件内容') if web_source_name in ('中原云商', '中原云商电子招投标平台'): # 修复某些公告没地区 web_source_name = '河南中原云商' province_l, city_l, district_l = self.find_whole_areas('%s %s'%(title, addr_project), self.pettern, self.area_variance_dic, self.full_dic) pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic, self.short_dic, self.idx_dic) big_area_1, pred_pro_1, pred_city_1, pred_dis_1, prob_1, max_score, code_dic_1 = self.get_final_addr(pro_ids, city_ids, dis_ids, self.idx_dic) big_area, pred_pro, pred_city, pred_dis, prob, code_dic = big_area_1, pred_pro_1, pred_city_1, pred_dis_1, prob_1, code_dic_1 # print('关键词1:', province_l, city_l, district_l) # print('输入:', '标题:%s; 项目地址:%s'%(title, addr_project)) # print('分数:', pro_ids, city_ids, dis_ids, prob, max_score) msc += '第一次预测,标题:%s; 项目地址:%s; 预测为:%s %s ##'%(title, addr_project, pred_pro, pred_city) if pred_city_1 == "" or prob < 0.7 or max_score<2: ree, addr = self.get_ree_addr(prem) have_bus, bus_dic = get_business_data(ree) if ree != '' else False, {} if ree in title: ree = '##' rule_ree_addr = self.get_role_address(content) if rule_ree_addr: addr = rule_ree_addr # addr = content # ree = '' province_l2, city_l2, district_l2 = self.find_whole_areas('%s %s %s' % (ree, addr, addr_delivery), self.pettern, self.area_variance_dic, self.full_dic, weight=1) province_l.extend(province_l2) city_l.extend(city_l2) district_l.extend(district_l2) pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic, self.short_dic, self.idx_dic) big_area_2, pred_pro_2, pred_city_2, pred_dis_2, prob_2, max_score, code_dic_2 = self.get_final_addr(pro_ids, city_ids, dis_ids, self.idx_dic) big_area, pred_pro, pred_city, pred_dis, prob, code_dic = big_area_2, pred_pro_2, pred_city_2, pred_dis_2, prob_2, code_dic_2 # print('关键词2:', province_l, city_l, district_l) # print('输入:', '招标人:%s; 招标人地址:%s; 收货地址:%s' % (ree, addr, addr_delivery)) # print('分数:', pro_ids, city_ids, dis_ids, prob, max_score) msc += '第二次预测,招标人:%s; 招标人地址:%s; 收货地址:%s; 预测为:%s %s ##' % (ree, addr, addr_delivery, pred_pro, pred_city) if re.search('省|市|县|自治', addr_project) and prob_1 !=0.5 and pred_pro_1 != '' and pred_pro_1 != pred_pro_2: # 如果有项目地址使用项目地址 要有省市县等 275127622 工程地点为狮山镇颜峰综合区岐山至人和段道路, 提错 岐山 not_sure = False big_area, pred_pro, pred_city, pred_dis, code_dic = big_area_1, pred_pro_1, pred_city_1, pred_dis_1, code_dic_1 msc += "有项目地址,一二次预测省份不同,改为第一次结果。" if not_sure and (pred_city_2 == "" or prob < 0.7 or max_score<2): addr_bus = '%s %s'%(bus_dic.get('province', ''), bus_dic.get('city', '')) province_l3, city_l3, district_l3 = self.find_whole_areas('%s %s; %s; %s'%(addr_bus, addr_contact, addr_bidopen, addr_bidsend), self.pettern, self.area_variance_dic, self.full_dic, weight=0.6) province_l.extend(province_l3) city_l.extend(city_l3) district_l.extend(district_l3) pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic, self.short_dic, self.idx_dic) big_area_3, pred_pro_3, pred_city_3, pred_dis_3, prob_3, max_score, code_dic_3 = self.get_final_addr(pro_ids, city_ids, dis_ids, self.idx_dic) big_area, pred_pro, pred_city, pred_dis, prob, code_dic = big_area_3, pred_pro_3, pred_city_3, pred_dis_3, prob_3, code_dic_3 # print('关键词3:', province_l, city_l, district_l) # print('输入:', '联系:%s, 开标:%s, 邮寄:%s'%(addr_contact, addr_bidopen, addr_bidsend)) # print('分数:', pro_ids, city_ids, dis_ids, prob, max_score) msc += '第三次预测,工商地址:%s;联系:%s; 开标:%s; 邮寄:%s; 预测为:%s %s ##' % (addr_bus, addr_contact, addr_bidopen, addr_bidsend, pred_pro, pred_city) if pred_city_2 != "" and prob_2 !=0.5 and pred_city_2 != pred_city_3: not_sure = False big_area, pred_pro, pred_city, pred_dis, code_dic = big_area_2, pred_pro_2, pred_city_2, pred_dis_2, code_dic_2 # 如果招标人、招标人地址、收货地址与开标地址、联系地址等不一致,取招标人地址 msc += "二三次预测城市不一致,改为第二次结果。" if not_sure and (pred_city_3 == "" or prob < 0.6 or max_score < 2): all_addr, tenderees = self.get_all_addr(list_entity, filter_attach) province_l4, city_l4, district_l4 = self.find_whole_areas('%s %s %s' % (web_source_name, tenderees, all_addr), self.pettern, self.area_variance_dic, self.full_dic, weight=0.3) province_l.extend(province_l4) city_l.extend(city_l4) district_l.extend(district_l4) pro_ids, city_ids, dis_ids = self.merge_score(province_l, city_l, district_l, self.full_dic, self.short_dic, self.idx_dic) big_area_4, pred_pro_4, pred_city_4, pred_dis_4, prob_4, max_score, code_dic_4 = self.get_final_addr(pro_ids, city_ids,dis_ids, self.idx_dic) big_area, pred_pro, pred_city, pred_dis, prob, code_dic = big_area_4, pred_pro_4, pred_city_4, pred_dis_4, prob_4, code_dic_4 if pred_city_3 != "" and prob_3 !=0.5 and pred_city_3 != pred_city_4: not_sure = False big_area, pred_pro, pred_city, pred_dis, code_dic = big_area_3, pred_pro_3, pred_city_3, pred_dis_3, code_dic_3 # 如果开标地址等提取的城市与所有地址提取的城市不一致,取开标地址等 msc += "三四次预测城市不一致,改为第三次结果。" if pred_pro_3 != pred_pro_4 and (prob < 0.6 or max_score < 2) or max_score < 1: in_content = True # print('关键词4:', province_l, city_l, district_l) # print('输入:', '站源:%s, 角色:%s, 地址:%s' % (web_source_name, tenderees, all_addr)) # print('分数:', pro_ids, city_ids, dis_ids, prob, max_score) msc += '第四次预测,站源:%s; 招标、代理角色:%s; 所有地址:%s; 预测为:%s %s ##' % (web_source_name, tenderees, all_addr, pred_pro, pred_city) if big_area != "": area_dic['area'] = big_area if pred_pro != "": area_dic['province'] = pred_pro if pred_city != "": area_dic['city'] = pred_city if pred_dis != "": area_dic['district'] = pred_dis for k, v in code_dic.items(): if v != '': area_dic[k] = v if pred_city in ['北京', '天津', '上海', '重庆']: # 直辖市调整县级 area_dic['city'] = area_dic['district'] area_dic['district'] = '未知' if 'district_code' in area_dic: area_dic['city_code'] = area_dic['district_code'] area_dic.pop('district_code') if area_dic['city'] == '未知' and 'city_code' in area_dic: area_dic.pop('city_code') area_dic['is_in_text'] = in_content # area_dic['prob'] = prob # area_dic['max_score'] = max_score # print('最终地址:', pred_pro, pred_city, pred_dis) if prob < 0.6 or max_score < 2: log('地区匹配预测,最终结果:%s %s %s, 预测:%s, docid:%s'%(pred_pro, pred_city, pred_dis, msc, docid)) return {'district': area_dic} def get_area(self, text, web_name, in_content=False): p_pro, p_city, p_dis, idx_dic, full_dic, short_dic = self.p_pro, self.p_city, self.p_dis, self.idx_dic, self.full_dic, self.short_dic def get_final_addr(pro_ids, city_ids, dis_ids): ''' 先把所有匹配的全称、简称转为id,如果省份不为空,城市不为空且有城市属于省份的取该城市 :param province_l: 匹配到的所有省份 :param city_l: 匹配到的所有城市 :param district_l: 匹配到的所有区县 :return: ''' big_area = "" pred_pro = "" pred_city = "" pred_dis = "" final_pro = "" final_city = "" pro_prob = 0 city_prob = 0 if len(pro_ids) >= 1: pro_l = sorted([(k, v) for k, v in pro_ids.items()], key=lambda x: x[1], reverse=True) scores = [it[1] for it in pro_l] pro_prob = max(scores)/sum(scores) final_pro, score = pro_l[0] if score >= 0.01: pred_pro = idx_dic[final_pro]['返回名称'] big_area = idx_dic[final_pro]['大区'] # else: # print("得分过低,过滤掉", idx_dic[final_pro]['返回名称'], score) if pred_pro != "" and len(city_ids) >= 1: city_l = sorted([(k, v) for k, v in city_ids.items()], key=lambda x: x[1], reverse=True) scores = [it[1] for it in city_l] city_prob = max(scores) / sum(scores) for it in city_l: if idx_dic[it[0]]['省'] == final_pro: final_city = it[0] pred_city = idx_dic[final_city]['返回名称'] break if final_city != "" and len(set(dis_ids)) >= 1: dis_l = sorted([(k, v) for k, v in dis_ids.items()], key=lambda x: x[1], reverse=True) for it in dis_l: if idx_dic[it[0]]['市'] == final_city: pred_dis = idx_dic[it[0]]['返回名称'] elif pred_pro != "" and pred_city == "" and len(set(dis_ids)) >= 1: # 20241111 省份不为空,市为空,如果区县在省份下,补充对应的市县 dis_l = sorted([(k, v) for k, v in dis_ids.items()], key=lambda x: x[1], reverse=True) for it in dis_l: if idx_dic[it[0]]['省'] == final_pro: pred_city = idx_dic[idx_dic[it[0]]['市']]['返回名称'] pred_dis = idx_dic[it[0]]['返回名称'] # print('20241111 省份不为空,市为空,如果区县在省份下,补充对应的市县: ', pred_city, pred_dis) if pred_city in ['北京', '天津', '上海', '重庆']: pred_city = pred_dis pred_dis = "" return big_area, pred_pro, pred_city, pred_dis def find_areas(pettern, text): ''' 通过正则匹配字符串返回地址 :param pettern: 地址正则 广东省|广西省|... :param text: 待匹配文本 :return: ''' addr = [] for it in re.finditer(pettern, text): if re.search('[省市区县旗盟]$', it.group(0)) == None and re.search( '^([东南西北中一二三四五六七八九十大小]?(村|镇|街|路|道|社区)|酒店|宾馆)', text[it.end():]): continue if it.group(0) == '站前': # 20240314 修复类似 中铁二局新建沪苏湖铁路工程站前VI标项目 错识别为 省份:辽宁, 城市:营口,区县:站前 continue if re.search('^(经济开发区|开发区|新区)', text[it.end():]) and re.search('广州市', pettern): # 城市不匹配为区的地址 修复 滨州北海经济开发区 北海新区 等提取为北海 continue addr.append((it.group(0), it.start(), it.end())) if re.search('^([分支](公司|局|行|校|院|干?线)|\w{,3}段|地铁|(火车|高铁)?站|\w{,3}项目)', text[it.end():]): addr.append((it.group(0), it.start(), it.end())) return addr def chage_area2score(group_list, max_len): ''' 把匹配的的地址转为分数 :param group_list: [('name', b, e)] :return: ''' area_list = [] if group_list != []: for it in group_list: name, b, e = it area_list.append((name, (e - b + e) / max_len / 2)) return area_list def find_whole_areas(text): ''' 通过正则匹配字符串返回地址 :param pettern: 地址正则 广东省|广西省|... :param text: 待匹配文本 :return: ''' pettern = "((?P%s)(?P%s)?(?P%s)?)|((?P%s)(?P%s)?)|(?P%s)" % ( p_pro, p_city, p_dis, p_city, p_dis, p_dis) province_l, city_l, district_l = [], [], [] for it in re.finditer(pettern, text): if re.search('[省市区县旗盟]', it.group(0)) == None and re.search( '^([东南西北中一二三四五六七八九十大小]?(村|镇|街|路|道|社区)|酒店|宾馆)', text[it.end():]): continue if it.group(0) == '站前': # 20240314 修复类似 中铁二局新建沪苏湖铁路工程站前VI标项目 错识别为 省份:辽宁, 城市:营口,区县:站前 continue for k, v in it.groupdict().items(): if v != None: if k in ['prov']: province_l.append((it.group(k), it.start(k), it.end(k))) elif k in ['city', 'city1']: if re.search('^(经济开发区|开发区|新区)', text[it.end(k):]): # 城市不匹配为区的地址 修复 滨州北海经济开发区 北海新区 等提取为北海 continue city_l.append((it.group(k), it.start(k), it.end(k))) if re.search('^([分支](公司|局|行|校|院|干?线)|\w{,3}段|地铁|(火车|高铁)?站|\w{,3}项目)', text[it.end(k):]): city_l.append((it.group(k), it.start(k), it.end(k))) elif k in ['dist', 'dist1', 'dist2']: if it.group(k)=='昌江' and '景德镇' not in it.group(0): district_l.append(('昌江黎族', it.start(k), it.end(k))) else: district_l.append((it.group(k), it.start(k), it.end(k))) return province_l, city_l, district_l def get_pro_city_dis_score(text, text_weight=1): text = re.sub('复合肥|海南岛|兴业银行|双河口|阳光|杭州湾|新城区|中粮屯河|老城(区|改造|更新|升级|翻新)|沙县小吃|北京时间', ' ', text) # 544151395 赤壁市老城区燃气管道老化更新改造 text = re.sub('珠海城市', '珠海', text) # 修复 426624023 珠海城市 预测为海城市 text = re.sub('怒江州', '怒江傈僳族自治州', text) # 修复 423589589 所属地域:怒江州 识别为广西 - 崇左 - 江州 text = re.sub('茂名滨海新区', '茂名市', text) text = re.sub('中山([东南西][部区环]|黄圃|南头|东凤|小榄|石岐|翠亨|南朗)', '中山市', text) text = re.sub('横州市', '横县', text) # 例:547363890 修复广西南宁横州 不在地区表问题 ser = re.search('海南(昌江|白沙|乐东|陵水|保亭|琼中)(黎族)?', text) if ser and '黎族' not in ser.group(0): text = text.replace(ser.group(0), ser.group(0)+'黎族') for k, v in self.area_variance_dic.items(): # 20241113 根据地区变更信息替换文本 text = text.replace(k, v) # province_l = find_areas(p_pro, text) # city_l = find_areas(p_city, text) # district_l = find_areas(p_dis, text) province_l, city_l, district_l = find_whole_areas(text) # 20240703 优化地址提取,解决类似 海南昌江 得到 海南 南昌 结果 # if len(province_l) == len(city_l) == 0: # district_l = [it for it in district_l if # re.search('[市县旗区]$', it[0])] # 20240428去掉只有区县地址且不是全称的匹配,避免错误 例 凌云工业股份有限公司 提取地区为广西白色凌云 province_l = chage_area2score(province_l, max_len=len(text)) city_l = chage_area2score(city_l, max_len=len(text)) district_l = chage_area2score(district_l, max_len=len(text)) pro_ids = dict() city_ids = dict() dis_ids = dict() for pro in province_l: name, score = pro assert (name in full_dic['province'] or name in short_dic['province']) if name in full_dic['province']: idx = full_dic['province'][name] if idx not in pro_ids: pro_ids[idx] = 0 pro_ids[idx] += (score + 1) else: idx = short_dic['province'][name] if idx not in pro_ids: pro_ids[idx] = 0 pro_ids[idx] += (score + 0) for city in city_l: name, score = city if name in full_dic['city']: w = 0.1 if len(full_dic['city'][name]) > 1 else 1 for idx in full_dic['city'][name]: if idx not in city_ids: city_ids[idx] = 0 # weight = idx_dic[idx]['权重'] city_ids[idx] += (score + 2) * w pro_idx = idx_dic[idx]['省'] if pro_idx in pro_ids: pro_ids[pro_idx] += (score + 2) * w else: pro_ids[pro_idx] = (score + 2) * w * 0.5 elif name in short_dic['city']: w = 0.1 if len(short_dic['city'][name]) > 1 else 1 for idx in short_dic['city'][name]: if idx not in city_ids: city_ids[idx] = 0 weight = idx_dic[idx]['权重'] city_ids[idx] += (score + 1) * w * weight pro_idx = idx_dic[idx]['省'] if pro_idx in pro_ids: pro_ids[pro_idx] += (score + 1) * w * weight else: pro_ids[pro_idx] = (score + 1) * w * weight * 0.5 for dis in district_l: name, score = dis if name in full_dic['district']: w = 0.1 if len(full_dic['district'][name]) > 1 else 1 for idx in full_dic['district'][name]: if idx not in dis_ids: dis_ids[idx] = 0 # weight = idx_dic[idx]['权重'] dis_ids[idx] += (score + 1) * w pro_idx = idx_dic[idx]['省'] if pro_idx in pro_ids: pro_ids[pro_idx] += (score + 1) * w else: pro_ids[pro_idx] = (score + 1) * w * 0.5 city_idx = idx_dic[idx]['市'] if city_idx in city_ids: city_ids[city_idx] += (score + 1) * w else: city_ids[city_idx] = (score + 1) * w * 0.5 elif name in short_dic['district']: w = 0.1 if len(short_dic['district'][name]) > 1 else 1 for idx in short_dic['district'][name]: if idx not in dis_ids: dis_ids[idx] = 0 weight = idx_dic[idx]['权重'] dis_ids[idx] += (score + 0) * w if idx_dic[idx]['市'] not in city_ids and idx_dic[idx]['省'] not in pro_ids: # 20241111 区县简称不在获取到的省、市范围内的过滤掉 continue pro_idx = idx_dic[idx]['省'] if pro_idx in pro_ids: pro_ids[pro_idx] += (score + 0) * w * weight # else: # 20241015 注销 区县简称且不在提取的省市下面,不加分,避免提取错误 例:536550843 # pro_ids[pro_idx] = (score + 0) * w * weight * 0.5 city_idx = idx_dic[idx]['市'] if city_idx in city_ids: city_ids[city_idx] += (score + 0) * w * weight # else: # 20241015 注销 区县简称且不在提取的省市下面,不加分,避免提取错误 例:536550843 # city_ids[city_idx] = (score + 0) * w * weight * 0.1 elif pro_idx in pro_ids: city_ids[city_idx] = (score + 0) * w * weight * 0.1 for k, v in pro_ids.items(): pro_ids[k] = v * text_weight for k, v in city_ids.items(): city_ids[k] = v * text_weight for k, v in dis_ids.items(): dis_ids[k] = v * text_weight return pro_ids, city_ids, dis_ids area_dic = {'area': '全国', 'province': '全国', 'city': '未知', 'district': '未知', "is_in_text": False} pro_ids, city_ids, dis_ids = get_pro_city_dis_score(text) pro_ids1, city_ids1, dis_ids1 = get_pro_city_dis_score(web_name, text_weight=0.01) # 20240422 修改为站源名称只取前三字,避免类似 459056219 中金岭南阳光采购平台 错提取阳光 for k in pro_ids1: if k in pro_ids: pro_ids[k] += pro_ids1[k] else: pro_ids[k] = pro_ids1[k] for k in city_ids1: if k in city_ids: city_ids[k] += city_ids1[k] else: city_ids[k] = city_ids1[k] for k in dis_ids1: if k in dis_ids: dis_ids[k] += dis_ids1[k] else: dis_ids[k] = dis_ids1[k] big_area, pred_pro, pred_city, pred_dis = get_final_addr(pro_ids, city_ids, dis_ids) if big_area != "": area_dic['area'] = big_area if pred_pro != "": area_dic['province'] = pred_pro if pred_city != "": area_dic['city'] = pred_city if pred_dis != "": area_dic['district'] = pred_dis if in_content: area_dic['is_in_text'] = True return {'district': area_dic} def predict(self, project_name, prem, title, list_articles, web_source_name = "", list_entitys=""): ''' 先匹配 project_name+tenderee+tenderee_address, 如果缺少省或市 再匹配 title+content :param project_name: :param prem: :param title: :param list_articles: :param web_source_name: :return: ''' def get_ree_addr(prem): tenderee = "" tenderee_address = "" try: for v in prem[0]['prem'].values(): for link in v['roleList']: if link['role_name'] == 'tenderee' and tenderee == "": tenderee = link['role_text'] tenderee_address = link['address'] except Exception as e: print('解析prem 获取招标人、及地址出错') return tenderee, tenderee_address def get_role_address(text): '''正则匹配获取招标人地址 3:地址直接在招标人后面 招标人:xxx,地址:xxx 4:招标、代理一起,两个地址一起 招标人:xxx, 代理人:xxx, 地址:xxx, 地址:xxx. ''' p3 = '(招标|采购|甲)(人|方|单位)(信息:|(甲方))?(名称)?:[\w()]{4,15},(联系)?地址:(?P(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+|\w{2,15}[,。])' p4 = '(招标|采购|甲)(人|方|单位)(信息:|(甲方))?(名称)?:[\w()]{4,15},(招标|采购)?代理(人|机构)(名称)?:[\w()]{4,15},(联系)?地址:(?P(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+|\w{2,15}[,。])' p5 = '(采购|招标)(人|单位)(联系)?地址:(?P(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+|\w{2,15}[,。])' if re.search(p3, text): return re.search(p3, text).group('addr') elif re.search(p4, text): return re.search(p4, text).group('addr') elif re.search(p5, text): return re.search(p5, text).group('addr') else: return '' def get_project_addr(text): p1 = '(项目|施工|实施|建设|工程|服务|交货|送货|收货|展示|看样|拍卖)(地址|地点|位置|所在地区?)(位于)?:(?P(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+([\w()]{,20}[,。])?|\w{2,15}[,。])' p2 = '项目位于(?P\w{2}市\w{2,4}区)' if re.search(p1, text): return re.search(p1, text).group('addr') elif re.search(p2, text): return re.search(p2, text).group('addr') else: return '' def get_bid_addr(text): p2 = '(磋商|谈判|开标|投标|评标|报名|递交|评审|发售|所属)(地址|地点|所在地区?|地域):(?P(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+|\w{2,15}[,。])' if re.search(p2, text): return re.search(p2, text).group('addr') else: return '' def get_all_addr(list_entitys): tenderee_l = [] addr_l = [] for ent in list_entitys[0]: if ent.entity_type == 'location' and len(ent.entity_text) > 2: addr_l.append(ent.entity_text) elif ent.entity_type in ['org', 'company']: if ent.label in [0, 1]: # 加招标或代理 tenderee_l.append(ent.entity_text) return ' '.join(addr_l), ' '.join(tenderee_l) def get_title_addr(text): p1 = '(?P(\w{1,13}(自治[区州县旗]|地区|[省市区县旗盟])[^\w]*)+|\w{2,15}[,。])' if re.search(p1, text): return re.search(p1, text).group('addr') else: return '' if '##attachment##' in list_articles[0].content: content, attachment = list_articles[0].content.split('##attachment##') if len(content) < 200: content += attachment else: content = list_articles[0].content tenderee, tenderee_address = get_ree_addr(prem) msc = "" pro_addr = get_project_addr(content) if pro_addr != "" and re.search('(采购人|招标人)?指定地点', pro_addr)==None: # 排除错误项目地址 例:554024168 1.5服务地点:采购人指定地点。 msc += '使用规则提取的项目地址;' tenderee_address = pro_addr else: role_addr = get_role_address(content) if role_addr != "" and re.search('(采购人|招标人)?指定地点', role_addr)==None: msc += '使用规则提取的联系人地址;' tenderee_address = role_addr if tenderee_address == "": title_addr = get_title_addr(title) if title_addr != "": msc += '使用规则提取的标题地址;' tenderee_address = title_addr else: bid_addr = get_bid_addr(content) if bid_addr != "": msc += '使用规则提取的开标地址;' tenderee_address = bid_addr project_name = str(project_name) tenderee = str(tenderee) # print('招标人地址',role_addr, tenderee_address) project_name = project_name + title if project_name not in title else title # project_name = project_name.replace(tenderee, '') if len(project_name)>3: entity_list = getNers([project_name],useselffool=False) # 2024/4/26 修改为去重项目名称中所有公司名称 for tup in entity_list[0]: if tup[2] in ['org', 'company']: project_name = project_name.replace(tup[3], '') text1 = "{0} {1} {2}".format(tenderee, tenderee_address, project_name) web_source_name = str(web_source_name) # 修复某些不是字符串类型造成报错 text1 = re.sub('复合肥|铁路|公路|新会计', ' ', text1) # 预防提取错 合肥 路南 新会 等地区 if pro_addr and re.search('\w{2,}([市县旗盟]|自治[区州县旗])', pro_addr): if re.search('[市县旗盟]', pro_addr)==None: # 修复 486623506 项目地址不完整 pro_addr = text1 + ' '+ pro_addr msc += '## 使用项目地址输入:%s ##;' % pro_addr rs = self.get_area(pro_addr, '') msc += '预测结果:省份:%s, 城市:%s,区县:%s;' % ( rs['district']['province'], rs['district']['city'], rs['district']['district']) if rs['district']['province'] != '全国' and rs['district']['city'] != '未知': # print('地区匹配:', msc) return rs # print('text1:', text1) msc += '## 第一次预测输入:%s ##;' % text1 rs = self.get_area(text1, '') # 2024/4/22 调整第一次输入不带站源名称,避免出错 msc += '预测结果:省份:%s, 城市:%s,区县:%s;' % ( rs['district']['province'], rs['district']['city'], rs['district']['district']) # self.f.write('%s %s \n' % (list_articles[0].id, msc)) # print('地区匹配:', msc) if rs['district']['province'] == '全国' or rs['district']['city'] == '未知': # msc = "" all_addr, tenderees = get_all_addr(list_entitys) text2 = tenderees + " " + all_addr + ' ' + title msc += '使用实体列表所有招标人+所有地址;' # text2 += title + content if len(content)<2000 else title + content[:1000] + content[-1000:] text2 = re.sub('复合肥|铁路|公路|新会计', ' ', text2) # print('text2:', text2) msc += '## 第二次预测输入:%s %s##' % (text2,web_source_name) rs2 = self.get_area(text2, web_source_name, in_content=True) # rs2['district']['is_in_text'] = True if rs['district']['province'] == '全国' and rs2['district']['province'] != '全国': rs = rs2 elif rs['district']['province'] == rs2['district']['province'] and rs2['district']['city'] != '未知': rs = rs2 msc += '预测结果:省份:%s, 城市:%s,区县:%s' % ( rs['district']['province'], rs['district']['city'], rs['district']['district']) # self.f.write('%s %s \n'%(list_articles[0].id, msc)) # print('地区匹配:', msc) return rs