# -*- coding: utf-8 -*- """标签类预测器。 按 ARCHITECTURE.md Phase 5 拆分建议,从 ``interface/predictor.py`` 迁出以下 标签相关类: - ``ProjectLabel`` — 项目标签提取(原 predictor.py 第 5585-5891 行) - ``IndustryLabel`` — 行业标签分类(原 5895-5942 行) - ``PropertyLabel`` — 产权分类二级标签(原 6041-6201 行) - ``BiddingScore`` — 评分表投标人得分/排名提取(原 9835-10039 行) 原 ``from common.Utils import *`` / ``from interface.modelFactory import *`` 已替换为显式 import;``os.path.dirname(__file__)`` 路径引用替换为 ``predictors._common.INTERFACE_DIR``。 ``interface/predictor.py`` 仍保留原定义,老 import 不受影响。 """ from __future__ import absolute_import import re import pandas as pd from bs4 import BeautifulSoup from BiddingKG.dl.interface.classification_process import classify_text, process_rules from BiddingKG.dl.predictors._common import INTERFACE_DIR, get_role from BiddingKG.dl.predictors.table_prem import TableTag2List, get_header_line __all__ = [ "ProjectLabel", "IndustryLabel", "PropertyLabel", "BiddingScore", ] class ProjectLabel(): def __init__(self, ): self.keyword_list = self.get_label_keywords() self.kongjing_keyword_list = self.get_kongjing_keywords() self.ICT_smart_compute_keyword_list = self.get_ICT_smart_compute_keywords() def get_label_keywords(self): import csv path = INTERFACE_DIR+'/project_label_keywords.csv' with open(path, 'r',encoding='utf-8') as f: reader = csv.reader(f) key_word_list = [] for r in reader: if r[0] == '类型': continue type = r[0] key_wrod = r[1] key_paichuci = str(r[2]) key_paichuci = key_paichuci if key_paichuci and key_paichuci != 'nan' else "" type_paichuci = str(r[3]) type_paichuci = type_paichuci if type_paichuci and type_paichuci != 'nan' else "" key_word_list.append((type, key_wrod, key_paichuci, type_paichuci)) return key_word_list def get_kongjing_keywords(self): import csv path = INTERFACE_DIR+'/kongjing_label_keywords.csv' with open(path, 'r',encoding='utf-8') as f: reader = csv.reader(f) key_word_list = [] for r in reader: if r[0] == '关键词': continue key_wrod = r[0] key_wrod2 = str(r[1]) key_wrod2 = key_wrod2 if key_wrod2 and key_wrod2 != 'nan' else "" search_type = r[2] info_type_list = str(r[3]) info_type_list = info_type_list if info_type_list and info_type_list != 'nan' else "" key_word_list.append((key_wrod, key_wrod2, search_type, info_type_list)) return key_word_list def get_ICT_smart_compute_keywords(self): import csv path = INTERFACE_DIR+'/ICT智算_label_keywords.csv' with open(path, 'r',encoding='utf-8') as f: reader = csv.reader(f) key_word_list = [] for r in reader: if r[0] == '关键词': continue key_wrod = r[0] key_wrod2 = str(r[1]) key_wrod2 = key_wrod2 if key_wrod2 and key_wrod2 != 'nan' else "" search_type = r[2] info_type_list = str(r[3]) info_type_list = info_type_list if info_type_list and info_type_list != 'nan' else "" key_word_list.append((key_wrod, key_wrod2, search_type, info_type_list)) return key_word_list def extract_core_text(self,all_text,tenderee="",agency=""): # 剔除 招标单位、代理机构名称 if tenderee: all_text = all_text.replace(tenderee, " ") if agency: all_text = all_text.replace(agency, " ") # 定义需要匹配的关键词列表 keywords = [ '项目名称', '工程名称', '采购名称', '标段名称', '项目的名称', '设备名称', '申购主题', '申购单主题', '标的', '商品名称', '二级目录', '招标内容', '项目内容', '商品清单', '标的名称', '采购内容', '集成要求', '概况介绍', '品目分类', '招标范围', '采购范围', '项目采购分', '采购合同', '招标合同','产品名称','服务内容','采购品目名称','货物名称', '采购需求概况','项目概况','招标范围','采购条目名称','物资名称','物料名称','建设规模', '建设内容','采购项目概况','采购包名称','物料描述','商品信息','服务品目','标项名称', '规格描述','采购标的','服务名称','采购单名称','明细信息','申购主题','需求详情' ] # 创建正则表达式模式,匹配任意一个关键词 pattern = r'(' + '|'.join(re.escape(kw) for kw in keywords) + r')' # 查找所有匹配位置 matches = list(re.finditer(pattern, all_text)) if not matches: return "" # 没有找到关键词 all_match_text = [] for _match in matches: start_pos = _match.end() # 关键词结束位置 # 提取关键词之后的内容 after_text = all_text[start_pos:] # 提取最多45个汉字 chinese_chars = [] count = 0 for char in after_text: # 判断是否为汉字 (Unicode范围) if '\u4e00' <= char <= '\u9fff': count += 1 if count > 45: break chinese_chars.append(char) chinese_chars = ''.join(chinese_chars).strip() if chinese_chars and chinese_chars not in all_match_text: all_match_text.append(chinese_chars) # 将字符列表组合成字符串 main_content_text = ','.join(all_match_text).strip() return main_content_text def predict(self, doctitle,product,project_name,prem,all_text): doctitle = doctitle if doctitle else "" product = product if product else "" product = ",".join(set(product.split(','))) # 产品词去重 project_name = project_name if project_name else "" tenderee = "" agency = "" sub_project_names = [] # 标段名称 try: for k,v in prem[0]['prem'].items(): # sub_project_names.append(k) sub_project_names.append(v.get("name","")) for link in v['roleList']: if link['role_name'] == 'tenderee' and tenderee == "": tenderee = link['role_text'] if link['role_name'] == 'agency' and agency == "": agency = link['role_text'] except Exception as e: # print('解析prem 获取招标人、代理人出错') pass sub_project_names = ";".join(sub_project_names) main_content_text = self.extract_core_text(all_text,tenderee,agency) # 核心字段:标题+产品词+项目名称+标段名称 # main_text = ",".join([doctitle, product, project_name, sub_project_names]) # 核心字段:标题+项目名称+产品词+正文定位词后45个字 main_text = ",".join([doctitle, project_name, product, main_content_text]) # 剔除 招标单位、代理机构名称 if tenderee: doctitle = doctitle.replace(tenderee, " ") main_text = main_text.replace(tenderee, " ") if agency: doctitle = doctitle.replace(agency, " ") main_text = main_text.replace(agency, " ") doctitle_dict = dict() main_text_dict = dict() for item in self.keyword_list: _type = item[0] key_wrod = item[1] # 关键词排除词 key_paichuci = item[2] key_paichuci_s = "|".join([re.escape(word) for word in key_paichuci.strip('、').split('、')]) # 类型排除词 type_paichuci = item[3] if type_paichuci: paichuci_split = type_paichuci.strip('、').split('、') if re.search("|".join([re.escape(word) for word in paichuci_split]), main_text): continue if doctitle: if key_wrod in doctitle: if not key_paichuci_s or (key_paichuci_s and not re.search(key_paichuci_s, doctitle)): key_wrod_count1 = doctitle.count(key_wrod) if _type not in doctitle_dict: # doctitle_dict[_type] = {'关键词': [], '排除词': type_paichuci} doctitle_dict[_type] = [] doctitle_dict[_type].append((key_wrod, key_wrod_count1)) if main_text: if key_wrod in main_text: if not key_paichuci_s or (key_paichuci_s and not re.search(key_paichuci_s, main_text)): key_wrod_count2 = main_text.count(key_wrod) if _type not in main_text_dict: # main_text_dict[_type] = {'关键词': [], '排除词': type_paichuci} main_text_dict[_type] = [] main_text_dict[_type].append((key_wrod, key_wrod_count2)) # 排序 doctitle for k, v in doctitle_dict.items(): doctitle_dict[k].sort(key=lambda x: x[1], reverse=True) # 按匹配次数保留前10个标签 if len(doctitle_dict) > 10: doctitle_labels = [(k, sum(w[1] for w in doctitle_dict[k])) for k in doctitle_dict] doctitle_labels.sort(key=lambda x: x[1], reverse=True) for item in doctitle_labels[10:]: doctitle_dict.pop(item[0]) # main_text pop_list = [] for k, v in main_text_dict.items(): if sum([j[1] for j in main_text_dict[k]]) == 1: # 关键词匹配次数等于1的标签 pop_list.append(k) main_text_dict[k].sort(key=lambda x: x[1], reverse=True) # 核心字段标签,若存在同一个标签的关键词匹配次数大于1,则只保留关键词匹配次数大于1的标签,关键词匹配次数等于1的标签不要 if len(pop_list) < len(main_text_dict): for k in pop_list: main_text_dict.pop(k) # 按匹配次数保留前10个标签 if len(main_text_dict) > 10: main_text_labels = [(k, sum(w[1] for w in main_text_dict[k])) for k in main_text_dict] main_text_labels.sort(key=lambda x: x[1], reverse=True) for item in main_text_labels[10:]: main_text_dict.pop(item[0]) return {"标题":doctitle_dict,"核心字段":main_text_dict},main_content_text def predict_other(self,project_label,industry,doctitle,project_name,product,list_articles,main_content_text): # doctextcon 取正文内容 doctextcon = list_articles[0].content.split('##attachment##')[0] info_type = industry.get('industry',{}).get("class_name","") doctitle = doctitle if doctitle else "" product = product if product else "" product = ",".join(set(product.split(','))) # 产品词去重 project_name = project_name if project_name else "" main_content_text = main_content_text if main_content_text else "" # 空净通 get_kongjing_label = False keywords_list = [] for item in self.kongjing_keyword_list: key_wrod = item[0] key_wrod2 = item[1] search_type = item[2] info_type_list = item[3] info_type_list = info_type_list.strip('|').split("|") if info_type_list else [] search_text = "" if search_type=='正文': search_text = ",".join([doctextcon,doctitle,project_name,product]) elif search_type=='产品': search_text = ",".join([doctitle,project_name,product]) if search_type=='行业': # ’行业’类型直接用info_type匹配关键词 if info_type==key_wrod: # 匹配关键词记录 keywords_list.append(key_wrod) get_kongjing_label = True break else: if key_wrod in search_text: if key_wrod2 and key_wrod2 not in search_text: continue if info_type_list and info_type not in info_type_list: continue # 匹配关键词记录 if key_wrod2: keywords_list.append(key_wrod+'+'+key_wrod2) else: keywords_list.append(key_wrod) get_kongjing_label = True break if get_kongjing_label: project_label["核心字段"]["空净通"] = [[word,1] for word in keywords_list][:10] # ICT智算 get_ICT_label = False keywords_list = [] for item in self.ICT_smart_compute_keyword_list: key_wrod = item[0] key_wrod2 = item[1] search_type = item[2] info_type_list = item[3] info_type_list = info_type_list.strip('|').split("|") if info_type_list else [] search_text = "" if search_type=='正文': search_text = ",".join([doctextcon,doctitle,project_name,product]) elif search_type=='产品': search_text = ",".join([doctitle,project_name,product]) elif search_type=='core_text': # core_text规则:标题+项目名称+产品词+正文定位词后45个字 search_text = doctitle + ',' + project_name + "," + product + "," + main_content_text if search_type=='行业': # ’行业’类型直接用info_type匹配关键词 if info_type==key_wrod: # 匹配关键词记录 keywords_list.append(key_wrod) get_ICT_label = True break else: if key_wrod in search_text: if key_wrod2 and key_wrod2 not in search_text: continue if info_type_list and info_type not in info_type_list: continue # 匹配关键词记录 if key_wrod2: keywords_list.append(key_wrod+'+'+key_wrod2) else: keywords_list.append(key_wrod) get_ICT_label = True break if info_type in ['计算机设备','监控设备','通信设备','信息系统集成和物联网技术服务','运行维护服务','信息处理和存储支持服务','互联网安全服务','互联网接入及相关服务','电信']: # 新增规则info_type符合范围直接判定 get_ICT_label = True if get_ICT_label: project_label["核心字段"]["ICT智算"] = [[word,1] for word in keywords_list][:10] return project_label # 行业标签 class IndustryLabel(): def __init__(self): # self.keyword_list = self.get_label_keywords() pass def predict(self,doctitle,article,product,prem): doctitle = doctitle if doctitle else "" product = product if product else "" product = ",".join(set(product.split(','))) # 产品词去重 all_text = article.content all_text = re.sub('\s+', ' ', all_text) tenderee = "" agency = "" try: for k,v in prem[0]['prem'].items(): for link in v['roleList']: if link['role_name'] == 'tenderee' and tenderee == "": tenderee = link['role_text'] if link['role_name'] == 'agency' and agency == "": agency = link['role_text'] except Exception as e: # print('解析prem 获取招标人、代理人出错') pass # 剔除 招标单位、代理机构名称 if tenderee: doctitle = doctitle.replace(tenderee, " ") all_text = all_text.replace(tenderee, " ") if agency: doctitle = doctitle.replace(agency, " ") all_text = all_text.replace(agency, " ") # category_1, category_2, category_3, matched_keywords, rule_id = product_classify_process(doctitle, all_text, product) category_1, category_2, category_3, matched_keywords, rule_id = classify_text(doctitle, all_text, product) # print(category_1, category_2, category_3, matched_keywords, rule_id) tenderee_label = process_rules(tenderee) new_tenderee_label = [] for k in tenderee_label.keys(): if '-' in k: new_tenderee_label.append({"first_level":k.split('-')[0],"second_level":k.split('-')[1]}) # new_tenderee_label.append({"first_level":k.split('-')[0],"second_level":k.split('-')[1],"code":k.split('-')[2]}) # print(new_tenderee_label) if category_2=="标题排除": category_1 = "其他" category_2 = "" rule_id = "" return {"first_level":category_1,"second_level":category_2},new_tenderee_label # def get_label_keywords(self): # import csv # path = os.path.dirname(__file__)+'/industry_label_keywords.csv' # with open(path, 'r',encoding='utf-8') as f: # reader = csv.reader(f) # key_word_list = [] # for r in reader: # if r[0] == '一级标签': # continue # first_level = r[0] # second_level = str(r[1]) # second_level = second_level.strip() if second_level and second_level != 'nan' else "" # key_word = str(r[2]).strip() # all_paichuci = str(r[3]) # all_paichuci = all_paichuci.strip() if all_paichuci and all_paichuci != 'nan' else "" # title_paichuci = str(r[4]) # title_paichuci = title_paichuci.strip() if title_paichuci and title_paichuci != 'nan' else "" # product_paichuci = str(r[5]) # product_paichuci = product_paichuci.strip() if product_paichuci and product_paichuci != 'nan' else "" # key_word_list.append((first_level, second_level, key_word, all_paichuci,title_paichuci,product_paichuci)) # return key_word_list # # def predict(self, doctitle,article,product,prem): # # doctitle = doctitle if doctitle else "" # product = product if product else "" # product = ",".join(set(product.split(','))) # 产品词去重 # all_text = article.content # tenderee = "" # agency = "" # try: # for k,v in prem[0]['prem'].items(): # for link in v['roleList']: # if link['role_name'] == 'tenderee' and tenderee == "": # tenderee = link['role_text'] # if link['role_name'] == 'agency' and agency == "": # agency = link['role_text'] # except Exception as e: # # print('解析prem 获取招标人、代理人出错') # pass # # 剔除 招标单位、代理机构名称 # if tenderee: # doctitle = doctitle.replace(tenderee, " ") # all_text = all_text.replace(tenderee, " ") # if agency: # doctitle = doctitle.replace(agency, " ") # all_text = all_text.replace(agency, " ") # # label_list = [] # for item in self.keyword_list: # first_level = item[0] # second_level = item[1] # key_word = item[2] # key_word = key_word.strip('、').split('、') # # 全文排除词 # all_paichuci = item[3] # all_paichuci = "|".join([re.escape(word) for word in all_paichuci.strip('、').split('、')]) # # 标题排除词 # title_paichuci = item[4] # title_paichuci = "|".join([re.escape(word) for word in title_paichuci.strip('、').split('、')]) # # 产品排除词 # product_paichuci = item[5] # product_paichuci = "|".join([re.escape(word) for word in product_paichuci.strip('、').split('、')]) # # # if doctitle and title_paichuci: # if re.search(title_paichuci,doctitle): # continue # if product and product_paichuci: # if re.search(product_paichuci,product): # continue # if all_text: # if all_paichuci: # if re.search(all_paichuci,all_text): # continue # get_label = False # for _keyword in key_word: # if '+' not in _keyword: # if _keyword in all_text: # get_label = True # break # else: # get_keyword = True # for _word in _keyword.split("+"): # if _word not in all_text: # get_keyword = False # break # if get_keyword: # get_label = True # break # if get_label: # label_list.append({"first_level":first_level,"second_level":second_level}) # # return label_list # 产权分类二级标签 class PropertyLabel(): ''' 产权分类二级标签 全部类别: 股权, 债权, 知识产权, 矿权, 房产, 土地, 交通运输工具, 闲置物资、设备、材料, 其他 ''' def __init__(self, ): car = "比亚迪|奇瑞|奥迪|宝马|菲尼迪|雷克萨斯|三菱|铃木|马自达|奔驰|劳斯莱斯|北京现代|" \ "宾利|兰博基尼|布加迪|保时捷|斯柯达|雪佛兰|别克|凯迪拉克|庞蒂亚克|克尔维特|福特|林肯|克莱斯勒|道奇|JEEP品牌" self.keywords_dict = { "房产": "房产|住宅|公寓|商铺|车位|写字楼|办公楼|别墅|综合楼|在建工程|厂房|车库|车房|房转让|房屋|商品房|商业用房|" "宅基地|[\u4e00-\u9fa5]{,2}用房|店面|商业房|门[面市]房|仓库|铺位|地下室|\d号?(房|室|门市|门面|商?铺|单元|户)|不动产|" "自建房|铺面|商务楼|商住楼|阁楼|(杂物|储物|储藏)(房|间|室)|套房|[\da-zA-Z](栋|棟|幢|层|座|号?楼|单元)\d{1,4}(号|房|室|商?铺|户)|" "[\da-zA-Z](栋|棟|幢|层|座|号?楼|单元)\d{2,}|门面+转让|楼+变卖|房地产", "交通运输工具": "车辆|轿车|汽车(?!用品|库|位|衡)|公车|客车|货车|面包车|SUV|新能源车|二手车|车辆|商用车|机动车|观光车|巴车|" "船舶|四驱" + "|" + car, "股权": "\d.?股|股权(?!交易中心)|\d%(比例)?.?股|\d万.?股|\d.?元/股|增资(?!源)|扩股|股(转让|出售)|百分之[一二三四五六七八九十]{1,3}股", "债权": "债权|债权转让|债权人|债务人|原债权人|新债权人|金融资产", "土地": "住宅用地|商业用地|工业用地|国有[\u4e00-\u9fa5]{,3}[土用]地|集体土地|划拨|流转|地块编号|" "土地使用权证|土地经营权|土地证|土地[发承]包|[\u4e00-\u9fa5]{,2}用地|土地\d{1,3}(亩|公?顷)|\d{1,3}(亩|公?顷)(使用|经营)权|" "承包土地|(地块|土地)承包|水面经营权|[鱼水]塘|鱼池|(水面|旱田)[\u4e00-\u9fa5]{,3}[发承]包|水面资源|(水面|水田)[\u4e00-\u9fa5]{,3}权|" "四荒|林地|林场|林木所有权|采伐权|水利设施所有权|水利设施使用权|海域|滩涂|林业产权|旱田|水田|机动田|机动地|耕地|荒地|农田|" "苗圃地|塘口", "矿权": "矿权|矿业权|采矿许可|探矿权|采矿权|开采权|矿产资源处置|矿[\u4e00-\u9fa5]{1,3}开[发采]", "知识产权": "知识产权(?!局)|商标|专利|著作权|版权|商业秘密|科研成果", "闲置物资、设备、材料": "(废旧|报废|废|闲置|二手|淘汰)(物资|资产|机械|设备|仪器|汽车|车|钢铁|钢材|钢|金属|塑料|材料|导管|漆|渣|有色|品|[\u4e00-\u9fa5]{,2}车|偶头)|" "(金属|机械|设备|仪器|汽车|钢铁|钢材|钢|塑料|有色|)废料|废液|废旧|报废|边角料|残次品|(热轧|冷轧|酸洗|镀铝|热镀|镀锌|镀镁)|" "机[器械]设备|医疗设备|生产设备|办公设备|仪器|仪表|设备出租|设备租赁|拖拉机|收割机|插秧机|挖机|车床|挖掘机|电机|" "戒指|弃渣|电解质块|茶杯|装置|花瓶|女表|手表|男表|硫磺|物资|书画|茶叶|油茶|红茶|[茗名]茶|白酒|红酒|酒水|酒品|名酒|毛石|[石金木铁矿铜锌铝钢]料|" "零部件", "经营权": "经营权", "租赁": "房+租|市场+续约|资产+出租|租赁|续租|招租|出租|租金|房租" } self.cqjy_keywords = self.get_cqjy_keywords() self.score_idx = ["股权", "债权", "知识产权", "矿权", "房产", "土地", "交通运输工具", "闲置物资、设备、材料"] def get_cqjy_keywords(self): import csv path = INTERFACE_DIR+'/property_label_products.csv' with open(path, 'r',encoding='utf-8') as f: reader = csv.reader(f) key_word_list = [] for r in reader: if r[0] == 'product': continue key_wrod = r[0] _type = r[1] key_word_list.append((_type, key_wrod)) return key_word_list def get_type(self, text): keyword_list = [] for key, value in self.keywords_dict.items(): keyword = "|".join([i for i in value.split("|") if '+' not in i]) keyword2 = [i for i in value.split("|") if '+' in i] if re.search(keyword, text): re1 = [i for i in re.finditer(keyword, text)][-1] keyword_list.append((key, re1.start())) else: # 组合词 查询 for k in keyword2: k1, k2 = k.split('+') if re.search(k1, text) and re.search(k2, text): keyword_list.append((key, re.search(k2, text).start())) break return keyword_list def get_type2(self, text, cqjy_type_list): have_type = [i[0] for i in cqjy_type_list] for item in self.cqjy_keywords: _type = item[0] key_wrod = item[1] if _type not in have_type: if '+' in key_wrod: k1, k2 = key_wrod.split('+') if re.search(k1, text) and re.search(k2, text): cqjy_type_list.append((_type, re.search(k2, text).start())) have_type.append(_type) else: if key_wrod in text: cqjy_type_list.append((_type, text.index(key_wrod))) have_type.append(_type) return cqjy_type_list def predict(self, doctitle,product,project_name,prem,channel_dic): docchannel = channel_dic['docchannel']['doctype'] # print('docchannel',docchannel) if docchannel not in ['土地矿产', '拍卖出让', '产权交易']: return "" doctitle = doctitle if doctitle else "" product = product if product else "" product = ",".join(set(product.split(','))) # 产品词去重 project_name = project_name if project_name else "" tenderee = "" agency = "" try: for k,v in prem[0]['prem'].items(): for link in v['roleList']: if link['role_name'] == 'tenderee' and tenderee == "": tenderee = link['role_text'] if link['role_name'] == 'agency' and agency == "": agency = link['role_text'] except Exception as e: # print('解析prem 获取招标人、代理人出错') pass cqjy_type = [] idx = 0 for text in [doctitle, project_name, product]: if tenderee: text = text.replace(tenderee, "") if agency: text = text.replace(agency, "") cqjy_type = self.get_type(text) if not cqjy_type: cqjy_type = self.get_type2(text, cqjy_type) idx += 1 if idx == 2: # project_name if len(re.split("[,、]", text)) > 9: cqjy_type = [] if idx == 3: # product if len(text.split(",")) > 15: cqjy_type = [] if cqjy_type: break cqjy_type2 = [i[0] for i in cqjy_type] if cqjy_type: # 类别优先级调整 if "租赁" in cqjy_type2: cqjy_type2 = ['租赁'] elif "经营权" in cqjy_type2: cqjy_type2 = ['经营权'] elif "股权" in cqjy_type2 or "债权" in cqjy_type2 or "知识产权" in cqjy_type2: cqjy_type.sort(key=lambda x: self.score_idx.index(x[0])) cqjy_type = cqjy_type[0] cqjy_type2 = [cqjy_type[0]] elif len(cqjy_type2) == 2 and "房产" in cqjy_type2 and "土地" in cqjy_type2: cqjy_type2 = ['房产'] else: # 权重排序,取第一位 if idx in [1, 2]: # doctitle, project_name cqjy_type.sort(key=lambda x: x[1], reverse=True) cqjy_type = cqjy_type[0] cqjy_type2 = [cqjy_type[0]] else: cqjy_type.sort(key=lambda x: self.score_idx.index(x[0])) cqjy_type = cqjy_type[0] cqjy_type2 = [cqjy_type[0]] cqjy_type2 = ",".join(cqjy_type2) if not cqjy_type2: cqjy_type2 = '其他' return cqjy_type2 class BiddingScore(): def __init__(self): self.head_rule_dic = { "tenderer": "((候选|入围|入选|投标|应答|响应)(供应商库)?的?(人|人?单位|机构|供应商|供货商|服务商|投标人|(中标)?公司|(中标)?企业|银行)|(通过)?名单|中标候选人)(名称|名单|全称|\d)?$|^供应商(名称|信息)?$|投标个人/单位", #补充 368295593 投标个人/单位 提取 "score_price": "(价格|报价|单价|总价|经济)(部分|\w{,2})?([得评]分|评审)", "score_technical": "技术(部分|\w{,2})?标?([得评]分|评审)", "score_commercial": "商务(部分|\w{,2})?标?([得评]分|评审)", "score_integrity": "诚信(部分|\w{,2})?([得评]分|评审)", "score_comprehensive": "(综合(标|评估)?|总|最终)得?分$", "ranking": "(得分)?排名", "qualification_review": "资格性审查|是否通过资格", "compliance_review": "符合性审查|是否通过符合" } self.tb = TableTag2List() def get_table_info(self, df, nlp_enterprise): def get_header_index(datas): ''' 根据表格表头判断结果0/1 得到哪些行和列是表头 :param datas: 表格内容表头判断结果数据[[1,1,1,1],[0,0,0,0]] :return: 表头所在的行和列序号 ''' header_row = [] header_col = [] df_h = pd.DataFrame(datas) # 表头判断数据 , columns=columns for i in df_h.index: line = df_h.loc[i].values if sum(line) == len(line): header_row.append((i, sum(line) / len(line))) elif sum(line) / len(line) > 0.8: header_row.append((i, sum(line) / len(line))) elif len(line) > 3 and len(re.findall('11', ''.join([str(it) for it in line]))) > len( re.findall('10', ''.join([str(it) for it in line]))): header_row.append((i, sum(line) / len(line))) for i in df_h.columns: col = df_h[i].values if sum(col) == len(col): header_col.append((i, sum(col) / len(col))) elif sum(col) / len(col) > 0.8: header_col.append((i, sum(col) / len(col))) elif len(col) > 3 and len(re.findall('11', ''.join([str(it) for it in line]))) > len( re.findall('10', ''.join([str(it) for it in line]))): header_col.append((i, sum(col) / len(col))) return header_row, header_col def get_header(l, head_rule_dic): header_dic = {} for i in range(len(l)): text = l[i] num = 0 tmp_dic = {} for k, v in head_rule_dic.items(): # print('k : ', k) if re.search(v, text): tmp_dic[k] = i num += 1 # if num > 1: # if tmp_dic.keys() == set(['qualification_review', 'compliance_review']): # for k, v in tmp_dic.items(): # if k not in header_dic: # header_dic[k] = v # elif tmp_dic: for k, v in tmp_dic.items(): if k not in header_dic: header_dic[k] = v return header_dic def get_score(text): text = text.strip() if re.search('^\d{1,2}(\.\d{2})$', text): return text elif re.search('^\d{1,2}(\.\d{2})?[\d,,;\.]*$', text): return text return '' result_l = [] datas = [] for i in df.index: line = get_header_line(df.loc[i].values) datas.append(line) header_row, header_col = get_header_index(datas) if len(header_col) == 1 and header_col[0][0] > 1: # 列表头不可能在第1列后面开始 header_col = [] if len(header_row) >= 1 and len(header_col) == 0: # 有行表头无列表头 i = 0 while i < len(header_row): idx, ratio = header_row[i] if idx + 1 >= len(df): break header_dic = get_header(df.loc[idx].values, self.head_rule_dic) i += 1 range_from = idx + 1 range_to = len(df) if i < len(header_row): next_header = i for j in range(i, len(header_row)): idx2, ratio2 = header_row[j] if idx2 - idx == 1: header_dic2 = get_header(df.loc[idx2].values, self.head_rule_dic) if set(df.loc[idx].values) & set(df.loc[idx2].values) != set(): header_dic.update(header_dic2) else: header_dic = header_dic2 range_from = idx2 + 1 range_to = len(df) next_header = j + 1 idx = idx2 else: range_from = idx + 1 range_to = idx2 next_header = j break i = next_header if len(header_dic) >= 2 and 'tenderer' in header_dic: for index in range(range_from, range_to): tmp_dic = {} for k, v in header_dic.items(): if k.startswith('score'): content = get_score(df.loc[index, v]) elif k == 'tenderer': content = get_role(df.loc[index, v], nlp_enterprise) elif k == 'ranking': content = df.loc[index, v] if re.search('^第?[\d一二三四五六七八九十]+名?$',df.loc[index, v]) else '' else: content = df.loc[index, v] if content != '': tmp_dic[k] = content if len(tmp_dic) > 1 and 'tenderer' in tmp_dic and tmp_dic not in result_l: result_l.append(tmp_dic) elif len(header_row) == 0 and len(header_col) >= 1: i = 0 while i < len(header_col): idx, ratio = header_col[i] if idx + 1 >= len(df.columns): break header_dic = get_header(df[idx].values, self.head_rule_dic) i += 1 range_from = idx + 1 range_to = len(df.columns) if i < len(header_col): next_header = i for j in range(i, len(header_col)): idx2, ratio2 = header_col[j] if idx2 - idx == 1: header_dic2 = get_header(df[idx2].values, self.head_rule_dic) if set(df[idx].values) & set(df[idx2].values) != set(): header_dic.update(header_dic2) else: header_dic = header_dic2 range_from = idx2 + 1 range_to = len(df.columns) next_header = j + 1 idx = idx2 else: range_from = idx + 1 range_to = idx2 next_header = j break i = next_header if len(header_dic.keys()&set(['tenderer','score_technical', 'score_commercial', 'score_price', 'score_comprehensive'])) >= 2 and 'tenderer' in header_dic: for index in range(range_from, range_to): tmp_dic = {} for k, v in header_dic.items(): if k.startswith('score'): content = get_score(df.loc[v, index]) elif k == 'tenderer': content = get_role(df.loc[v, index], nlp_enterprise) elif k == 'ranking': content = df.loc[v, index] if re.search('^第?[\d一二三四五六七八九十]+名?$', df.loc[v, index]) else '' else: content = df.loc[v, index] if content != '': tmp_dic[k] = content if len(tmp_dic) > 2 and 'tenderer' in tmp_dic and tmp_dic not in result_l: result_l.append(tmp_dic) elif len(header_row) == 1 and len(header_col) == 1: pass return result_l def predict(self, html, nlp_enterprise=[]): html = re.sub("|||", "", html) html = re.sub("##attachment##", "", html) soup = BeautifulSoup(html, 'lxml') richText = soup.find(name='div', attrs={'class': 'richTextFetch'}) self.nlp_enterprise = nlp_enterprise if richText: richText = richText.extract() # 过滤掉附件 tables = soup.find_all('table') if len(tables) == 0 and richText: tables = richText.find_all('table') tables.reverse() rs_dic = {} for table in tables: trs = self.tb.table2list(table) if len(trs)>1 and len(trs[0])>1 and len(set([len(tr) for tr in trs])) == 1: df = pd.DataFrame(trs) rs_l = self.get_table_info(df, nlp_enterprise) for d in rs_l: if d['tenderer'] not in rs_dic: rs_dic[d['tenderer']] = d elif len(d) > len(rs_dic[d['tenderer']]): rs_dic[d['tenderer']] = d table.extract() return list(rs_dic.values())