| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731732733734735736737738739740741742743744745746747748749750751752753754755756757758759760761762763764765766767768769770771772773774775776777778779780781782783784785786787788789790791792793794795796797798799800801802803804805806807808809810811812813814815816817818819820821822823824825826827828829830831832833834835836837838839840841842843844845846847848849850851852853854855856 |
- # -*- coding: utf-8 -*-
- """标签类预测器。
- 按 ARCHITECTURE.md Phase 5 拆分建议,从 ``interface/predictor.py`` 迁出以下
- 标签相关类:
- - ``ProjectLabel`` — 项目标签提取(原 predictor.py 第 5585-5891 行)
- - ``IndustryLabel`` — 行业标签分类(原 5895-5942 行)
- - ``PropertyLabel`` — 产权分类二级标签(原 6041-6201 行)
- - ``BiddingScore`` — 评分表投标人得分/排名提取(原 9835-10039 行)
- 原 ``from common.Utils import *`` / ``from interface.modelFactory import *``
- 已替换为显式 import;``os.path.dirname(__file__)`` 路径引用替换为
- ``predictors._common.INTERFACE_DIR``。
- ``interface/predictor.py`` 仍保留原定义,老 import 不受影响。
- """
- from __future__ import absolute_import
- import re
- import pandas as pd
- from bs4 import BeautifulSoup
- from BiddingKG.dl.interface.classification_process import classify_text, process_rules
- from BiddingKG.dl.predictors._common import INTERFACE_DIR, get_role
- from BiddingKG.dl.predictors.table_prem import TableTag2List, get_header_line
- __all__ = [
- "ProjectLabel",
- "IndustryLabel",
- "PropertyLabel",
- "BiddingScore",
- ]
- class ProjectLabel():
- def __init__(self, ):
- self.keyword_list = self.get_label_keywords()
- self.kongjing_keyword_list = self.get_kongjing_keywords()
- self.ICT_smart_compute_keyword_list = self.get_ICT_smart_compute_keywords()
- def get_label_keywords(self):
- import csv
- path = INTERFACE_DIR+'/project_label_keywords.csv'
- with open(path, 'r',encoding='utf-8') as f:
- reader = csv.reader(f)
- key_word_list = []
- for r in reader:
- if r[0] == '类型':
- continue
- type = r[0]
- key_wrod = r[1]
- key_paichuci = str(r[2])
- key_paichuci = key_paichuci if key_paichuci and key_paichuci != 'nan' else ""
- type_paichuci = str(r[3])
- type_paichuci = type_paichuci if type_paichuci and type_paichuci != 'nan' else ""
- key_word_list.append((type, key_wrod, key_paichuci, type_paichuci))
- return key_word_list
- def get_kongjing_keywords(self):
- import csv
- path = INTERFACE_DIR+'/kongjing_label_keywords.csv'
- with open(path, 'r',encoding='utf-8') as f:
- reader = csv.reader(f)
- key_word_list = []
- for r in reader:
- if r[0] == '关键词':
- continue
- key_wrod = r[0]
- key_wrod2 = str(r[1])
- key_wrod2 = key_wrod2 if key_wrod2 and key_wrod2 != 'nan' else ""
- search_type = r[2]
- info_type_list = str(r[3])
- info_type_list = info_type_list if info_type_list and info_type_list != 'nan' else ""
- key_word_list.append((key_wrod, key_wrod2, search_type, info_type_list))
- return key_word_list
- def get_ICT_smart_compute_keywords(self):
- import csv
- path = INTERFACE_DIR+'/ICT智算_label_keywords.csv'
- with open(path, 'r',encoding='utf-8') as f:
- reader = csv.reader(f)
- key_word_list = []
- for r in reader:
- if r[0] == '关键词':
- continue
- key_wrod = r[0]
- key_wrod2 = str(r[1])
- key_wrod2 = key_wrod2 if key_wrod2 and key_wrod2 != 'nan' else ""
- search_type = r[2]
- info_type_list = str(r[3])
- info_type_list = info_type_list if info_type_list and info_type_list != 'nan' else ""
- key_word_list.append((key_wrod, key_wrod2, search_type, info_type_list))
- return key_word_list
- def extract_core_text(self,all_text,tenderee="",agency=""):
- # 剔除 招标单位、代理机构名称
- if tenderee:
- all_text = all_text.replace(tenderee, " ")
- if agency:
- all_text = all_text.replace(agency, " ")
- # 定义需要匹配的关键词列表
- keywords = [
- '项目名称', '工程名称', '采购名称', '标段名称', '项目的名称', '设备名称', '申购主题',
- '申购单主题', '标的', '商品名称', '二级目录', '招标内容', '项目内容', '商品清单',
- '标的名称', '采购内容', '集成要求', '概况介绍', '品目分类', '招标范围', '采购范围',
- '项目采购分', '采购合同', '招标合同','产品名称','服务内容','采购品目名称','货物名称',
- '采购需求概况','项目概况','招标范围','采购条目名称','物资名称','物料名称','建设规模',
- '建设内容','采购项目概况','采购包名称','物料描述','商品信息','服务品目','标项名称',
- '规格描述','采购标的','服务名称','采购单名称','明细信息','申购主题','需求详情'
- ]
- # 创建正则表达式模式,匹配任意一个关键词
- pattern = r'(' + '|'.join(re.escape(kw) for kw in keywords) + r')'
- # 查找所有匹配位置
- matches = list(re.finditer(pattern, all_text))
- if not matches:
- return "" # 没有找到关键词
- all_match_text = []
- for _match in matches:
- start_pos = _match.end() # 关键词结束位置
- # 提取关键词之后的内容
- after_text = all_text[start_pos:]
- # 提取最多45个汉字
- chinese_chars = []
- count = 0
- for char in after_text:
- # 判断是否为汉字 (Unicode范围)
- if '\u4e00' <= char <= '\u9fff':
- count += 1
- if count > 45:
- break
- chinese_chars.append(char)
- chinese_chars = ''.join(chinese_chars).strip()
- if chinese_chars and chinese_chars not in all_match_text:
- all_match_text.append(chinese_chars)
- # 将字符列表组合成字符串
- main_content_text = ','.join(all_match_text).strip()
- return main_content_text
- def predict(self, doctitle,product,project_name,prem,all_text):
- doctitle = doctitle if doctitle else ""
- product = product if product else ""
- product = ",".join(set(product.split(','))) # 产品词去重
- project_name = project_name if project_name else ""
- tenderee = ""
- agency = ""
- sub_project_names = [] # 标段名称
- try:
- for k,v in prem[0]['prem'].items():
- # sub_project_names.append(k)
- sub_project_names.append(v.get("name",""))
- for link in v['roleList']:
- if link['role_name'] == 'tenderee' and tenderee == "":
- tenderee = link['role_text']
- if link['role_name'] == 'agency' and agency == "":
- agency = link['role_text']
- except Exception as e:
- # print('解析prem 获取招标人、代理人出错')
- pass
- sub_project_names = ";".join(sub_project_names)
- main_content_text = self.extract_core_text(all_text,tenderee,agency)
- # 核心字段:标题+产品词+项目名称+标段名称
- # main_text = ",".join([doctitle, product, project_name, sub_project_names])
- # 核心字段:标题+项目名称+产品词+正文定位词后45个字
- main_text = ",".join([doctitle, project_name, product, main_content_text])
- # 剔除 招标单位、代理机构名称
- if tenderee:
- doctitle = doctitle.replace(tenderee, " ")
- main_text = main_text.replace(tenderee, " ")
- if agency:
- doctitle = doctitle.replace(agency, " ")
- main_text = main_text.replace(agency, " ")
- doctitle_dict = dict()
- main_text_dict = dict()
- for item in self.keyword_list:
- _type = item[0]
- key_wrod = item[1]
- # 关键词排除词
- key_paichuci = item[2]
- key_paichuci_s = "|".join([re.escape(word) for word in key_paichuci.strip('、').split('、')])
- # 类型排除词
- type_paichuci = item[3]
- if type_paichuci:
- paichuci_split = type_paichuci.strip('、').split('、')
- if re.search("|".join([re.escape(word) for word in paichuci_split]), main_text):
- continue
- if doctitle:
- if key_wrod in doctitle:
- if not key_paichuci_s or (key_paichuci_s and not re.search(key_paichuci_s, doctitle)):
- key_wrod_count1 = doctitle.count(key_wrod)
- if _type not in doctitle_dict:
- # doctitle_dict[_type] = {'关键词': [], '排除词': type_paichuci}
- doctitle_dict[_type] = []
- doctitle_dict[_type].append((key_wrod, key_wrod_count1))
- if main_text:
- if key_wrod in main_text:
- if not key_paichuci_s or (key_paichuci_s and not re.search(key_paichuci_s, main_text)):
- key_wrod_count2 = main_text.count(key_wrod)
- if _type not in main_text_dict:
- # main_text_dict[_type] = {'关键词': [], '排除词': type_paichuci}
- main_text_dict[_type] = []
- main_text_dict[_type].append((key_wrod, key_wrod_count2))
- # 排序 doctitle
- for k, v in doctitle_dict.items():
- doctitle_dict[k].sort(key=lambda x: x[1], reverse=True)
- # 按匹配次数保留前10个标签
- if len(doctitle_dict) > 10:
- doctitle_labels = [(k, sum(w[1] for w in doctitle_dict[k])) for k in doctitle_dict]
- doctitle_labels.sort(key=lambda x: x[1], reverse=True)
- for item in doctitle_labels[10:]:
- doctitle_dict.pop(item[0])
- # main_text
- pop_list = []
- for k, v in main_text_dict.items():
- if sum([j[1] for j in main_text_dict[k]]) == 1:
- # 关键词匹配次数等于1的标签
- pop_list.append(k)
- main_text_dict[k].sort(key=lambda x: x[1], reverse=True)
- # 核心字段标签,若存在同一个标签的关键词匹配次数大于1,则只保留关键词匹配次数大于1的标签,关键词匹配次数等于1的标签不要
- if len(pop_list) < len(main_text_dict):
- for k in pop_list:
- main_text_dict.pop(k)
- # 按匹配次数保留前10个标签
- if len(main_text_dict) > 10:
- main_text_labels = [(k, sum(w[1] for w in main_text_dict[k])) for k in main_text_dict]
- main_text_labels.sort(key=lambda x: x[1], reverse=True)
- for item in main_text_labels[10:]:
- main_text_dict.pop(item[0])
- return {"标题":doctitle_dict,"核心字段":main_text_dict},main_content_text
- def predict_other(self,project_label,industry,doctitle,project_name,product,list_articles,main_content_text):
- # doctextcon 取正文内容
- doctextcon = list_articles[0].content.split('##attachment##')[0]
- info_type = industry.get('industry',{}).get("class_name","")
- doctitle = doctitle if doctitle else ""
- product = product if product else ""
- product = ",".join(set(product.split(','))) # 产品词去重
- project_name = project_name if project_name else ""
- main_content_text = main_content_text if main_content_text else ""
- # 空净通
- get_kongjing_label = False
- keywords_list = []
- for item in self.kongjing_keyword_list:
- key_wrod = item[0]
- key_wrod2 = item[1]
- search_type = item[2]
- info_type_list = item[3]
- info_type_list = info_type_list.strip('|').split("|") if info_type_list else []
- search_text = ""
- if search_type=='正文':
- search_text = ",".join([doctextcon,doctitle,project_name,product])
- elif search_type=='产品':
- search_text = ",".join([doctitle,project_name,product])
- if search_type=='行业':
- # ’行业’类型直接用info_type匹配关键词
- if info_type==key_wrod:
- # 匹配关键词记录
- keywords_list.append(key_wrod)
- get_kongjing_label = True
- break
- else:
- if key_wrod in search_text:
- if key_wrod2 and key_wrod2 not in search_text:
- continue
- if info_type_list and info_type not in info_type_list:
- continue
- # 匹配关键词记录
- if key_wrod2:
- keywords_list.append(key_wrod+'+'+key_wrod2)
- else:
- keywords_list.append(key_wrod)
- get_kongjing_label = True
- break
- if get_kongjing_label:
- project_label["核心字段"]["空净通"] = [[word,1] for word in keywords_list][:10]
- # ICT智算
- get_ICT_label = False
- keywords_list = []
- for item in self.ICT_smart_compute_keyword_list:
- key_wrod = item[0]
- key_wrod2 = item[1]
- search_type = item[2]
- info_type_list = item[3]
- info_type_list = info_type_list.strip('|').split("|") if info_type_list else []
- search_text = ""
- if search_type=='正文':
- search_text = ",".join([doctextcon,doctitle,project_name,product])
- elif search_type=='产品':
- search_text = ",".join([doctitle,project_name,product])
- elif search_type=='core_text':
- # core_text规则:标题+项目名称+产品词+正文定位词后45个字
- search_text = doctitle + ',' + project_name + "," + product + "," + main_content_text
- if search_type=='行业':
- # ’行业’类型直接用info_type匹配关键词
- if info_type==key_wrod:
- # 匹配关键词记录
- keywords_list.append(key_wrod)
- get_ICT_label = True
- break
- else:
- if key_wrod in search_text:
- if key_wrod2 and key_wrod2 not in search_text:
- continue
- if info_type_list and info_type not in info_type_list:
- continue
- # 匹配关键词记录
- if key_wrod2:
- keywords_list.append(key_wrod+'+'+key_wrod2)
- else:
- keywords_list.append(key_wrod)
- get_ICT_label = True
- break
- if info_type in ['计算机设备','监控设备','通信设备','信息系统集成和物联网技术服务','运行维护服务','信息处理和存储支持服务','互联网安全服务','互联网接入及相关服务','电信']:
- # 新增规则info_type符合范围直接判定
- get_ICT_label = True
- if get_ICT_label:
- project_label["核心字段"]["ICT智算"] = [[word,1] for word in keywords_list][:10]
- return project_label
- # 行业标签
- class IndustryLabel():
- def __init__(self):
- # self.keyword_list = self.get_label_keywords()
- pass
- def predict(self,doctitle,article,product,prem):
- doctitle = doctitle if doctitle else ""
- product = product if product else ""
- product = ",".join(set(product.split(','))) # 产品词去重
- all_text = article.content
- all_text = re.sub('\s+', ' ', all_text)
- tenderee = ""
- agency = ""
- try:
- for k,v in prem[0]['prem'].items():
- for link in v['roleList']:
- if link['role_name'] == 'tenderee' and tenderee == "":
- tenderee = link['role_text']
- if link['role_name'] == 'agency' and agency == "":
- agency = link['role_text']
- except Exception as e:
- # print('解析prem 获取招标人、代理人出错')
- pass
- # 剔除 招标单位、代理机构名称
- if tenderee:
- doctitle = doctitle.replace(tenderee, " ")
- all_text = all_text.replace(tenderee, " ")
- if agency:
- doctitle = doctitle.replace(agency, " ")
- all_text = all_text.replace(agency, " ")
- # category_1, category_2, category_3, matched_keywords, rule_id = product_classify_process(doctitle, all_text, product)
- category_1, category_2, category_3, matched_keywords, rule_id = classify_text(doctitle, all_text, product)
- # print(category_1, category_2, category_3, matched_keywords, rule_id)
- tenderee_label = process_rules(tenderee)
- new_tenderee_label = []
- for k in tenderee_label.keys():
- if '-' in k:
- new_tenderee_label.append({"first_level":k.split('-')[0],"second_level":k.split('-')[1]})
- # new_tenderee_label.append({"first_level":k.split('-')[0],"second_level":k.split('-')[1],"code":k.split('-')[2]})
- # print(new_tenderee_label)
- if category_2=="标题排除":
- category_1 = "其他"
- category_2 = ""
- rule_id = ""
- return {"first_level":category_1,"second_level":category_2},new_tenderee_label
- # def get_label_keywords(self):
- # import csv
- # path = os.path.dirname(__file__)+'/industry_label_keywords.csv'
- # with open(path, 'r',encoding='utf-8') as f:
- # reader = csv.reader(f)
- # key_word_list = []
- # for r in reader:
- # if r[0] == '一级标签':
- # continue
- # first_level = r[0]
- # second_level = str(r[1])
- # second_level = second_level.strip() if second_level and second_level != 'nan' else ""
- # key_word = str(r[2]).strip()
- # all_paichuci = str(r[3])
- # all_paichuci = all_paichuci.strip() if all_paichuci and all_paichuci != 'nan' else ""
- # title_paichuci = str(r[4])
- # title_paichuci = title_paichuci.strip() if title_paichuci and title_paichuci != 'nan' else ""
- # product_paichuci = str(r[5])
- # product_paichuci = product_paichuci.strip() if product_paichuci and product_paichuci != 'nan' else ""
- # key_word_list.append((first_level, second_level, key_word, all_paichuci,title_paichuci,product_paichuci))
- # return key_word_list
- #
- # def predict(self, doctitle,article,product,prem):
- #
- # doctitle = doctitle if doctitle else ""
- # product = product if product else ""
- # product = ",".join(set(product.split(','))) # 产品词去重
- # all_text = article.content
- # tenderee = ""
- # agency = ""
- # try:
- # for k,v in prem[0]['prem'].items():
- # for link in v['roleList']:
- # if link['role_name'] == 'tenderee' and tenderee == "":
- # tenderee = link['role_text']
- # if link['role_name'] == 'agency' and agency == "":
- # agency = link['role_text']
- # except Exception as e:
- # # print('解析prem 获取招标人、代理人出错')
- # pass
- # # 剔除 招标单位、代理机构名称
- # if tenderee:
- # doctitle = doctitle.replace(tenderee, " ")
- # all_text = all_text.replace(tenderee, " ")
- # if agency:
- # doctitle = doctitle.replace(agency, " ")
- # all_text = all_text.replace(agency, " ")
- #
- # label_list = []
- # for item in self.keyword_list:
- # first_level = item[0]
- # second_level = item[1]
- # key_word = item[2]
- # key_word = key_word.strip('、').split('、')
- # # 全文排除词
- # all_paichuci = item[3]
- # all_paichuci = "|".join([re.escape(word) for word in all_paichuci.strip('、').split('、')])
- # # 标题排除词
- # title_paichuci = item[4]
- # title_paichuci = "|".join([re.escape(word) for word in title_paichuci.strip('、').split('、')])
- # # 产品排除词
- # product_paichuci = item[5]
- # product_paichuci = "|".join([re.escape(word) for word in product_paichuci.strip('、').split('、')])
- #
- #
- # if doctitle and title_paichuci:
- # if re.search(title_paichuci,doctitle):
- # continue
- # if product and product_paichuci:
- # if re.search(product_paichuci,product):
- # continue
- # if all_text:
- # if all_paichuci:
- # if re.search(all_paichuci,all_text):
- # continue
- # get_label = False
- # for _keyword in key_word:
- # if '+' not in _keyword:
- # if _keyword in all_text:
- # get_label = True
- # break
- # else:
- # get_keyword = True
- # for _word in _keyword.split("+"):
- # if _word not in all_text:
- # get_keyword = False
- # break
- # if get_keyword:
- # get_label = True
- # break
- # if get_label:
- # label_list.append({"first_level":first_level,"second_level":second_level})
- #
- # return label_list
- # 产权分类二级标签
- class PropertyLabel():
- '''
- 产权分类二级标签
- 全部类别:
- 股权, 债权, 知识产权, 矿权, 房产, 土地, 交通运输工具, 闲置物资、设备、材料, 其他
- '''
- def __init__(self, ):
- car = "比亚迪|奇瑞|奥迪|宝马|菲尼迪|雷克萨斯|三菱|铃木|马自达|奔驰|劳斯莱斯|北京现代|" \
- "宾利|兰博基尼|布加迪|保时捷|斯柯达|雪佛兰|别克|凯迪拉克|庞蒂亚克|克尔维特|福特|林肯|克莱斯勒|道奇|JEEP品牌"
- self.keywords_dict = {
- "房产": "房产|住宅|公寓|商铺|车位|写字楼|办公楼|别墅|综合楼|在建工程|厂房|车库|车房|房转让|房屋|商品房|商业用房|"
- "宅基地|[\u4e00-\u9fa5]{,2}用房|店面|商业房|门[面市]房|仓库|铺位|地下室|\d号?(房|室|门市|门面|商?铺|单元|户)|不动产|"
- "自建房|铺面|商务楼|商住楼|阁楼|(杂物|储物|储藏)(房|间|室)|套房|[\da-zA-Z](栋|棟|幢|层|座|号?楼|单元)\d{1,4}(号|房|室|商?铺|户)|"
- "[\da-zA-Z](栋|棟|幢|层|座|号?楼|单元)\d{2,}|门面+转让|楼+变卖|房地产",
- "交通运输工具": "车辆|轿车|汽车(?!用品|库|位|衡)|公车|客车|货车|面包车|SUV|新能源车|二手车|车辆|商用车|机动车|观光车|巴车|"
- "船舶|四驱" + "|" + car,
- "股权": "\d.?股|股权(?!交易中心)|\d%(比例)?.?股|\d万.?股|\d.?元/股|增资(?!源)|扩股|股(转让|出售)|百分之[一二三四五六七八九十]{1,3}股",
- "债权": "债权|债权转让|债权人|债务人|原债权人|新债权人|金融资产",
- "土地": "住宅用地|商业用地|工业用地|国有[\u4e00-\u9fa5]{,3}[土用]地|集体土地|划拨|流转|地块编号|"
- "土地使用权证|土地经营权|土地证|土地[发承]包|[\u4e00-\u9fa5]{,2}用地|土地\d{1,3}(亩|公?顷)|\d{1,3}(亩|公?顷)(使用|经营)权|"
- "承包土地|(地块|土地)承包|水面经营权|[鱼水]塘|鱼池|(水面|旱田)[\u4e00-\u9fa5]{,3}[发承]包|水面资源|(水面|水田)[\u4e00-\u9fa5]{,3}权|"
- "四荒|林地|林场|林木所有权|采伐权|水利设施所有权|水利设施使用权|海域|滩涂|林业产权|旱田|水田|机动田|机动地|耕地|荒地|农田|"
- "苗圃地|塘口",
- "矿权": "矿权|矿业权|采矿许可|探矿权|采矿权|开采权|矿产资源处置|矿[\u4e00-\u9fa5]{1,3}开[发采]",
- "知识产权": "知识产权(?!局)|商标|专利|著作权|版权|商业秘密|科研成果",
- "闲置物资、设备、材料": "(废旧|报废|废|闲置|二手|淘汰)(物资|资产|机械|设备|仪器|汽车|车|钢铁|钢材|钢|金属|塑料|材料|导管|漆|渣|有色|品|[\u4e00-\u9fa5]{,2}车|偶头)|"
- "(金属|机械|设备|仪器|汽车|钢铁|钢材|钢|塑料|有色|)废料|废液|废旧|报废|边角料|残次品|(热轧|冷轧|酸洗|镀铝|热镀|镀锌|镀镁)|"
- "机[器械]设备|医疗设备|生产设备|办公设备|仪器|仪表|设备出租|设备租赁|拖拉机|收割机|插秧机|挖机|车床|挖掘机|电机|"
- "戒指|弃渣|电解质块|茶杯|装置|花瓶|女表|手表|男表|硫磺|物资|书画|茶叶|油茶|红茶|[茗名]茶|白酒|红酒|酒水|酒品|名酒|毛石|[石金木铁矿铜锌铝钢]料|"
- "零部件",
- "经营权": "经营权",
- "租赁": "房+租|市场+续约|资产+出租|租赁|续租|招租|出租|租金|房租"
- }
- self.cqjy_keywords = self.get_cqjy_keywords()
- self.score_idx = ["股权", "债权", "知识产权", "矿权", "房产", "土地", "交通运输工具", "闲置物资、设备、材料"]
- def get_cqjy_keywords(self):
- import csv
- path = INTERFACE_DIR+'/property_label_products.csv'
- with open(path, 'r',encoding='utf-8') as f:
- reader = csv.reader(f)
- key_word_list = []
- for r in reader:
- if r[0] == 'product':
- continue
- key_wrod = r[0]
- _type = r[1]
- key_word_list.append((_type, key_wrod))
- return key_word_list
- def get_type(self, text):
- keyword_list = []
- for key, value in self.keywords_dict.items():
- keyword = "|".join([i for i in value.split("|") if '+' not in i])
- keyword2 = [i for i in value.split("|") if '+' in i]
- if re.search(keyword, text):
- re1 = [i for i in re.finditer(keyword, text)][-1]
- keyword_list.append((key, re1.start()))
- else:
- # 组合词 查询
- for k in keyword2:
- k1, k2 = k.split('+')
- if re.search(k1, text) and re.search(k2, text):
- keyword_list.append((key, re.search(k2, text).start()))
- break
- return keyword_list
- def get_type2(self, text, cqjy_type_list):
- have_type = [i[0] for i in cqjy_type_list]
- for item in self.cqjy_keywords:
- _type = item[0]
- key_wrod = item[1]
- if _type not in have_type:
- if '+' in key_wrod:
- k1, k2 = key_wrod.split('+')
- if re.search(k1, text) and re.search(k2, text):
- cqjy_type_list.append((_type, re.search(k2, text).start()))
- have_type.append(_type)
- else:
- if key_wrod in text:
- cqjy_type_list.append((_type, text.index(key_wrod)))
- have_type.append(_type)
- return cqjy_type_list
- def predict(self, doctitle,product,project_name,prem,channel_dic):
- docchannel = channel_dic['docchannel']['doctype']
- # print('docchannel',docchannel)
- if docchannel not in ['土地矿产', '拍卖出让', '产权交易']:
- return ""
- doctitle = doctitle if doctitle else ""
- product = product if product else ""
- product = ",".join(set(product.split(','))) # 产品词去重
- project_name = project_name if project_name else ""
- tenderee = ""
- agency = ""
- try:
- for k,v in prem[0]['prem'].items():
- for link in v['roleList']:
- if link['role_name'] == 'tenderee' and tenderee == "":
- tenderee = link['role_text']
- if link['role_name'] == 'agency' and agency == "":
- agency = link['role_text']
- except Exception as e:
- # print('解析prem 获取招标人、代理人出错')
- pass
- cqjy_type = []
- idx = 0
- for text in [doctitle, project_name, product]:
- if tenderee:
- text = text.replace(tenderee, "")
- if agency:
- text = text.replace(agency, "")
- cqjy_type = self.get_type(text)
- if not cqjy_type:
- cqjy_type = self.get_type2(text, cqjy_type)
- idx += 1
- if idx == 2: # project_name
- if len(re.split("[,、]", text)) > 9:
- cqjy_type = []
- if idx == 3: # product
- if len(text.split(",")) > 15:
- cqjy_type = []
- if cqjy_type:
- break
- cqjy_type2 = [i[0] for i in cqjy_type]
- if cqjy_type:
- # 类别优先级调整
- if "租赁" in cqjy_type2:
- cqjy_type2 = ['租赁']
- elif "经营权" in cqjy_type2:
- cqjy_type2 = ['经营权']
- elif "股权" in cqjy_type2 or "债权" in cqjy_type2 or "知识产权" in cqjy_type2:
- cqjy_type.sort(key=lambda x: self.score_idx.index(x[0]))
- cqjy_type = cqjy_type[0]
- cqjy_type2 = [cqjy_type[0]]
- elif len(cqjy_type2) == 2 and "房产" in cqjy_type2 and "土地" in cqjy_type2:
- cqjy_type2 = ['房产']
- else:
- # 权重排序,取第一位
- if idx in [1, 2]: # doctitle, project_name
- cqjy_type.sort(key=lambda x: x[1], reverse=True)
- cqjy_type = cqjy_type[0]
- cqjy_type2 = [cqjy_type[0]]
- else:
- cqjy_type.sort(key=lambda x: self.score_idx.index(x[0]))
- cqjy_type = cqjy_type[0]
- cqjy_type2 = [cqjy_type[0]]
- cqjy_type2 = ",".join(cqjy_type2)
- if not cqjy_type2:
- cqjy_type2 = '其他'
- return cqjy_type2
- class BiddingScore():
- def __init__(self):
- self.head_rule_dic = {
- "tenderer": "((候选|入围|入选|投标|应答|响应)(供应商库)?的?(人|人?单位|机构|供应商|供货商|服务商|投标人|(中标)?公司|(中标)?企业|银行)|(通过)?名单|中标候选人)(名称|名单|全称|\d)?$|^供应商(名称|信息)?$|投标个人/单位", #补充 368295593 投标个人/单位 提取
- "score_price": "(价格|报价|单价|总价|经济)(部分|\w{,2})?([得评]分|评审)",
- "score_technical": "技术(部分|\w{,2})?标?([得评]分|评审)",
- "score_commercial": "商务(部分|\w{,2})?标?([得评]分|评审)",
- "score_integrity": "诚信(部分|\w{,2})?([得评]分|评审)",
- "score_comprehensive": "(综合(标|评估)?|总|最终)得?分$",
- "ranking": "(得分)?排名",
- "qualification_review": "资格性审查|是否通过资格",
- "compliance_review": "符合性审查|是否通过符合"
- }
- self.tb = TableTag2List()
- def get_table_info(self, df, nlp_enterprise):
- def get_header_index(datas):
- '''
- 根据表格表头判断结果0/1 得到哪些行和列是表头
- :param datas: 表格内容表头判断结果数据[[1,1,1,1],[0,0,0,0]]
- :return: 表头所在的行和列序号
- '''
- header_row = []
- header_col = []
- df_h = pd.DataFrame(datas) # 表头判断数据 , columns=columns
- for i in df_h.index:
- line = df_h.loc[i].values
- if sum(line) == len(line):
- header_row.append((i, sum(line) / len(line)))
- elif sum(line) / len(line) > 0.8:
- header_row.append((i, sum(line) / len(line)))
- elif len(line) > 3 and len(re.findall('11', ''.join([str(it) for it in line]))) > len(
- re.findall('10', ''.join([str(it) for it in line]))):
- header_row.append((i, sum(line) / len(line)))
- for i in df_h.columns:
- col = df_h[i].values
- if sum(col) == len(col):
- header_col.append((i, sum(col) / len(col)))
- elif sum(col) / len(col) > 0.8:
- header_col.append((i, sum(col) / len(col)))
- elif len(col) > 3 and len(re.findall('11', ''.join([str(it) for it in line]))) > len(
- re.findall('10', ''.join([str(it) for it in line]))):
- header_col.append((i, sum(col) / len(col)))
- return header_row, header_col
- def get_header(l, head_rule_dic):
- header_dic = {}
- for i in range(len(l)):
- text = l[i]
- num = 0
- tmp_dic = {}
- for k, v in head_rule_dic.items():
- # print('k : ', k)
- if re.search(v, text):
- tmp_dic[k] = i
- num += 1
- # if num > 1:
- # if tmp_dic.keys() == set(['qualification_review', 'compliance_review']):
- # for k, v in tmp_dic.items():
- # if k not in header_dic:
- # header_dic[k] = v
- # elif tmp_dic:
- for k, v in tmp_dic.items():
- if k not in header_dic:
- header_dic[k] = v
- return header_dic
- def get_score(text):
- text = text.strip()
- if re.search('^\d{1,2}(\.\d{2})$', text):
- return text
- elif re.search('^\d{1,2}(\.\d{2})?[\d,,;\.]*$', text):
- return text
- return ''
- result_l = []
- datas = []
- for i in df.index:
- line = get_header_line(df.loc[i].values)
- datas.append(line)
- header_row, header_col = get_header_index(datas)
- if len(header_col) == 1 and header_col[0][0] > 1: # 列表头不可能在第1列后面开始
- header_col = []
- if len(header_row) >= 1 and len(header_col) == 0: # 有行表头无列表头
- i = 0
- while i < len(header_row):
- idx, ratio = header_row[i]
- if idx + 1 >= len(df):
- break
- header_dic = get_header(df.loc[idx].values, self.head_rule_dic)
- i += 1
- range_from = idx + 1
- range_to = len(df)
- if i < len(header_row):
- next_header = i
- for j in range(i, len(header_row)):
- idx2, ratio2 = header_row[j]
- if idx2 - idx == 1:
- header_dic2 = get_header(df.loc[idx2].values, self.head_rule_dic)
- if set(df.loc[idx].values) & set(df.loc[idx2].values) != set():
- header_dic.update(header_dic2)
- else:
- header_dic = header_dic2
- range_from = idx2 + 1
- range_to = len(df)
- next_header = j + 1
- idx = idx2
- else:
- range_from = idx + 1
- range_to = idx2
- next_header = j
- break
- i = next_header
- if len(header_dic) >= 2 and 'tenderer' in header_dic:
- for index in range(range_from, range_to):
- tmp_dic = {}
- for k, v in header_dic.items():
- if k.startswith('score'):
- content = get_score(df.loc[index, v])
- elif k == 'tenderer':
- content = get_role(df.loc[index, v], nlp_enterprise)
- elif k == 'ranking':
- content = df.loc[index, v] if re.search('^第?[\d一二三四五六七八九十]+名?$',df.loc[index, v]) else ''
- else:
- content = df.loc[index, v]
- if content != '':
- tmp_dic[k] = content
- if len(tmp_dic) > 1 and 'tenderer' in tmp_dic and tmp_dic not in result_l:
- result_l.append(tmp_dic)
- elif len(header_row) == 0 and len(header_col) >= 1:
- i = 0
- while i < len(header_col):
- idx, ratio = header_col[i]
- if idx + 1 >= len(df.columns):
- break
- header_dic = get_header(df[idx].values, self.head_rule_dic)
- i += 1
- range_from = idx + 1
- range_to = len(df.columns)
- if i < len(header_col):
- next_header = i
- for j in range(i, len(header_col)):
- idx2, ratio2 = header_col[j]
- if idx2 - idx == 1:
- header_dic2 = get_header(df[idx2].values, self.head_rule_dic)
- if set(df[idx].values) & set(df[idx2].values) != set():
- header_dic.update(header_dic2)
- else:
- header_dic = header_dic2
- range_from = idx2 + 1
- range_to = len(df.columns)
- next_header = j + 1
- idx = idx2
- else:
- range_from = idx + 1
- range_to = idx2
- next_header = j
- break
- i = next_header
- if len(header_dic.keys()&set(['tenderer','score_technical', 'score_commercial', 'score_price', 'score_comprehensive'])) >= 2 and 'tenderer' in header_dic:
- for index in range(range_from, range_to):
- tmp_dic = {}
- for k, v in header_dic.items():
- if k.startswith('score'):
- content = get_score(df.loc[v, index])
- elif k == 'tenderer':
- content = get_role(df.loc[v, index], nlp_enterprise)
- elif k == 'ranking':
- content = df.loc[v, index] if re.search('^第?[\d一二三四五六七八九十]+名?$', df.loc[v, index]) else ''
- else:
- content = df.loc[v, index]
- if content != '':
- tmp_dic[k] = content
- if len(tmp_dic) > 2 and 'tenderer' in tmp_dic and tmp_dic not in result_l:
- result_l.append(tmp_dic)
- elif len(header_row) == 1 and len(header_col) == 1:
- pass
- return result_l
- def predict(self, html, nlp_enterprise=[]):
- html = re.sub("<html>|</html>|<body>|</body>", "", html)
- html = re.sub("##attachment##", "", html)
- soup = BeautifulSoup(html, 'lxml')
- richText = soup.find(name='div', attrs={'class': 'richTextFetch'})
- self.nlp_enterprise = nlp_enterprise
- if richText:
- richText = richText.extract() # 过滤掉附件
- tables = soup.find_all('table')
- if len(tables) == 0 and richText:
- tables = richText.find_all('table')
- tables.reverse()
- rs_dic = {}
- for table in tables:
- trs = self.tb.table2list(table)
- if len(trs)>1 and len(trs[0])>1 and len(set([len(tr) for tr in trs])) == 1:
- df = pd.DataFrame(trs)
- rs_l = self.get_table_info(df, nlp_enterprise)
- for d in rs_l:
- if d['tenderer'] not in rs_dic:
- rs_dic[d['tenderer']] = d
- elif len(d) > len(rs_dic[d['tenderer']]):
- rs_dic[d['tenderer']] = d
- table.extract()
- return list(rs_dic.values())
|