| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731732733734735736737738739740741742743744745746747748749750751752753754755756757758759760761762763764765766767768769770771772773774775776777778779780781782783784785786787788789790791792793794795796797798799800801802803804805806807808809810811812813814815816817818819820821822823824825826827828829830831832833834835836837838839840841842843844845846847848849850851852853854855856857858859860861862863864865866867868869870871872873874875876877878879880881882883884885886887888889890891892 |
- # -*- coding: utf-8 -*-
- '''
- Created on 2019年1月4日
- @author: User
- '''
- import os
- from bs4 import BeautifulSoup, Comment
- import copy
- import re
- import sys
- import os
- import codecs
- import requests
- import time
- from unicodedata import normalize
- _time1 = time.time()
- sys.path.append(os.path.abspath("../.."))
- from BiddingKG.dl.common.Utils import *
- import BiddingKG.dl.entityLink.entityLink as entityLink
- import BiddingKG.dl.interface.predictor as predictor
- import BiddingKG.dl.interface.Preprocessing as Preprocessing
- import BiddingKG.dl.interface.getAttributes as getAttributes
- import BiddingKG.dl.complaint.punish_predictor as punish_rule
- import json
- from BiddingKG.dl.money.re_money_total_unit import extract_total_money, extract_unit_money
- from BiddingKG.dl.ratio.re_ratio import extract_ratio
- from BiddingKG.dl.interface.outline_extractor import ParseDocument, extract_parameters, extract_sentence_list, extract_addr
- from BiddingKG.dl.interface.get_label_dic import get_all_label
- from BiddingKG.dl.channel.channel_bert import merge_channel
- from BiddingKG.dl.interface.kvtree_search import get_kvtree_value
- from BiddingKG.dl.interface.special_debt_extract import get_debt_info
- from BiddingKG.dl.fingerprint.documentFingerprint import getFingerprint
- import BiddingKG.dl.template_extract.template_extractor as template_extractor
- import BiddingKG.dl.template_extract.common_table_prem_extractor as common_table_prem_extractor
- from BiddingKG.dl.interface.error_attachments_filter import delete_error_attachment
- from BiddingKG.dl.predictors.role_context import build_role_contexts
- # 自定义jsonEncoder
- class MyEncoder(json.JSONEncoder):
- def default(self, obj):
- if isinstance(obj, np.ndarray):
- return obj.tolist()
- elif isinstance(obj, bytes):
- return str(obj, encoding='utf-8')
- elif isinstance(obj, (np.float_, np.float16, np.float32,
- np.float64)):
- return float(obj)
- elif isinstance(obj,str):
- return obj
- return json.JSONEncoder.default(self, obj)
- def get_login_web_set():
- file = os.path.join(os.path.dirname(__file__),"login_weblist.txt")
- list_web = []
- try:
- if os.path.exists(file):
- with open(file,"r",encoding="utf8") as f:
- while 1:
- line = f.readline()
- if not line:
- break
- line = line.strip()
- if line:
- list_web.append(line)
- except Exception as e:
- traceback.print_exc()
- _set = set(list_web)
- log("get_login_web_set length %d"%(len(_set)))
- return _set
- set_login_web = get_login_web_set()
- def extractCount(extract_dict,page_attachments,web_source_name,page_time):
- # time_pattern = "\d{4}\-\d{2}\-\d{2}.*"
- if len(extract_dict):
- _extract = extract_dict
- else:
- _extract = {}
- # print(_extract)
- dict_pack = _extract.get("prem",{})
- extract_count = 0
- list_code = _extract.get("code",[])
- word_count = _extract.get("word_count",{})
- if word_count.get("正文",0)>500:
- extract_count += 3
- if len(list_code)>0:
- project_code = list_code[0]
- else:
- project_code = ""
- project_name = _extract.get("name","")
- bidding_budget = ""
- win_tenderer = ""
- win_bid_price = ""
- linklist_count = 0
- time_getFileEnd = _extract.get("time_getFileEnd","")
- time_bidopen = _extract.get("time_bidopen","")
- time_bidclose = _extract.get("time_bidclose","")
- time_publicityEnd = _extract.get("time_publicityEnd","")
- docchannel_dict = _extract.get("docchannel",{})
- if page_time!="":
- if docchannel_dict.get("docchannel","") in ["招标公告","招标预告","招标文件","采购意向"] and docchannel_dict.get("doctype","")=="采招数据":
- if time_getFileEnd!="" and time_getFileEnd[:10]<page_time:
- extract_count -= 5
- if time_bidopen!="" and time_bidopen[:10]<page_time:
- extract_count -= 5
- if time_bidclose!="" and time_bidclose[:10]<page_time:
- extract_count -= 5
- if time_publicityEnd!="" and time_publicityEnd[:10]<page_time:
- extract_count -= 5
- for _key in dict_pack.keys():
- if "tendereeMoney" in dict_pack[_key] and dict_pack[_key]["tendereeMoney"]!='' and float(dict_pack[_key]["tendereeMoney"])>0:
- extract_count += 1
- if bidding_budget=="":
- bidding_budget = str(float(dict_pack[_key]["tendereeMoney"]))
- for _role in dict_pack[_key]["roleList"]:
- if isinstance(_role,list):
- extract_count += 1
- if _role[2]!='' and float(_role[2])>0:
- extract_count += 1
- if _role[0]=="tenderee":
- tenderee = _role[1]
- if _role[0]=="win_tenderer":
- if _role[1] is not None and _role[1]!="":
- extract_count += 2
- if win_tenderer=="":
- win_tenderer = _role[1]
- if _role[2]!='' and float(_role[2])>0:
- extract_count += 2
- if win_bid_price=="":
- win_bid_price = str(float(_role[2]))
- if _role[0]=="agency":
- agency = _role[1]
- if isinstance(_role,dict):
- extract_count += 1
- if "role_money" in _role:
- if str(_role["role_money"].get("money",""))!='' and float(_role["role_money"].get("money",""))>0:
- extract_count += 1
- if _role.get("role_name")=="tenderee":
- tenderee = _role["role_text"]
- if _role.get("role_name")=="win_tenderer":
- if _role["role_text"] is not None and _role["role_text"]!="":
- extract_count += 2
- if win_tenderer=="":
- win_tenderer = _role["role_text"]
- if "role_money" in _role:
- if str(_role["role_money"]["money"])!='' and float(_role["role_money"]["money"])>0:
- extract_count += 2
- if win_bid_price=="":
- win_bid_price = str(float(_role["role_money"]["money"]))
- if _role["role_name"]=="agency":
- agency = _role["role_text"]
- linklist = _role.get("linklist",[])
- for link in linklist:
- for l in link:
- if l!="":
- linklist_count += 1
- extract_count += linklist_count//2
- if project_code!="":
- extract_count += 1
- if project_name!="":
- extract_count += 1
- if page_attachments is not None and page_attachments!='':
- try:
- _attachments = json.loads(page_attachments)
- set_md5 = set()
- has_zhaobiao = False
- has_qingdan = False
- if len(_attachments)>0:
- for _atta in _attachments:
- classification = _atta.get("classification","")
- set_md5.add(_atta.get("fileMd5"))
- if str(classification)=='招标文件':
- has_zhaobiao = True
- if str(classification)=='采购清单':
- has_qingdan = True
- extract_count += len(set_md5)//2+1
- if has_zhaobiao:
- extract_count += 2
- if has_qingdan:
- extract_count += 1
- except Exception as e:
- traceback.print_exc()
- pass
- list_approval_dict = _extract.get("approval",[])
- for _dict in list_approval_dict:
- for k,v in _dict.items():
- if v is not None and v!='' and v!="未知":
- extract_count += 1
- punish_dict = _extract.get("punish",{})
- for k,v in punish_dict.items():
- if v is not None and v!='' and v!="未知":
- extract_count += 1
- if web_source_name in set_login_web:
- extract_count -= 3
- product = _extract.get("product","")
- extract_count += len(str(product).split(","))//5
- product_attrs = _extract.get("product_attrs",{})
- product_attrs_data = product_attrs.get("data",[])
- product_attrs_len = 0
- for _product in product_attrs_data:
- if _product.get("brand","") or _product.get("specs",""):
- # 有品牌型号的产品
- product_attrs_len += 1
- if product_attrs_len>0:
- extract_count += max((product_attrs_len//5)//2,1)
- return extract_count
- def time_entity_check(prem,channel_dic,page_time):
- # print('time_dict:',[(k,v) for k,v in prem[0].items() if re.search("^time_",k) and v])
- time_bidclose = prem[0].get("time_bidclose","")
- time_bidopen = prem[0].get("time_bidopen","")
- time_getFileStart = prem[0].get("time_getFileStart","")
- time_registrationEnd = prem[0].get("time_registrationEnd","")
- if page_time:
- if channel_dic['docchannel']['docchannel'] in ["招标公告","招标预告","招标文件","采购意向"]:
- if time_bidclose and time_bidclose[:10]<=page_time:
- prem[0]["time_bidclose"] = ""
- time_bidclose = ""
- if time_bidopen and time_bidopen[:10]<=page_time:
- prem[0]["time_bidopen"] = ""
- time_bidopen = ""
- if time_getFileStart and time_getFileStart[:10]<=page_time:
- prem[0]["time_getFileStart"] = ""
- time_getFileStart = ""
- if time_registrationEnd and time_registrationEnd[:10]<=page_time:
- prem[0]["time_registrationEnd"] = ""
- time_registrationEnd = ""
- if time_bidopen and (time_bidclose or time_registrationEnd):
- if time_bidclose:
- if time_bidclose[:10]>=time_registrationEnd[:10] and time_bidopen[:10]<time_bidclose[:10]:
- prem[0]["time_bidopen"] = ""
- time_bidopen = ""
- elif time_registrationEnd:
- if time_bidopen[:10]<time_registrationEnd[:10]:
- prem[0]["time_bidopen"] = ""
- time_bidopen = ""
- return prem
- # 字符编码标准化
- def str_normalize(text):
- # time1 = time.time()
- cn_punctuation = "¥,。:;{}!?()<ⅠⅡⅢⅣⅤⅥⅦⅧⅨⅩⅪⅫ①②③④⑤⑥⑦⑧⑨⑩⑪⑫⑬⑭⑮⑯⑰⑱⑲⑳"
- text_split = re.split("([{}])+".format(cn_punctuation),text)
- # print(text_split)
- new_text = ""
- for s in text_split:
- if re.search("^[{}]+$".format(cn_punctuation),s):
- new_text += s
- else:
- new_text += normalize('NFKD', s)
- # print("str_normalize cost time %s"%str(time.time()-time1))
- # print(new_text)
- return new_text
- # 修复prem中地区前缀不完整实体
- def repair_entity(prem,district_dict,list_articles):
- district_dict = district_dict['district']
- province = district_dict['province'] if district_dict['province'] and district_dict['province'] not in ['未知','全国'] else ""
- city = district_dict['city'] if district_dict['city'] and district_dict['city']!='未知' else ""
- district = district_dict['district'] if district_dict['district'] and district_dict['district']!='未知' else ""
- content_text = list_articles[0].content
- autonomous_region_dict = {
- "新疆":"新疆维吾尔",
- "西藏":"西藏",
- "内蒙古":"内蒙古",
- "广西":"广西壮族",
- "宁夏":"宁夏回族"
- }
- for package,_prem in prem[0]['prem'].items():
- for role in _prem['roleList']:
- if role['role_name'] in ['tenderee','agency']:
- role_text = role['role_text']
- if re.search("^[省市县区]",role_text):
- if role_text[0]=='省' and role_text[:2] not in ['省道']:
- role['role_text'] = province + role_text
- elif role_text[0]=='市' and role_text[:2] not in ['市政','市场']:
- if district+'市' in content_text:
- # 县级市
- role['role_text'] = district + role_text
- else:
- role['role_text'] = city + role_text
- elif role_text[0] in ['县','区']:
- role['role_text'] = district + role_text
- elif re.search("^自治[区州县]",role_text):
- if role_text[:3]=='自治区':
- role['role_text'] = autonomous_region_dict.get(province,"") + role_text
- elif role_text[:3] in ['自治县',"自治州"]:
- if re.search("自治[县州]?$",district):
- role['role_text'] = re.sub("自治[县州]?","",district) + role_text
- elif re.search("族$",district):
- role['role_text'] = district + role_text
- elif re.search("自治[县州]?$",city):
- role['role_text'] = re.sub("自治[县州]?","",city) + role_text
- elif re.search("族$",city):
- role['role_text'] = city + role_text
- def fix_table_structure_preserve_order(html):
- """
- 修复table结构中tr与tbody平级的问题
- 保持原有行顺序不变
- """
- soup = BeautifulSoup(html, 'html.parser')
- for table in soup.find_all('table'):
- if table.find_all('tr', recursive=False) != []:
- # 获取table下所有直接子节点
- children = list(table.children)
- tbody_new = soup.new_tag('tbody')
- table.append(tbody_new)
- for child in children:
- if child.name:
- if child.name == 'tbody':
- for tag in list(child.children):
- tbody_new.append(tag.extract())
- child.extract()
- else:
- tbody_new.append(child.extract())
- return str(soup)
- def clean_binary_chars(text):
- """
- 移除可能被识别为二进制的字符,包括:
- - 八进制转义控制字符(\00-\377)
- - 十六进制转义控制字符(\x00-\x1F、\x7F-\xFF)
- - ASCII 不可见控制字符(0-31、127)
- - 无效的多字节编码字符(如�)
- """
- if not isinstance(text, str):
- text = str(text) # 确保输入为字符串
- # 组合正则规则,覆盖更多场景
- pattern = re.compile(r'''
- # 八进制转义符(\ddd,如\037)
- \\[0-7]{1,3}
- |
- # 十六进制转义符(\xhh,如\x0B、\xFF)
- \\x[0-9A-Fa-f]{2}
- |
- # ASCII控制字符(0-31、127)
- [\x00-\x1F\x7F]
- |
- # Unicode替换字符(�,表示编码错误)
- \ufffd
- |
- # 高字节非文本字符(\x80-\xFF,排除常见符号)
- [\x80-\xFF](?![\x80-\xFF]) # 单独出现的高字节(非多字节序列)
- |
- ?+ # 多个重复问号异常数据
- ''', re.VERBOSE)
- cleaned_text = pattern.sub('', text)
- # 附件中标签参数转义内容,例如:<td colspan='\"1\"' rowspan='\"2\"'>
- cleaned_text = cleaned_text.replace("'\\\"",'"').replace("\\\"'",'"')
- return cleaned_text
- def predict(doc_id,text,title="",page_time="",web_source_no='',web_source_name="",original_docchannel='',page_attachments='[]',**kwargs):
- cost_time = dict()
- if web_source_no == None:
- web_source_no = ''
- if web_source_name == None:
- web_source_name = ''
- page_time = page_time[:10] if page_time else ""
- start_time = time.time()
- log("start process doc %s"%(str(doc_id)))
- # 字符编码标准化
- title = clean_binary_chars(title) # 去除异常编码字符 修复 244125382 标题有异常转义字符
- text = clean_binary_chars(text) # 去除异常编码字符 修复 653086830 内容有转义字符
- text = str_normalize(text)
- text = re.sub("<html>|</html>|<body>|</body>","",text) # 20260205 删除多个重复标签,避免解析获取节点异常
- text = fix_table_structure_preserve_order(text) # 20250331 修复表格tr tbody平级问题
- text = merge_single_row_tables(text) # 20260123 修复站源14132表头内容拆分多个表格 例子:685993450, 620634112
- original_text = text
- fingerprint = getFingerprint(title+original_text)
- text = delete_error_attachment(text, page_attachments) # 删除无关附件内容
- list_articles,list_sentences,list_entitys,list_outlines,_cost_time = Preprocessing.get_preprocessed([[doc_id,text,"","",title,page_time, web_source_no]],useselffool=True)
- log("get preprocessed done of doc_id%s"%(doc_id))
- cost_time["preprocess"] = round(time.time()-start_time,2)
- cost_time.update(_cost_time)
- '''大纲提取及大纲内容相关提取'''
- start_time = time.time()
- sentence2_list, sentence2_list_attach = extract_sentence_list(list_sentences[0])
- parse_document = ParseDocument(text, True,list_obj=sentence2_list)
- tup_main = extract_parameters(parse_document)
- if sentence2_list_attach!=[] and tup_main[0] == '' and tup_main[1] == '' and tup_main[2] =="":
- parse_document = ParseDocument(text, True, list_obj=sentence2_list_attach)
- tup_att = extract_parameters(parse_document)
- requirement_text, aptitude_text, addr_bidopen_text, addr_bidsend_text, out_lines, requirement_scope, pinmu_name, list_policy, winter_scope, correction_content = tuple(v1 if v1 else v2 for v1, v2 in zip(tup_main, tup_att))
- else:
- requirement_text, aptitude_text, addr_bidopen_text, addr_bidsend_text, out_lines, requirement_scope, pinmu_name, list_policy, winter_scope, correction_content = tup_main
- # print('out_lines',out_lines)
- # if addr_bidopen_text == '':
- # addr_bidopen_text = extract_addr(list_articles[0].content)
- addr_dic, time_dic, code_investment = predictor.getPredictor('entity_type_rule').predict(list_entitys, list_sentences, list_articles)
- if addr_bidopen_text != '' and 'addr_bidopen' not in addr_dic:
- addr_dic['addr_bidopen'] = addr_bidopen_text
- if addr_bidsend_text != '' and 'addr_bidsend' not in addr_dic:
- addr_dic['addr_bidsend'] = addr_bidsend_text
- log("get outline done of doc_id%s"%(doc_id))
- cost_time["outline"] = round(time.time()-start_time,2)
- '''从 kvtree 正则匹配要素'''
- start_time = time.time()
- kv_single_dic, kv_addr_dic = get_kvtree_value(text)
- log("get kvtree done of doc_id%s"%(doc_id))
- cost_time["kvtree"] = round(time.time()-start_time,2)
- # 过滤掉Redis里值为0的错误实体
- # list_entitys[0] = entityLink.enterprise_filter(list_entitys[0])
- # #依赖句子顺序
- # start_time = time.time() # 公告类型/生命周期提取 此处作废 换到后面预测 2022/4/29
- # channel_dic = predictor.getPredictor("channel").predict(title=title, list_sentence=list_sentences[0],
- # web_source_no=web_source_no,original_docchannel=original_docchannel)
- # cost_time["channel"] = round(time.time()-start_time,2)
- start_time = time.time() # 项目编号、名称提取
- codeName = predictor.getPredictor("codeName").predict(list_sentences,MAX_AREA=5000,list_entitys=list_entitys, doctitle = title)
- if re.search('破产清算案', title):
- end = re.search('破产清算案', title).end()
- codeName[0]['name'] = title[:end]
- log("get codename done of doc_id%s"%(doc_id))
- cost_time["codename"] = round(time.time()-start_time,2)
- start_time = time.time() # 公告类别预测
- channel_dic, msc = predictor.getPredictor("channel").predict_merge(title, list_sentences[0], text,original_docchannel, web_source_no)
- if channel_dic['docchannel']['docchannel'] == "":
- channel_dic = merge_channel(list_articles, channel_dic, original_docchannel,page_time,web_source_no=web_source_no)
- cost_time["rule_channel"] = round(time.time() - start_time, 2)
- start_time = time.time() # 角色金额模型提取
- # Phase A:角色链路入口构建一次预计算上下文(实体-句子配对 + 切片缓存),
- # prem 角色/金额两个模型共享同一份,消除原来各自的重复配对与切片构建。
- role_contexts = build_role_contexts(list_sentences, list_entitys)
- predictor.getPredictor("prem").predict(list_sentences,list_entitys, contexts=role_contexts)
- log("get prem done of doc_id%s"%(doc_id))
- cost_time["prem"] = round(time.time()-start_time,2)
- # start_time = time.time() # 产品名称及废标原因提取 此处作废 换到后面预测 2022/4/29
- # fail = channel_dic['docchannel']['docchannel'] == "废标公告"
- # fail_reason = predictor.getPredictor("product").predict(list_sentences,list_entitys,list_articles, fail) #只返回失败原因,产品已加入到Entity类
- # # predictor.getPredictor("product").predict(list_sentences, list_entitys)
- # log("get product done of doc_id%s"%(doc_id))
- # cost_time["product"] = round(time.time()-start_time,2)
- start_time = time.time() # 产品相关要素正则提取 单价、数量、品牌规格 ; 项目、需求、预算、时间
- product_attrs, total_product_money, total_budget = predictor.getPredictor("product_attrs").predict(doc_id, text, page_time)
- log("get product attributes done of doc_id%s"%(doc_id))
- cost_time["product_attrs"] = round(time.time()-start_time,2)
- # 是否为存款类项目
- deposit_project = is_deposit_project(title, codeName[0]['name'], requirement_text)
- start_time = time.time() #正则角色提取
- # Phase E:roleRule / tendereeRuleRecall 复用入口构建的 RoleContext(与 prem 共享)
- all_tenderer = predictor.getPredictor("roleRule").predict(list_articles,list_sentences, list_entitys,codeName, channel_dic, title,all_winner=is_all_winner(title), req_scope=requirement_scope, deposit_project=deposit_project, contexts=role_contexts)
- cost_time["rule"] = round(time.time()-start_time,2)
- '''正则补充最后一句实体日期格式为招标或代理 2021/12/30;正则最后补充角色及去掉包含 公共资源交易中心 的招标人'''
- start_time = time.time() #正则角色提取
- predictor.getPredictor("roleRuleFinal").predict(list_articles,list_sentences,list_entitys, codeName)
- cost_time["roleRuleFinal"] = round(time.time()-start_time,2)
- start_time = time.time() #正则招标人召回
- predictor.getPredictor("tendereeRuleRecall").predict(list_articles,list_sentences,list_entitys, codeName, contexts=role_contexts)
- cost_time["tendereeRuleRecall"] = round(time.time()-start_time,2)
- '''规则调整角色概率'''
- start_time = time.time() #
- predictor.getPredictor("rolegrade").predict(list_sentences,list_entitys,doc_id, original_docchannel, title, outlines = out_lines)
- cost_time["rolegrade"] = round(time.time()-start_time,2)
- '''规则调整金额概率'''
- start_time = time.time() #
- predictor.getPredictor("moneygrade").predict(list_sentences,list_entitys)
- cost_time["moneygrade"] = round(time.time()-start_time,2)
- start_time = time.time() #联系人模型提取
- predictor.getPredictor("epc").predict(list_sentences,list_entitys)
- log("get epc done of doc_id%s"%(doc_id))
- cost_time["person"] = round(time.time()-start_time,2)
- start_time = time.time() # 时间类别提取
- predictor.getPredictor("time").predict(list_sentences, list_entitys)
- log("get time done of doc_id%s"%(doc_id))
- cost_time["time"] = round(time.time()-start_time,2)
- start_time = time.time() # 保证金支付方式
- payment_way_dic = predictor.getPredictor("deposit_payment_way").predict(content=list_articles[0].content)
- cost_time["deposit"] = round(time.time()-start_time,2)
- # 需在getPredictor("prem").predict后 getAttributes.getPREMs 前 规则调整 监理|施工|设计|勘察类别公告的费用 为招标或中标金额
- predictor.getPredictor("prem").correct_money_by_rule(title, list_entitys, list_articles)
- # 2021-12-29新增:提取:总价,单价
- start_time = time.time() # 总价单价提取
- predictor.getPredictor("total_unit_money").predict(list_sentences, list_entitys)
- cost_time["total_unit_money"] = round(time.time()-start_time, 2)
- # 依赖句子顺序
- start_time = time.time() # 实体链接
- entityLink.link_entitys(list_entitys)
- doctitle_refine = entityLink.doctitle_refine(title)
- nlp_enterprise,nlp_enterprise_attachment, dict_enterprise = entityLink.get_nlp_enterprise(list_entitys[0])
- prem = getAttributes.getPREMs(list_sentences,list_entitys,list_articles,list_outlines,page_time,winter_scope)
- log("get attributes done of doc_id%s"%(doc_id))
- cost_time["attrs"] = round(time.time()-start_time,2)
- # if original_docchannel != 302: # 审批项目不做下面提取
- # '''表格要素提取'''
- # table_prem, in_attachment = predictor.getPredictor("tableprem").predict(text, nlp_enterprise+nlp_enterprise_attachment, web_source_name, is_all_winner(title), doc_id)
- # print('表格提取中标人:', table_prem)
- # print('原提取角色:', prem[0]['prem'])
- # if table_prem:
- # getAttributes.update_prem(old_prem=prem[0]['prem'], new_prem=table_prem, in_attachment=in_attachment)
- #
- # '''候选人提取'''
- # candidate_top3_prem, candidate_dic, in_attachment = predictor.getPredictor("candidate").predict(text, list_sentences, list_entitys, nlp_enterprise+nlp_enterprise_attachment,doc_id)
- # # print('表格提取候选人:', candidate_top3_prem)
- # getAttributes.update_prem(old_prem=prem[0]['prem'], new_prem=candidate_top3_prem, in_attachment=in_attachment)
- '''合并中标人、候选人表格提取''' # 20260311取消上面表格要素提取及候选人提取,合并为此方法提取
- common_prem, in_attachment = common_table_prem_extractor.table_prem_extract(text, page_time=page_time, enterprise=dict_enterprise, all_winner = is_all_winner(title), docid=doc_id)
- # print('公共表格提取:', in_attachment, common_prem)
- # print('非表格提取:', prem[0]['prem'])
- getAttributes.update_prem(old_prem=prem[0]['prem'], new_prem=common_prem, in_attachment=in_attachment)
- # print('表格非表格合并后:', prem[0]['prem'])
- '''获取联合体信息'''
- getAttributes.get_win_joint(prem, list_entitys, list_sentences, list_articles)
- '''修正采购公告表格形式多种采购产品中标价格;中标金额小于所有产品总金额则改为总金额'''
- getAttributes.correct_rolemoney(doc_id, prem, total_product_money, total_budget, list_articles)
- '''模板提取要素内容'''
- result_template = template_extractor.template_predict(text, web_source_no, enterprise = dict_enterprise, page_time=page_time, docid=doc_id)
- if result_template:
- product_attrs, codeName, prem = template_extractor.update_template_result(product_attrs, codeName, prem, result_template, doc_id)
- '''修正channel预测类别为招标公告却有中标人及预测为中标信息却无中标关键词的类别''' # 依赖 prem
- start_time = time.time()
- # content = list_articles[0].content
- # channel_dic = predictor.getPredictor("channel").predict_rule(title, content, channel_dic, prem_dic=prem[0]['prem'])
- if original_docchannel == 302:
- channel_dic = {"docchannel":
- { "docchannel": "审批项目", "doctype": "审批项目", "life_docchannel": "审批项目" }
- }
- else:
- channel_dic, msc = predictor.getPredictor("channel").final_change(channel_dic, prem[0], original_docchannel,nlp_enterprise,nlp_enterprise_attachment, msc)
- # print('msc', msc)
- if channel_dic['docchannel'].get('docchannel') == '':
- channel_dic = merge_channel(list_articles,channel_dic,original_docchannel,page_time,prem=prem[0],web_source_no=web_source_no,is_finnal=True) # channel_dic 根据新模型预测结合判断,整合结果
- channel_dic, msc = predictor.getPredictor("channel").final_change(channel_dic, prem[0], original_docchannel, nlp_enterprise, nlp_enterprise_attachment, msc)
- else:
- channel_dic = merge_channel(list_articles,channel_dic,original_docchannel,page_time,prem=prem[0],web_source_no=web_source_no,is_finnal=True) # channel_dic 根据新模型预测结合判断,整合结果
- cost_time["rule_channel2"] = round(time.time()-start_time,2)
- '''时间类提取结果校验'''
- prem = time_entity_check(prem,channel_dic,page_time)
- '''一包多中标人提取及所有金额提取'''
- all_moneys = getAttributes.get_multi_winner_and_money(channel_dic, prem, list_entitys,list_sentences, is_all_winner(title))
- start_time = time.time() # 产品名称及废标原因提取 #依赖 docchannel结果
- fail = channel_dic['docchannel']['docchannel'] == "废标公告"
- fail_reason, product_list = predictor.getPredictor("product").predict(list_sentences,list_entitys,list_articles, fail,out_lines=out_lines) #只返回失败原因,产品已加入到Entity类 #2022/7/29补充返回产品,方便行业分类调用
- # predictor.getPredictor("product").predict(list_sentences, list_entitys)
- log("get product done of doc_id%s"%(doc_id))
- cost_time["product"] = round(time.time()-start_time,2)
- prem[0].update(getAttributes.getOtherAttributes(list_entitys[0],page_time,prem,channel_dic))
- '''更新单一来源招标公告中标角色为预中标'''
- getAttributes.fix_single_source(prem[0], channel_dic, original_docchannel)
- '''公告无表格格式时,采购意向预测''' #依赖 docchannel结果 依赖产品及prem
- '''把产品要素提取结果在项目名称的添加到 采购需求,预算时间,采购时间 要素中'''
- predictor.getPredictor("product_attrs").add_product_attrs(channel_dic, product_attrs, list_sentences,list_entitys,list_outlines,product_list,codeName,prem,text,page_time)
- '''行业分类提取,需要用标题、项目名称、产品、及prem 里面的角色'''
- industry = predictor.getPredictor('industry').predict(title, project=codeName[0]['name'], product=','.join(product_list), prem=prem, product_attrs=product_attrs)
- '''根据数据源最后召回招标人角色'''
- prem = predictor.getPredictor('websource_tenderee').get_websource_tenderee(doc_id, web_source_no, web_source_name, prem)
- '''地区获取'''
- start_time = time.time()
- # district = predictor.getPredictor('district').predict(project_name=codeName[0]['name'], prem=prem,title=title, list_articles=list_articles, web_source_name=web_source_name, list_entitys=list_entitys)
- district = predictor.getPredictor('district').predict_area(doc_id, title, list_articles[0].content, web_source_name, prem=prem[0]['prem'], addr_dic=addr_dic, list_entity=list_entitys[0])
- cost_time["district"] = round(time.time() - start_time, 2)
- '''根据district提取结果修复实体'''
- repair_entity(prem,district,list_articles)
- '''规则补充招标无招标人中标无中标人角色'''
- getAttributes.rule_add_role(doc_id,prem[0]['prem'], channel_dic, list_articles[0].content, web_source_no, nlp_enterprise)
- '''最终验证prem'''
- candidate_dic = getAttributes.confirm_prem(doc_id, prem[0]['prem'], channel_dic, list_articles[0].content, deposit_project, prem[0]['total_tendereeMoney'], all_tenderer)
- '''通过产品补充标段包名20241203'''
- getAttributes.add_package_name(prem[0]['prem'], list_entitys[0], product_list, name=codeName[0]['name'])
- # 提取拟在建所需字段
- start_time = time.time()
- pb_json = predictor.getPredictor('pb_extract').predict(prem, list_articles, list_sentences, list_entitys, title, codeName[0], text, web_source_name, industry, original_text)
- log("pb_extract done of doc_id%s"%(doc_id))
- cost_time["pb_extract"] = round(time.time() - start_time, 2)
- '''打标签'''
- label_dic = get_all_label(title, list_articles[0].content, prem[0]['prem'])
- '''评标评分提取'''
- bid_score = predictor.getPredictor('bid_score').predict(text, nlp_enterprise+nlp_enterprise_attachment)
- # data_res = Preprocessing.union_result(Preprocessing.union_result(codeName, prem),list_punish_dic)[0]
- # data_res = Preprocessing.union_result(Preprocessing.union_result(Preprocessing.union_result(codeName, prem),list_punish_dic), list_channel_dic)[0]
- version_date = {'version_date': '2026-08-13'}
- data_res = dict(codeName[0], **prem[0], **channel_dic, **product_attrs[0], **product_attrs[1], **payment_way_dic, **fail_reason, **industry, **district, **candidate_dic, **version_date, **all_moneys, **pb_json)
- if original_docchannel == 302:
- approval = predictor.getPredictor("approval").predict(list_sentences, list_entitys, text, nlp_enterprise=nlp_enterprise+nlp_enterprise_attachment)
- approval = predictor.getPredictor("approval").add_ree2approval(approval , prem[0]['prem'])
- approval = predictor.getPredictor("approval").add_codename2approval(approval , codeName)
- data_res['prem'] = {} # 审批项目不要这项
- data_res['approval'] = approval[:100] # 20250217 限制获取最多100个项目
- if web_source_no == 'XM6486':
- debt_dic = get_debt_info(text) # 专项债信息提取
- if debt_dic.get('district', '') != '':
- district = predictor.getPredictor('district').predict_area(doc_id, debt_dic['district'], '', web_source_name)
- debt_dic['district'] = district['district']
- data_res['district'] = district['district']
- # 提取专项债信息
- data_res['debt_dic'] = debt_dic
- data_res['docchannel'] = { "docchannel": "审批项目", "doctype": "审批项目", "life_docchannel": "审批项目" }
- if channel_dic['docchannel']['doctype'] == '处罚公告': # 20240627 处罚公告进行失信要素提取
- start_time = time.time() #失信数据要素提取
- punish_dic = predictor.getPredictor("punish").get_punish_extracts(list_articles,list_sentences, list_entitys)
- cost_time["punish"] = round(time.time()-start_time,2)
- data_res['punish'] = punish_dic
- if "Project" in data_res['prem']:
- for d in data_res['prem']['Project']['roleList']:
- if d['role_name'] == 'tenderee' and d.get('role_prob', 0.6) < 0.6: # 处罚公告 去掉低概率招标人
- data_res['prem']['Project']['roleList'] = [d for d in data_res['prem']['Project']['roleList'] if d['role_name'] != 'tenderee']
- break
- if len(data_res['prem']['Project']['roleList']) == 0 and data_res['prem']['Project'].get('tendereeMoney', 0) in [0, '0']: # 删除空包
- data_res['prem'].pop('Project')
- # 把产品属性里面的产品补充到产品列表
- if len(data_res['product_attrs']['data']) > 0: # 20241108 如果产品单价数量提取到产品的,原来提取的产品只保留标题中的
- data_res['product'] = [it for it in data_res['product'] if it in title]
- for d in data_res['product_attrs']['data']:
- if isinstance(d['product'], str) and d['product'] not in data_res['product']:
- data_res['product'].append(d['product'])
- # 标签生成,使用data_res['product']更新后的字段
- '''根据关键词表生成项目标签'''
- project_label, main_content_text = predictor.getPredictor('project_label').predict(title,product=','.join(data_res['product']),
- project_name=codeName[0]['name'],prem=prem,all_text=list_articles[0].content)
- # 额外需求的标签
- project_label = predictor.getPredictor('project_label').predict_other(project_label, industry, title,codeName[0]['name'],
- ','.join(data_res['product']),list_articles,main_content_text)
- '''行业关键词标签'''
- industry_label,tenderee_label = predictor.getPredictor('industry_label').predict(title, list_articles[0],product=','.join(data_res['product']), prem=prem)
- '''产权分类二级标签'''
- property_label = predictor.getPredictor('property_label').predict(title, product=','.join(data_res['product']),project_name=codeName[0]['name'],
- prem=prem,channel_dic=channel_dic)
- '''最终检查修正招标、中标金额'''
- getAttributes.limit_maximum_amount(data_res, list_entitys[0])
- '''添加候选人项目负责人、经理等'''
- getAttributes.add_manager(doc_id, prem[0]['prem'],list_entitys, list_sentences)
- '''利用采购意向需求信息补充项目'''
- if channel_dic['docchannel']['docchannel'] == '采购意向':
- getAttributes.demand_to_prem(data_res.get('demand_info', {}), prem[0]['prem'])
- '''提取标题包号及实体表招标金额、中标金额,项目合并用'''
- ree_moneys, win_moneys, title_packages = predictor.get_package_moneys(title, list_entitys)
- data_res['packages_title'] = list(set(title_packages))
- data_res['moneys_tenderee'] = list(set(ree_moneys))
- data_res['moneys_tenderer'] = list(set(win_moneys))
- data_res["project_label"] = project_label
- data_res["main_content_text"] = main_content_text
- data_res["industry_label"] = industry_label
- data_res["tenderee_label"] = tenderee_label
- data_res["property_label"] = property_label
- data_res["doctitle_refine"] = doctitle_refine
- data_res["nlp_enterprise"] = nlp_enterprise
- data_res["nlp_enterprise_attachment"] = nlp_enterprise_attachment
- data_res["dict_enterprise"] = dict_enterprise
- data_res["fingerprint"] = fingerprint
- # 要素的个数
- data_res['extract_count'] = extractCount(data_res,page_attachments,web_source_name,page_time)
- # 是否有表格
- data_res['exist_table'] = 1 if re.search("<td",text) else 0
- data_res["cost_time"] = cost_time
- data_res["success"] = True
- # 拟在建需建索引字段
- data_res["proportion"] = pb_json.get('pb').get('proportion', '')
- data_res["pb_project_name"] = pb_json.get('pb').get('project_name_refind', '')
- data_res["industry_codes"] = pb_json.get('pb').get('industry_codes', '')
- # 更正内容
- data_res['change_content'] = correction_content[:500]
- # 资质要求
- data_res['aptitude'] = aptitude_text[:1500]
- # 采购内容
- data_res['requirement'] = requirement_text[:1500]
- # 打标签
- data_res['label_dic'] = label_dic
- # 开标、投标、项目、收货等地址
- data_res['addr_dic'] = addr_dic
- # 字数
- text_main, text_attn = 0, 0
- for sentence in list_sentences[0]:
- if sentence.in_attachment:
- text_attn += len(re.sub("##attachment##[,。]?", "", sentence.sentence_text))
- else:
- text_main += len(sentence.sentence_text)
- data_res['word_count'] = {'正文': text_main, '附件': text_attn}
- # 限制产品数量
- data_res['product'] = data_res['product'][:500]
- data_res['product_attrs']['data'] = data_res['product_attrs']['data'][:500]
- # 是否为存款项目
- data_res['is_deposit_project'] = deposit_project
- data_res['pinmu_name'] = pinmu_name # 品目名称
- data_res['policies'] = list_policy # 政策法规
- data_res['bid_score'] = bid_score # 评标得分
- data_res['time_planned'] = time_dic.get('time_planned', '') # 预计招标时间
- data_res['code_investment'] = code_investment # 投资项目编号
- if data_res['product_attrs'].get('data', []) == []: # 20250619 为空的直接返回空字典
- data_res['product_attrs'] = {}
- if data_res['demand_info'].get('data', []) == []:
- data_res['demand_info'] = {}
- for k, v in kv_single_dic.items(): # 没获取到的用kv_tree补充
- if data_res.get(k, '') == '':
- data_res[k] = v
- for k, v in kv_addr_dic.items(): # 没获取到地址的用kv_tree补充
- if data_res['addr_dic'].get(k, '') == '' or re.search('时间:', data_res['addr_dic'][k]):
- data_res['addr_dic'][k] = v
- # for _article in list_articles:
- # log(_article.content)
- #
- # for list_entity in list_entitys:
- # for _entity in list_entity:
- # log("type:%s,text:%s,label:%s,values:%s,sentence:%s,begin_index:%s,end_index:%s"%
- # (str(_entity.entity_type),str(_entity.entity_text),str(_entity.label),str(_entity.values),str(_entity.sentence_index),
- # str(_entity.begin_index),str(_entity.end_index)))
- _extract_json = json.dumps(data_res,cls=MyEncoder,sort_keys=True,indent=4,ensure_ascii=False)
- _extract_json = _extract_json.replace("\x06", "").replace("\x05", "").replace("\x07", "")
- return _extract_json#, list_articles[0].content, get_ent_context(list_sentences, list_entitys)
- def test1(name,content):
- user = {
- "content": content,
- "id":name
- }
- myheaders = {'Content-Type': 'application/json'}
- # Phase 1: URL 走 config/settings.yaml + 环境变量 INTERNAL_ARTICLE_EXTRACT_URL
- # 原硬编码:http://192.168.2.102:15030/article_extract
- from BiddingKG.dl.infra import config as _infra_config
- _url = _infra_config.internal_api_url("article_extract_url")
- if not _url:
- _url = "http://127.0.0.1:15030/article_extract"
- _resp = requests.post(_url, json=user, headers=myheaders, verify=True)
- resp_json = _resp.content.decode("utf-8")
- # print(resp_json)
- return resp_json
- def get_ent_context(list_sentences, list_entitys):
- rs_list = []
- sentences = sorted(list_sentences[0], key=lambda x:x.sentence_index)
- for list_entity in list_entitys:
- for _entity in list_entity:
- if _entity.entity_type in ['org', 'company', 'money']:
- s = sentences[_entity.sentence_index].sentence_text
- b = _entity.wordOffset_begin
- e = _entity.wordOffset_end
- # print("%s %d %.4f; %s %s %s"%(_entity.entity_type, _entity.label, _entity.values[_entity.label], s[max(0, b-10):b], _entity.entity_text, s[e:e+10]))
- rs_list.append("%s %d %.4f; %s ## %s ## %s"%(_entity.entity_type, _entity.label, _entity.values[_entity.label], s[max(0, b-10):b], _entity.entity_text, s[e:e+10]))
- return '\n'.join(rs_list)
- def get_role_context(docid, list_sentences, list_entitys):
- rs_list = []
- sentences = sorted(list_sentences[0], key=lambda x:x.sentence_index)
- for list_entity in list_entitys:
- for _entity in list_entity:
- if _entity.entity_type in ['org', 'company']:
- idx = _entity.entity_id
- sentence = sentences[_entity.sentence_index]
- # _span = spanWindow(tokens=sentence.tokens, begin_index=_entity.begin_index, end_index=_entity.end_index, size=20,
- # center_include=False, word_flag=True, text=_entity.entity_text)
- _span = get_context(sentence.sentence_text, _entity.wordOffset_begin, _entity.wordOffset_end, size=40, center_include=False)
- rs_list.append((docid,idx, _entity.entity_type, _entity.label, '%.4f'%_entity.values[_entity.label], _span[0],
- _entity.entity_text, _span[1]))
- return rs_list
- if __name__=="__main__":
- t1 = time.time()
- with open('d:/html/2.html', 'r', encoding='utf-8') as f:
- html = f.read()
- with open('d:/html/要素提取输入要素.json', 'r', encoding='utf-8') as f:
- docid, title, page_time, web_source_no, web_source_name, original_docchannel, page_attachments = json.load(f)
- # title = ''
- # web_source_name = 'DX000002'
- # web_source_no = 'XX0182'
- # print('title, web_source_name: ', title, web_source_name)
- rs = predict(docid, html, title, page_time, web_source_no, web_source_name, original_docchannel, page_attachments)
- print('耗时:', time.time()-t1)
- print(rs)
- # import pandas as pd
- # df = pd.read_csv('E:/模版提取/医院大学站源公告.csv')
- # df2 = pd.read_csv('E:\模版提取/医院大学站源公告_html.csv')
- #
- # df = pd.read_csv('E:/模版提取/中国南方电网-站源公告.csv')
- # df2 = pd.read_csv('E:/模版提取/中国南方电网-站源公告_html.csv')
- #
- # # df = pd.read_csv('E:/模版提取/站源数据_易派客1122.csv')
- # # df2 = pd.read_csv('E:/模版提取/站源数据_易派客1122_html.csv')
- # #
- # # # df = pd.read_csv('E:\公告输入要素/待检查_中标人为空,但二三中标人存在数据_站源去重_其他要素.csv') # 待检查_有二三候选人无中标人公告
- # # # df2 = pd.read_csv('E:\公告输入要素/待检查_中标人为空,但二三中标人存在数据_站源去重_html.csv')
- # #
- # # df = pd.read_csv('E:\模版提取/表头AI判断有角色金额目前表格提取情况_html.csv')
- # # df2 = pd.read_csv('E:\模版提取/表头AI判断有角色金额目前表格提取情况_其他要素.csv')
- #
- # df = df.merge(df2, on='docid', how='inner')[:]
- # print(df.columns)
- # df = df[df['docid'].isin([673800683])]
- # # df = df[df['web_source_no'].str.contains('XX0757-')]
- # # df.drop_duplicates(subset=['web_source_no','page_time','original_docchannel'], inplace=True)
- # print(len(df))
- # datas = []
- # for docid, html, page_time, web_no, web_name, channel in zip(df['docid'], df['dochtmlcon'],
- # # df['doctitle'],
- # df['page_time'],
- # df['web_source_no'],
- # df['web_source_name'],
- # df['original_docchannel'],
- # ):
- # # html = '更正内容:本项目原成交供应商“合肥万展家具有限公司”因被质疑后放弃成交资格,采购人依据中华人民共和国财政部令第94号《政府采购质疑和投诉办法》中第十六条规定,按照评审报告推荐的成交候选人名单排序,确定第二成交候选人“六安万仞科技有限公司”为成交供应商。,'
- # # if '-' in web_no:
- # # web_no = web_no.split('-')[0]
- # # if web_no not in ['00198', '30306', 'DX000002', 'DX006116', 'DX007098', 'DX013702', 'DX016887', 'DX020336', 'shared_template', 'XX0182', 'XX0757', 'XX1068', 'XX7523']:
- # # continue
- # rs = predict(docid, html, '', page_time, web_no, web_name, channel, '')
- # print(rs)
- # # # datas.append((docid, rs))
- # # # df = pd.DataFrame(datas, columns=['docid', 'rs'])
- # # # df.to_csv('E:/模版提取/中国南方电网_模板提取结果.csv', index=False)
|