| 12345678910111213141516171819202122232425262728293031323334353637383940414243444546474849505152535455565758596061626364656667686970717273747576777879808182838485868788899091929394959697989910010110210310410510610710810911011111211311411511611711811912012112212312412512612712812913013113213313413513613713813914014114214314414514614714814915015115215315415515615715815916016116216316416516616716816917017117217317417517617717817918018118218318418518618718818919019119219319419519619719819920020120220320420520620720820921021121221321421521621721821922022122222322422522622722822923023123223323423523623723823924024124224324424524624724824925025125225325425525625725825926026126226326426526626726826927027127227327427527627727827928028128228328428528628728828929029129229329429529629729829930030130230330430530630730830931031131231331431531631731831932032132232332432532632732832933033133233333433533633733833934034134234334434534634734834935035135235335435535635735835936036136236336436536636736836937037137237337437537637737837938038138238338438538638738838939039139239339439539639739839940040140240340440540640740840941041141241341441541641741841942042142242342442542642742842943043143243343443543643743843944044144244344444544644744844945045145245345445545645745845946046146246346446546646746846947047147247347447547647747847948048148248348448548648748848949049149249349449549649749849950050150250350450550650750850951051151251351451551651751851952052152252352452552652752852953053153253353453553653753853954054154254354454554654754854955055155255355455555655755855956056156256356456556656756856957057157257357457557657757857958058158258358458558658758858959059159259359459559659759859960060160260360460560660760860961061161261361461561661761861962062162262362462562662762862963063163263363463563663763863964064164264364464564664764864965065165265365465565665765865966066166266366466566666766866967067167267367467567667767867968068168268368468568668768868969069169269369469569669769869970070170270370470570670770870971071171271371471571671771871972072172272372472572672772872973073173273373473573673773873974074174274374474574674774874975075175275375475575675775875976076176276376476576676776876977077177277377477577677777877978078178278378478578678778878979079179279379479579679779879980080180280380480580680780880981081181281381481581681781881982082182282382482582682782882983083183283383483583683783883984084184284384484584684784884985085185285385485585685785885986086186286386486586686786886987087187287387487587687787887988088188288388488588688788888989089189289389489589689789889990090190290390490590690790890991091191291391491591691791891992092192292392492592692792892993093193293393493593693793893994094194294394494594694794894995095195295395495595695795895996096196296396496596696796896997097197297397497597697797897998098198298398498598698798898999099199299399499599699799899910001001100210031004100510061007100810091010101110121013101410151016101710181019102010211022102310241025102610271028102910301031103210331034103510361037103810391040104110421043104410451046104710481049105010511052105310541055105610571058105910601061106210631064106510661067106810691070107110721073107410751076107710781079108010811082108310841085108610871088108910901091109210931094109510961097109810991100110111021103110411051106110711081109111011111112111311141115111611171118111911201121112211231124112511261127112811291130113111321133113411351136113711381139114011411142114311441145114611471148114911501151115211531154115511561157115811591160116111621163116411651166116711681169117011711172117311741175117611771178117911801181118211831184118511861187118811891190119111921193119411951196119711981199120012011202120312041205120612071208120912101211121212131214121512161217121812191220122112221223122412251226122712281229123012311232123312341235123612371238123912401241124212431244124512461247124812491250125112521253125412551256125712581259126012611262126312641265126612671268126912701271127212731274127512761277127812791280128112821283128412851286128712881289129012911292129312941295129612971298129913001301130213031304130513061307130813091310131113121313131413151316131713181319132013211322132313241325132613271328132913301331133213331334133513361337133813391340134113421343134413451346134713481349135013511352135313541355135613571358135913601361136213631364136513661367136813691370137113721373137413751376137713781379138013811382138313841385138613871388138913901391139213931394139513961397139813991400140114021403140414051406140714081409141014111412141314141415141614171418141914201421142214231424142514261427142814291430 |
- # -*- coding: utf-8 -*-
- """产品属性预测器(ProductAttributesPredictor)。
- 按 ARCHITECTURE.md Phase 5 拆分建议,从 ``interface/predictor.py`` 迁出以下
- 产品属性提取相关类:
- - ``TableResult`` — 表格提取结果数据类(原 predictor.py 第 3330-3368 行)
- - ``ProductAttributesPredictor`` — 产品数量/单价/品牌/规格/表格要素提取
- (原 predictor.py 第 3371-4719 行)
- 原 ``from common.Utils import *`` / ``from common.nerUtils import *`` 已替换为
- 显式 import;``os.path.dirname(__file__)`` 路径引用替换为
- ``predictors._common.INTERFACE_DIR``。
- ``interface/predictor.py`` 仍 re-export 以上全部名称,老 import 不受影响。
- """
- from __future__ import absolute_import
- import os
- import re
- import copy
- import pickle
- import calendar
- import datetime
- from bs4 import BeautifulSoup
- from dataclasses import dataclass, field
- from typing import List, Dict, Set, Any
- from BiddingKG.dl.common.logging import log
- from BiddingKG.dl.common.context_utils import money_process, spanWindow
- from BiddingKG.dl.common.Utils import del_tabel_achievement
- from BiddingKG.dl.predictors._common import INTERFACE_DIR
- from BiddingKG.dl.predictors.table_prem import TableTag2List
- __all__ = [
- "TableResult",
- "ProductAttributesPredictor",
- ]
- @dataclass
- class TableResult:
- table_index: int = 0
- product_confirm: List[int] = field(default_factory=list)
- headers: List[str] = field(default_factory=list)
- headers_demand: List[str] = field(default_factory=list)
- header_col: List[str] = field(default_factory=list)
- product_link: List[Dict[str, Any]] = field(default_factory=list)
- demand_link: List[Dict[str, Any]] = field(default_factory=list)
- product_set: Set[tuple] = field(default_factory=set)
- total_product_money: float = 0
- unit_price_list: List[str] = field(default_factory=list)
- total_price_list: list = field(default_factory=list)
- budget_list: list = field(default_factory=list)
- @property
- def product_names(self) -> Set[str]:
- return {p.get('product', '') for p in self.product_link if p.get('product', '')}
- @property
- def header_signature(self) -> str:
- return '|'.join(sorted(set(self.headers)))
- @property
- def demand_header_signature(self) -> str:
- return '|'.join(sorted(set(self.headers_demand)))
- def product_count(self) -> int:
- return len(self.product_link)
- def demand_count(self) -> int:
- return len(self.demand_link)
- def avg_attrs_per_product(self) -> float:
- if not self.product_link:
- return 0
- total = sum(len([v for v in p.values() if v != '']) for p in self.product_link)
- return total / len(self.product_link)
- # 产品数量单价品牌规格提取 #2021/11/10 添加表格中的项目、需求、预算、时间要素提取
- class ProductAttributesPredictor():
- MAX_UNIT_PRICE = 100000000
- MAX_TOTAL_PRICE = 50000000000
- MAX_BUDGET = 50000000000
- MAX_PRODUCT_NAME_LEN = 100
- MAX_BRAND_LEN = 50
- MAX_SPECS_LEN = 500
- MAX_PARAM_LEN = 500
- MAX_TENDEREE_LEN = 30
- MAX_QUANTITY_FOR_CALC = 50000
- FUTURE_YEAR_LIMIT = 2050
- def __init__(self,):
- # self.pat_category = '(类别|类型|物类|目录|类目|分类)(名称|$)|^品名|^品类|^品目|(标项|分项|项目|计划|包组|标段|[分子]?包|子目|服务|招标|中标|成交|工程|招标内容)(名称|内容|描述)'
- self.pat_category = '(品目|品类)名称?|采购(品目|品类)$|^品名|^品类|^品目$'
- # self.pat_product_primary = '(标的|维修|系统|报价构成|商品|产品|物料|物资|货物|设备|采购品|采购条目|物品|材料|印刷品?|采购|物装|配件|资产|耗材|清单|器材|仪器|器械|备件|拍卖物|标的物|物件|药品|药材|药械|货品|食品|食材|品目|^品名|气体)[\))的]?(名称|内容|描述)'
- self.pat_product_primary = '(标的|商品|产品|物料|物资|货物|设备|采购品|采购条目|物品|材料|印刷品?|采购|物装|配件|资产|耗材|清单|器材|仪器|器械|备件|拍卖物|标的物|物件|药品|药材|药械|疫苗|货品|食品|食材|品目|^品名|气体)[\))的]?(名称|内容|描述)'
- # self.pat_product_secondary = '标的|标项|项目$|商品|产品|物料|物资|货物|设备|采购品|采购条目|物品|材料|印刷品|物装|配件|资产|招标内容|耗材|清单|器材|仪器|器械|备件|拍卖物|标的物|物件|药品|药材|药械|货品|食品|食材|菜名|^品目$|^品名$|^名称|^内容$|(标项|分项|项目|计划|包组|标段|[分子]?包|子目|服务|招标|中标|成交|工程|招标内容)(名称|内容|描述)'
- self.pat_product_secondary = '标的|商品|产品|物料|物资|货物|设备|采购品|采购条目|物品|材料|印刷品|物装|配件|资产|耗材|清单|器材|仪器|器械|备件|拍卖物|标的物|物件|药品|药材|药械|货品|食品|食材|菜名|^品目$|^品名$'
- self.pat_project = '(标项|分项|项目|计划|包组|标段|[分子]?包|子目|服务|招标|中标|成交|工程)(名称|内容|描述)|^名称|采购类别'
- with open(os.path.join(INTERFACE_DIR, 'header_set.pkl'), 'rb') as f:
- self.header_set = pickle.load(f)
- self.tb = TableTag2List()
- def isTrueTable(self, table):
- '''真假表格规则:
- 1、包含<caption>或<th>标签为真
- 2、包含大量链接、表单、图片或嵌套表格为假
- 3、表格尺寸太小为假
- 4、外层<table>嵌套子<table>,一般子为真,外为假'''
- if table.find_all(['caption', 'th']) != []:
- return True
- # elif len(table.find_all(['form', 'a', 'img'])) > 5: # 20260602 去掉,某些表格可能有多个链接 例子:437440313
- # # print('过滤表格:包含链接图片等大于5的为假表格')
- # return False
- elif len(table.find_all(['tr'])) < 2:
- # print('过滤表格:行数小于2的为假表格')
- return False
- elif len(table.find_all(['table'])) >= 1:
- # print('过滤表格:包含多个表格的为假表格')
- inner_table_num = len(table.find_all(['table']))
- text_num = 0 # 表格只有一格作文本框的数量,docid:631910513
- for inner_table in table.find_all(['table']):
- if len(inner_table.find_all(['tr']))==0 or (len(inner_table.find_all(['tr']))==1 and len(inner_table.find_all(['tr'])[0].find_all(['td']))<=1):
- text_num += 1
- if inner_table_num - text_num > 0:
- return False
- else:
- return True
- else:
- return True
- def getTrs(self, tbody):
- # 获取所有的tr
- trs = []
- objs = tbody.find_all(recursive=False)
- for obj in objs:
- if obj.name == "tr":
- trs.append(obj)
- if obj.name == "tbody":
- for tr in obj.find_all("tr", recursive=False):
- trs.append(tr)
- return trs
- def getTable(self, tbody):
- trs = self.getTrs(tbody)
- inner_table = []
- if len(trs) < 2:
- return inner_table
- for tr in trs:
- tr_line = []
- tds = tr.findChildren(['td', 'th'], recursive=False)
- if len(tds) < 2:
- continue
- for td in tds:
- # td_text = re.sub('\s+|…', ' ', td.get_text()).strip()
- td_text = re.sub('…', '', td.get_text()).strip()
- td_text = re.sub('\n+|\s+', ' ', td_text) # 20250626 去掉\n等避免存OTS后去掉转义导致json解析错误
- td_text = td_text.replace("\x06", "").replace("\x05", "").replace("\x07", "").replace('\\', '/').replace('"', '') # 修复272144312 # 产品单价数量提取结果有特殊符号\ 气动执行装置备件\密封组件\NBR+PT
- td_text = td_text.replace("(", "(").replace(")", ")").replace(':', ':')
- tr_line.append(td_text)
- inner_table.append(tr_line)
- return inner_table
- def fixSpan(self, tbody):
- # 处理colspan, rowspan信息补全问题
- trs = self.getTrs(tbody)
- ths_len = 0
- ths = list()
- trs_set = set()
- # 修改为先进行列补全再进行行补全,否则可能会出现表格解析混乱
- # 遍历每一个tr
- for indtr, tr in enumerate(trs):
- ths_tmp = tr.findChildren('th', recursive=False)
- if len(tr.findChildren('table')) > 0:
- continue
- if len(ths_tmp) > 0:
- for th in ths_tmp:
- ths.append(th)
- trs_set.add(tr)
- # 遍历每行中的element
- tds = tr.findChildren(recursive=False)
- if len(tds) < 3:
- continue # 列数太少的不补全
- for indtd, td in enumerate(tds):
- # 若有colspan 则补全同一行下一个位置
- if 'colspan' in td.attrs and str(re.sub("[^0-9]", "", str(td['colspan']))) != "":
- col = int(re.sub("[^0-9]", "", str(td['colspan'])))
- if col < 10 and len(td.get_text()) < 500:
- td['colspan'] = 1
- for i in range(1, col, 1):
- td.insert_after(copy.copy(td))
- for indtr, tr in enumerate(trs):
- ths_tmp = tr.findChildren('th', recursive=False)
- # 不补全含有表格的tr
- if len(tr.findChildren('table')) > 0:
- continue
- if len(ths_tmp) > 0:
- ths_len = ths_len + len(ths_tmp)
- for th in ths_tmp:
- ths.append(th)
- trs_set.add(tr)
- # 遍历每行中的element
- tds = tr.findChildren(recursive=False)
- same_span = 0
- if len(tds) > 1 and 'rowspan' in tds[0].attrs:
- span0 = tds[0].attrs['rowspan']
- for td in tds:
- if 'rowspan' in td.attrs and td.attrs['rowspan'] == span0:
- same_span += 1
- if same_span == len(tds):
- continue
- for indtd, td in enumerate(tds):
- # 若有rowspan 则补全下一行同样位置
- if 'rowspan' in td.attrs and str(re.sub("[^0-9]", "", str(td['rowspan']))) != "":
- row = int(re.sub("[^0-9]", "", str(td['rowspan'])))
- td['rowspan'] = 1
- for i in range(1, row, 1):
- # 获取下一行的所有td, 在对应的位置插入
- if indtr + i < len(trs):
- tds1 = trs[indtr + i].findChildren(['td', 'th'], recursive=False)
- if len(tds1) >= (indtd) and len(tds1) > 0:
- if indtd > 0:
- tds1[indtd - 1].insert_after(copy.copy(td))
- else:
- tds1[0].insert_before(copy.copy(td))
- elif len(tds1) > 0 and len(tds1) == indtd - 1:
- tds1[indtd - 2].insert_after(copy.copy(td))
- def get_monthlen(self, year, month):
- '''输入年份、月份 int类型 得到该月份天数'''
- try:
- weekday, num = calendar.monthrange(int(year), int(month))
- except (ValueError, TypeError):
- num = 30
- return str(num)
- def _validate_date_range(self, order_begin, order_end, page_time):
- if not order_end > page_time:
- return "", ""
- if order_begin != "" and order_end != "":
- order_begin_year = int(order_begin.split("-")[0])
- order_end_year = int(order_end.split("-")[0])
- if order_begin_year >= self.FUTURE_YEAR_LIMIT or order_end_year >= self.FUTURE_YEAR_LIMIT:
- return "", ""
- return order_begin, order_end
- def _build_month_range(self, year, month):
- month = str(month).zfill(2)
- num = self.get_monthlen(year, month).zfill(2)
- order_begin = "%s-%s-01" % (year, month)
- order_end = "%s-%s-%s" % (year, month, num)
- return order_begin, order_end
- def _resolve_year_for_month(self, html, page_time):
- year = re.search('(\d{4})年(.{,12}采购意向)?', html)
- if year:
- return year.group(1)
- if page_time != "":
- year = re.search('\d{4}', page_time)
- if year:
- return year.group(0)
- return str(datetime.datetime.now().year)
- def fix_time(self, text, html, page_time):
- for it in [('十二', '12'),('十一', '11'),('十','10'),('九','9'),('八','8'),('七','7'),
- ('六','6'),('五','5'),('四','4'),('三','3'),('二','2'),('一','1')]:
- if it[0] in text:
- text = text.replace(it[0], it[1])
- if re.search('^\d{1,2}月$', text):
- m = re.search('^(\d{1,2})月$', text).group(1)
- y = self._resolve_year_for_month(html, page_time)
- return self._build_month_range(y, m)
- t1 = re.search('^(\d{4})(年|/|\.|-)(\d{1,2})月?$', text)
- if t1:
- return self._build_month_range(t1.group(1), t1.group(3))
- t2 = re.search('^(\d{4})(年|/|\.|-)(\d{1,2})(月|/|\.|-)(\d{1,2})日?$', text)
- if t2:
- y = t2.group(1)
- m = t2.group(3).zfill(2)
- d = t2.group(5).zfill(2)
- order_begin = order_end = "%s-%s-%s"%(y,m,d)
- return order_begin, order_end
- t3 = re.search("^(20\d{2})(\d{1,2})$",text)
- if t3:
- year = t3.group(1)
- month = t3.group(2)
- if int(month)>0 and int(month)<=12:
- return self._build_month_range(year, month)
- t4 = re.search("^(20\d{2})(\d{2})(\d{2})$", text)
- if t4:
- year = t4.group(1)
- month = t4.group(2)
- day = t4.group(3)
- if int(month) > 0 and int(month) <= 12 and int(day)>0 and int(day)<=31:
- order_begin = order_end = "%s-%s-%s"%(year,month,day)
- return order_begin, order_end
- all_match = re.finditer('^(?P<y1>\d{4})(年|/|\.)(?P<m1>\d{1,2})(?:(月|/|\.)(?:(?P<d1>\d{1,2})日)?)?'
- '(到|至|-)(?:(?P<y2>\d{4})(年|/|\.))?(?P<m2>\d{1,2})(?:(月|/|\.)'
- '(?:(?P<d2>\d{1,2})日)?)?$', text)
- y1 = m1 = d1 = y2 = m2 = d2 = ""
- found_math = False
- for _match in all_match:
- if len(_match.group()) > 0:
- found_math = True
- for k, v in _match.groupdict().items():
- if v!="" and v is not None:
- if k == 'y1':
- y1 = v
- elif k == 'm1':
- m1 = v
- elif k == 'd1':
- d1 = v
- elif k == 'y2':
- y2 = v
- elif k == 'm2':
- m2 = v
- elif k == 'd2':
- d2 = v
- if not found_math:
- return "", ""
- y2 = y1 if y2 == "" else y2
- d1 = '1' if d1 == "" else d1
- d2 = self.get_monthlen(y2, m2) if d2 == "" else d2
- m1 = '0' + m1 if len(m1) < 2 else m1
- m2 = '0' + m2 if len(m2) < 2 else m2
- d1 = '0' + d1 if len(d1) < 2 else d1
- d2 = '0' + d2 if len(d2) < 2 else d2
- order_begin = "%s-%s-%s"%(y1,m1,d1)
- order_end = "%s-%s-%s"%(y2,m2,d2)
- return order_begin, order_end
- def fix_quantity(self, quantity_text, header_quan_unit):
- '''
- 产品数量标准化,统一为数值型字符串
- :param quantity_text: 原始数量字符串
- :param header_quan_unit: 表头数量单位字符串
- :return: 返回数量及单位
- '''
- quantity = quantity_text
- quantity = re.sub('[一壹]', '1', quantity)
- quantity = re.sub('[,,约]|(\d+)', '', quantity)
- ser = re.search('^(\d+\.?\d*)(?([㎡\w/]{,5})', quantity)
- if ser:
- quantity = str(ser.group(1))
- quantity_unit = ser.group(2)
- if quantity_unit == "" and header_quan_unit != "":
- quantity_unit = header_quan_unit
- else:
- quantity = ""
- quantity_unit = ""
- return quantity, quantity_unit
- def find_header(self, items,p0, p1, p2, pat_project):
- '''
- inner_table 每行正则检查是否为表头,是则返回表头所在列序号,及表头内容
- :param items: 列表,内容为每个td 文本内容
- :param p1: 优先表头正则
- :param p2: 第二表头正则
- :param pat_project: 项目表头正则
- :return: 表头所在列序号,是否表头,表头内容
- '''
- items = [re.sub('\s', '', it) for it in items]
- flag = False
- header_dic = {'产品为项目': False,'名称': '', '数量': '', '单位': '', '单价': '', '品牌': '', '规格': '', '需求': '', '预算': '', '时间': '', '总价': '', '品目': '', '参数': '', '采购人':'', '备注':'','发布日期':'', '品目号':'', '品目名':''}
- product = "" # 产品
- quantity = "" # 数量
- quantity_unit = "" # 数量单位
- unitPrice = "" # 单价
- brand = "" # 品牌
- specs = "" # 规格
- demand = "" # 采购需求
- budget = "" # 预算金额
- order_time = "" # 采购时间
- total_price = "" # 总价
- category = "" # 品目
- parameter = "" # 参数
- tenderee = "" # 采购人
- notes = "" # 备注 2024/3/27 达仁 需求
- issue_date = "" # 发布日期 2024/3/27 达仁 需求
- pinmu_no = "" # 品目号
- pinmu_name = "" # 品目名称
- product_secondary = ''
- product_idx = ''
- project = ''
- project_idx = ''
- for i in range(int(len(items)*0.75)):
- it = items[i]
- if len(it) < 15 and re.search(p0, it):
- flag = True
- if category != "" and category != it:
- continue
- category = it
- header_dic['品目'] = i
- elif len(it) < 15 and re.search(p1, it):
- flag = True
- if product !='' and product != it:
- break
- product = it
- header_dic['名称'] = i
- # break
- if len(it) < 15 and it != category and re.search(p2, it) and (re.search('^名称|^品名|^品目$', it) or re.search(
- '编号|编码|号|情况|报名|单位|位置|地址|数量|单价|价格|金额|品牌|规格类型|型号|公司|中标人|企业|供应商|候选人', it) == None):
- product_secondary = it
- product_idx = i
- if len(it) < 15 and re.search(pat_project, it) and '项目名称' not in project:
- project = it
- project_idx = i
- if product == "":
- if product_secondary:
- flag = True
- product = product_secondary
- header_dic['名称'] = product_idx
- elif category:
- flag = True
- product = category
- header_dic['名称'] = header_dic['品目']
- header_dic['品目'] = ''
- category = ''
- elif project:
- flag = True
- product = project
- header_dic['名称'] = project_idx
- header_dic['产品为项目'] = True
- if flag == False and len(items)>3 and re.search('^第[一二三四五六七八九十](包|标段)$', items[0]):
- product = items[0]
- header_dic['名称'] = 0
- flag = True
- if flag:
- for j in range(len(items)):
- if header_dic['品目号'] == "" and re.search('(品目|品类)(编?号|编码|序号)', items[j]):
- header_dic['品目号'] = j
- pinmu_no = items[j]
- elif header_dic['品目名'] == "" and re.search('(品目|品类)名称|采购(品目|品类)$', items[j]):
- header_dic['品目名'] = j
- pinmu_name = items[j]
- if items[j] in [product, category]:
- continue
- if len(items[j]) > 20 and len(re.sub('[\((].*[)\)]|[^\u4e00-\u9fa5]', '', items[j])) > 10:
- continue
- if header_dic['数量']=="" and re.search('数量|采购量', items[j]) and re.search('单价|用途|要求|规格|型号|运输|承运', items[j])==None:
- header_dic['数量'] = j
- quantity = items[j]
- quantity = re.sub('\d', '', quantity)
- elif header_dic['单位']=="" and re.search('^(数量单位|计量单位|单位)$', items[j]):
- header_dic['单位'] = j
- quantity_unit = items[j]
- elif re.search('单价', items[j]) and re.search('数量|规格|型号|品牌|供应商', items[j])==None:
- header_dic['单价'] = j
- unitPrice = items[j]
- unitPrice = re.sub('\d', '', unitPrice)
- elif re.search('品牌', items[j]):
- header_dic['品牌'] = j
- brand = items[j]
- elif re.search('规格|型号', items[j]):
- header_dic['规格'] = j
- specs = items[j]
- elif re.search('参数', items[j]):
- header_dic['参数'] = j
- parameter = items[j]
- elif re.search('预算单位|(采购|招标|购买)(单位|人|方|主体)|项目业主|采购商|申购单位|需求单位|业主单位',items[j]) and len(items[j])<=8:
- header_dic['采购人'] = j
- tenderee = items[j]
- elif re.search('需求|服务要求|服务标准', items[j]):
- header_dic['需求'] = j
- demand = items[j]
- elif re.search('(采购|招标|投资|项目)(预算|估算)|(预算|控制|投资|项目|采购|招标)金额|(最高|招标)(\w{,2})限价|拦标价', items[j]) and not re.search('预算单位',items[j]):
- header_dic['预算'] = j
- budget = items[j]
- elif re.search('时间|采购时间|采购实施月份|采购月份|采购日期|(预计|计划)(招标|采购|发标|发包)(时间|月份)', items[j]):
- header_dic['时间'] = j
- order_time = items[j]
- elif re.search('总价|(成交|中标|验收|合同|预算|控制|总|合计))?([金总]额|价格?)|最高限价|价格|金额', items[j]) and re.search('数量|规格|型号|品牌|供应商', items[j])==None:
- header_dic['总价'] = j
- total_price = items[j]
- total_price = re.sub('\d', '', total_price)
- elif re.search('^备\s*注$|资质要求|预留面向中小企业|是否适宜中小企业采购预算预留|公开征集信息', items[j]):
- header_dic['备注'] = j
- notes = items[j]
- elif re.search('^\w{,4}发布(时间|日期)$', items[j]):
- header_dic['发布日期'] = j
- issue_date = items[j]
- if header_dic.get('名称', "") != "" or header_dic.get('品目', "") != "":
- # num = 0
- # for it in (quantity, unitPrice, brand, specs, product, demand, budget, order_time, total_price):
- # if it != "":
- # num += 1
- # if num >=2:
- # return header_dic, flag, (product, quantity, quantity_unit, unitPrice, brand, specs, total_price, category, parameter), (product, demand, budget, order_time)
- if set([quantity, brand, specs, unitPrice, total_price])!=set([""]) or set([demand, budget])!=set([""]):
- # if header_dic['产品为项目'] and (brand or specs):
- # header_dic['产品为项目'] = False
- return header_dic, flag, (product, quantity, quantity_unit, unitPrice, brand, specs, total_price, category, parameter, pinmu_no, pinmu_name), (product, demand, budget, order_time,tenderee, notes,issue_date)
- flag = False
- return header_dic, flag, (product, quantity, quantity_unit, unitPrice, brand, specs, total_price, category, parameter, pinmu_no, pinmu_name), (product, demand, budget, order_time,tenderee,notes,issue_date)
- def predict(self, docid='', html='', page_time=""):
- html = html.replace('<br>', '\n').replace('<br/>', '\n')
- html = re.sub("<html>|</html>|<body>|</body>","",html)
- html = re.sub("##attachment##","",html)
- soup = BeautifulSoup(html, 'lxml')
- del_tabel_achievement(soup)
- richText = soup.find(name='div', attrs={'class': 'richTextFetch'})
- if richText:
- richText = richText.extract() # 过滤掉附件
- def extract_product(soup):
- flag_yx = True
- tables = soup.find_all(['table'])
- table_results: List[TableResult] = []
- for table_idx in range(len(tables)):
- table = tables[table_idx]
- if table.parent.name == 'td' and len(table.find_all('td')) <= 3:
- table.string = table.get_text()
- table.name = 'turntable'
- continue
- if not self.isTrueTable(table):
- continue
- self.fixSpan(table)
- inner_table = self.getTable(table)
- table.extract()
- tr = self._extract_from_single_table(
- table_idx, inner_table, html, page_time, flag_yx
- )
- if tr.product_link and sum(tr.product_confirm) == 0:
- products = [re.sub('([^)]+)', '', it[0]) for it in tr.product_set]
- is_project = [1 if re.search('.{10,}(工程|项目|施工)', product) else 0 for product in products]
- if sum(is_project)>len(is_project)*0.5 or len(tr.product_link[0]) < 2: # 不确定产品,一半产品包含项目工程关键词或只有产品的去掉。
- tr.product_link = []
- tr.headers = []
- tr.total_product_money = 0
- log('去除项目工程名称作产品:%s, docid:%s'%(' '.join(products), docid))
- if tr.product_link or tr.demand_link:
- table_results.append(tr)
- if len(inner_table) > 0:
- last_tds = inner_table[-1] if inner_table else []
- if len(last_tds) >= 2 and len(set(last_tds)) in [2, 3] and re.search('订单总价',
- last_tds[0]) and re.search(
- '\d+[\d,\.]*', last_tds[1]):
- money_, unit_ = money_process(last_tds[1], last_tds[0])
- if table_results:
- table_results[-1].total_product_money = max(money_, table_results[-1].total_product_money)
- return table_results
- table_results = extract_product(soup)
- if (len(table_results) < 1 or sum([it for l in table_results for it in l.product_confirm])==0) and richText: # 正文没提取或提取产品不确定补充附件提取
- table_results_richText = extract_product(richText)
- table_results.extend(table_results_richText)
- merged = self._merge_table_results(table_results)
- product_link = merged.product_link
- demand_link = merged.demand_link
- total_product_money = merged.total_product_money
- headers = merged.headers
- headers_demand = merged.headers_demand
- header_col = merged.header_col
- budget_list = merged.budget_list
- result = self._post_process(product_link, demand_link, merged.total_price_list,
- merged.unit_price_list, budget_list)
- if result == 0:
- total_product_money = 0
- if len(product_link)>0:
- product_link = [{k:v for k,v in d.items() if v!=''} for d in product_link]
- attr_dic = {'product_attrs':{'data':product_link, 'header':headers, 'header_col':header_col}}
- attr_dic['product_attrs']['product_confirm'] = sum(merged.product_confirm) # 是否明确产品
- total_budget = sum(budget_list) if len(budget_list) == len(product_link) else 0
- if not attr_dic['product_attrs']['product_confirm']:
- if len(product_link[0]) < 3: # 产品不确定且要素少于3个的去掉
- attr_dic = {'product_attrs': {'data': [], 'header': [], 'header_col': []}}
- else:
- log('产品属性提取不确定,产品数量:%d,要素数量:%d,docid:%s'%(len(product_link), len(product_link[0]), docid))
- else:
- attr_dic = {'product_attrs': {'data': [], 'header': [], 'header_col': []}}
- total_budget = 0
- if len(demand_link)>0:
- demand_link = [{k: v for k, v in d.items() if v != ''} for d in demand_link]
- demand_dic = {'demand_info':{'data':demand_link, 'header':headers_demand, 'header_col':header_col}}
- else:
- demand_dic = {'demand_info':{'data':[], 'header':[], 'header_col':[]}}
- return [attr_dic, demand_dic], total_product_money, total_budget
- def _extract_from_single_table(self, table_idx, inner_table, html, page_time, flag_yx) -> TableResult:
- tr = TableResult(table_index=table_idx)
- found_header = False
- header_quan_unit = ""
- header_colnum = 0
- header_dic = {}
- header_list = header_list2 = ()
- if flag_yx:
- col0_l, col1_l = [], []
- for tds in inner_table:
- if len(tds) == 2:
- col0_l.append(re.sub('[::]', '', tds[0]))
- col1_l.append(tds[1])
- elif len(tds)>=4 and len(inner_table)==2:
- col0_l = inner_table[0]
- col1_l = inner_table[1]
- break
- if len(set(col0_l) & self.header_set) > len(col0_l) * 0.2 and len(col0_l)==len(col1_l):
- d_links, d_headers = self._extract_demand_from_2col(col0_l, col1_l, html, page_time)
- if d_links:
- tr.demand_link.extend(d_links)
- tr.headers_demand.extend(d_headers)
- return tr
- if len(inner_table)>3 and len(inner_table[0])==2 and len(inner_table[1])==2:
- col0_l, col1_l = [], []
- for tds in inner_table:
- if len(tds) == 2:
- col0_l.append(re.sub('[::]', '', tds[0]))
- col1_l.append(tds[1])
- else:
- break
- if len(set(col0_l) & self.header_set) > len(col0_l) * 0.5 and len(col0_l) == len(col1_l):
- inner_table = [col0_l, col1_l]
- elif len(inner_table)>2 and len(inner_table[0])==4 and len(inner_table[1])==4 and len(set(inner_table[0]) & self.header_set)==2:
- col0_l, col1_l, col2_l, col3_l = [], [], [], []
- for tds in inner_table:
- if len(tds) == 4 and len(set(tds))>2:
- col0_l.append(re.sub('[::]', '', tds[0]))
- col1_l.append(tds[1])
- col2_l.append(re.sub('[::]', '', tds[2]))
- col3_l.append(tds[3])
- else:
- break
- if len(set(col0_l) & self.header_set) > len(col0_l) * 0.5 and len(set(col2_l) & self.header_set) > len(col2_l) * 0.5:
- inner_table = [col0_l+col2_l, col1_l+col3_l]
- row_idx = 0
- while row_idx < (len(inner_table)):
- tds = inner_table[row_idx]
- not_empty = [it for it in tds if re.sub('\s', '', it) != ""]
- if len(set(not_empty))<2 or len(set(tds))<2 or (len(set(tds))==2 and re.search('总计|合计|汇总|总价', tds[0])):
- row_idx += 1
- continue
- product = ""
- quantity = ""
- quantity_unit = ""
- unitPrice = ""
- brand = ""
- specs = ""
- demand = ""
- budget = ""
- order_time = ""
- order_begin = ""
- order_end = ""
- total_price = ""
- parameter = ""
- tenderee = ""
- notes = ''
- issue_date = ''
- pinmu_no = ''
- pinmu_name = ''
- if len(set([re.sub('[::\s]','',td) for td in tds]) & self.header_set) > len(tds) * 0.4:
- header_dic, found_header, header_list, header_list2 = self.find_header(tds, self.pat_category, self.pat_product_primary, self.pat_product_secondary, self.pat_project)
- if found_header:
- header_colnum = len(tds)
- if found_header and isinstance(header_list, tuple) and len(header_list) > 2:
- quantity_header = header_list[1].replace('单位:', '')
- if re.search('(([\w/]{,5}))', quantity_header):
- header_quan_unit = re.search('(([\w/]{,5}))', quantity_header).group(1)
- else:
- header_quan_unit = ""
- if found_header and ('_'.join(header_list) not in tr.headers or '_'.join(header_list2) not in tr.headers_demand):
- tr.headers.append('_'.join(header_list))
- tr.headers_demand.append('_'.join(header_list2))
- tr.header_col.append('_'.join(tds))
- is_confirm = 0 if header_dic['产品为项目'] else 1
- tr.product_confirm.append(is_confirm)
- row_idx += 1
- continue
- elif found_header:
- if len(tds) > header_colnum or len(tds)-1<max([it for it in header_dic.values() if it!=""]):
- row_idx += 1
- continue
- id0 = header_dic.get('品目', "")
- id1 = header_dic.get('名称', "")
- id2 = header_dic.get('数量', "")
- id2_2 = header_dic.get('单位', "")
- id3 = header_dic.get('单价', "")
- id4 = header_dic.get('品牌', "")
- id5 = header_dic.get('规格', "")
- id6 = header_dic.get('需求', "")
- id7 = header_dic.get('预算', "")
- id8 = header_dic.get('时间', "")
- id9 = header_dic.get("总价", "")
- id10 = header_dic.get('参数', "")
- id11 = header_dic.get('采购人', "")
- id12 = header_dic.get('备注', "")
- id13 = header_dic.get('发布日期', "")
- id14 = header_dic.get('品目号', "")
- id15 = header_dic.get('品目名', "")
- not_attr = 0
- for k, v in header_dic.items():
- if isinstance(v, int):
- if v >= len(tds) or tds[v] in self.header_set:
- not_attr = 1
- if not_attr>=2:
- row_idx += 1
- found_header = False
- continue
- if id1!="" and re.search('[a-zA-Z\u4e00-\u9fa5]', tds[id1]) and tds[id1] not in self.header_set and \
- re.search('备注|汇总|合计|总价|价格|金额|^详见|无$|xxx', tds[id1].replace(' ', '')) == None:
- product = re.sub('\s+', '', tds[id1])
- if id0!="" and re.search('[a-zA-Z\u4e00-\u9fa5]', tds[id0]) and tds[id0] not in self.header_set and \
- re.search('备注|汇总|合计|总价|价格|金额|^详见|无$|xxx', tds[id0].replace(' ', '')) == None:
- category = re.sub('\s', '', tds[id0])
- # product = "%s_%s"%(category, product) if product!="" and product!=category else category # 20260528 去掉名称组合 修复 776939340 名称组合不像产品
- if product == '':
- product = category
- if product and re.match( '【?(工程类|服务类|货物类|工程|服务|货物|运费)】?', product) == None:
- if id2 != "":
- if re.search('\d+|[壹贰叁肆伍陆柒捌玖拾一二三四五六七八九十]', tds[id2]):
- quantity = tds[id2]
- elif re.search('\w{5,}', tds[id2]) and re.search('^详见|^详情', tds[id2])==None:
- row_idx += 1
- continue
- if id2_2 != "":
- if re.search('^\w{1,4}$', tds[id2_2]) and re.search('元', tds[id2_2])==None:
- quantity_unit = tds[id2_2]
- if id3 != "":
- if re.search('[零壹贰叁肆伍陆柒捌玖拾佰仟萬億十百千万亿元角分]{3,}', tds[id3]):
- unitPrice = tds[id3]
- elif re.search('^[\d,.亿万元人民币欧美日金额:()();;、,\n]+$|¥|¥|RMB|USD|EUR|JPY|CNY|元$', tds[id3].strip()):
- unitPrice = tds[id3]
- elif len(re.sub('[金额万元()()::零壹贰叁肆伍陆柒捌玖拾佰仟萬億圆十百千万亿元角分¥整\d,.]', '', tds[id3])) > 5 and re.search('^详见|^详情', tds[id3])==None:
- row_idx += 1
- continue
- else:
- unitPrice = tds[id3]
- if id4 != "":
- if re.search('\w', tds[id4]):
- brand = tds[id4]
- if re.match('^详见|^详情', brand.strip()):
- brand = ""
- else:
- brand = ""
- if id5 != "":
- if re.search('\w', tds[id5]):
- specs = tds[id5][:self.MAX_SPECS_LEN]
- if re.match('^详见|^详情', specs.strip()):
- specs = ""
- else:
- specs = ""
- if id6 != "":
- if re.search('\w', tds[id6]):
- demand = tds[id6]
- else:
- demand = ""
- if id7 != "":
- if re.search('\d+|[零壹贰叁肆伍陆柒捌玖拾佰仟萬億十百千万亿元角分]{3,}', tds[id7]):
- budget = tds[id7]
- if id8 != "":
- if re.search('\w', tds[id8]):
- order_time = tds[id8].strip()
- order_begin, order_end = self.fix_time(order_time, html, page_time)
- if id9 != "":
- if re.search('[零壹贰叁肆伍陆柒捌玖拾佰仟萬億十百千万亿元角分]{3,}', tds[id9]):
- total_price = tds[id9]
- elif re.search('^[\d,.亿万元人民币欧美日金额:()();;、,\n]+$|¥|¥|RMB|USD|EUR|JPY|CNY|元$', tds[id9].strip()):
- total_price = tds[id9]
- elif len(re.sub('[金额万元()()::零壹贰叁肆伍陆柒捌玖拾佰仟萬億圆十百千万亿元角分¥整\d,.]', '', tds[id9])) > 5 and re.search('^详见|^详情', tds[id9])==None:
- row_idx += 1
- continue
- if id10 != "":
- parameter = tds[id10][:self.MAX_PARAM_LEN]
- if re.match('^详见|^详情', parameter.strip()):
- parameter = ""
- if id11 != "":
- tenderee = re.sub("\s","",tds[id11])
- if len(tenderee) > self.MAX_TENDEREE_LEN:
- tenderee = ""
- if id12 != "":
- notes = tds[id12].strip()
- if id13 != "":
- issue_date = self.fix_time(tds[id13].strip(), '', '')[0]
- if id14 != "":
- pinmu_no = tds[id14].strip()
- if id15 != "":
- pinmu_name = tds[id15].strip()
- if quantity != "" or unitPrice != "" or brand != "" or specs != "" or total_price or '单价' in header_dic or '总价' in header_dic:
- if id1!="" and id2 != "" and id3 != "" and len(re.split('[;;、,\n]+', tds[id2])) > 1 and len(re.split('[;;、,\n]+', tds[id1])) == len(re.split('[;;、,\n]+', tds[id2])):
- products = re.split('[;;、,\n]+', tds[id1])
- quantitys = re.split('[;;、,\n]+', tds[id2])
- unitPrices = re.split('[;;、,\n]+', tds[id3])
- total_prices = re.split('[;;、,\n]+', total_price)
- brands = re.split('[;;、,\n]+', brand) if re.search('等$', brand)==None else [brand]
- specses = re.split('[;;、,\n]+', specs) if re.search('等$', specs)==None else [specs]
- parameters = re.split('[;;、,\n]+', parameter) if re.search('等$', parameter)==None else [parameter]
- unitPrices = [""]*len(products) if len(unitPrices)==1 else unitPrices
- total_prices = [""]*len(products) if len(total_prices)==1 else total_prices
- brands = brands*len(products) if len(brands)==1 else brands
- specses = specses*len(products) if len(specses)==1 else specses
- brands = [brand]*len(products) if len(brands) < len(products) else brands
- specses = [specs] * len(products) if len(specses) < len(products) else specses
- parameters = parameters*len(products) if len(parameters)==1 else parameters
- if len(products) == len(quantitys) == len(unitPrices) == len(brands) == len(specses):
- for product, quantity, unitPrice, brand, specs, total_price, parameter in zip(products,quantitys,unitPrices, brands, specses, total_prices, parameters):
- if product.strip() == '':
- continue
- if quantity != "":
- quantity, quantity_unit_ = self.fix_quantity(quantity, header_quan_unit)
- quantity_unit = quantity_unit_ if quantity_unit_ != "" else quantity_unit
- if unitPrice != "":
- unitPrice, _money_unit = money_process(unitPrice, header_list[3])
- unitPrice = str(unitPrice) if unitPrice != 0 and unitPrice<self.MAX_UNIT_PRICE else ""
- if budget != "":
- budget, _money_unit = money_process(budget, header_list2[2])
- budget = str(budget) if budget != 0 and budget<self.MAX_BUDGET else ''
- if total_price != "":
- total_price, _money_unit = money_process(total_price, header_list[6])
- tr.total_price_list.append(total_price)
- total_price = str(total_price) if total_price != 0 and total_price<self.MAX_TOTAL_PRICE else ""
- link = {'product': product, 'quantity': quantity,
- 'quantity_unit': quantity_unit, 'unitPrice': unitPrice,
- 'brand': brand[:self.MAX_BRAND_LEN], 'specs': specs, 'total_price': total_price, 'parameter': parameter}
- if (product, specs, unitPrice, quantity) not in tr.product_set:# and not header_dic['产品为项目']:
- tr.product_set.add((product, specs, unitPrice, quantity))
- tr.product_link.append(link)
- if budget != '' and float(budget) > 0:
- tr.budget_list.append(float(budget))
- if link['unitPrice'] != "" and link['quantity'] != '':
- try:
- tr.total_product_money += float(link['unitPrice']) * float(
- link['quantity']) if float(link['quantity']) < self.MAX_QUANTITY_FOR_CALC else 0
- except (ValueError, TypeError):
- log('产品属性单价数量相乘出错, 单价: %s, 数量: %s' % (
- link['unitPrice'], link['quantity']))
- elif len(product)>self.MAX_PRODUCT_NAME_LEN:
- row_idx += 1
- continue
- else:
- if quantity != "":
- quantity, quantity_unit_ = self.fix_quantity(quantity, header_quan_unit)
- quantity_unit = quantity_unit_ if quantity_unit_ != "" else quantity_unit
- if unitPrice != "":
- unitPrice, _money_unit = money_process(unitPrice, header_list[3])
- unitPrice = str(unitPrice) if unitPrice != 0 and unitPrice<self.MAX_UNIT_PRICE else ""
- if budget != "":
- budget, _money_unit = money_process(budget, header_list2[2])
- budget = str(budget) if budget != 0 and budget<self.MAX_BUDGET else ''
- if total_price != "":
- total_price, _money_unit = money_process(total_price, header_list[6])
- tr.total_price_list.append(total_price)
- total_price = str(total_price) if total_price != 0 and total_price<self.MAX_TOTAL_PRICE else ""
- link = {'product': product, 'quantity': quantity, 'quantity_unit': quantity_unit, 'unitPrice': unitPrice,
- 'brand': brand[:self.MAX_BRAND_LEN], 'specs':specs, 'total_price': total_price, 'parameter': parameter,
- 'pinmu_no': pinmu_no, 'pinmu_name': pinmu_name}
- if (product, unitPrice,) not in tr.product_set:# and not header_dic['产品为项目']:
- tr.product_set.add((product, unitPrice))
- tr.product_link.append(link)
- if budget != '' and float(budget) > 0:
- tr.budget_list.append(float(budget))
- if link['unitPrice']:
- tr.unit_price_list.append(link['unitPrice'])
- if link['unitPrice'] != "" and link['quantity'] != '':
- try:
- tr.total_product_money += float(link['unitPrice'])*float(link['quantity']) if float(link['quantity'])<self.MAX_QUANTITY_FOR_CALC else 0
- if float(link['unitPrice'])>10000 and float(link['quantity'])>100:
- tr.total_product_money = 0
- except (ValueError, TypeError):
- log('产品属性单价数量相乘出错, 单价: %s, 数量: %s'%(link['unitPrice'], link['quantity']))
- order_begin, order_end = self._validate_date_range(order_begin, order_end, page_time)
- if budget != "" and order_end != "":
- link = {'project_name': product, 'product':[], 'demand': demand, 'budget': budget, 'order_begin':order_begin, 'order_end':order_end, 'tenderee':tenderee,'notes':notes,'issue_date':issue_date}
- if link not in tr.demand_link:
- tr.demand_link.append(link)
- row_idx += 1
- else:
- row_idx += 1
- return tr
- def _extract_demand_from_2col(self, col0_l, col1_l, html, page_time):
- header_list2 = []
- product = demand = budget = order_begin = order_end = ""
- tenderee = ""
- notes = ''
- issue_date = ''
- demand_links = []
- for i in range(len(col0_l)):
- if re.search('项目名称', col0_l[i]):
- header_list2.append(col0_l[i])
- product = col1_l[i]
- elif re.search('采购需求|需求概况|招标内容|项目概况', col0_l[i]):
- header_list2.append(col0_l[i])
- demand = col1_l[i]
- elif re.search('(采购|招标|投资|项目)(预算|估算)|(预算|控制|投资|项目|采购|招标)金额|(最高|招标)(\w{,2})限价|拦标价', col0_l[i]):
- header_list2.append(col0_l[i])
- _budget = col1_l[i]
- re_price = re.findall("[零壹贰叁肆伍陆柒捌玖拾佰仟萬億圆十百千万亿元角分]{3,}|\d[\d,]*(?:\.\d+)?万?", _budget)
- if re_price:
- _budget, _money_unit = money_process(_budget, col0_l[i])
- budget = str(_budget)
- if '.' in budget:
- budget = budget.rstrip('0').rstrip('.')
- if float(budget)>= 500*100000000:
- budget = ""
- elif re.search('预算单位|(采购|招标|购买)(单位|人|方|主体)|项目业主|采购商|申购单位|需求单位|业主单位', col0_l[i]):
- header_list2.append(col0_l[i])
- tenderee = re.sub("\s","",col1_l[i])
- if len(tenderee) > 20:
- tenderee = ""
- elif re.search('时间|采购时间|采购实施月份|采购月份|采购日期|(预计|计划)(招标|采购|发标|发包)(时间|月份)', col0_l[i]):
- header_list2.append(col0_l[i])
- order_time = col1_l[i].strip()
- order_begin, order_end = self.fix_time(order_time, html, page_time)
- elif re.search('^备\s*注$|资质要求|预留面向中小企业|是否适宜中小企业采购预算预留|公开征集信息', col0_l[i]):
- header_list2.append(col0_l[i])
- notes = col1_l[i].strip()
- elif re.search('^\w{,4}发布(时间|日期)$', col0_l[i]):
- header_list2.append(col0_l[i])
- issue_date = self.fix_time(col1_l[i].strip(), '', '')[0]
- order_begin, order_end = self._validate_date_range(order_begin, order_end, page_time)
- if product!= "" and demand != "" and budget!="" and order_end != "":
- link = {'project_name': product, 'product': [], 'demand': demand, 'budget': budget,
- 'order_begin': order_begin, 'order_end': order_end ,'tenderee':tenderee, 'notes':notes, 'issue_date':issue_date}
- if link not in demand_links:
- demand_links.append(link)
- return demand_links, header_list2
- def _merge_table_results(self, table_results: List[TableResult]) -> TableResult:
- if not table_results:
- return TableResult()
- if len(table_results) == 1:
- return table_results[0]
- merged = TableResult()
- merged_product_set: Set[tuple] = set()
- merged_demand_set: Set[str] = set()
- budget_list = []
- total_product_money_list = []
- for i, tr in enumerate(table_results):
- if not tr.product_link and not tr.demand_link:
- continue
- overlap_ratio = self._calc_product_overlap(tr.product_names, merged_product_set)
- if overlap_ratio > 0.5 and self._same_header_group(tr, table_results[:i]):
- self._merge_with_dedup(tr, merged, merged_product_set, merged_demand_set, prefer_richer=True)
- elif overlap_ratio > 0:
- self._merge_with_dedup(tr, merged, merged_product_set, merged_demand_set, prefer_richer=True)
- else:
- self._merge_with_dedup(tr, merged, merged_product_set, merged_demand_set, prefer_richer=False)
- merged.headers.extend(tr.headers)
- merged.headers_demand.extend(tr.headers_demand)
- merged.header_col.extend(tr.header_col)
- merged.total_price_list.extend(tr.total_price_list)
- merged.unit_price_list.extend(tr.unit_price_list)
- merged.budget_list.extend(tr.budget_list)
- merged.product_confirm.extend(tr.product_confirm)
- if tr.budget_list and tr.budget_list not in budget_list:
- budget_list.append(tr.budget_list)
- if tr.total_product_money and tr.total_product_money not in total_product_money_list:
- total_product_money_list.append(tr.total_product_money)
- merged.total_product_money += tr.total_product_money
- if len(budget_list) > 1:
- merged.budget_list = []
- if len(total_product_money_list) > 1:
- merged.total_product_money = 0
- merged.headers = list(set(merged.headers))
- merged.headers_demand = list(set(merged.headers_demand))
- merged.header_col = list(set(merged.header_col))
- return merged
- @staticmethod
- def _calc_product_overlap(new_names: Set[str], existing_set: Set[tuple]) -> float:
- if not new_names or not existing_set:
- return 0.0
- existing_names = {p[0] for p in existing_set if p}
- if not existing_names:
- return 0.0
- overlap = new_names & existing_names
- return len(overlap) / min(len(new_names), len(existing_names))
- @staticmethod
- def _same_header_group(tr: TableResult, previous_results: List[TableResult]) -> bool:
- for prev in previous_results:
- if tr.header_signature and tr.header_signature == prev.header_signature:
- return True
- if tr.demand_header_signature and tr.demand_header_signature == prev.demand_header_signature:
- return True
- return False
- @staticmethod
- def _count_non_empty(d: dict) -> int:
- return len([v for v in d.values() if v != '' and v is not None])
- def _merge_with_dedup(self, new_tr: TableResult, merged: TableResult,
- merged_product_set: Set[tuple], merged_demand_set: Set[str],
- prefer_richer: bool) -> None:
- for link in new_tr.product_link:
- product = link.get('product', '')
- unit_price = link.get('unitPrice', '')
- specs = link.get('specs', '')
- quantity = link.get('quantity', '')
- key = (product, unit_price)
- if key in merged_product_set:
- if prefer_richer:
- existing_idx = None
- for idx, existing in enumerate(merged.product_link):
- if existing.get('product', '') == product and existing.get('unitPrice', '') == unit_price:
- existing_idx = idx
- break
- if existing_idx is not None:
- existing = merged.product_link[existing_idx]
- if self._count_non_empty(link) > self._count_non_empty(existing):
- merged.product_link[existing_idx] = link
- continue
- merged_product_set.add(key)
- merged.product_link.append(link)
- for link in new_tr.demand_link:
- project_name = link.get('project_name', '')
- budget = link.get('budget', '')
- order_end = link.get('order_end', '')
- demand_key = f"{project_name}_{budget}_{order_end}"
- if demand_key in merged_demand_set:
- if prefer_richer:
- existing_idx = None
- for idx, existing in enumerate(merged.demand_link):
- ex_key = f"{existing.get('project_name', '')}_{existing.get('budget', '')}_{existing.get('order_end', '')}"
- if ex_key == demand_key:
- existing_idx = idx
- break
- if existing_idx is not None:
- existing = merged.demand_link[existing_idx]
- if self._count_non_empty(link) > self._count_non_empty(existing):
- merged.demand_link[existing_idx] = link
- continue
- merged_demand_set.add(demand_key)
- merged.demand_link.append(link)
- def _post_process(self, product_link, demand_link, total_price_list, unit_price_list, budget_list):
- if len(total_price_list)>1 and len(set(total_price_list))/len(total_price_list)<=0.5:
- for link in product_link:
- if 'total_price' in link:
- link['total_price'] = ""
- if len(demand_link) > 2 and demand_link[0].get('budget', '') != '' and len(set([d.get('budget', '') for d in demand_link])) == 1:
- for d in demand_link:
- if 'budget' in d:
- d['budget'] = ""
- if len(unit_price_list)>0 and len(unit_price_list)==len(product_link) and len(set(unit_price_list))/len(unit_price_list)<=0.5:
- return 0
- return None
- def predict_without_table(self,product_attrs,list_sentences,list_entitys,codeName,prem, html='', page_time=""):
- if len(prem[0]['prem'])==1:
- list_sentences[0].sort(key=lambda x:x.sentence_index)
- list_sentence = list_sentences[0]
- list_entity = list_entitys[0]
- _data = product_attrs[1]['demand_info']['data']
- re_bidding_time = re.compile("(采购|采购实施|(预计|计划)(招标|采购|发标|发包))(时间|月份|日期)[::,].{0,2}$")
- order_times = []
- for entity in list_entity:
- if entity.entity_type=='time':
- # print('time',entity.entity_text)
- sentence = list_sentence[entity.sentence_index]
- s = spanWindow(tokens=sentence.tokens, begin_index=entity.begin_index,
- end_index=entity.end_index,size=20)
- entity_left = "".join(s[0])
- if re.search(re_bidding_time,entity_left):
- time_text = entity.entity_text.strip()
- standard_time = re.compile("((?P<year>\d{4}|\d{2})\s*[-\/年\.]\s*(?P<month>\d{1,2})\s*[-\/月\.]\s*((?P<day>\d{1,2})日?)?)")
- time_match = re.search(standard_time,time_text)
- # print(time_text, time_match)
- if time_match:
- time_text = time_match.group()
- order_times.append(time_text)
- # print(order_times)
- order_times = [tuple(self.fix_time(order_time, html, page_time)) for order_time in order_times]
- order_times = [order_time for order_time in order_times if order_time[0]!=""]
- if len(set(order_times))==1:
- order_begin,order_end = order_times[0]
- order_begin, order_end = self._validate_date_range(order_begin, order_end, page_time)
- if order_end!="":
- project_name = codeName[0]['name']
- pack_info = [pack for pack in prem[0]['prem'].values()]
- budget = pack_info[0].get('tendereeMoney',0)
- product = prem[0]['product']
- link = {'project_name': project_name, 'product': product, 'demand': project_name, 'budget': budget,
- 'order_begin': order_begin, 'order_end': order_end}
- _data.append(link)
- product_attrs[1]['demand_info']['data'] = _data
- # print('predict_without_table: ', product_attrs)
- return product_attrs
- def predict_by_text(self,product_attrs,html,list_outlines,product_list,page_time=""):
- product_entity_list = list(set(product_list))
- list_outline = list_outlines[0]
- get_product_attrs = False
- for _outline in list_outline:
- if re.search("信息|情况|清单|概况",_outline.outline_summary):
- outline_text = _outline.outline_text
- outline_text = outline_text.replace(_outline.outline_summary,"")
- key_value_list = [_split for _split in re.split("[,。;]",outline_text) if re.search("[::]",_split)]
- if not key_value_list:
- continue
- head_list = []
- head_value_list = []
- for key_value in key_value_list:
- key_value = re.sub("^[一二三四五六七八九十]{1,3}[、.]|^[\d]{1,2}[、.]\d{,2}|^[\((]?[一二三四五六七八九十]{1,3}[\))][、]?","",key_value)
- temp = re.split("[::]",key_value)
- if len(temp)>2:
- if temp[0] in head_list:
- key = temp[0]
- value = "".join(temp[1:])
- else:
- key = temp[-2]
- value = temp[-1]
- else:
- key = temp[0]
- value = temp[1]
- key = re.sub("^[一二三四五六七八九十]{1,3}[、.]|^[\d]{1,2}[、.]\d{,2}|^[\((]?[一二三四五六七八九十]{1,3}[\))][、]?","",key)
- head_list.append(key)
- head_value_list.append(value)
- head_set = set(head_list)
- # print('head_set',head_set)
- if len(head_set & self.header_set) > len(head_set)*0.2:
- loop_list = []
- begin_list = [0]
- for index,head in enumerate(head_list):
- if head not in loop_list:
- if re.search('第[一二三四五六七八九十](包|标段)', head) and re.search('第[一二三四五六七八九十](包|标段)', '|'.join(loop_list)):
- begin_list.append(index)
- loop_list = []
- loop_list.append(head)
- else:
- loop_list.append(head)
- else:
- begin_list.append(index)
- loop_list = []
- loop_list.append(head)
- headers = []
- headers_demand = []
- header_col = []
- product_link = []
- demand_link = []
- product_set = set()
- for idx in range(len(begin_list)):
- if idx==len(begin_list)-1:
- deal_list = head_value_list[begin_list[idx]:]
- tmp_head_list = head_list[begin_list[idx]:]
- else:
- deal_list = head_value_list[begin_list[idx]:begin_list[idx+1]]
- tmp_head_list = head_list[begin_list[idx]:begin_list[idx+1]]
- product = "" # 产品
- quantity = "" # 数量
- quantity_unit = "" # 单位
- unitPrice = "" # 单价
- brand = "" # 品牌
- specs = "" # 规格
- demand = "" # 采购需求
- budget = "" # 预算金额
- order_time = "" # 采购时间
- order_begin = ""
- order_end = ""
- total_price = "" # 总金额
- parameter = "" # 参数
- header_dic, found_header, header_list, header_list2 = self.find_header(tmp_head_list, self.pat_category, self.pat_product_primary, self.pat_product_secondary, self.pat_project)
- if found_header:
- headers.append('_'.join(header_list))
- headers_demand.append('_'.join(header_list2))
- header_col.append('_'.join(tmp_head_list))
- # print('header_dic: ',header_dic)
- id0 = header_dic.get('品目', "")
- id1 = header_dic.get('名称', "")
- id2 = header_dic.get('数量', "")
- id2_2 = header_dic.get('单位', "")
- id3 = header_dic.get('单价', "")
- id4 = header_dic.get('品牌', "")
- id5 = header_dic.get('规格', "")
- id6 = header_dic.get('需求', "")
- id7 = header_dic.get('预算', "")
- id8 = header_dic.get('时间', "")
- id9 = header_dic.get("总价", "")
- id10 = header_dic.get('参数', "")
- if id1!='' and re.search('[a-zA-Z\u4e00-\u9fa5]', deal_list[id1]) and deal_list[id1] not in self.header_set and \
- re.search('备注|汇总|合计|总价|价格|金额|公司|附件|详见|无$|xxx', deal_list[id1]) == None:
- product = deal_list[id1]
- if id0 != "" and re.search('[a-zA-Z\u4e00-\u9fa5]', deal_list[id0]) and deal_list[id0] not in self.header_set and \
- re.search('备注|汇总|合计|总价|价格|金额|公司|附件|详见|无$|xxx', deal_list[id0]) == None:
- category = deal_list[id0]
- product = "%s_%s" % (category, product) if product != "" else category
- if product == "":
- # print(deal_list[id4],deal_list[id5],tmp_head_list,deal_list)
- if (id4 != "" and deal_list[id4] != "") or (id5 != "" and deal_list[id5] != ""):
- for head,value in zip(tmp_head_list,deal_list):
- if value and value in product_entity_list:
- product = value
- break
- if product and re.match('【?(工程类|服务类|货物类|工程|服务|货物|运费)】?', product) == None:
- if id2 != "":
- if re.search('\d+|[壹贰叁肆伍陆柒捌玖拾一二三四五六七八九十]', deal_list[id2]):
- quantity = deal_list[id2]
- quantity = re.sub('[()(),,约]', '', quantity)
- quantity = re.sub('[一壹]', '1', quantity)
- ser = re.search('^(\d+(?:\.\d+)?)([㎡\w/]{,5})', quantity)
- if ser:
- quantity = str(ser.group(1))
- quantity_unit = ser.group(2)
- if float(quantity)>=10000*10000:
- quantity = ""
- quantity_unit = ""
- else:
- quantity = ""
- quantity_unit = ""
- if id2_2 != "":
- if re.search('^\w{1,4}$', deal_list[id2_2]):
- quantity_unit = deal_list[id2_2]
- else:
- quantity_unit = ""
- # if id2 != "":
- # if re.search('\d+|[壹贰叁肆伍陆柒捌玖拾一二三四五六七八九十]', deal_list[id2]):
- # quantity = deal_list[id2]
- # else:
- # quantity = ""
- if id3 != "":
- if re.search('\d+|[零壹贰叁肆伍陆柒捌玖拾佰仟萬億十百千万亿元角分]{3,}', deal_list[id3]):
- _unitPrice = deal_list[id3]
- re_price = re.findall("[零壹贰叁肆伍陆柒捌玖拾佰仟萬億圆十百千万亿元角分]{3,}|\d[\d,]*(?:\.\d+)?万?",_unitPrice)
- if re_price:
- # _unitPrice = re_price[0]
- # if '万元' in header_list[3] and '万' not in _unitPrice:
- # _unitPrice += '万元'
- # unitPrice = getUnifyMoney(_unitPrice)
- # if unitPrice>=10000*10000:
- # unitPrice = ""
- # unitPrice = str(unitPrice)
- _unitPrice, _money_unit = money_process(_unitPrice, header_list[3])
- if _unitPrice >= 10000 * 10000:
- _unitPrice = ""
- unitPrice = str(_unitPrice)
- if '.' in unitPrice:
- unitPrice = unitPrice.rstrip('0').rstrip('.')
- if id4 != "":
- if re.search('\w', deal_list[id4]):
- brand = deal_list[id4]
- if re.match('^详见|^详情', brand.strip()):
- brand = ""
- else:
- brand = ""
- if id5 != "":
- if re.search('\w', deal_list[id5]):
- specs = deal_list[id5][:self.MAX_SPECS_LEN]
- if re.match('^详见|^详情', specs.strip()):
- specs = ""
- else:
- specs = ""
- if id6 != "":
- if re.search('\w', deal_list[id6]):
- demand = deal_list[id6]
- else:
- demand = ""
- if id7 != "":
- if re.search('\d+|[零壹贰叁肆伍陆柒捌玖拾佰仟萬億十百千万亿元角分]{3,}', deal_list[id7]):
- _budget = deal_list[id7]
- re_price = re.findall("[零壹贰叁肆伍陆柒捌玖拾佰仟萬億圆十百千万亿元角分]{3,}|\d[\d,]*(?:\.\d+)?万?",_budget)
- if re_price:
- # _budget = re_price[0]
- # if '万元' in header_list2[2] and '万' not in _budget:
- # _budget += '万元'
- # budget = str(getUnifyMoney(_budget))
- _budget, _money_unit = money_process(_budget, header_list2[2])
- budget = str(_budget)
- if '.' in budget:
- budget = budget.rstrip('0').rstrip('.')
- if float(budget)>= 100000*10000:
- budget = ""
- if id8 != "":
- if re.search('\w', deal_list[id8]) and re.search("(采购|采购实施|(预计|计划)(招标|采购|发标|发包))(时间|月份|日期)",header_list2[3]):
- order_time = deal_list[id8].strip()
- order_begin, order_end = self.fix_time(order_time, html, page_time)
- if id9 != "":
- if re.search('[零壹贰叁肆伍陆柒捌玖拾佰仟萬億十百千万亿元角分]{3,}', deal_list[id9]):
- total_price = deal_list[id9]
- elif re.search('^[\d,.亿万元人民币欧美日金额:()();;、,\n]+$', deal_list[id9].strip()):
- total_price = deal_list[id9]
- if id10 != "":
- parameter = deal_list[id10][:self.MAX_PARAM_LEN]
- if re.match('^详见|^详情', parameter.strip()):
- parameter = ""
- if quantity != "" or unitPrice != "" or brand != "" or specs != "" or total_price:
- if id1 != "" and id2 != "" and id3 != "" and len(re.split('[;;、,\n]', deal_list[id2])) > 1 and len(
- re.split('[;;、,\n]', deal_list[id1])) == len(re.split('[;;、,\n]', deal_list[id2])): # 处理一个空格包含多个产品,逗号或空格分割情况 例子 292846806 292650743
- products = re.split('[;;、,\n]', deal_list[id1])
- quantitys = re.split('[;;、,\n]', deal_list[id2])
- unitPrices = re.split('[;;、,\n]', deal_list[id3])
- total_prices = re.split('[;;、,\n]', total_price)
- brands = re.split('[;;、,\n]', brand) if re.search('等$', brand) == None else [brand]
- specses = re.split('[;;、,\n]', specs) if re.search('等$', specs) == None else [specs]
- parameters = re.split('[;;、,\n]', parameter) if re.search('等$', parameter) == None else [parameter]
- unitPrices = [""] * len(products) if len(unitPrices) == 1 else unitPrices
- total_prices = [""] * len(products) if len(total_prices) == 1 else total_prices
- brands = brands * len(products) if len(brands) == 1 else brands
- specses = specses * len(products) if len(specses) == 1 else specses
- parameters = parameters * len(products) if len(parameters) == 1 else parameters
- if len(products) == len(quantitys) == len(unitPrices) == len(brands) == len(
- specses):
- for product, quantity, unitPrice, brand, specs, total_price, parameter in zip(
- products, quantitys, unitPrices, brands, specses, total_prices,
- parameters):
- if quantity != "":
- quantity, quantity_unit_ = self.fix_quantity(quantity,quantity_unit)
- quantity_unit = quantity_unit_ if quantity_unit_ != "" else quantity_unit
- if unitPrice != "":
- unitPrice, _money_unit = money_process(unitPrice, header_list[3])
- unitPrice = str(unitPrice) if unitPrice != 0 and unitPrice<self.MAX_UNIT_PRICE else ""
- if budget != "":
- budget, _money_unit = money_process(budget, header_list2[2])
- budget = str(budget) if budget != 0 and budget<self.MAX_BUDGET else ''
- if total_price != "":
- total_price, _money_unit = money_process(total_price,
- header_list[6])
- total_price = str(total_price) if total_price != 0 and total_price<self.MAX_TOTAL_PRICE else ""
- link = {'product': product, 'quantity': quantity,
- 'quantity_unit': quantity_unit, 'unitPrice': unitPrice,
- 'brand': brand[:self.MAX_BRAND_LEN], 'specs': specs, 'total_price': total_price,
- 'parameter': parameter}
- if (product, specs, unitPrice, quantity) not in product_set: # and not header_dic['产品为项目']:
- product_set.add((product, specs, unitPrice, quantity))
- product_link.append(link)
- elif len(unitPrice) > 15 or len(product) > self.MAX_PRODUCT_NAME_LEN:
- # i += 1
- continue
- else:
- if quantity != "":
- quantity, quantity_unit_ = self.fix_quantity(quantity, quantity_unit)
- quantity_unit = quantity_unit_ if quantity_unit_ != "" else quantity_unit
- if unitPrice != "":
- unitPrice, _money_unit = money_process(unitPrice, header_list[3])
- unitPrice = str(unitPrice) if unitPrice != 0 and unitPrice<self.MAX_UNIT_PRICE else ""
- if budget != "":
- budget, _money_unit = money_process(budget, header_list2[2])
- budget = str(budget) if budget != 0 and budget<self.MAX_BUDGET else ''
- if total_price != "":
- total_price, _money_unit = money_process(total_price, header_list[6])
- total_price = str(total_price) if total_price != 0 and total_price<self.MAX_TOTAL_PRICE else ""
- link = {'product': product, 'quantity': quantity,
- 'quantity_unit': quantity_unit, 'unitPrice': unitPrice,
- 'brand': brand[:self.MAX_BRAND_LEN], 'specs': specs, 'total_price': total_price,
- 'parameter': parameter}
- if (product, specs, unitPrice, quantity) not in product_set: # and not header_dic['产品为项目']:
- product_set.add((product, specs, unitPrice, quantity))
- product_link.append(link)
- order_begin, order_end = self._validate_date_range(order_begin, order_end, page_time)
- if order_begin != "" and order_end != "":
- order_begin_year = int(order_begin.split("-")[0])
- order_end_year = int(order_end.split("-")[0])
- if order_begin_year < 2000 or order_end_year < 2000:
- order_begin = order_end = ""
- if budget != "" and order_end != "":
- link = {'project_name': product, 'product': [], 'demand': demand, 'budget': budget,
- 'order_begin': order_begin, 'order_end': order_end}
- if link not in demand_link:
- demand_link.append(link)
- if len(product_link) > 0:
- attr_dic = {'product_attrs': {'data': product_link, 'header': list(set(headers)), 'header_col': list(set(header_col))}}
- get_product_attrs = True
- else:
- attr_dic = {'product_attrs': {'data': [], 'header': [], 'header_col': []}}
- if len(demand_link) > 0:
- demand_dic = {'demand_info': {'data': demand_link, 'header': headers_demand, 'header_col': header_col}}
- else:
- demand_dic = {'demand_info': {'data': [], 'header': [], 'header_col': []}}
- product_attrs[0] = attr_dic
- if len(product_attrs[1]['demand_info']['data']) == 0:
- product_attrs[1] = demand_dic
- if get_product_attrs:
- break
- # print('predict_by_text: ', product_attrs)
- return product_attrs
- def add_product_attrs(self,channel_dic, product_attrs, list_sentences,list_entitys,list_outlines,product_list,codeName,prem,text,page_time):
- # print(1,product_attrs[1]['demand_info']['data'])
- if channel_dic['docchannel']['docchannel']=="采购意向" and len(product_attrs[1]['demand_info']['data']) == 0:
- product_attrs = self.predict_without_table(product_attrs, list_sentences,list_entitys,codeName,prem,text,page_time)
- # print(2,product_attrs[1]['demand_info']['data'])
- if len(product_attrs[0]['product_attrs']['data']) == 0:
- product_attrs = self.predict_by_text(product_attrs,text,list_outlines,product_list,page_time)
- # print(3,product_attrs[1]['demand_info']['data'])
- if len(product_attrs[1]['demand_info']['data'])>0:
- for d in product_attrs[1]['demand_info']['data']:
- for product in set(prem[0]['product']):
- if product in d['project_name'] and product not in d['product']:
- d['product'].append(product) #把产品在项目名称中的添加进需求要素中
|