# -*- coding: utf-8 -*- """产品属性预测器(ProductAttributesPredictor)。 按 ARCHITECTURE.md Phase 5 拆分建议,从 ``interface/predictor.py`` 迁出以下 产品属性提取相关类: - ``TableResult`` — 表格提取结果数据类(原 predictor.py 第 3330-3368 行) - ``ProductAttributesPredictor`` — 产品数量/单价/品牌/规格/表格要素提取 (原 predictor.py 第 3371-4719 行) 原 ``from common.Utils import *`` / ``from common.nerUtils import *`` 已替换为 显式 import;``os.path.dirname(__file__)`` 路径引用替换为 ``predictors._common.INTERFACE_DIR``。 ``interface/predictor.py`` 仍 re-export 以上全部名称,老 import 不受影响。 """ from __future__ import absolute_import import os import re import copy import pickle import calendar import datetime from bs4 import BeautifulSoup from dataclasses import dataclass, field from typing import List, Dict, Set, Any from BiddingKG.dl.common.logging import log from BiddingKG.dl.common.context_utils import money_process, spanWindow from BiddingKG.dl.common.Utils import del_tabel_achievement from BiddingKG.dl.predictors._common import INTERFACE_DIR from BiddingKG.dl.predictors.table_prem import TableTag2List __all__ = [ "TableResult", "ProductAttributesPredictor", ] @dataclass class TableResult: table_index: int = 0 product_confirm: List[int] = field(default_factory=list) headers: List[str] = field(default_factory=list) headers_demand: List[str] = field(default_factory=list) header_col: List[str] = field(default_factory=list) product_link: List[Dict[str, Any]] = field(default_factory=list) demand_link: List[Dict[str, Any]] = field(default_factory=list) product_set: Set[tuple] = field(default_factory=set) total_product_money: float = 0 unit_price_list: List[str] = field(default_factory=list) total_price_list: list = field(default_factory=list) budget_list: list = field(default_factory=list) @property def product_names(self) -> Set[str]: return {p.get('product', '') for p in self.product_link if p.get('product', '')} @property def header_signature(self) -> str: return '|'.join(sorted(set(self.headers))) @property def demand_header_signature(self) -> str: return '|'.join(sorted(set(self.headers_demand))) def product_count(self) -> int: return len(self.product_link) def demand_count(self) -> int: return len(self.demand_link) def avg_attrs_per_product(self) -> float: if not self.product_link: return 0 total = sum(len([v for v in p.values() if v != '']) for p in self.product_link) return total / len(self.product_link) # 产品数量单价品牌规格提取 #2021/11/10 添加表格中的项目、需求、预算、时间要素提取 class ProductAttributesPredictor(): MAX_UNIT_PRICE = 100000000 MAX_TOTAL_PRICE = 50000000000 MAX_BUDGET = 50000000000 MAX_PRODUCT_NAME_LEN = 100 MAX_BRAND_LEN = 50 MAX_SPECS_LEN = 500 MAX_PARAM_LEN = 500 MAX_TENDEREE_LEN = 30 MAX_QUANTITY_FOR_CALC = 50000 FUTURE_YEAR_LIMIT = 2050 def __init__(self,): # self.pat_category = '(类别|类型|物类|目录|类目|分类)(名称|$)|^品名|^品类|^品目|(标项|分项|项目|计划|包组|标段|[分子]?包|子目|服务|招标|中标|成交|工程|招标内容)(名称|内容|描述)' self.pat_category = '(品目|品类)名称?|采购(品目|品类)$|^品名|^品类|^品目$' # self.pat_product_primary = '(标的|维修|系统|报价构成|商品|产品|物料|物资|货物|设备|采购品|采购条目|物品|材料|印刷品?|采购|物装|配件|资产|耗材|清单|器材|仪器|器械|备件|拍卖物|标的物|物件|药品|药材|药械|货品|食品|食材|品目|^品名|气体)[\))的]?(名称|内容|描述)' self.pat_product_primary = '(标的|商品|产品|物料|物资|货物|设备|采购品|采购条目|物品|材料|印刷品?|采购|物装|配件|资产|耗材|清单|器材|仪器|器械|备件|拍卖物|标的物|物件|药品|药材|药械|疫苗|货品|食品|食材|品目|^品名|气体)[\))的]?(名称|内容|描述)' # self.pat_product_secondary = '标的|标项|项目$|商品|产品|物料|物资|货物|设备|采购品|采购条目|物品|材料|印刷品|物装|配件|资产|招标内容|耗材|清单|器材|仪器|器械|备件|拍卖物|标的物|物件|药品|药材|药械|货品|食品|食材|菜名|^品目$|^品名$|^名称|^内容$|(标项|分项|项目|计划|包组|标段|[分子]?包|子目|服务|招标|中标|成交|工程|招标内容)(名称|内容|描述)' self.pat_product_secondary = '标的|商品|产品|物料|物资|货物|设备|采购品|采购条目|物品|材料|印刷品|物装|配件|资产|耗材|清单|器材|仪器|器械|备件|拍卖物|标的物|物件|药品|药材|药械|货品|食品|食材|菜名|^品目$|^品名$' self.pat_project = '(标项|分项|项目|计划|包组|标段|[分子]?包|子目|服务|招标|中标|成交|工程)(名称|内容|描述)|^名称|采购类别' with open(os.path.join(INTERFACE_DIR, 'header_set.pkl'), 'rb') as f: self.header_set = pickle.load(f) self.tb = TableTag2List() def isTrueTable(self, table): '''真假表格规则: 1、包含或标签为真 2、包含大量链接、表单、图片或嵌套表格为假 3、表格尺寸太小为假 4、外层嵌套子
,一般子为真,外为假''' if table.find_all(['caption', 'th']) != []: return True # elif len(table.find_all(['form', 'a', 'img'])) > 5: # 20260602 去掉,某些表格可能有多个链接 例子:437440313 # # print('过滤表格:包含链接图片等大于5的为假表格') # return False elif len(table.find_all(['tr'])) < 2: # print('过滤表格:行数小于2的为假表格') return False elif len(table.find_all(['table'])) >= 1: # print('过滤表格:包含多个表格的为假表格') inner_table_num = len(table.find_all(['table'])) text_num = 0 # 表格只有一格作文本框的数量,docid:631910513 for inner_table in table.find_all(['table']): if len(inner_table.find_all(['tr']))==0 or (len(inner_table.find_all(['tr']))==1 and len(inner_table.find_all(['tr'])[0].find_all(['td']))<=1): text_num += 1 if inner_table_num - text_num > 0: return False else: return True else: return True def getTrs(self, tbody): # 获取所有的tr trs = [] objs = tbody.find_all(recursive=False) for obj in objs: if obj.name == "tr": trs.append(obj) if obj.name == "tbody": for tr in obj.find_all("tr", recursive=False): trs.append(tr) return trs def getTable(self, tbody): trs = self.getTrs(tbody) inner_table = [] if len(trs) < 2: return inner_table for tr in trs: tr_line = [] tds = tr.findChildren(['td', 'th'], recursive=False) if len(tds) < 2: continue for td in tds: # td_text = re.sub('\s+|…', ' ', td.get_text()).strip() td_text = re.sub('…', '', td.get_text()).strip() td_text = re.sub('\n+|\s+', ' ', td_text) # 20250626 去掉\n等避免存OTS后去掉转义导致json解析错误 td_text = td_text.replace("\x06", "").replace("\x05", "").replace("\x07", "").replace('\\', '/').replace('"', '') # 修复272144312 # 产品单价数量提取结果有特殊符号\ 气动执行装置备件\密封组件\NBR+PT td_text = td_text.replace("(", "(").replace(")", ")").replace(':', ':') tr_line.append(td_text) inner_table.append(tr_line) return inner_table def fixSpan(self, tbody): # 处理colspan, rowspan信息补全问题 trs = self.getTrs(tbody) ths_len = 0 ths = list() trs_set = set() # 修改为先进行列补全再进行行补全,否则可能会出现表格解析混乱 # 遍历每一个tr for indtr, tr in enumerate(trs): ths_tmp = tr.findChildren('th', recursive=False) if len(tr.findChildren('table')) > 0: continue if len(ths_tmp) > 0: for th in ths_tmp: ths.append(th) trs_set.add(tr) # 遍历每行中的element tds = tr.findChildren(recursive=False) if len(tds) < 3: continue # 列数太少的不补全 for indtd, td in enumerate(tds): # 若有colspan 则补全同一行下一个位置 if 'colspan' in td.attrs and str(re.sub("[^0-9]", "", str(td['colspan']))) != "": col = int(re.sub("[^0-9]", "", str(td['colspan']))) if col < 10 and len(td.get_text()) < 500: td['colspan'] = 1 for i in range(1, col, 1): td.insert_after(copy.copy(td)) for indtr, tr in enumerate(trs): ths_tmp = tr.findChildren('th', recursive=False) # 不补全含有表格的tr if len(tr.findChildren('table')) > 0: continue if len(ths_tmp) > 0: ths_len = ths_len + len(ths_tmp) for th in ths_tmp: ths.append(th) trs_set.add(tr) # 遍历每行中的element tds = tr.findChildren(recursive=False) same_span = 0 if len(tds) > 1 and 'rowspan' in tds[0].attrs: span0 = tds[0].attrs['rowspan'] for td in tds: if 'rowspan' in td.attrs and td.attrs['rowspan'] == span0: same_span += 1 if same_span == len(tds): continue for indtd, td in enumerate(tds): # 若有rowspan 则补全下一行同样位置 if 'rowspan' in td.attrs and str(re.sub("[^0-9]", "", str(td['rowspan']))) != "": row = int(re.sub("[^0-9]", "", str(td['rowspan']))) td['rowspan'] = 1 for i in range(1, row, 1): # 获取下一行的所有td, 在对应的位置插入 if indtr + i < len(trs): tds1 = trs[indtr + i].findChildren(['td', 'th'], recursive=False) if len(tds1) >= (indtd) and len(tds1) > 0: if indtd > 0: tds1[indtd - 1].insert_after(copy.copy(td)) else: tds1[0].insert_before(copy.copy(td)) elif len(tds1) > 0 and len(tds1) == indtd - 1: tds1[indtd - 2].insert_after(copy.copy(td)) def get_monthlen(self, year, month): '''输入年份、月份 int类型 得到该月份天数''' try: weekday, num = calendar.monthrange(int(year), int(month)) except (ValueError, TypeError): num = 30 return str(num) def _validate_date_range(self, order_begin, order_end, page_time): if not order_end > page_time: return "", "" if order_begin != "" and order_end != "": order_begin_year = int(order_begin.split("-")[0]) order_end_year = int(order_end.split("-")[0]) if order_begin_year >= self.FUTURE_YEAR_LIMIT or order_end_year >= self.FUTURE_YEAR_LIMIT: return "", "" return order_begin, order_end def _build_month_range(self, year, month): month = str(month).zfill(2) num = self.get_monthlen(year, month).zfill(2) order_begin = "%s-%s-01" % (year, month) order_end = "%s-%s-%s" % (year, month, num) return order_begin, order_end def _resolve_year_for_month(self, html, page_time): year = re.search('(\d{4})年(.{,12}采购意向)?', html) if year: return year.group(1) if page_time != "": year = re.search('\d{4}', page_time) if year: return year.group(0) return str(datetime.datetime.now().year) def fix_time(self, text, html, page_time): for it in [('十二', '12'),('十一', '11'),('十','10'),('九','9'),('八','8'),('七','7'), ('六','6'),('五','5'),('四','4'),('三','3'),('二','2'),('一','1')]: if it[0] in text: text = text.replace(it[0], it[1]) if re.search('^\d{1,2}月$', text): m = re.search('^(\d{1,2})月$', text).group(1) y = self._resolve_year_for_month(html, page_time) return self._build_month_range(y, m) t1 = re.search('^(\d{4})(年|/|\.|-)(\d{1,2})月?$', text) if t1: return self._build_month_range(t1.group(1), t1.group(3)) t2 = re.search('^(\d{4})(年|/|\.|-)(\d{1,2})(月|/|\.|-)(\d{1,2})日?$', text) if t2: y = t2.group(1) m = t2.group(3).zfill(2) d = t2.group(5).zfill(2) order_begin = order_end = "%s-%s-%s"%(y,m,d) return order_begin, order_end t3 = re.search("^(20\d{2})(\d{1,2})$",text) if t3: year = t3.group(1) month = t3.group(2) if int(month)>0 and int(month)<=12: return self._build_month_range(year, month) t4 = re.search("^(20\d{2})(\d{2})(\d{2})$", text) if t4: year = t4.group(1) month = t4.group(2) day = t4.group(3) if int(month) > 0 and int(month) <= 12 and int(day)>0 and int(day)<=31: order_begin = order_end = "%s-%s-%s"%(year,month,day) return order_begin, order_end all_match = re.finditer('^(?P\d{4})(年|/|\.)(?P\d{1,2})(?:(月|/|\.)(?:(?P\d{1,2})日)?)?' '(到|至|-)(?:(?P\d{4})(年|/|\.))?(?P\d{1,2})(?:(月|/|\.)' '(?:(?P\d{1,2})日)?)?$', text) y1 = m1 = d1 = y2 = m2 = d2 = "" found_math = False for _match in all_match: if len(_match.group()) > 0: found_math = True for k, v in _match.groupdict().items(): if v!="" and v is not None: if k == 'y1': y1 = v elif k == 'm1': m1 = v elif k == 'd1': d1 = v elif k == 'y2': y2 = v elif k == 'm2': m2 = v elif k == 'd2': d2 = v if not found_math: return "", "" y2 = y1 if y2 == "" else y2 d1 = '1' if d1 == "" else d1 d2 = self.get_monthlen(y2, m2) if d2 == "" else d2 m1 = '0' + m1 if len(m1) < 2 else m1 m2 = '0' + m2 if len(m2) < 2 else m2 d1 = '0' + d1 if len(d1) < 2 else d1 d2 = '0' + d2 if len(d2) < 2 else d2 order_begin = "%s-%s-%s"%(y1,m1,d1) order_end = "%s-%s-%s"%(y2,m2,d2) return order_begin, order_end def fix_quantity(self, quantity_text, header_quan_unit): ''' 产品数量标准化,统一为数值型字符串 :param quantity_text: 原始数量字符串 :param header_quan_unit: 表头数量单位字符串 :return: 返回数量及单位 ''' quantity = quantity_text quantity = re.sub('[一壹]', '1', quantity) quantity = re.sub('[,,约]|(\d+)', '', quantity) ser = re.search('^(\d+\.?\d*)(?([㎡\w/]{,5})', quantity) if ser: quantity = str(ser.group(1)) quantity_unit = ser.group(2) if quantity_unit == "" and header_quan_unit != "": quantity_unit = header_quan_unit else: quantity = "" quantity_unit = "" return quantity, quantity_unit def find_header(self, items,p0, p1, p2, pat_project): ''' inner_table 每行正则检查是否为表头,是则返回表头所在列序号,及表头内容 :param items: 列表,内容为每个td 文本内容 :param p1: 优先表头正则 :param p2: 第二表头正则 :param pat_project: 项目表头正则 :return: 表头所在列序号,是否表头,表头内容 ''' items = [re.sub('\s', '', it) for it in items] flag = False header_dic = {'产品为项目': False,'名称': '', '数量': '', '单位': '', '单价': '', '品牌': '', '规格': '', '需求': '', '预算': '', '时间': '', '总价': '', '品目': '', '参数': '', '采购人':'', '备注':'','发布日期':'', '品目号':'', '品目名':''} product = "" # 产品 quantity = "" # 数量 quantity_unit = "" # 数量单位 unitPrice = "" # 单价 brand = "" # 品牌 specs = "" # 规格 demand = "" # 采购需求 budget = "" # 预算金额 order_time = "" # 采购时间 total_price = "" # 总价 category = "" # 品目 parameter = "" # 参数 tenderee = "" # 采购人 notes = "" # 备注 2024/3/27 达仁 需求 issue_date = "" # 发布日期 2024/3/27 达仁 需求 pinmu_no = "" # 品目号 pinmu_name = "" # 品目名称 product_secondary = '' product_idx = '' project = '' project_idx = '' for i in range(int(len(items)*0.75)): it = items[i] if len(it) < 15 and re.search(p0, it): flag = True if category != "" and category != it: continue category = it header_dic['品目'] = i elif len(it) < 15 and re.search(p1, it): flag = True if product !='' and product != it: break product = it header_dic['名称'] = i # break if len(it) < 15 and it != category and re.search(p2, it) and (re.search('^名称|^品名|^品目$', it) or re.search( '编号|编码|号|情况|报名|单位|位置|地址|数量|单价|价格|金额|品牌|规格类型|型号|公司|中标人|企业|供应商|候选人', it) == None): product_secondary = it product_idx = i if len(it) < 15 and re.search(pat_project, it) and '项目名称' not in project: project = it project_idx = i if product == "": if product_secondary: flag = True product = product_secondary header_dic['名称'] = product_idx elif category: flag = True product = category header_dic['名称'] = header_dic['品目'] header_dic['品目'] = '' category = '' elif project: flag = True product = project header_dic['名称'] = project_idx header_dic['产品为项目'] = True if flag == False and len(items)>3 and re.search('^第[一二三四五六七八九十](包|标段)$', items[0]): product = items[0] header_dic['名称'] = 0 flag = True if flag: for j in range(len(items)): if header_dic['品目号'] == "" and re.search('(品目|品类)(编?号|编码|序号)', items[j]): header_dic['品目号'] = j pinmu_no = items[j] elif header_dic['品目名'] == "" and re.search('(品目|品类)名称|采购(品目|品类)$', items[j]): header_dic['品目名'] = j pinmu_name = items[j] if items[j] in [product, category]: continue if len(items[j]) > 20 and len(re.sub('[\((].*[)\)]|[^\u4e00-\u9fa5]', '', items[j])) > 10: continue if header_dic['数量']=="" and re.search('数量|采购量', items[j]) and re.search('单价|用途|要求|规格|型号|运输|承运', items[j])==None: header_dic['数量'] = j quantity = items[j] quantity = re.sub('\d', '', quantity) elif header_dic['单位']=="" and re.search('^(数量单位|计量单位|单位)$', items[j]): header_dic['单位'] = j quantity_unit = items[j] elif re.search('单价', items[j]) and re.search('数量|规格|型号|品牌|供应商', items[j])==None: header_dic['单价'] = j unitPrice = items[j] unitPrice = re.sub('\d', '', unitPrice) elif re.search('品牌', items[j]): header_dic['品牌'] = j brand = items[j] elif re.search('规格|型号', items[j]): header_dic['规格'] = j specs = items[j] elif re.search('参数', items[j]): header_dic['参数'] = j parameter = items[j] elif re.search('预算单位|(采购|招标|购买)(单位|人|方|主体)|项目业主|采购商|申购单位|需求单位|业主单位',items[j]) and len(items[j])<=8: header_dic['采购人'] = j tenderee = items[j] elif re.search('需求|服务要求|服务标准', items[j]): header_dic['需求'] = j demand = items[j] elif re.search('(采购|招标|投资|项目)(预算|估算)|(预算|控制|投资|项目|采购|招标)金额|(最高|招标)(\w{,2})限价|拦标价', items[j]) and not re.search('预算单位',items[j]): header_dic['预算'] = j budget = items[j] elif re.search('时间|采购时间|采购实施月份|采购月份|采购日期|(预计|计划)(招标|采购|发标|发包)(时间|月份)', items[j]): header_dic['时间'] = j order_time = items[j] elif re.search('总价|(成交|中标|验收|合同|预算|控制|总|合计))?([金总]额|价格?)|最高限价|价格|金额', items[j]) and re.search('数量|规格|型号|品牌|供应商', items[j])==None: header_dic['总价'] = j total_price = items[j] total_price = re.sub('\d', '', total_price) elif re.search('^备\s*注$|资质要求|预留面向中小企业|是否适宜中小企业采购预算预留|公开征集信息', items[j]): header_dic['备注'] = j notes = items[j] elif re.search('^\w{,4}发布(时间|日期)$', items[j]): header_dic['发布日期'] = j issue_date = items[j] if header_dic.get('名称', "") != "" or header_dic.get('品目', "") != "": # num = 0 # for it in (quantity, unitPrice, brand, specs, product, demand, budget, order_time, total_price): # if it != "": # num += 1 # if num >=2: # return header_dic, flag, (product, quantity, quantity_unit, unitPrice, brand, specs, total_price, category, parameter), (product, demand, budget, order_time) if set([quantity, brand, specs, unitPrice, total_price])!=set([""]) or set([demand, budget])!=set([""]): # if header_dic['产品为项目'] and (brand or specs): # header_dic['产品为项目'] = False return header_dic, flag, (product, quantity, quantity_unit, unitPrice, brand, specs, total_price, category, parameter, pinmu_no, pinmu_name), (product, demand, budget, order_time,tenderee, notes,issue_date) flag = False return header_dic, flag, (product, quantity, quantity_unit, unitPrice, brand, specs, total_price, category, parameter, pinmu_no, pinmu_name), (product, demand, budget, order_time,tenderee,notes,issue_date) def predict(self, docid='', html='', page_time=""): html = html.replace('
', '\n').replace('
', '\n') html = re.sub("|||","",html) html = re.sub("##attachment##","",html) soup = BeautifulSoup(html, 'lxml') del_tabel_achievement(soup) richText = soup.find(name='div', attrs={'class': 'richTextFetch'}) if richText: richText = richText.extract() # 过滤掉附件 def extract_product(soup): flag_yx = True tables = soup.find_all(['table']) table_results: List[TableResult] = [] for table_idx in range(len(tables)): table = tables[table_idx] if table.parent.name == 'td' and len(table.find_all('td')) <= 3: table.string = table.get_text() table.name = 'turntable' continue if not self.isTrueTable(table): continue self.fixSpan(table) inner_table = self.getTable(table) table.extract() tr = self._extract_from_single_table( table_idx, inner_table, html, page_time, flag_yx ) if tr.product_link and sum(tr.product_confirm) == 0: products = [re.sub('([^)]+)', '', it[0]) for it in tr.product_set] is_project = [1 if re.search('.{10,}(工程|项目|施工)', product) else 0 for product in products] if sum(is_project)>len(is_project)*0.5 or len(tr.product_link[0]) < 2: # 不确定产品,一半产品包含项目工程关键词或只有产品的去掉。 tr.product_link = [] tr.headers = [] tr.total_product_money = 0 log('去除项目工程名称作产品:%s, docid:%s'%(' '.join(products), docid)) if tr.product_link or tr.demand_link: table_results.append(tr) if len(inner_table) > 0: last_tds = inner_table[-1] if inner_table else [] if len(last_tds) >= 2 and len(set(last_tds)) in [2, 3] and re.search('订单总价', last_tds[0]) and re.search( '\d+[\d,\.]*', last_tds[1]): money_, unit_ = money_process(last_tds[1], last_tds[0]) if table_results: table_results[-1].total_product_money = max(money_, table_results[-1].total_product_money) return table_results table_results = extract_product(soup) if (len(table_results) < 1 or sum([it for l in table_results for it in l.product_confirm])==0) and richText: # 正文没提取或提取产品不确定补充附件提取 table_results_richText = extract_product(richText) table_results.extend(table_results_richText) merged = self._merge_table_results(table_results) product_link = merged.product_link demand_link = merged.demand_link total_product_money = merged.total_product_money headers = merged.headers headers_demand = merged.headers_demand header_col = merged.header_col budget_list = merged.budget_list result = self._post_process(product_link, demand_link, merged.total_price_list, merged.unit_price_list, budget_list) if result == 0: total_product_money = 0 if len(product_link)>0: product_link = [{k:v for k,v in d.items() if v!=''} for d in product_link] attr_dic = {'product_attrs':{'data':product_link, 'header':headers, 'header_col':header_col}} attr_dic['product_attrs']['product_confirm'] = sum(merged.product_confirm) # 是否明确产品 total_budget = sum(budget_list) if len(budget_list) == len(product_link) else 0 if not attr_dic['product_attrs']['product_confirm']: if len(product_link[0]) < 3: # 产品不确定且要素少于3个的去掉 attr_dic = {'product_attrs': {'data': [], 'header': [], 'header_col': []}} else: log('产品属性提取不确定,产品数量:%d,要素数量:%d,docid:%s'%(len(product_link), len(product_link[0]), docid)) else: attr_dic = {'product_attrs': {'data': [], 'header': [], 'header_col': []}} total_budget = 0 if len(demand_link)>0: demand_link = [{k: v for k, v in d.items() if v != ''} for d in demand_link] demand_dic = {'demand_info':{'data':demand_link, 'header':headers_demand, 'header_col':header_col}} else: demand_dic = {'demand_info':{'data':[], 'header':[], 'header_col':[]}} return [attr_dic, demand_dic], total_product_money, total_budget def _extract_from_single_table(self, table_idx, inner_table, html, page_time, flag_yx) -> TableResult: tr = TableResult(table_index=table_idx) found_header = False header_quan_unit = "" header_colnum = 0 header_dic = {} header_list = header_list2 = () if flag_yx: col0_l, col1_l = [], [] for tds in inner_table: if len(tds) == 2: col0_l.append(re.sub('[::]', '', tds[0])) col1_l.append(tds[1]) elif len(tds)>=4 and len(inner_table)==2: col0_l = inner_table[0] col1_l = inner_table[1] break if len(set(col0_l) & self.header_set) > len(col0_l) * 0.2 and len(col0_l)==len(col1_l): d_links, d_headers = self._extract_demand_from_2col(col0_l, col1_l, html, page_time) if d_links: tr.demand_link.extend(d_links) tr.headers_demand.extend(d_headers) return tr if len(inner_table)>3 and len(inner_table[0])==2 and len(inner_table[1])==2: col0_l, col1_l = [], [] for tds in inner_table: if len(tds) == 2: col0_l.append(re.sub('[::]', '', tds[0])) col1_l.append(tds[1]) else: break if len(set(col0_l) & self.header_set) > len(col0_l) * 0.5 and len(col0_l) == len(col1_l): inner_table = [col0_l, col1_l] elif len(inner_table)>2 and len(inner_table[0])==4 and len(inner_table[1])==4 and len(set(inner_table[0]) & self.header_set)==2: col0_l, col1_l, col2_l, col3_l = [], [], [], [] for tds in inner_table: if len(tds) == 4 and len(set(tds))>2: col0_l.append(re.sub('[::]', '', tds[0])) col1_l.append(tds[1]) col2_l.append(re.sub('[::]', '', tds[2])) col3_l.append(tds[3]) else: break if len(set(col0_l) & self.header_set) > len(col0_l) * 0.5 and len(set(col2_l) & self.header_set) > len(col2_l) * 0.5: inner_table = [col0_l+col2_l, col1_l+col3_l] row_idx = 0 while row_idx < (len(inner_table)): tds = inner_table[row_idx] not_empty = [it for it in tds if re.sub('\s', '', it) != ""] if len(set(not_empty))<2 or len(set(tds))<2 or (len(set(tds))==2 and re.search('总计|合计|汇总|总价', tds[0])): row_idx += 1 continue product = "" quantity = "" quantity_unit = "" unitPrice = "" brand = "" specs = "" demand = "" budget = "" order_time = "" order_begin = "" order_end = "" total_price = "" parameter = "" tenderee = "" notes = '' issue_date = '' pinmu_no = '' pinmu_name = '' if len(set([re.sub('[::\s]','',td) for td in tds]) & self.header_set) > len(tds) * 0.4: header_dic, found_header, header_list, header_list2 = self.find_header(tds, self.pat_category, self.pat_product_primary, self.pat_product_secondary, self.pat_project) if found_header: header_colnum = len(tds) if found_header and isinstance(header_list, tuple) and len(header_list) > 2: quantity_header = header_list[1].replace('单位:', '') if re.search('(([\w/]{,5}))', quantity_header): header_quan_unit = re.search('(([\w/]{,5}))', quantity_header).group(1) else: header_quan_unit = "" if found_header and ('_'.join(header_list) not in tr.headers or '_'.join(header_list2) not in tr.headers_demand): tr.headers.append('_'.join(header_list)) tr.headers_demand.append('_'.join(header_list2)) tr.header_col.append('_'.join(tds)) is_confirm = 0 if header_dic['产品为项目'] else 1 tr.product_confirm.append(is_confirm) row_idx += 1 continue elif found_header: if len(tds) > header_colnum or len(tds)-1= len(tds) or tds[v] in self.header_set: not_attr = 1 if not_attr>=2: row_idx += 1 found_header = False continue if id1!="" and re.search('[a-zA-Z\u4e00-\u9fa5]', tds[id1]) and tds[id1] not in self.header_set and \ re.search('备注|汇总|合计|总价|价格|金额|^详见|无$|xxx', tds[id1].replace(' ', '')) == None: product = re.sub('\s+', '', tds[id1]) if id0!="" and re.search('[a-zA-Z\u4e00-\u9fa5]', tds[id0]) and tds[id0] not in self.header_set and \ re.search('备注|汇总|合计|总价|价格|金额|^详见|无$|xxx', tds[id0].replace(' ', '')) == None: category = re.sub('\s', '', tds[id0]) # product = "%s_%s"%(category, product) if product!="" and product!=category else category # 20260528 去掉名称组合 修复 776939340 名称组合不像产品 if product == '': product = category if product and re.match( '【?(工程类|服务类|货物类|工程|服务|货物|运费)】?', product) == None: if id2 != "": if re.search('\d+|[壹贰叁肆伍陆柒捌玖拾一二三四五六七八九十]', tds[id2]): quantity = tds[id2] elif re.search('\w{5,}', tds[id2]) and re.search('^详见|^详情', tds[id2])==None: row_idx += 1 continue if id2_2 != "": if re.search('^\w{1,4}$', tds[id2_2]) and re.search('元', tds[id2_2])==None: quantity_unit = tds[id2_2] if id3 != "": if re.search('[零壹贰叁肆伍陆柒捌玖拾佰仟萬億十百千万亿元角分]{3,}', tds[id3]): unitPrice = tds[id3] elif re.search('^[\d,.亿万元人民币欧美日金额:()();;、,\n]+$|¥|¥|RMB|USD|EUR|JPY|CNY|元$', tds[id3].strip()): unitPrice = tds[id3] elif len(re.sub('[金额万元()()::零壹贰叁肆伍陆柒捌玖拾佰仟萬億圆十百千万亿元角分¥整\d,.]', '', tds[id3])) > 5 and re.search('^详见|^详情', tds[id3])==None: row_idx += 1 continue else: unitPrice = tds[id3] if id4 != "": if re.search('\w', tds[id4]): brand = tds[id4] if re.match('^详见|^详情', brand.strip()): brand = "" else: brand = "" if id5 != "": if re.search('\w', tds[id5]): specs = tds[id5][:self.MAX_SPECS_LEN] if re.match('^详见|^详情', specs.strip()): specs = "" else: specs = "" if id6 != "": if re.search('\w', tds[id6]): demand = tds[id6] else: demand = "" if id7 != "": if re.search('\d+|[零壹贰叁肆伍陆柒捌玖拾佰仟萬億十百千万亿元角分]{3,}', tds[id7]): budget = tds[id7] if id8 != "": if re.search('\w', tds[id8]): order_time = tds[id8].strip() order_begin, order_end = self.fix_time(order_time, html, page_time) if id9 != "": if re.search('[零壹贰叁肆伍陆柒捌玖拾佰仟萬億十百千万亿元角分]{3,}', tds[id9]): total_price = tds[id9] elif re.search('^[\d,.亿万元人民币欧美日金额:()();;、,\n]+$|¥|¥|RMB|USD|EUR|JPY|CNY|元$', tds[id9].strip()): total_price = tds[id9] elif len(re.sub('[金额万元()()::零壹贰叁肆伍陆柒捌玖拾佰仟萬億圆十百千万亿元角分¥整\d,.]', '', tds[id9])) > 5 and re.search('^详见|^详情', tds[id9])==None: row_idx += 1 continue if id10 != "": parameter = tds[id10][:self.MAX_PARAM_LEN] if re.match('^详见|^详情', parameter.strip()): parameter = "" if id11 != "": tenderee = re.sub("\s","",tds[id11]) if len(tenderee) > self.MAX_TENDEREE_LEN: tenderee = "" if id12 != "": notes = tds[id12].strip() if id13 != "": issue_date = self.fix_time(tds[id13].strip(), '', '')[0] if id14 != "": pinmu_no = tds[id14].strip() if id15 != "": pinmu_name = tds[id15].strip() if quantity != "" or unitPrice != "" or brand != "" or specs != "" or total_price or '单价' in header_dic or '总价' in header_dic: if id1!="" and id2 != "" and id3 != "" and len(re.split('[;;、,\n]+', tds[id2])) > 1 and len(re.split('[;;、,\n]+', tds[id1])) == len(re.split('[;;、,\n]+', tds[id2])): products = re.split('[;;、,\n]+', tds[id1]) quantitys = re.split('[;;、,\n]+', tds[id2]) unitPrices = re.split('[;;、,\n]+', tds[id3]) total_prices = re.split('[;;、,\n]+', total_price) brands = re.split('[;;、,\n]+', brand) if re.search('等$', brand)==None else [brand] specses = re.split('[;;、,\n]+', specs) if re.search('等$', specs)==None else [specs] parameters = re.split('[;;、,\n]+', parameter) if re.search('等$', parameter)==None else [parameter] unitPrices = [""]*len(products) if len(unitPrices)==1 else unitPrices total_prices = [""]*len(products) if len(total_prices)==1 else total_prices brands = brands*len(products) if len(brands)==1 else brands specses = specses*len(products) if len(specses)==1 else specses brands = [brand]*len(products) if len(brands) < len(products) else brands specses = [specs] * len(products) if len(specses) < len(products) else specses parameters = parameters*len(products) if len(parameters)==1 else parameters if len(products) == len(quantitys) == len(unitPrices) == len(brands) == len(specses): for product, quantity, unitPrice, brand, specs, total_price, parameter in zip(products,quantitys,unitPrices, brands, specses, total_prices, parameters): if product.strip() == '': continue if quantity != "": quantity, quantity_unit_ = self.fix_quantity(quantity, header_quan_unit) quantity_unit = quantity_unit_ if quantity_unit_ != "" else quantity_unit if unitPrice != "": unitPrice, _money_unit = money_process(unitPrice, header_list[3]) unitPrice = str(unitPrice) if unitPrice != 0 and unitPrice 0: tr.budget_list.append(float(budget)) if link['unitPrice'] != "" and link['quantity'] != '': try: tr.total_product_money += float(link['unitPrice']) * float( link['quantity']) if float(link['quantity']) < self.MAX_QUANTITY_FOR_CALC else 0 except (ValueError, TypeError): log('产品属性单价数量相乘出错, 单价: %s, 数量: %s' % ( link['unitPrice'], link['quantity'])) elif len(product)>self.MAX_PRODUCT_NAME_LEN: row_idx += 1 continue else: if quantity != "": quantity, quantity_unit_ = self.fix_quantity(quantity, header_quan_unit) quantity_unit = quantity_unit_ if quantity_unit_ != "" else quantity_unit if unitPrice != "": unitPrice, _money_unit = money_process(unitPrice, header_list[3]) unitPrice = str(unitPrice) if unitPrice != 0 and unitPrice 0: tr.budget_list.append(float(budget)) if link['unitPrice']: tr.unit_price_list.append(link['unitPrice']) if link['unitPrice'] != "" and link['quantity'] != '': try: tr.total_product_money += float(link['unitPrice'])*float(link['quantity']) if float(link['quantity'])10000 and float(link['quantity'])>100: tr.total_product_money = 0 except (ValueError, TypeError): log('产品属性单价数量相乘出错, 单价: %s, 数量: %s'%(link['unitPrice'], link['quantity'])) order_begin, order_end = self._validate_date_range(order_begin, order_end, page_time) if budget != "" and order_end != "": link = {'project_name': product, 'product':[], 'demand': demand, 'budget': budget, 'order_begin':order_begin, 'order_end':order_end, 'tenderee':tenderee,'notes':notes,'issue_date':issue_date} if link not in tr.demand_link: tr.demand_link.append(link) row_idx += 1 else: row_idx += 1 return tr def _extract_demand_from_2col(self, col0_l, col1_l, html, page_time): header_list2 = [] product = demand = budget = order_begin = order_end = "" tenderee = "" notes = '' issue_date = '' demand_links = [] for i in range(len(col0_l)): if re.search('项目名称', col0_l[i]): header_list2.append(col0_l[i]) product = col1_l[i] elif re.search('采购需求|需求概况|招标内容|项目概况', col0_l[i]): header_list2.append(col0_l[i]) demand = col1_l[i] elif re.search('(采购|招标|投资|项目)(预算|估算)|(预算|控制|投资|项目|采购|招标)金额|(最高|招标)(\w{,2})限价|拦标价', col0_l[i]): header_list2.append(col0_l[i]) _budget = col1_l[i] re_price = re.findall("[零壹贰叁肆伍陆柒捌玖拾佰仟萬億圆十百千万亿元角分]{3,}|\d[\d,]*(?:\.\d+)?万?", _budget) if re_price: _budget, _money_unit = money_process(_budget, col0_l[i]) budget = str(_budget) if '.' in budget: budget = budget.rstrip('0').rstrip('.') if float(budget)>= 500*100000000: budget = "" elif re.search('预算单位|(采购|招标|购买)(单位|人|方|主体)|项目业主|采购商|申购单位|需求单位|业主单位', col0_l[i]): header_list2.append(col0_l[i]) tenderee = re.sub("\s","",col1_l[i]) if len(tenderee) > 20: tenderee = "" elif re.search('时间|采购时间|采购实施月份|采购月份|采购日期|(预计|计划)(招标|采购|发标|发包)(时间|月份)', col0_l[i]): header_list2.append(col0_l[i]) order_time = col1_l[i].strip() order_begin, order_end = self.fix_time(order_time, html, page_time) elif re.search('^备\s*注$|资质要求|预留面向中小企业|是否适宜中小企业采购预算预留|公开征集信息', col0_l[i]): header_list2.append(col0_l[i]) notes = col1_l[i].strip() elif re.search('^\w{,4}发布(时间|日期)$', col0_l[i]): header_list2.append(col0_l[i]) issue_date = self.fix_time(col1_l[i].strip(), '', '')[0] order_begin, order_end = self._validate_date_range(order_begin, order_end, page_time) if product!= "" and demand != "" and budget!="" and order_end != "": link = {'project_name': product, 'product': [], 'demand': demand, 'budget': budget, 'order_begin': order_begin, 'order_end': order_end ,'tenderee':tenderee, 'notes':notes, 'issue_date':issue_date} if link not in demand_links: demand_links.append(link) return demand_links, header_list2 def _merge_table_results(self, table_results: List[TableResult]) -> TableResult: if not table_results: return TableResult() if len(table_results) == 1: return table_results[0] merged = TableResult() merged_product_set: Set[tuple] = set() merged_demand_set: Set[str] = set() budget_list = [] total_product_money_list = [] for i, tr in enumerate(table_results): if not tr.product_link and not tr.demand_link: continue overlap_ratio = self._calc_product_overlap(tr.product_names, merged_product_set) if overlap_ratio > 0.5 and self._same_header_group(tr, table_results[:i]): self._merge_with_dedup(tr, merged, merged_product_set, merged_demand_set, prefer_richer=True) elif overlap_ratio > 0: self._merge_with_dedup(tr, merged, merged_product_set, merged_demand_set, prefer_richer=True) else: self._merge_with_dedup(tr, merged, merged_product_set, merged_demand_set, prefer_richer=False) merged.headers.extend(tr.headers) merged.headers_demand.extend(tr.headers_demand) merged.header_col.extend(tr.header_col) merged.total_price_list.extend(tr.total_price_list) merged.unit_price_list.extend(tr.unit_price_list) merged.budget_list.extend(tr.budget_list) merged.product_confirm.extend(tr.product_confirm) if tr.budget_list and tr.budget_list not in budget_list: budget_list.append(tr.budget_list) if tr.total_product_money and tr.total_product_money not in total_product_money_list: total_product_money_list.append(tr.total_product_money) merged.total_product_money += tr.total_product_money if len(budget_list) > 1: merged.budget_list = [] if len(total_product_money_list) > 1: merged.total_product_money = 0 merged.headers = list(set(merged.headers)) merged.headers_demand = list(set(merged.headers_demand)) merged.header_col = list(set(merged.header_col)) return merged @staticmethod def _calc_product_overlap(new_names: Set[str], existing_set: Set[tuple]) -> float: if not new_names or not existing_set: return 0.0 existing_names = {p[0] for p in existing_set if p} if not existing_names: return 0.0 overlap = new_names & existing_names return len(overlap) / min(len(new_names), len(existing_names)) @staticmethod def _same_header_group(tr: TableResult, previous_results: List[TableResult]) -> bool: for prev in previous_results: if tr.header_signature and tr.header_signature == prev.header_signature: return True if tr.demand_header_signature and tr.demand_header_signature == prev.demand_header_signature: return True return False @staticmethod def _count_non_empty(d: dict) -> int: return len([v for v in d.values() if v != '' and v is not None]) def _merge_with_dedup(self, new_tr: TableResult, merged: TableResult, merged_product_set: Set[tuple], merged_demand_set: Set[str], prefer_richer: bool) -> None: for link in new_tr.product_link: product = link.get('product', '') unit_price = link.get('unitPrice', '') specs = link.get('specs', '') quantity = link.get('quantity', '') key = (product, unit_price) if key in merged_product_set: if prefer_richer: existing_idx = None for idx, existing in enumerate(merged.product_link): if existing.get('product', '') == product and existing.get('unitPrice', '') == unit_price: existing_idx = idx break if existing_idx is not None: existing = merged.product_link[existing_idx] if self._count_non_empty(link) > self._count_non_empty(existing): merged.product_link[existing_idx] = link continue merged_product_set.add(key) merged.product_link.append(link) for link in new_tr.demand_link: project_name = link.get('project_name', '') budget = link.get('budget', '') order_end = link.get('order_end', '') demand_key = f"{project_name}_{budget}_{order_end}" if demand_key in merged_demand_set: if prefer_richer: existing_idx = None for idx, existing in enumerate(merged.demand_link): ex_key = f"{existing.get('project_name', '')}_{existing.get('budget', '')}_{existing.get('order_end', '')}" if ex_key == demand_key: existing_idx = idx break if existing_idx is not None: existing = merged.demand_link[existing_idx] if self._count_non_empty(link) > self._count_non_empty(existing): merged.demand_link[existing_idx] = link continue merged_demand_set.add(demand_key) merged.demand_link.append(link) def _post_process(self, product_link, demand_link, total_price_list, unit_price_list, budget_list): if len(total_price_list)>1 and len(set(total_price_list))/len(total_price_list)<=0.5: for link in product_link: if 'total_price' in link: link['total_price'] = "" if len(demand_link) > 2 and demand_link[0].get('budget', '') != '' and len(set([d.get('budget', '') for d in demand_link])) == 1: for d in demand_link: if 'budget' in d: d['budget'] = "" if len(unit_price_list)>0 and len(unit_price_list)==len(product_link) and len(set(unit_price_list))/len(unit_price_list)<=0.5: return 0 return None def predict_without_table(self,product_attrs,list_sentences,list_entitys,codeName,prem, html='', page_time=""): if len(prem[0]['prem'])==1: list_sentences[0].sort(key=lambda x:x.sentence_index) list_sentence = list_sentences[0] list_entity = list_entitys[0] _data = product_attrs[1]['demand_info']['data'] re_bidding_time = re.compile("(采购|采购实施|(预计|计划)(招标|采购|发标|发包))(时间|月份|日期)[::,].{0,2}$") order_times = [] for entity in list_entity: if entity.entity_type=='time': # print('time',entity.entity_text) sentence = list_sentence[entity.sentence_index] s = spanWindow(tokens=sentence.tokens, begin_index=entity.begin_index, end_index=entity.end_index,size=20) entity_left = "".join(s[0]) if re.search(re_bidding_time,entity_left): time_text = entity.entity_text.strip() standard_time = re.compile("((?P\d{4}|\d{2})\s*[-\/年\.]\s*(?P\d{1,2})\s*[-\/月\.]\s*((?P\d{1,2})日?)?)") time_match = re.search(standard_time,time_text) # print(time_text, time_match) if time_match: time_text = time_match.group() order_times.append(time_text) # print(order_times) order_times = [tuple(self.fix_time(order_time, html, page_time)) for order_time in order_times] order_times = [order_time for order_time in order_times if order_time[0]!=""] if len(set(order_times))==1: order_begin,order_end = order_times[0] order_begin, order_end = self._validate_date_range(order_begin, order_end, page_time) if order_end!="": project_name = codeName[0]['name'] pack_info = [pack for pack in prem[0]['prem'].values()] budget = pack_info[0].get('tendereeMoney',0) product = prem[0]['product'] link = {'project_name': project_name, 'product': product, 'demand': project_name, 'budget': budget, 'order_begin': order_begin, 'order_end': order_end} _data.append(link) product_attrs[1]['demand_info']['data'] = _data # print('predict_without_table: ', product_attrs) return product_attrs def predict_by_text(self,product_attrs,html,list_outlines,product_list,page_time=""): product_entity_list = list(set(product_list)) list_outline = list_outlines[0] get_product_attrs = False for _outline in list_outline: if re.search("信息|情况|清单|概况",_outline.outline_summary): outline_text = _outline.outline_text outline_text = outline_text.replace(_outline.outline_summary,"") key_value_list = [_split for _split in re.split("[,。;]",outline_text) if re.search("[::]",_split)] if not key_value_list: continue head_list = [] head_value_list = [] for key_value in key_value_list: key_value = re.sub("^[一二三四五六七八九十]{1,3}[、.]|^[\d]{1,2}[、.]\d{,2}|^[\((]?[一二三四五六七八九十]{1,3}[\))][、]?","",key_value) temp = re.split("[::]",key_value) if len(temp)>2: if temp[0] in head_list: key = temp[0] value = "".join(temp[1:]) else: key = temp[-2] value = temp[-1] else: key = temp[0] value = temp[1] key = re.sub("^[一二三四五六七八九十]{1,3}[、.]|^[\d]{1,2}[、.]\d{,2}|^[\((]?[一二三四五六七八九十]{1,3}[\))][、]?","",key) head_list.append(key) head_value_list.append(value) head_set = set(head_list) # print('head_set',head_set) if len(head_set & self.header_set) > len(head_set)*0.2: loop_list = [] begin_list = [0] for index,head in enumerate(head_list): if head not in loop_list: if re.search('第[一二三四五六七八九十](包|标段)', head) and re.search('第[一二三四五六七八九十](包|标段)', '|'.join(loop_list)): begin_list.append(index) loop_list = [] loop_list.append(head) else: loop_list.append(head) else: begin_list.append(index) loop_list = [] loop_list.append(head) headers = [] headers_demand = [] header_col = [] product_link = [] demand_link = [] product_set = set() for idx in range(len(begin_list)): if idx==len(begin_list)-1: deal_list = head_value_list[begin_list[idx]:] tmp_head_list = head_list[begin_list[idx]:] else: deal_list = head_value_list[begin_list[idx]:begin_list[idx+1]] tmp_head_list = head_list[begin_list[idx]:begin_list[idx+1]] product = "" # 产品 quantity = "" # 数量 quantity_unit = "" # 单位 unitPrice = "" # 单价 brand = "" # 品牌 specs = "" # 规格 demand = "" # 采购需求 budget = "" # 预算金额 order_time = "" # 采购时间 order_begin = "" order_end = "" total_price = "" # 总金额 parameter = "" # 参数 header_dic, found_header, header_list, header_list2 = self.find_header(tmp_head_list, self.pat_category, self.pat_product_primary, self.pat_product_secondary, self.pat_project) if found_header: headers.append('_'.join(header_list)) headers_demand.append('_'.join(header_list2)) header_col.append('_'.join(tmp_head_list)) # print('header_dic: ',header_dic) id0 = header_dic.get('品目', "") id1 = header_dic.get('名称', "") id2 = header_dic.get('数量', "") id2_2 = header_dic.get('单位', "") id3 = header_dic.get('单价', "") id4 = header_dic.get('品牌', "") id5 = header_dic.get('规格', "") id6 = header_dic.get('需求', "") id7 = header_dic.get('预算', "") id8 = header_dic.get('时间', "") id9 = header_dic.get("总价", "") id10 = header_dic.get('参数', "") if id1!='' and re.search('[a-zA-Z\u4e00-\u9fa5]', deal_list[id1]) and deal_list[id1] not in self.header_set and \ re.search('备注|汇总|合计|总价|价格|金额|公司|附件|详见|无$|xxx', deal_list[id1]) == None: product = deal_list[id1] if id0 != "" and re.search('[a-zA-Z\u4e00-\u9fa5]', deal_list[id0]) and deal_list[id0] not in self.header_set and \ re.search('备注|汇总|合计|总价|价格|金额|公司|附件|详见|无$|xxx', deal_list[id0]) == None: category = deal_list[id0] product = "%s_%s" % (category, product) if product != "" else category if product == "": # print(deal_list[id4],deal_list[id5],tmp_head_list,deal_list) if (id4 != "" and deal_list[id4] != "") or (id5 != "" and deal_list[id5] != ""): for head,value in zip(tmp_head_list,deal_list): if value and value in product_entity_list: product = value break if product and re.match('【?(工程类|服务类|货物类|工程|服务|货物|运费)】?', product) == None: if id2 != "": if re.search('\d+|[壹贰叁肆伍陆柒捌玖拾一二三四五六七八九十]', deal_list[id2]): quantity = deal_list[id2] quantity = re.sub('[()(),,约]', '', quantity) quantity = re.sub('[一壹]', '1', quantity) ser = re.search('^(\d+(?:\.\d+)?)([㎡\w/]{,5})', quantity) if ser: quantity = str(ser.group(1)) quantity_unit = ser.group(2) if float(quantity)>=10000*10000: quantity = "" quantity_unit = "" else: quantity = "" quantity_unit = "" if id2_2 != "": if re.search('^\w{1,4}$', deal_list[id2_2]): quantity_unit = deal_list[id2_2] else: quantity_unit = "" # if id2 != "": # if re.search('\d+|[壹贰叁肆伍陆柒捌玖拾一二三四五六七八九十]', deal_list[id2]): # quantity = deal_list[id2] # else: # quantity = "" if id3 != "": if re.search('\d+|[零壹贰叁肆伍陆柒捌玖拾佰仟萬億十百千万亿元角分]{3,}', deal_list[id3]): _unitPrice = deal_list[id3] re_price = re.findall("[零壹贰叁肆伍陆柒捌玖拾佰仟萬億圆十百千万亿元角分]{3,}|\d[\d,]*(?:\.\d+)?万?",_unitPrice) if re_price: # _unitPrice = re_price[0] # if '万元' in header_list[3] and '万' not in _unitPrice: # _unitPrice += '万元' # unitPrice = getUnifyMoney(_unitPrice) # if unitPrice>=10000*10000: # unitPrice = "" # unitPrice = str(unitPrice) _unitPrice, _money_unit = money_process(_unitPrice, header_list[3]) if _unitPrice >= 10000 * 10000: _unitPrice = "" unitPrice = str(_unitPrice) if '.' in unitPrice: unitPrice = unitPrice.rstrip('0').rstrip('.') if id4 != "": if re.search('\w', deal_list[id4]): brand = deal_list[id4] if re.match('^详见|^详情', brand.strip()): brand = "" else: brand = "" if id5 != "": if re.search('\w', deal_list[id5]): specs = deal_list[id5][:self.MAX_SPECS_LEN] if re.match('^详见|^详情', specs.strip()): specs = "" else: specs = "" if id6 != "": if re.search('\w', deal_list[id6]): demand = deal_list[id6] else: demand = "" if id7 != "": if re.search('\d+|[零壹贰叁肆伍陆柒捌玖拾佰仟萬億十百千万亿元角分]{3,}', deal_list[id7]): _budget = deal_list[id7] re_price = re.findall("[零壹贰叁肆伍陆柒捌玖拾佰仟萬億圆十百千万亿元角分]{3,}|\d[\d,]*(?:\.\d+)?万?",_budget) if re_price: # _budget = re_price[0] # if '万元' in header_list2[2] and '万' not in _budget: # _budget += '万元' # budget = str(getUnifyMoney(_budget)) _budget, _money_unit = money_process(_budget, header_list2[2]) budget = str(_budget) if '.' in budget: budget = budget.rstrip('0').rstrip('.') if float(budget)>= 100000*10000: budget = "" if id8 != "": if re.search('\w', deal_list[id8]) and re.search("(采购|采购实施|(预计|计划)(招标|采购|发标|发包))(时间|月份|日期)",header_list2[3]): order_time = deal_list[id8].strip() order_begin, order_end = self.fix_time(order_time, html, page_time) if id9 != "": if re.search('[零壹贰叁肆伍陆柒捌玖拾佰仟萬億十百千万亿元角分]{3,}', deal_list[id9]): total_price = deal_list[id9] elif re.search('^[\d,.亿万元人民币欧美日金额:()();;、,\n]+$', deal_list[id9].strip()): total_price = deal_list[id9] if id10 != "": parameter = deal_list[id10][:self.MAX_PARAM_LEN] if re.match('^详见|^详情', parameter.strip()): parameter = "" if quantity != "" or unitPrice != "" or brand != "" or specs != "" or total_price: if id1 != "" and id2 != "" and id3 != "" and len(re.split('[;;、,\n]', deal_list[id2])) > 1 and len( re.split('[;;、,\n]', deal_list[id1])) == len(re.split('[;;、,\n]', deal_list[id2])): # 处理一个空格包含多个产品,逗号或空格分割情况 例子 292846806 292650743 products = re.split('[;;、,\n]', deal_list[id1]) quantitys = re.split('[;;、,\n]', deal_list[id2]) unitPrices = re.split('[;;、,\n]', deal_list[id3]) total_prices = re.split('[;;、,\n]', total_price) brands = re.split('[;;、,\n]', brand) if re.search('等$', brand) == None else [brand] specses = re.split('[;;、,\n]', specs) if re.search('等$', specs) == None else [specs] parameters = re.split('[;;、,\n]', parameter) if re.search('等$', parameter) == None else [parameter] unitPrices = [""] * len(products) if len(unitPrices) == 1 else unitPrices total_prices = [""] * len(products) if len(total_prices) == 1 else total_prices brands = brands * len(products) if len(brands) == 1 else brands specses = specses * len(products) if len(specses) == 1 else specses parameters = parameters * len(products) if len(parameters) == 1 else parameters if len(products) == len(quantitys) == len(unitPrices) == len(brands) == len( specses): for product, quantity, unitPrice, brand, specs, total_price, parameter in zip( products, quantitys, unitPrices, brands, specses, total_prices, parameters): if quantity != "": quantity, quantity_unit_ = self.fix_quantity(quantity,quantity_unit) quantity_unit = quantity_unit_ if quantity_unit_ != "" else quantity_unit if unitPrice != "": unitPrice, _money_unit = money_process(unitPrice, header_list[3]) unitPrice = str(unitPrice) if unitPrice != 0 and unitPrice 15 or len(product) > self.MAX_PRODUCT_NAME_LEN: # i += 1 continue else: if quantity != "": quantity, quantity_unit_ = self.fix_quantity(quantity, quantity_unit) quantity_unit = quantity_unit_ if quantity_unit_ != "" else quantity_unit if unitPrice != "": unitPrice, _money_unit = money_process(unitPrice, header_list[3]) unitPrice = str(unitPrice) if unitPrice != 0 and unitPrice 0: attr_dic = {'product_attrs': {'data': product_link, 'header': list(set(headers)), 'header_col': list(set(header_col))}} get_product_attrs = True else: attr_dic = {'product_attrs': {'data': [], 'header': [], 'header_col': []}} if len(demand_link) > 0: demand_dic = {'demand_info': {'data': demand_link, 'header': headers_demand, 'header_col': header_col}} else: demand_dic = {'demand_info': {'data': [], 'header': [], 'header_col': []}} product_attrs[0] = attr_dic if len(product_attrs[1]['demand_info']['data']) == 0: product_attrs[1] = demand_dic if get_product_attrs: break # print('predict_by_text: ', product_attrs) return product_attrs def add_product_attrs(self,channel_dic, product_attrs, list_sentences,list_entitys,list_outlines,product_list,codeName,prem,text,page_time): # print(1,product_attrs[1]['demand_info']['data']) if channel_dic['docchannel']['docchannel']=="采购意向" and len(product_attrs[1]['demand_info']['data']) == 0: product_attrs = self.predict_without_table(product_attrs, list_sentences,list_entitys,codeName,prem,text,page_time) # print(2,product_attrs[1]['demand_info']['data']) if len(product_attrs[0]['product_attrs']['data']) == 0: product_attrs = self.predict_by_text(product_attrs,text,list_outlines,product_list,page_time) # print(3,product_attrs[1]['demand_info']['data']) if len(product_attrs[1]['demand_info']['data'])>0: for d in product_attrs[1]['demand_info']['data']: for product in set(prem[0]['product']): if product in d['project_name'] and product not in d['product']: d['product'].append(product) #把产品在项目名称中的添加进需求要素中