product_attrs.py 83 KB

12345678910111213141516171819202122232425262728293031323334353637383940414243444546474849505152535455565758596061626364656667686970717273747576777879808182838485868788899091929394959697989910010110210310410510610710810911011111211311411511611711811912012112212312412512612712812913013113213313413513613713813914014114214314414514614714814915015115215315415515615715815916016116216316416516616716816917017117217317417517617717817918018118218318418518618718818919019119219319419519619719819920020120220320420520620720820921021121221321421521621721821922022122222322422522622722822923023123223323423523623723823924024124224324424524624724824925025125225325425525625725825926026126226326426526626726826927027127227327427527627727827928028128228328428528628728828929029129229329429529629729829930030130230330430530630730830931031131231331431531631731831932032132232332432532632732832933033133233333433533633733833934034134234334434534634734834935035135235335435535635735835936036136236336436536636736836937037137237337437537637737837938038138238338438538638738838939039139239339439539639739839940040140240340440540640740840941041141241341441541641741841942042142242342442542642742842943043143243343443543643743843944044144244344444544644744844945045145245345445545645745845946046146246346446546646746846947047147247347447547647747847948048148248348448548648748848949049149249349449549649749849950050150250350450550650750850951051151251351451551651751851952052152252352452552652752852953053153253353453553653753853954054154254354454554654754854955055155255355455555655755855956056156256356456556656756856957057157257357457557657757857958058158258358458558658758858959059159259359459559659759859960060160260360460560660760860961061161261361461561661761861962062162262362462562662762862963063163263363463563663763863964064164264364464564664764864965065165265365465565665765865966066166266366466566666766866967067167267367467567667767867968068168268368468568668768868969069169269369469569669769869970070170270370470570670770870971071171271371471571671771871972072172272372472572672772872973073173273373473573673773873974074174274374474574674774874975075175275375475575675775875976076176276376476576676776876977077177277377477577677777877978078178278378478578678778878979079179279379479579679779879980080180280380480580680780880981081181281381481581681781881982082182282382482582682782882983083183283383483583683783883984084184284384484584684784884985085185285385485585685785885986086186286386486586686786886987087187287387487587687787887988088188288388488588688788888989089189289389489589689789889990090190290390490590690790890991091191291391491591691791891992092192292392492592692792892993093193293393493593693793893994094194294394494594694794894995095195295395495595695795895996096196296396496596696796896997097197297397497597697797897998098198298398498598698798898999099199299399499599699799899910001001100210031004100510061007100810091010101110121013101410151016101710181019102010211022102310241025102610271028102910301031103210331034103510361037103810391040104110421043104410451046104710481049105010511052105310541055105610571058105910601061106210631064106510661067106810691070107110721073107410751076107710781079108010811082108310841085108610871088108910901091109210931094109510961097109810991100110111021103110411051106110711081109111011111112111311141115111611171118111911201121112211231124112511261127112811291130113111321133113411351136113711381139114011411142114311441145114611471148114911501151115211531154115511561157115811591160116111621163116411651166116711681169117011711172117311741175117611771178117911801181118211831184118511861187118811891190119111921193119411951196119711981199120012011202120312041205120612071208120912101211121212131214121512161217121812191220122112221223122412251226122712281229123012311232123312341235123612371238123912401241124212431244124512461247124812491250125112521253125412551256125712581259126012611262126312641265126612671268126912701271127212731274127512761277127812791280128112821283128412851286128712881289129012911292129312941295129612971298129913001301130213031304130513061307130813091310131113121313131413151316131713181319132013211322132313241325132613271328132913301331133213331334133513361337133813391340134113421343134413451346134713481349135013511352135313541355135613571358135913601361136213631364136513661367136813691370137113721373137413751376137713781379138013811382138313841385138613871388138913901391139213931394139513961397139813991400140114021403140414051406140714081409141014111412141314141415141614171418141914201421142214231424142514261427142814291430
  1. # -*- coding: utf-8 -*-
  2. """产品属性预测器(ProductAttributesPredictor)。
  3. 按 ARCHITECTURE.md Phase 5 拆分建议,从 ``interface/predictor.py`` 迁出以下
  4. 产品属性提取相关类:
  5. - ``TableResult`` — 表格提取结果数据类(原 predictor.py 第 3330-3368 行)
  6. - ``ProductAttributesPredictor`` — 产品数量/单价/品牌/规格/表格要素提取
  7. (原 predictor.py 第 3371-4719 行)
  8. 原 ``from common.Utils import *`` / ``from common.nerUtils import *`` 已替换为
  9. 显式 import;``os.path.dirname(__file__)`` 路径引用替换为
  10. ``predictors._common.INTERFACE_DIR``。
  11. ``interface/predictor.py`` 仍 re-export 以上全部名称,老 import 不受影响。
  12. """
  13. from __future__ import absolute_import
  14. import os
  15. import re
  16. import copy
  17. import pickle
  18. import calendar
  19. import datetime
  20. from bs4 import BeautifulSoup
  21. from dataclasses import dataclass, field
  22. from typing import List, Dict, Set, Any
  23. from BiddingKG.dl.common.logging import log
  24. from BiddingKG.dl.common.context_utils import money_process, spanWindow
  25. from BiddingKG.dl.common.Utils import del_tabel_achievement
  26. from BiddingKG.dl.predictors._common import INTERFACE_DIR
  27. from BiddingKG.dl.predictors.table_prem import TableTag2List
  28. __all__ = [
  29. "TableResult",
  30. "ProductAttributesPredictor",
  31. ]
  32. @dataclass
  33. class TableResult:
  34. table_index: int = 0
  35. product_confirm: List[int] = field(default_factory=list)
  36. headers: List[str] = field(default_factory=list)
  37. headers_demand: List[str] = field(default_factory=list)
  38. header_col: List[str] = field(default_factory=list)
  39. product_link: List[Dict[str, Any]] = field(default_factory=list)
  40. demand_link: List[Dict[str, Any]] = field(default_factory=list)
  41. product_set: Set[tuple] = field(default_factory=set)
  42. total_product_money: float = 0
  43. unit_price_list: List[str] = field(default_factory=list)
  44. total_price_list: list = field(default_factory=list)
  45. budget_list: list = field(default_factory=list)
  46. @property
  47. def product_names(self) -> Set[str]:
  48. return {p.get('product', '') for p in self.product_link if p.get('product', '')}
  49. @property
  50. def header_signature(self) -> str:
  51. return '|'.join(sorted(set(self.headers)))
  52. @property
  53. def demand_header_signature(self) -> str:
  54. return '|'.join(sorted(set(self.headers_demand)))
  55. def product_count(self) -> int:
  56. return len(self.product_link)
  57. def demand_count(self) -> int:
  58. return len(self.demand_link)
  59. def avg_attrs_per_product(self) -> float:
  60. if not self.product_link:
  61. return 0
  62. total = sum(len([v for v in p.values() if v != '']) for p in self.product_link)
  63. return total / len(self.product_link)
  64. # 产品数量单价品牌规格提取 #2021/11/10 添加表格中的项目、需求、预算、时间要素提取
  65. class ProductAttributesPredictor():
  66. MAX_UNIT_PRICE = 100000000
  67. MAX_TOTAL_PRICE = 50000000000
  68. MAX_BUDGET = 50000000000
  69. MAX_PRODUCT_NAME_LEN = 100
  70. MAX_BRAND_LEN = 50
  71. MAX_SPECS_LEN = 500
  72. MAX_PARAM_LEN = 500
  73. MAX_TENDEREE_LEN = 30
  74. MAX_QUANTITY_FOR_CALC = 50000
  75. FUTURE_YEAR_LIMIT = 2050
  76. def __init__(self,):
  77. # self.pat_category = '(类别|类型|物类|目录|类目|分类)(名称|$)|^品名|^品类|^品目|(标项|分项|项目|计划|包组|标段|[分子]?包|子目|服务|招标|中标|成交|工程|招标内容)(名称|内容|描述)'
  78. self.pat_category = '(品目|品类)名称?|采购(品目|品类)$|^品名|^品类|^品目$'
  79. # self.pat_product_primary = '(标的|维修|系统|报价构成|商品|产品|物料|物资|货物|设备|采购品|采购条目|物品|材料|印刷品?|采购|物装|配件|资产|耗材|清单|器材|仪器|器械|备件|拍卖物|标的物|物件|药品|药材|药械|货品|食品|食材|品目|^品名|气体)[\))的]?(名称|内容|描述)'
  80. self.pat_product_primary = '(标的|商品|产品|物料|物资|货物|设备|采购品|采购条目|物品|材料|印刷品?|采购|物装|配件|资产|耗材|清单|器材|仪器|器械|备件|拍卖物|标的物|物件|药品|药材|药械|疫苗|货品|食品|食材|品目|^品名|气体)[\))的]?(名称|内容|描述)'
  81. # self.pat_product_secondary = '标的|标项|项目$|商品|产品|物料|物资|货物|设备|采购品|采购条目|物品|材料|印刷品|物装|配件|资产|招标内容|耗材|清单|器材|仪器|器械|备件|拍卖物|标的物|物件|药品|药材|药械|货品|食品|食材|菜名|^品目$|^品名$|^名称|^内容$|(标项|分项|项目|计划|包组|标段|[分子]?包|子目|服务|招标|中标|成交|工程|招标内容)(名称|内容|描述)'
  82. self.pat_product_secondary = '标的|商品|产品|物料|物资|货物|设备|采购品|采购条目|物品|材料|印刷品|物装|配件|资产|耗材|清单|器材|仪器|器械|备件|拍卖物|标的物|物件|药品|药材|药械|货品|食品|食材|菜名|^品目$|^品名$'
  83. self.pat_project = '(标项|分项|项目|计划|包组|标段|[分子]?包|子目|服务|招标|中标|成交|工程)(名称|内容|描述)|^名称|采购类别'
  84. with open(os.path.join(INTERFACE_DIR, 'header_set.pkl'), 'rb') as f:
  85. self.header_set = pickle.load(f)
  86. self.tb = TableTag2List()
  87. def isTrueTable(self, table):
  88. '''真假表格规则:
  89. 1、包含<caption>或<th>标签为真
  90. 2、包含大量链接、表单、图片或嵌套表格为假
  91. 3、表格尺寸太小为假
  92. 4、外层<table>嵌套子<table>,一般子为真,外为假'''
  93. if table.find_all(['caption', 'th']) != []:
  94. return True
  95. # elif len(table.find_all(['form', 'a', 'img'])) > 5: # 20260602 去掉,某些表格可能有多个链接 例子:437440313
  96. # # print('过滤表格:包含链接图片等大于5的为假表格')
  97. # return False
  98. elif len(table.find_all(['tr'])) < 2:
  99. # print('过滤表格:行数小于2的为假表格')
  100. return False
  101. elif len(table.find_all(['table'])) >= 1:
  102. # print('过滤表格:包含多个表格的为假表格')
  103. inner_table_num = len(table.find_all(['table']))
  104. text_num = 0 # 表格只有一格作文本框的数量,docid:631910513
  105. for inner_table in table.find_all(['table']):
  106. if len(inner_table.find_all(['tr']))==0 or (len(inner_table.find_all(['tr']))==1 and len(inner_table.find_all(['tr'])[0].find_all(['td']))<=1):
  107. text_num += 1
  108. if inner_table_num - text_num > 0:
  109. return False
  110. else:
  111. return True
  112. else:
  113. return True
  114. def getTrs(self, tbody):
  115. # 获取所有的tr
  116. trs = []
  117. objs = tbody.find_all(recursive=False)
  118. for obj in objs:
  119. if obj.name == "tr":
  120. trs.append(obj)
  121. if obj.name == "tbody":
  122. for tr in obj.find_all("tr", recursive=False):
  123. trs.append(tr)
  124. return trs
  125. def getTable(self, tbody):
  126. trs = self.getTrs(tbody)
  127. inner_table = []
  128. if len(trs) < 2:
  129. return inner_table
  130. for tr in trs:
  131. tr_line = []
  132. tds = tr.findChildren(['td', 'th'], recursive=False)
  133. if len(tds) < 2:
  134. continue
  135. for td in tds:
  136. # td_text = re.sub('\s+|…', ' ', td.get_text()).strip()
  137. td_text = re.sub('…', '', td.get_text()).strip()
  138. td_text = re.sub('\n+|\s+', ' ', td_text) # 20250626 去掉\n等避免存OTS后去掉转义导致json解析错误
  139. td_text = td_text.replace("\x06", "").replace("\x05", "").replace("\x07", "").replace('\\', '/').replace('"', '') # 修复272144312 # 产品单价数量提取结果有特殊符号\ 气动执行装置备件\密封组件\NBR+PT
  140. td_text = td_text.replace("(", "(").replace(")", ")").replace(':', ':')
  141. tr_line.append(td_text)
  142. inner_table.append(tr_line)
  143. return inner_table
  144. def fixSpan(self, tbody):
  145. # 处理colspan, rowspan信息补全问题
  146. trs = self.getTrs(tbody)
  147. ths_len = 0
  148. ths = list()
  149. trs_set = set()
  150. # 修改为先进行列补全再进行行补全,否则可能会出现表格解析混乱
  151. # 遍历每一个tr
  152. for indtr, tr in enumerate(trs):
  153. ths_tmp = tr.findChildren('th', recursive=False)
  154. if len(tr.findChildren('table')) > 0:
  155. continue
  156. if len(ths_tmp) > 0:
  157. for th in ths_tmp:
  158. ths.append(th)
  159. trs_set.add(tr)
  160. # 遍历每行中的element
  161. tds = tr.findChildren(recursive=False)
  162. if len(tds) < 3:
  163. continue # 列数太少的不补全
  164. for indtd, td in enumerate(tds):
  165. # 若有colspan 则补全同一行下一个位置
  166. if 'colspan' in td.attrs and str(re.sub("[^0-9]", "", str(td['colspan']))) != "":
  167. col = int(re.sub("[^0-9]", "", str(td['colspan'])))
  168. if col < 10 and len(td.get_text()) < 500:
  169. td['colspan'] = 1
  170. for i in range(1, col, 1):
  171. td.insert_after(copy.copy(td))
  172. for indtr, tr in enumerate(trs):
  173. ths_tmp = tr.findChildren('th', recursive=False)
  174. # 不补全含有表格的tr
  175. if len(tr.findChildren('table')) > 0:
  176. continue
  177. if len(ths_tmp) > 0:
  178. ths_len = ths_len + len(ths_tmp)
  179. for th in ths_tmp:
  180. ths.append(th)
  181. trs_set.add(tr)
  182. # 遍历每行中的element
  183. tds = tr.findChildren(recursive=False)
  184. same_span = 0
  185. if len(tds) > 1 and 'rowspan' in tds[0].attrs:
  186. span0 = tds[0].attrs['rowspan']
  187. for td in tds:
  188. if 'rowspan' in td.attrs and td.attrs['rowspan'] == span0:
  189. same_span += 1
  190. if same_span == len(tds):
  191. continue
  192. for indtd, td in enumerate(tds):
  193. # 若有rowspan 则补全下一行同样位置
  194. if 'rowspan' in td.attrs and str(re.sub("[^0-9]", "", str(td['rowspan']))) != "":
  195. row = int(re.sub("[^0-9]", "", str(td['rowspan'])))
  196. td['rowspan'] = 1
  197. for i in range(1, row, 1):
  198. # 获取下一行的所有td, 在对应的位置插入
  199. if indtr + i < len(trs):
  200. tds1 = trs[indtr + i].findChildren(['td', 'th'], recursive=False)
  201. if len(tds1) >= (indtd) and len(tds1) > 0:
  202. if indtd > 0:
  203. tds1[indtd - 1].insert_after(copy.copy(td))
  204. else:
  205. tds1[0].insert_before(copy.copy(td))
  206. elif len(tds1) > 0 and len(tds1) == indtd - 1:
  207. tds1[indtd - 2].insert_after(copy.copy(td))
  208. def get_monthlen(self, year, month):
  209. '''输入年份、月份 int类型 得到该月份天数'''
  210. try:
  211. weekday, num = calendar.monthrange(int(year), int(month))
  212. except (ValueError, TypeError):
  213. num = 30
  214. return str(num)
  215. def _validate_date_range(self, order_begin, order_end, page_time):
  216. if not order_end > page_time:
  217. return "", ""
  218. if order_begin != "" and order_end != "":
  219. order_begin_year = int(order_begin.split("-")[0])
  220. order_end_year = int(order_end.split("-")[0])
  221. if order_begin_year >= self.FUTURE_YEAR_LIMIT or order_end_year >= self.FUTURE_YEAR_LIMIT:
  222. return "", ""
  223. return order_begin, order_end
  224. def _build_month_range(self, year, month):
  225. month = str(month).zfill(2)
  226. num = self.get_monthlen(year, month).zfill(2)
  227. order_begin = "%s-%s-01" % (year, month)
  228. order_end = "%s-%s-%s" % (year, month, num)
  229. return order_begin, order_end
  230. def _resolve_year_for_month(self, html, page_time):
  231. year = re.search('(\d{4})年(.{,12}采购意向)?', html)
  232. if year:
  233. return year.group(1)
  234. if page_time != "":
  235. year = re.search('\d{4}', page_time)
  236. if year:
  237. return year.group(0)
  238. return str(datetime.datetime.now().year)
  239. def fix_time(self, text, html, page_time):
  240. for it in [('十二', '12'),('十一', '11'),('十','10'),('九','9'),('八','8'),('七','7'),
  241. ('六','6'),('五','5'),('四','4'),('三','3'),('二','2'),('一','1')]:
  242. if it[0] in text:
  243. text = text.replace(it[0], it[1])
  244. if re.search('^\d{1,2}月$', text):
  245. m = re.search('^(\d{1,2})月$', text).group(1)
  246. y = self._resolve_year_for_month(html, page_time)
  247. return self._build_month_range(y, m)
  248. t1 = re.search('^(\d{4})(年|/|\.|-)(\d{1,2})月?$', text)
  249. if t1:
  250. return self._build_month_range(t1.group(1), t1.group(3))
  251. t2 = re.search('^(\d{4})(年|/|\.|-)(\d{1,2})(月|/|\.|-)(\d{1,2})日?$', text)
  252. if t2:
  253. y = t2.group(1)
  254. m = t2.group(3).zfill(2)
  255. d = t2.group(5).zfill(2)
  256. order_begin = order_end = "%s-%s-%s"%(y,m,d)
  257. return order_begin, order_end
  258. t3 = re.search("^(20\d{2})(\d{1,2})$",text)
  259. if t3:
  260. year = t3.group(1)
  261. month = t3.group(2)
  262. if int(month)>0 and int(month)<=12:
  263. return self._build_month_range(year, month)
  264. t4 = re.search("^(20\d{2})(\d{2})(\d{2})$", text)
  265. if t4:
  266. year = t4.group(1)
  267. month = t4.group(2)
  268. day = t4.group(3)
  269. if int(month) > 0 and int(month) <= 12 and int(day)>0 and int(day)<=31:
  270. order_begin = order_end = "%s-%s-%s"%(year,month,day)
  271. return order_begin, order_end
  272. all_match = re.finditer('^(?P<y1>\d{4})(年|/|\.)(?P<m1>\d{1,2})(?:(月|/|\.)(?:(?P<d1>\d{1,2})日)?)?'
  273. '(到|至|-)(?:(?P<y2>\d{4})(年|/|\.))?(?P<m2>\d{1,2})(?:(月|/|\.)'
  274. '(?:(?P<d2>\d{1,2})日)?)?$', text)
  275. y1 = m1 = d1 = y2 = m2 = d2 = ""
  276. found_math = False
  277. for _match in all_match:
  278. if len(_match.group()) > 0:
  279. found_math = True
  280. for k, v in _match.groupdict().items():
  281. if v!="" and v is not None:
  282. if k == 'y1':
  283. y1 = v
  284. elif k == 'm1':
  285. m1 = v
  286. elif k == 'd1':
  287. d1 = v
  288. elif k == 'y2':
  289. y2 = v
  290. elif k == 'm2':
  291. m2 = v
  292. elif k == 'd2':
  293. d2 = v
  294. if not found_math:
  295. return "", ""
  296. y2 = y1 if y2 == "" else y2
  297. d1 = '1' if d1 == "" else d1
  298. d2 = self.get_monthlen(y2, m2) if d2 == "" else d2
  299. m1 = '0' + m1 if len(m1) < 2 else m1
  300. m2 = '0' + m2 if len(m2) < 2 else m2
  301. d1 = '0' + d1 if len(d1) < 2 else d1
  302. d2 = '0' + d2 if len(d2) < 2 else d2
  303. order_begin = "%s-%s-%s"%(y1,m1,d1)
  304. order_end = "%s-%s-%s"%(y2,m2,d2)
  305. return order_begin, order_end
  306. def fix_quantity(self, quantity_text, header_quan_unit):
  307. '''
  308. 产品数量标准化,统一为数值型字符串
  309. :param quantity_text: 原始数量字符串
  310. :param header_quan_unit: 表头数量单位字符串
  311. :return: 返回数量及单位
  312. '''
  313. quantity = quantity_text
  314. quantity = re.sub('[一壹]', '1', quantity)
  315. quantity = re.sub('[,,约]|(\d+)', '', quantity)
  316. ser = re.search('^(\d+\.?\d*)(?([㎡\w/]{,5})', quantity)
  317. if ser:
  318. quantity = str(ser.group(1))
  319. quantity_unit = ser.group(2)
  320. if quantity_unit == "" and header_quan_unit != "":
  321. quantity_unit = header_quan_unit
  322. else:
  323. quantity = ""
  324. quantity_unit = ""
  325. return quantity, quantity_unit
  326. def find_header(self, items,p0, p1, p2, pat_project):
  327. '''
  328. inner_table 每行正则检查是否为表头,是则返回表头所在列序号,及表头内容
  329. :param items: 列表,内容为每个td 文本内容
  330. :param p1: 优先表头正则
  331. :param p2: 第二表头正则
  332. :param pat_project: 项目表头正则
  333. :return: 表头所在列序号,是否表头,表头内容
  334. '''
  335. items = [re.sub('\s', '', it) for it in items]
  336. flag = False
  337. header_dic = {'产品为项目': False,'名称': '', '数量': '', '单位': '', '单价': '', '品牌': '', '规格': '', '需求': '', '预算': '', '时间': '', '总价': '', '品目': '', '参数': '', '采购人':'', '备注':'','发布日期':'', '品目号':'', '品目名':''}
  338. product = "" # 产品
  339. quantity = "" # 数量
  340. quantity_unit = "" # 数量单位
  341. unitPrice = "" # 单价
  342. brand = "" # 品牌
  343. specs = "" # 规格
  344. demand = "" # 采购需求
  345. budget = "" # 预算金额
  346. order_time = "" # 采购时间
  347. total_price = "" # 总价
  348. category = "" # 品目
  349. parameter = "" # 参数
  350. tenderee = "" # 采购人
  351. notes = "" # 备注 2024/3/27 达仁 需求
  352. issue_date = "" # 发布日期 2024/3/27 达仁 需求
  353. pinmu_no = "" # 品目号
  354. pinmu_name = "" # 品目名称
  355. product_secondary = ''
  356. product_idx = ''
  357. project = ''
  358. project_idx = ''
  359. for i in range(int(len(items)*0.75)):
  360. it = items[i]
  361. if len(it) < 15 and re.search(p0, it):
  362. flag = True
  363. if category != "" and category != it:
  364. continue
  365. category = it
  366. header_dic['品目'] = i
  367. elif len(it) < 15 and re.search(p1, it):
  368. flag = True
  369. if product !='' and product != it:
  370. break
  371. product = it
  372. header_dic['名称'] = i
  373. # break
  374. if len(it) < 15 and it != category and re.search(p2, it) and (re.search('^名称|^品名|^品目$', it) or re.search(
  375. '编号|编码|号|情况|报名|单位|位置|地址|数量|单价|价格|金额|品牌|规格类型|型号|公司|中标人|企业|供应商|候选人', it) == None):
  376. product_secondary = it
  377. product_idx = i
  378. if len(it) < 15 and re.search(pat_project, it) and '项目名称' not in project:
  379. project = it
  380. project_idx = i
  381. if product == "":
  382. if product_secondary:
  383. flag = True
  384. product = product_secondary
  385. header_dic['名称'] = product_idx
  386. elif category:
  387. flag = True
  388. product = category
  389. header_dic['名称'] = header_dic['品目']
  390. header_dic['品目'] = ''
  391. category = ''
  392. elif project:
  393. flag = True
  394. product = project
  395. header_dic['名称'] = project_idx
  396. header_dic['产品为项目'] = True
  397. if flag == False and len(items)>3 and re.search('^第[一二三四五六七八九十](包|标段)$', items[0]):
  398. product = items[0]
  399. header_dic['名称'] = 0
  400. flag = True
  401. if flag:
  402. for j in range(len(items)):
  403. if header_dic['品目号'] == "" and re.search('(品目|品类)(编?号|编码|序号)', items[j]):
  404. header_dic['品目号'] = j
  405. pinmu_no = items[j]
  406. elif header_dic['品目名'] == "" and re.search('(品目|品类)名称|采购(品目|品类)$', items[j]):
  407. header_dic['品目名'] = j
  408. pinmu_name = items[j]
  409. if items[j] in [product, category]:
  410. continue
  411. if len(items[j]) > 20 and len(re.sub('[\((].*[)\)]|[^\u4e00-\u9fa5]', '', items[j])) > 10:
  412. continue
  413. if header_dic['数量']=="" and re.search('数量|采购量', items[j]) and re.search('单价|用途|要求|规格|型号|运输|承运', items[j])==None:
  414. header_dic['数量'] = j
  415. quantity = items[j]
  416. quantity = re.sub('\d', '', quantity)
  417. elif header_dic['单位']=="" and re.search('^(数量单位|计量单位|单位)$', items[j]):
  418. header_dic['单位'] = j
  419. quantity_unit = items[j]
  420. elif re.search('单价', items[j]) and re.search('数量|规格|型号|品牌|供应商', items[j])==None:
  421. header_dic['单价'] = j
  422. unitPrice = items[j]
  423. unitPrice = re.sub('\d', '', unitPrice)
  424. elif re.search('品牌', items[j]):
  425. header_dic['品牌'] = j
  426. brand = items[j]
  427. elif re.search('规格|型号', items[j]):
  428. header_dic['规格'] = j
  429. specs = items[j]
  430. elif re.search('参数', items[j]):
  431. header_dic['参数'] = j
  432. parameter = items[j]
  433. elif re.search('预算单位|(采购|招标|购买)(单位|人|方|主体)|项目业主|采购商|申购单位|需求单位|业主单位',items[j]) and len(items[j])<=8:
  434. header_dic['采购人'] = j
  435. tenderee = items[j]
  436. elif re.search('需求|服务要求|服务标准', items[j]):
  437. header_dic['需求'] = j
  438. demand = items[j]
  439. elif re.search('(采购|招标|投资|项目)(预算|估算)|(预算|控制|投资|项目|采购|招标)金额|(最高|招标)(\w{,2})限价|拦标价', items[j]) and not re.search('预算单位',items[j]):
  440. header_dic['预算'] = j
  441. budget = items[j]
  442. elif re.search('时间|采购时间|采购实施月份|采购月份|采购日期|(预计|计划)(招标|采购|发标|发包)(时间|月份)', items[j]):
  443. header_dic['时间'] = j
  444. order_time = items[j]
  445. elif re.search('总价|(成交|中标|验收|合同|预算|控制|总|合计))?([金总]额|价格?)|最高限价|价格|金额', items[j]) and re.search('数量|规格|型号|品牌|供应商', items[j])==None:
  446. header_dic['总价'] = j
  447. total_price = items[j]
  448. total_price = re.sub('\d', '', total_price)
  449. elif re.search('^备\s*注$|资质要求|预留面向中小企业|是否适宜中小企业采购预算预留|公开征集信息', items[j]):
  450. header_dic['备注'] = j
  451. notes = items[j]
  452. elif re.search('^\w{,4}发布(时间|日期)$', items[j]):
  453. header_dic['发布日期'] = j
  454. issue_date = items[j]
  455. if header_dic.get('名称', "") != "" or header_dic.get('品目', "") != "":
  456. # num = 0
  457. # for it in (quantity, unitPrice, brand, specs, product, demand, budget, order_time, total_price):
  458. # if it != "":
  459. # num += 1
  460. # if num >=2:
  461. # return header_dic, flag, (product, quantity, quantity_unit, unitPrice, brand, specs, total_price, category, parameter), (product, demand, budget, order_time)
  462. if set([quantity, brand, specs, unitPrice, total_price])!=set([""]) or set([demand, budget])!=set([""]):
  463. # if header_dic['产品为项目'] and (brand or specs):
  464. # header_dic['产品为项目'] = False
  465. return header_dic, flag, (product, quantity, quantity_unit, unitPrice, brand, specs, total_price, category, parameter, pinmu_no, pinmu_name), (product, demand, budget, order_time,tenderee, notes,issue_date)
  466. flag = False
  467. return header_dic, flag, (product, quantity, quantity_unit, unitPrice, brand, specs, total_price, category, parameter, pinmu_no, pinmu_name), (product, demand, budget, order_time,tenderee,notes,issue_date)
  468. def predict(self, docid='', html='', page_time=""):
  469. html = html.replace('<br>', '\n').replace('<br/>', '\n')
  470. html = re.sub("<html>|</html>|<body>|</body>","",html)
  471. html = re.sub("##attachment##","",html)
  472. soup = BeautifulSoup(html, 'lxml')
  473. del_tabel_achievement(soup)
  474. richText = soup.find(name='div', attrs={'class': 'richTextFetch'})
  475. if richText:
  476. richText = richText.extract() # 过滤掉附件
  477. def extract_product(soup):
  478. flag_yx = True
  479. tables = soup.find_all(['table'])
  480. table_results: List[TableResult] = []
  481. for table_idx in range(len(tables)):
  482. table = tables[table_idx]
  483. if table.parent.name == 'td' and len(table.find_all('td')) <= 3:
  484. table.string = table.get_text()
  485. table.name = 'turntable'
  486. continue
  487. if not self.isTrueTable(table):
  488. continue
  489. self.fixSpan(table)
  490. inner_table = self.getTable(table)
  491. table.extract()
  492. tr = self._extract_from_single_table(
  493. table_idx, inner_table, html, page_time, flag_yx
  494. )
  495. if tr.product_link and sum(tr.product_confirm) == 0:
  496. products = [re.sub('([^)]+)', '', it[0]) for it in tr.product_set]
  497. is_project = [1 if re.search('.{10,}(工程|项目|施工)', product) else 0 for product in products]
  498. if sum(is_project)>len(is_project)*0.5 or len(tr.product_link[0]) < 2: # 不确定产品,一半产品包含项目工程关键词或只有产品的去掉。
  499. tr.product_link = []
  500. tr.headers = []
  501. tr.total_product_money = 0
  502. log('去除项目工程名称作产品:%s, docid:%s'%(' '.join(products), docid))
  503. if tr.product_link or tr.demand_link:
  504. table_results.append(tr)
  505. if len(inner_table) > 0:
  506. last_tds = inner_table[-1] if inner_table else []
  507. if len(last_tds) >= 2 and len(set(last_tds)) in [2, 3] and re.search('订单总价',
  508. last_tds[0]) and re.search(
  509. '\d+[\d,\.]*', last_tds[1]):
  510. money_, unit_ = money_process(last_tds[1], last_tds[0])
  511. if table_results:
  512. table_results[-1].total_product_money = max(money_, table_results[-1].total_product_money)
  513. return table_results
  514. table_results = extract_product(soup)
  515. if (len(table_results) < 1 or sum([it for l in table_results for it in l.product_confirm])==0) and richText: # 正文没提取或提取产品不确定补充附件提取
  516. table_results_richText = extract_product(richText)
  517. table_results.extend(table_results_richText)
  518. merged = self._merge_table_results(table_results)
  519. product_link = merged.product_link
  520. demand_link = merged.demand_link
  521. total_product_money = merged.total_product_money
  522. headers = merged.headers
  523. headers_demand = merged.headers_demand
  524. header_col = merged.header_col
  525. budget_list = merged.budget_list
  526. result = self._post_process(product_link, demand_link, merged.total_price_list,
  527. merged.unit_price_list, budget_list)
  528. if result == 0:
  529. total_product_money = 0
  530. if len(product_link)>0:
  531. product_link = [{k:v for k,v in d.items() if v!=''} for d in product_link]
  532. attr_dic = {'product_attrs':{'data':product_link, 'header':headers, 'header_col':header_col}}
  533. attr_dic['product_attrs']['product_confirm'] = sum(merged.product_confirm) # 是否明确产品
  534. total_budget = sum(budget_list) if len(budget_list) == len(product_link) else 0
  535. if not attr_dic['product_attrs']['product_confirm']:
  536. if len(product_link[0]) < 3: # 产品不确定且要素少于3个的去掉
  537. attr_dic = {'product_attrs': {'data': [], 'header': [], 'header_col': []}}
  538. else:
  539. log('产品属性提取不确定,产品数量:%d,要素数量:%d,docid:%s'%(len(product_link), len(product_link[0]), docid))
  540. else:
  541. attr_dic = {'product_attrs': {'data': [], 'header': [], 'header_col': []}}
  542. total_budget = 0
  543. if len(demand_link)>0:
  544. demand_link = [{k: v for k, v in d.items() if v != ''} for d in demand_link]
  545. demand_dic = {'demand_info':{'data':demand_link, 'header':headers_demand, 'header_col':header_col}}
  546. else:
  547. demand_dic = {'demand_info':{'data':[], 'header':[], 'header_col':[]}}
  548. return [attr_dic, demand_dic], total_product_money, total_budget
  549. def _extract_from_single_table(self, table_idx, inner_table, html, page_time, flag_yx) -> TableResult:
  550. tr = TableResult(table_index=table_idx)
  551. found_header = False
  552. header_quan_unit = ""
  553. header_colnum = 0
  554. header_dic = {}
  555. header_list = header_list2 = ()
  556. if flag_yx:
  557. col0_l, col1_l = [], []
  558. for tds in inner_table:
  559. if len(tds) == 2:
  560. col0_l.append(re.sub('[::]', '', tds[0]))
  561. col1_l.append(tds[1])
  562. elif len(tds)>=4 and len(inner_table)==2:
  563. col0_l = inner_table[0]
  564. col1_l = inner_table[1]
  565. break
  566. if len(set(col0_l) & self.header_set) > len(col0_l) * 0.2 and len(col0_l)==len(col1_l):
  567. d_links, d_headers = self._extract_demand_from_2col(col0_l, col1_l, html, page_time)
  568. if d_links:
  569. tr.demand_link.extend(d_links)
  570. tr.headers_demand.extend(d_headers)
  571. return tr
  572. if len(inner_table)>3 and len(inner_table[0])==2 and len(inner_table[1])==2:
  573. col0_l, col1_l = [], []
  574. for tds in inner_table:
  575. if len(tds) == 2:
  576. col0_l.append(re.sub('[::]', '', tds[0]))
  577. col1_l.append(tds[1])
  578. else:
  579. break
  580. if len(set(col0_l) & self.header_set) > len(col0_l) * 0.5 and len(col0_l) == len(col1_l):
  581. inner_table = [col0_l, col1_l]
  582. elif len(inner_table)>2 and len(inner_table[0])==4 and len(inner_table[1])==4 and len(set(inner_table[0]) & self.header_set)==2:
  583. col0_l, col1_l, col2_l, col3_l = [], [], [], []
  584. for tds in inner_table:
  585. if len(tds) == 4 and len(set(tds))>2:
  586. col0_l.append(re.sub('[::]', '', tds[0]))
  587. col1_l.append(tds[1])
  588. col2_l.append(re.sub('[::]', '', tds[2]))
  589. col3_l.append(tds[3])
  590. else:
  591. break
  592. if len(set(col0_l) & self.header_set) > len(col0_l) * 0.5 and len(set(col2_l) & self.header_set) > len(col2_l) * 0.5:
  593. inner_table = [col0_l+col2_l, col1_l+col3_l]
  594. row_idx = 0
  595. while row_idx < (len(inner_table)):
  596. tds = inner_table[row_idx]
  597. not_empty = [it for it in tds if re.sub('\s', '', it) != ""]
  598. if len(set(not_empty))<2 or len(set(tds))<2 or (len(set(tds))==2 and re.search('总计|合计|汇总|总价', tds[0])):
  599. row_idx += 1
  600. continue
  601. product = ""
  602. quantity = ""
  603. quantity_unit = ""
  604. unitPrice = ""
  605. brand = ""
  606. specs = ""
  607. demand = ""
  608. budget = ""
  609. order_time = ""
  610. order_begin = ""
  611. order_end = ""
  612. total_price = ""
  613. parameter = ""
  614. tenderee = ""
  615. notes = ''
  616. issue_date = ''
  617. pinmu_no = ''
  618. pinmu_name = ''
  619. if len(set([re.sub('[::\s]','',td) for td in tds]) & self.header_set) > len(tds) * 0.4:
  620. header_dic, found_header, header_list, header_list2 = self.find_header(tds, self.pat_category, self.pat_product_primary, self.pat_product_secondary, self.pat_project)
  621. if found_header:
  622. header_colnum = len(tds)
  623. if found_header and isinstance(header_list, tuple) and len(header_list) > 2:
  624. quantity_header = header_list[1].replace('单位:', '')
  625. if re.search('(([\w/]{,5}))', quantity_header):
  626. header_quan_unit = re.search('(([\w/]{,5}))', quantity_header).group(1)
  627. else:
  628. header_quan_unit = ""
  629. if found_header and ('_'.join(header_list) not in tr.headers or '_'.join(header_list2) not in tr.headers_demand):
  630. tr.headers.append('_'.join(header_list))
  631. tr.headers_demand.append('_'.join(header_list2))
  632. tr.header_col.append('_'.join(tds))
  633. is_confirm = 0 if header_dic['产品为项目'] else 1
  634. tr.product_confirm.append(is_confirm)
  635. row_idx += 1
  636. continue
  637. elif found_header:
  638. if len(tds) > header_colnum or len(tds)-1<max([it for it in header_dic.values() if it!=""]):
  639. row_idx += 1
  640. continue
  641. id0 = header_dic.get('品目', "")
  642. id1 = header_dic.get('名称', "")
  643. id2 = header_dic.get('数量', "")
  644. id2_2 = header_dic.get('单位', "")
  645. id3 = header_dic.get('单价', "")
  646. id4 = header_dic.get('品牌', "")
  647. id5 = header_dic.get('规格', "")
  648. id6 = header_dic.get('需求', "")
  649. id7 = header_dic.get('预算', "")
  650. id8 = header_dic.get('时间', "")
  651. id9 = header_dic.get("总价", "")
  652. id10 = header_dic.get('参数', "")
  653. id11 = header_dic.get('采购人', "")
  654. id12 = header_dic.get('备注', "")
  655. id13 = header_dic.get('发布日期', "")
  656. id14 = header_dic.get('品目号', "")
  657. id15 = header_dic.get('品目名', "")
  658. not_attr = 0
  659. for k, v in header_dic.items():
  660. if isinstance(v, int):
  661. if v >= len(tds) or tds[v] in self.header_set:
  662. not_attr = 1
  663. if not_attr>=2:
  664. row_idx += 1
  665. found_header = False
  666. continue
  667. if id1!="" and re.search('[a-zA-Z\u4e00-\u9fa5]', tds[id1]) and tds[id1] not in self.header_set and \
  668. re.search('备注|汇总|合计|总价|价格|金额|^详见|无$|xxx', tds[id1].replace(' ', '')) == None:
  669. product = re.sub('\s+', '', tds[id1])
  670. if id0!="" and re.search('[a-zA-Z\u4e00-\u9fa5]', tds[id0]) and tds[id0] not in self.header_set and \
  671. re.search('备注|汇总|合计|总价|价格|金额|^详见|无$|xxx', tds[id0].replace(' ', '')) == None:
  672. category = re.sub('\s', '', tds[id0])
  673. # product = "%s_%s"%(category, product) if product!="" and product!=category else category # 20260528 去掉名称组合 修复 776939340 名称组合不像产品
  674. if product == '':
  675. product = category
  676. if product and re.match( '【?(工程类|服务类|货物类|工程|服务|货物|运费)】?', product) == None:
  677. if id2 != "":
  678. if re.search('\d+|[壹贰叁肆伍陆柒捌玖拾一二三四五六七八九十]', tds[id2]):
  679. quantity = tds[id2]
  680. elif re.search('\w{5,}', tds[id2]) and re.search('^详见|^详情', tds[id2])==None:
  681. row_idx += 1
  682. continue
  683. if id2_2 != "":
  684. if re.search('^\w{1,4}$', tds[id2_2]) and re.search('元', tds[id2_2])==None:
  685. quantity_unit = tds[id2_2]
  686. if id3 != "":
  687. if re.search('[零壹贰叁肆伍陆柒捌玖拾佰仟萬億十百千万亿元角分]{3,}', tds[id3]):
  688. unitPrice = tds[id3]
  689. elif re.search('^[\d,.亿万元人民币欧美日金额:()();;、,\n]+$|¥|¥|RMB|USD|EUR|JPY|CNY|元$', tds[id3].strip()):
  690. unitPrice = tds[id3]
  691. elif len(re.sub('[金额万元()()::零壹贰叁肆伍陆柒捌玖拾佰仟萬億圆十百千万亿元角分¥整\d,.]', '', tds[id3])) > 5 and re.search('^详见|^详情', tds[id3])==None:
  692. row_idx += 1
  693. continue
  694. else:
  695. unitPrice = tds[id3]
  696. if id4 != "":
  697. if re.search('\w', tds[id4]):
  698. brand = tds[id4]
  699. if re.match('^详见|^详情', brand.strip()):
  700. brand = ""
  701. else:
  702. brand = ""
  703. if id5 != "":
  704. if re.search('\w', tds[id5]):
  705. specs = tds[id5][:self.MAX_SPECS_LEN]
  706. if re.match('^详见|^详情', specs.strip()):
  707. specs = ""
  708. else:
  709. specs = ""
  710. if id6 != "":
  711. if re.search('\w', tds[id6]):
  712. demand = tds[id6]
  713. else:
  714. demand = ""
  715. if id7 != "":
  716. if re.search('\d+|[零壹贰叁肆伍陆柒捌玖拾佰仟萬億十百千万亿元角分]{3,}', tds[id7]):
  717. budget = tds[id7]
  718. if id8 != "":
  719. if re.search('\w', tds[id8]):
  720. order_time = tds[id8].strip()
  721. order_begin, order_end = self.fix_time(order_time, html, page_time)
  722. if id9 != "":
  723. if re.search('[零壹贰叁肆伍陆柒捌玖拾佰仟萬億十百千万亿元角分]{3,}', tds[id9]):
  724. total_price = tds[id9]
  725. elif re.search('^[\d,.亿万元人民币欧美日金额:()();;、,\n]+$|¥|¥|RMB|USD|EUR|JPY|CNY|元$', tds[id9].strip()):
  726. total_price = tds[id9]
  727. elif len(re.sub('[金额万元()()::零壹贰叁肆伍陆柒捌玖拾佰仟萬億圆十百千万亿元角分¥整\d,.]', '', tds[id9])) > 5 and re.search('^详见|^详情', tds[id9])==None:
  728. row_idx += 1
  729. continue
  730. if id10 != "":
  731. parameter = tds[id10][:self.MAX_PARAM_LEN]
  732. if re.match('^详见|^详情', parameter.strip()):
  733. parameter = ""
  734. if id11 != "":
  735. tenderee = re.sub("\s","",tds[id11])
  736. if len(tenderee) > self.MAX_TENDEREE_LEN:
  737. tenderee = ""
  738. if id12 != "":
  739. notes = tds[id12].strip()
  740. if id13 != "":
  741. issue_date = self.fix_time(tds[id13].strip(), '', '')[0]
  742. if id14 != "":
  743. pinmu_no = tds[id14].strip()
  744. if id15 != "":
  745. pinmu_name = tds[id15].strip()
  746. if quantity != "" or unitPrice != "" or brand != "" or specs != "" or total_price or '单价' in header_dic or '总价' in header_dic:
  747. if id1!="" and id2 != "" and id3 != "" and len(re.split('[;;、,\n]+', tds[id2])) > 1 and len(re.split('[;;、,\n]+', tds[id1])) == len(re.split('[;;、,\n]+', tds[id2])):
  748. products = re.split('[;;、,\n]+', tds[id1])
  749. quantitys = re.split('[;;、,\n]+', tds[id2])
  750. unitPrices = re.split('[;;、,\n]+', tds[id3])
  751. total_prices = re.split('[;;、,\n]+', total_price)
  752. brands = re.split('[;;、,\n]+', brand) if re.search('等$', brand)==None else [brand]
  753. specses = re.split('[;;、,\n]+', specs) if re.search('等$', specs)==None else [specs]
  754. parameters = re.split('[;;、,\n]+', parameter) if re.search('等$', parameter)==None else [parameter]
  755. unitPrices = [""]*len(products) if len(unitPrices)==1 else unitPrices
  756. total_prices = [""]*len(products) if len(total_prices)==1 else total_prices
  757. brands = brands*len(products) if len(brands)==1 else brands
  758. specses = specses*len(products) if len(specses)==1 else specses
  759. brands = [brand]*len(products) if len(brands) < len(products) else brands
  760. specses = [specs] * len(products) if len(specses) < len(products) else specses
  761. parameters = parameters*len(products) if len(parameters)==1 else parameters
  762. if len(products) == len(quantitys) == len(unitPrices) == len(brands) == len(specses):
  763. for product, quantity, unitPrice, brand, specs, total_price, parameter in zip(products,quantitys,unitPrices, brands, specses, total_prices, parameters):
  764. if product.strip() == '':
  765. continue
  766. if quantity != "":
  767. quantity, quantity_unit_ = self.fix_quantity(quantity, header_quan_unit)
  768. quantity_unit = quantity_unit_ if quantity_unit_ != "" else quantity_unit
  769. if unitPrice != "":
  770. unitPrice, _money_unit = money_process(unitPrice, header_list[3])
  771. unitPrice = str(unitPrice) if unitPrice != 0 and unitPrice<self.MAX_UNIT_PRICE else ""
  772. if budget != "":
  773. budget, _money_unit = money_process(budget, header_list2[2])
  774. budget = str(budget) if budget != 0 and budget<self.MAX_BUDGET else ''
  775. if total_price != "":
  776. total_price, _money_unit = money_process(total_price, header_list[6])
  777. tr.total_price_list.append(total_price)
  778. total_price = str(total_price) if total_price != 0 and total_price<self.MAX_TOTAL_PRICE else ""
  779. link = {'product': product, 'quantity': quantity,
  780. 'quantity_unit': quantity_unit, 'unitPrice': unitPrice,
  781. 'brand': brand[:self.MAX_BRAND_LEN], 'specs': specs, 'total_price': total_price, 'parameter': parameter}
  782. if (product, specs, unitPrice, quantity) not in tr.product_set:# and not header_dic['产品为项目']:
  783. tr.product_set.add((product, specs, unitPrice, quantity))
  784. tr.product_link.append(link)
  785. if budget != '' and float(budget) > 0:
  786. tr.budget_list.append(float(budget))
  787. if link['unitPrice'] != "" and link['quantity'] != '':
  788. try:
  789. tr.total_product_money += float(link['unitPrice']) * float(
  790. link['quantity']) if float(link['quantity']) < self.MAX_QUANTITY_FOR_CALC else 0
  791. except (ValueError, TypeError):
  792. log('产品属性单价数量相乘出错, 单价: %s, 数量: %s' % (
  793. link['unitPrice'], link['quantity']))
  794. elif len(product)>self.MAX_PRODUCT_NAME_LEN:
  795. row_idx += 1
  796. continue
  797. else:
  798. if quantity != "":
  799. quantity, quantity_unit_ = self.fix_quantity(quantity, header_quan_unit)
  800. quantity_unit = quantity_unit_ if quantity_unit_ != "" else quantity_unit
  801. if unitPrice != "":
  802. unitPrice, _money_unit = money_process(unitPrice, header_list[3])
  803. unitPrice = str(unitPrice) if unitPrice != 0 and unitPrice<self.MAX_UNIT_PRICE else ""
  804. if budget != "":
  805. budget, _money_unit = money_process(budget, header_list2[2])
  806. budget = str(budget) if budget != 0 and budget<self.MAX_BUDGET else ''
  807. if total_price != "":
  808. total_price, _money_unit = money_process(total_price, header_list[6])
  809. tr.total_price_list.append(total_price)
  810. total_price = str(total_price) if total_price != 0 and total_price<self.MAX_TOTAL_PRICE else ""
  811. link = {'product': product, 'quantity': quantity, 'quantity_unit': quantity_unit, 'unitPrice': unitPrice,
  812. 'brand': brand[:self.MAX_BRAND_LEN], 'specs':specs, 'total_price': total_price, 'parameter': parameter,
  813. 'pinmu_no': pinmu_no, 'pinmu_name': pinmu_name}
  814. if (product, unitPrice,) not in tr.product_set:# and not header_dic['产品为项目']:
  815. tr.product_set.add((product, unitPrice))
  816. tr.product_link.append(link)
  817. if budget != '' and float(budget) > 0:
  818. tr.budget_list.append(float(budget))
  819. if link['unitPrice']:
  820. tr.unit_price_list.append(link['unitPrice'])
  821. if link['unitPrice'] != "" and link['quantity'] != '':
  822. try:
  823. tr.total_product_money += float(link['unitPrice'])*float(link['quantity']) if float(link['quantity'])<self.MAX_QUANTITY_FOR_CALC else 0
  824. if float(link['unitPrice'])>10000 and float(link['quantity'])>100:
  825. tr.total_product_money = 0
  826. except (ValueError, TypeError):
  827. log('产品属性单价数量相乘出错, 单价: %s, 数量: %s'%(link['unitPrice'], link['quantity']))
  828. order_begin, order_end = self._validate_date_range(order_begin, order_end, page_time)
  829. if budget != "" and order_end != "":
  830. link = {'project_name': product, 'product':[], 'demand': demand, 'budget': budget, 'order_begin':order_begin, 'order_end':order_end, 'tenderee':tenderee,'notes':notes,'issue_date':issue_date}
  831. if link not in tr.demand_link:
  832. tr.demand_link.append(link)
  833. row_idx += 1
  834. else:
  835. row_idx += 1
  836. return tr
  837. def _extract_demand_from_2col(self, col0_l, col1_l, html, page_time):
  838. header_list2 = []
  839. product = demand = budget = order_begin = order_end = ""
  840. tenderee = ""
  841. notes = ''
  842. issue_date = ''
  843. demand_links = []
  844. for i in range(len(col0_l)):
  845. if re.search('项目名称', col0_l[i]):
  846. header_list2.append(col0_l[i])
  847. product = col1_l[i]
  848. elif re.search('采购需求|需求概况|招标内容|项目概况', col0_l[i]):
  849. header_list2.append(col0_l[i])
  850. demand = col1_l[i]
  851. elif re.search('(采购|招标|投资|项目)(预算|估算)|(预算|控制|投资|项目|采购|招标)金额|(最高|招标)(\w{,2})限价|拦标价', col0_l[i]):
  852. header_list2.append(col0_l[i])
  853. _budget = col1_l[i]
  854. re_price = re.findall("[零壹贰叁肆伍陆柒捌玖拾佰仟萬億圆十百千万亿元角分]{3,}|\d[\d,]*(?:\.\d+)?万?", _budget)
  855. if re_price:
  856. _budget, _money_unit = money_process(_budget, col0_l[i])
  857. budget = str(_budget)
  858. if '.' in budget:
  859. budget = budget.rstrip('0').rstrip('.')
  860. if float(budget)>= 500*100000000:
  861. budget = ""
  862. elif re.search('预算单位|(采购|招标|购买)(单位|人|方|主体)|项目业主|采购商|申购单位|需求单位|业主单位', col0_l[i]):
  863. header_list2.append(col0_l[i])
  864. tenderee = re.sub("\s","",col1_l[i])
  865. if len(tenderee) > 20:
  866. tenderee = ""
  867. elif re.search('时间|采购时间|采购实施月份|采购月份|采购日期|(预计|计划)(招标|采购|发标|发包)(时间|月份)', col0_l[i]):
  868. header_list2.append(col0_l[i])
  869. order_time = col1_l[i].strip()
  870. order_begin, order_end = self.fix_time(order_time, html, page_time)
  871. elif re.search('^备\s*注$|资质要求|预留面向中小企业|是否适宜中小企业采购预算预留|公开征集信息', col0_l[i]):
  872. header_list2.append(col0_l[i])
  873. notes = col1_l[i].strip()
  874. elif re.search('^\w{,4}发布(时间|日期)$', col0_l[i]):
  875. header_list2.append(col0_l[i])
  876. issue_date = self.fix_time(col1_l[i].strip(), '', '')[0]
  877. order_begin, order_end = self._validate_date_range(order_begin, order_end, page_time)
  878. if product!= "" and demand != "" and budget!="" and order_end != "":
  879. link = {'project_name': product, 'product': [], 'demand': demand, 'budget': budget,
  880. 'order_begin': order_begin, 'order_end': order_end ,'tenderee':tenderee, 'notes':notes, 'issue_date':issue_date}
  881. if link not in demand_links:
  882. demand_links.append(link)
  883. return demand_links, header_list2
  884. def _merge_table_results(self, table_results: List[TableResult]) -> TableResult:
  885. if not table_results:
  886. return TableResult()
  887. if len(table_results) == 1:
  888. return table_results[0]
  889. merged = TableResult()
  890. merged_product_set: Set[tuple] = set()
  891. merged_demand_set: Set[str] = set()
  892. budget_list = []
  893. total_product_money_list = []
  894. for i, tr in enumerate(table_results):
  895. if not tr.product_link and not tr.demand_link:
  896. continue
  897. overlap_ratio = self._calc_product_overlap(tr.product_names, merged_product_set)
  898. if overlap_ratio > 0.5 and self._same_header_group(tr, table_results[:i]):
  899. self._merge_with_dedup(tr, merged, merged_product_set, merged_demand_set, prefer_richer=True)
  900. elif overlap_ratio > 0:
  901. self._merge_with_dedup(tr, merged, merged_product_set, merged_demand_set, prefer_richer=True)
  902. else:
  903. self._merge_with_dedup(tr, merged, merged_product_set, merged_demand_set, prefer_richer=False)
  904. merged.headers.extend(tr.headers)
  905. merged.headers_demand.extend(tr.headers_demand)
  906. merged.header_col.extend(tr.header_col)
  907. merged.total_price_list.extend(tr.total_price_list)
  908. merged.unit_price_list.extend(tr.unit_price_list)
  909. merged.budget_list.extend(tr.budget_list)
  910. merged.product_confirm.extend(tr.product_confirm)
  911. if tr.budget_list and tr.budget_list not in budget_list:
  912. budget_list.append(tr.budget_list)
  913. if tr.total_product_money and tr.total_product_money not in total_product_money_list:
  914. total_product_money_list.append(tr.total_product_money)
  915. merged.total_product_money += tr.total_product_money
  916. if len(budget_list) > 1:
  917. merged.budget_list = []
  918. if len(total_product_money_list) > 1:
  919. merged.total_product_money = 0
  920. merged.headers = list(set(merged.headers))
  921. merged.headers_demand = list(set(merged.headers_demand))
  922. merged.header_col = list(set(merged.header_col))
  923. return merged
  924. @staticmethod
  925. def _calc_product_overlap(new_names: Set[str], existing_set: Set[tuple]) -> float:
  926. if not new_names or not existing_set:
  927. return 0.0
  928. existing_names = {p[0] for p in existing_set if p}
  929. if not existing_names:
  930. return 0.0
  931. overlap = new_names & existing_names
  932. return len(overlap) / min(len(new_names), len(existing_names))
  933. @staticmethod
  934. def _same_header_group(tr: TableResult, previous_results: List[TableResult]) -> bool:
  935. for prev in previous_results:
  936. if tr.header_signature and tr.header_signature == prev.header_signature:
  937. return True
  938. if tr.demand_header_signature and tr.demand_header_signature == prev.demand_header_signature:
  939. return True
  940. return False
  941. @staticmethod
  942. def _count_non_empty(d: dict) -> int:
  943. return len([v for v in d.values() if v != '' and v is not None])
  944. def _merge_with_dedup(self, new_tr: TableResult, merged: TableResult,
  945. merged_product_set: Set[tuple], merged_demand_set: Set[str],
  946. prefer_richer: bool) -> None:
  947. for link in new_tr.product_link:
  948. product = link.get('product', '')
  949. unit_price = link.get('unitPrice', '')
  950. specs = link.get('specs', '')
  951. quantity = link.get('quantity', '')
  952. key = (product, unit_price)
  953. if key in merged_product_set:
  954. if prefer_richer:
  955. existing_idx = None
  956. for idx, existing in enumerate(merged.product_link):
  957. if existing.get('product', '') == product and existing.get('unitPrice', '') == unit_price:
  958. existing_idx = idx
  959. break
  960. if existing_idx is not None:
  961. existing = merged.product_link[existing_idx]
  962. if self._count_non_empty(link) > self._count_non_empty(existing):
  963. merged.product_link[existing_idx] = link
  964. continue
  965. merged_product_set.add(key)
  966. merged.product_link.append(link)
  967. for link in new_tr.demand_link:
  968. project_name = link.get('project_name', '')
  969. budget = link.get('budget', '')
  970. order_end = link.get('order_end', '')
  971. demand_key = f"{project_name}_{budget}_{order_end}"
  972. if demand_key in merged_demand_set:
  973. if prefer_richer:
  974. existing_idx = None
  975. for idx, existing in enumerate(merged.demand_link):
  976. ex_key = f"{existing.get('project_name', '')}_{existing.get('budget', '')}_{existing.get('order_end', '')}"
  977. if ex_key == demand_key:
  978. existing_idx = idx
  979. break
  980. if existing_idx is not None:
  981. existing = merged.demand_link[existing_idx]
  982. if self._count_non_empty(link) > self._count_non_empty(existing):
  983. merged.demand_link[existing_idx] = link
  984. continue
  985. merged_demand_set.add(demand_key)
  986. merged.demand_link.append(link)
  987. def _post_process(self, product_link, demand_link, total_price_list, unit_price_list, budget_list):
  988. if len(total_price_list)>1 and len(set(total_price_list))/len(total_price_list)<=0.5:
  989. for link in product_link:
  990. if 'total_price' in link:
  991. link['total_price'] = ""
  992. if len(demand_link) > 2 and demand_link[0].get('budget', '') != '' and len(set([d.get('budget', '') for d in demand_link])) == 1:
  993. for d in demand_link:
  994. if 'budget' in d:
  995. d['budget'] = ""
  996. if len(unit_price_list)>0 and len(unit_price_list)==len(product_link) and len(set(unit_price_list))/len(unit_price_list)<=0.5:
  997. return 0
  998. return None
  999. def predict_without_table(self,product_attrs,list_sentences,list_entitys,codeName,prem, html='', page_time=""):
  1000. if len(prem[0]['prem'])==1:
  1001. list_sentences[0].sort(key=lambda x:x.sentence_index)
  1002. list_sentence = list_sentences[0]
  1003. list_entity = list_entitys[0]
  1004. _data = product_attrs[1]['demand_info']['data']
  1005. re_bidding_time = re.compile("(采购|采购实施|(预计|计划)(招标|采购|发标|发包))(时间|月份|日期)[::,].{0,2}$")
  1006. order_times = []
  1007. for entity in list_entity:
  1008. if entity.entity_type=='time':
  1009. # print('time',entity.entity_text)
  1010. sentence = list_sentence[entity.sentence_index]
  1011. s = spanWindow(tokens=sentence.tokens, begin_index=entity.begin_index,
  1012. end_index=entity.end_index,size=20)
  1013. entity_left = "".join(s[0])
  1014. if re.search(re_bidding_time,entity_left):
  1015. time_text = entity.entity_text.strip()
  1016. standard_time = re.compile("((?P<year>\d{4}|\d{2})\s*[-\/年\.]\s*(?P<month>\d{1,2})\s*[-\/月\.]\s*((?P<day>\d{1,2})日?)?)")
  1017. time_match = re.search(standard_time,time_text)
  1018. # print(time_text, time_match)
  1019. if time_match:
  1020. time_text = time_match.group()
  1021. order_times.append(time_text)
  1022. # print(order_times)
  1023. order_times = [tuple(self.fix_time(order_time, html, page_time)) for order_time in order_times]
  1024. order_times = [order_time for order_time in order_times if order_time[0]!=""]
  1025. if len(set(order_times))==1:
  1026. order_begin,order_end = order_times[0]
  1027. order_begin, order_end = self._validate_date_range(order_begin, order_end, page_time)
  1028. if order_end!="":
  1029. project_name = codeName[0]['name']
  1030. pack_info = [pack for pack in prem[0]['prem'].values()]
  1031. budget = pack_info[0].get('tendereeMoney',0)
  1032. product = prem[0]['product']
  1033. link = {'project_name': project_name, 'product': product, 'demand': project_name, 'budget': budget,
  1034. 'order_begin': order_begin, 'order_end': order_end}
  1035. _data.append(link)
  1036. product_attrs[1]['demand_info']['data'] = _data
  1037. # print('predict_without_table: ', product_attrs)
  1038. return product_attrs
  1039. def predict_by_text(self,product_attrs,html,list_outlines,product_list,page_time=""):
  1040. product_entity_list = list(set(product_list))
  1041. list_outline = list_outlines[0]
  1042. get_product_attrs = False
  1043. for _outline in list_outline:
  1044. if re.search("信息|情况|清单|概况",_outline.outline_summary):
  1045. outline_text = _outline.outline_text
  1046. outline_text = outline_text.replace(_outline.outline_summary,"")
  1047. key_value_list = [_split for _split in re.split("[,。;]",outline_text) if re.search("[::]",_split)]
  1048. if not key_value_list:
  1049. continue
  1050. head_list = []
  1051. head_value_list = []
  1052. for key_value in key_value_list:
  1053. key_value = re.sub("^[一二三四五六七八九十]{1,3}[、.]|^[\d]{1,2}[、.]\d{,2}|^[\((]?[一二三四五六七八九十]{1,3}[\))][、]?","",key_value)
  1054. temp = re.split("[::]",key_value)
  1055. if len(temp)>2:
  1056. if temp[0] in head_list:
  1057. key = temp[0]
  1058. value = "".join(temp[1:])
  1059. else:
  1060. key = temp[-2]
  1061. value = temp[-1]
  1062. else:
  1063. key = temp[0]
  1064. value = temp[1]
  1065. key = re.sub("^[一二三四五六七八九十]{1,3}[、.]|^[\d]{1,2}[、.]\d{,2}|^[\((]?[一二三四五六七八九十]{1,3}[\))][、]?","",key)
  1066. head_list.append(key)
  1067. head_value_list.append(value)
  1068. head_set = set(head_list)
  1069. # print('head_set',head_set)
  1070. if len(head_set & self.header_set) > len(head_set)*0.2:
  1071. loop_list = []
  1072. begin_list = [0]
  1073. for index,head in enumerate(head_list):
  1074. if head not in loop_list:
  1075. if re.search('第[一二三四五六七八九十](包|标段)', head) and re.search('第[一二三四五六七八九十](包|标段)', '|'.join(loop_list)):
  1076. begin_list.append(index)
  1077. loop_list = []
  1078. loop_list.append(head)
  1079. else:
  1080. loop_list.append(head)
  1081. else:
  1082. begin_list.append(index)
  1083. loop_list = []
  1084. loop_list.append(head)
  1085. headers = []
  1086. headers_demand = []
  1087. header_col = []
  1088. product_link = []
  1089. demand_link = []
  1090. product_set = set()
  1091. for idx in range(len(begin_list)):
  1092. if idx==len(begin_list)-1:
  1093. deal_list = head_value_list[begin_list[idx]:]
  1094. tmp_head_list = head_list[begin_list[idx]:]
  1095. else:
  1096. deal_list = head_value_list[begin_list[idx]:begin_list[idx+1]]
  1097. tmp_head_list = head_list[begin_list[idx]:begin_list[idx+1]]
  1098. product = "" # 产品
  1099. quantity = "" # 数量
  1100. quantity_unit = "" # 单位
  1101. unitPrice = "" # 单价
  1102. brand = "" # 品牌
  1103. specs = "" # 规格
  1104. demand = "" # 采购需求
  1105. budget = "" # 预算金额
  1106. order_time = "" # 采购时间
  1107. order_begin = ""
  1108. order_end = ""
  1109. total_price = "" # 总金额
  1110. parameter = "" # 参数
  1111. header_dic, found_header, header_list, header_list2 = self.find_header(tmp_head_list, self.pat_category, self.pat_product_primary, self.pat_product_secondary, self.pat_project)
  1112. if found_header:
  1113. headers.append('_'.join(header_list))
  1114. headers_demand.append('_'.join(header_list2))
  1115. header_col.append('_'.join(tmp_head_list))
  1116. # print('header_dic: ',header_dic)
  1117. id0 = header_dic.get('品目', "")
  1118. id1 = header_dic.get('名称', "")
  1119. id2 = header_dic.get('数量', "")
  1120. id2_2 = header_dic.get('单位', "")
  1121. id3 = header_dic.get('单价', "")
  1122. id4 = header_dic.get('品牌', "")
  1123. id5 = header_dic.get('规格', "")
  1124. id6 = header_dic.get('需求', "")
  1125. id7 = header_dic.get('预算', "")
  1126. id8 = header_dic.get('时间', "")
  1127. id9 = header_dic.get("总价", "")
  1128. id10 = header_dic.get('参数', "")
  1129. if id1!='' and re.search('[a-zA-Z\u4e00-\u9fa5]', deal_list[id1]) and deal_list[id1] not in self.header_set and \
  1130. re.search('备注|汇总|合计|总价|价格|金额|公司|附件|详见|无$|xxx', deal_list[id1]) == None:
  1131. product = deal_list[id1]
  1132. if id0 != "" and re.search('[a-zA-Z\u4e00-\u9fa5]', deal_list[id0]) and deal_list[id0] not in self.header_set and \
  1133. re.search('备注|汇总|合计|总价|价格|金额|公司|附件|详见|无$|xxx', deal_list[id0]) == None:
  1134. category = deal_list[id0]
  1135. product = "%s_%s" % (category, product) if product != "" else category
  1136. if product == "":
  1137. # print(deal_list[id4],deal_list[id5],tmp_head_list,deal_list)
  1138. if (id4 != "" and deal_list[id4] != "") or (id5 != "" and deal_list[id5] != ""):
  1139. for head,value in zip(tmp_head_list,deal_list):
  1140. if value and value in product_entity_list:
  1141. product = value
  1142. break
  1143. if product and re.match('【?(工程类|服务类|货物类|工程|服务|货物|运费)】?', product) == None:
  1144. if id2 != "":
  1145. if re.search('\d+|[壹贰叁肆伍陆柒捌玖拾一二三四五六七八九十]', deal_list[id2]):
  1146. quantity = deal_list[id2]
  1147. quantity = re.sub('[()(),,约]', '', quantity)
  1148. quantity = re.sub('[一壹]', '1', quantity)
  1149. ser = re.search('^(\d+(?:\.\d+)?)([㎡\w/]{,5})', quantity)
  1150. if ser:
  1151. quantity = str(ser.group(1))
  1152. quantity_unit = ser.group(2)
  1153. if float(quantity)>=10000*10000:
  1154. quantity = ""
  1155. quantity_unit = ""
  1156. else:
  1157. quantity = ""
  1158. quantity_unit = ""
  1159. if id2_2 != "":
  1160. if re.search('^\w{1,4}$', deal_list[id2_2]):
  1161. quantity_unit = deal_list[id2_2]
  1162. else:
  1163. quantity_unit = ""
  1164. # if id2 != "":
  1165. # if re.search('\d+|[壹贰叁肆伍陆柒捌玖拾一二三四五六七八九十]', deal_list[id2]):
  1166. # quantity = deal_list[id2]
  1167. # else:
  1168. # quantity = ""
  1169. if id3 != "":
  1170. if re.search('\d+|[零壹贰叁肆伍陆柒捌玖拾佰仟萬億十百千万亿元角分]{3,}', deal_list[id3]):
  1171. _unitPrice = deal_list[id3]
  1172. re_price = re.findall("[零壹贰叁肆伍陆柒捌玖拾佰仟萬億圆十百千万亿元角分]{3,}|\d[\d,]*(?:\.\d+)?万?",_unitPrice)
  1173. if re_price:
  1174. # _unitPrice = re_price[0]
  1175. # if '万元' in header_list[3] and '万' not in _unitPrice:
  1176. # _unitPrice += '万元'
  1177. # unitPrice = getUnifyMoney(_unitPrice)
  1178. # if unitPrice>=10000*10000:
  1179. # unitPrice = ""
  1180. # unitPrice = str(unitPrice)
  1181. _unitPrice, _money_unit = money_process(_unitPrice, header_list[3])
  1182. if _unitPrice >= 10000 * 10000:
  1183. _unitPrice = ""
  1184. unitPrice = str(_unitPrice)
  1185. if '.' in unitPrice:
  1186. unitPrice = unitPrice.rstrip('0').rstrip('.')
  1187. if id4 != "":
  1188. if re.search('\w', deal_list[id4]):
  1189. brand = deal_list[id4]
  1190. if re.match('^详见|^详情', brand.strip()):
  1191. brand = ""
  1192. else:
  1193. brand = ""
  1194. if id5 != "":
  1195. if re.search('\w', deal_list[id5]):
  1196. specs = deal_list[id5][:self.MAX_SPECS_LEN]
  1197. if re.match('^详见|^详情', specs.strip()):
  1198. specs = ""
  1199. else:
  1200. specs = ""
  1201. if id6 != "":
  1202. if re.search('\w', deal_list[id6]):
  1203. demand = deal_list[id6]
  1204. else:
  1205. demand = ""
  1206. if id7 != "":
  1207. if re.search('\d+|[零壹贰叁肆伍陆柒捌玖拾佰仟萬億十百千万亿元角分]{3,}', deal_list[id7]):
  1208. _budget = deal_list[id7]
  1209. re_price = re.findall("[零壹贰叁肆伍陆柒捌玖拾佰仟萬億圆十百千万亿元角分]{3,}|\d[\d,]*(?:\.\d+)?万?",_budget)
  1210. if re_price:
  1211. # _budget = re_price[0]
  1212. # if '万元' in header_list2[2] and '万' not in _budget:
  1213. # _budget += '万元'
  1214. # budget = str(getUnifyMoney(_budget))
  1215. _budget, _money_unit = money_process(_budget, header_list2[2])
  1216. budget = str(_budget)
  1217. if '.' in budget:
  1218. budget = budget.rstrip('0').rstrip('.')
  1219. if float(budget)>= 100000*10000:
  1220. budget = ""
  1221. if id8 != "":
  1222. if re.search('\w', deal_list[id8]) and re.search("(采购|采购实施|(预计|计划)(招标|采购|发标|发包))(时间|月份|日期)",header_list2[3]):
  1223. order_time = deal_list[id8].strip()
  1224. order_begin, order_end = self.fix_time(order_time, html, page_time)
  1225. if id9 != "":
  1226. if re.search('[零壹贰叁肆伍陆柒捌玖拾佰仟萬億十百千万亿元角分]{3,}', deal_list[id9]):
  1227. total_price = deal_list[id9]
  1228. elif re.search('^[\d,.亿万元人民币欧美日金额:()();;、,\n]+$', deal_list[id9].strip()):
  1229. total_price = deal_list[id9]
  1230. if id10 != "":
  1231. parameter = deal_list[id10][:self.MAX_PARAM_LEN]
  1232. if re.match('^详见|^详情', parameter.strip()):
  1233. parameter = ""
  1234. if quantity != "" or unitPrice != "" or brand != "" or specs != "" or total_price:
  1235. if id1 != "" and id2 != "" and id3 != "" and len(re.split('[;;、,\n]', deal_list[id2])) > 1 and len(
  1236. re.split('[;;、,\n]', deal_list[id1])) == len(re.split('[;;、,\n]', deal_list[id2])): # 处理一个空格包含多个产品,逗号或空格分割情况 例子 292846806 292650743
  1237. products = re.split('[;;、,\n]', deal_list[id1])
  1238. quantitys = re.split('[;;、,\n]', deal_list[id2])
  1239. unitPrices = re.split('[;;、,\n]', deal_list[id3])
  1240. total_prices = re.split('[;;、,\n]', total_price)
  1241. brands = re.split('[;;、,\n]', brand) if re.search('等$', brand) == None else [brand]
  1242. specses = re.split('[;;、,\n]', specs) if re.search('等$', specs) == None else [specs]
  1243. parameters = re.split('[;;、,\n]', parameter) if re.search('等$', parameter) == None else [parameter]
  1244. unitPrices = [""] * len(products) if len(unitPrices) == 1 else unitPrices
  1245. total_prices = [""] * len(products) if len(total_prices) == 1 else total_prices
  1246. brands = brands * len(products) if len(brands) == 1 else brands
  1247. specses = specses * len(products) if len(specses) == 1 else specses
  1248. parameters = parameters * len(products) if len(parameters) == 1 else parameters
  1249. if len(products) == len(quantitys) == len(unitPrices) == len(brands) == len(
  1250. specses):
  1251. for product, quantity, unitPrice, brand, specs, total_price, parameter in zip(
  1252. products, quantitys, unitPrices, brands, specses, total_prices,
  1253. parameters):
  1254. if quantity != "":
  1255. quantity, quantity_unit_ = self.fix_quantity(quantity,quantity_unit)
  1256. quantity_unit = quantity_unit_ if quantity_unit_ != "" else quantity_unit
  1257. if unitPrice != "":
  1258. unitPrice, _money_unit = money_process(unitPrice, header_list[3])
  1259. unitPrice = str(unitPrice) if unitPrice != 0 and unitPrice<self.MAX_UNIT_PRICE else ""
  1260. if budget != "":
  1261. budget, _money_unit = money_process(budget, header_list2[2])
  1262. budget = str(budget) if budget != 0 and budget<self.MAX_BUDGET else ''
  1263. if total_price != "":
  1264. total_price, _money_unit = money_process(total_price,
  1265. header_list[6])
  1266. total_price = str(total_price) if total_price != 0 and total_price<self.MAX_TOTAL_PRICE else ""
  1267. link = {'product': product, 'quantity': quantity,
  1268. 'quantity_unit': quantity_unit, 'unitPrice': unitPrice,
  1269. 'brand': brand[:self.MAX_BRAND_LEN], 'specs': specs, 'total_price': total_price,
  1270. 'parameter': parameter}
  1271. if (product, specs, unitPrice, quantity) not in product_set: # and not header_dic['产品为项目']:
  1272. product_set.add((product, specs, unitPrice, quantity))
  1273. product_link.append(link)
  1274. elif len(unitPrice) > 15 or len(product) > self.MAX_PRODUCT_NAME_LEN:
  1275. # i += 1
  1276. continue
  1277. else:
  1278. if quantity != "":
  1279. quantity, quantity_unit_ = self.fix_quantity(quantity, quantity_unit)
  1280. quantity_unit = quantity_unit_ if quantity_unit_ != "" else quantity_unit
  1281. if unitPrice != "":
  1282. unitPrice, _money_unit = money_process(unitPrice, header_list[3])
  1283. unitPrice = str(unitPrice) if unitPrice != 0 and unitPrice<self.MAX_UNIT_PRICE else ""
  1284. if budget != "":
  1285. budget, _money_unit = money_process(budget, header_list2[2])
  1286. budget = str(budget) if budget != 0 and budget<self.MAX_BUDGET else ''
  1287. if total_price != "":
  1288. total_price, _money_unit = money_process(total_price, header_list[6])
  1289. total_price = str(total_price) if total_price != 0 and total_price<self.MAX_TOTAL_PRICE else ""
  1290. link = {'product': product, 'quantity': quantity,
  1291. 'quantity_unit': quantity_unit, 'unitPrice': unitPrice,
  1292. 'brand': brand[:self.MAX_BRAND_LEN], 'specs': specs, 'total_price': total_price,
  1293. 'parameter': parameter}
  1294. if (product, specs, unitPrice, quantity) not in product_set: # and not header_dic['产品为项目']:
  1295. product_set.add((product, specs, unitPrice, quantity))
  1296. product_link.append(link)
  1297. order_begin, order_end = self._validate_date_range(order_begin, order_end, page_time)
  1298. if order_begin != "" and order_end != "":
  1299. order_begin_year = int(order_begin.split("-")[0])
  1300. order_end_year = int(order_end.split("-")[0])
  1301. if order_begin_year < 2000 or order_end_year < 2000:
  1302. order_begin = order_end = ""
  1303. if budget != "" and order_end != "":
  1304. link = {'project_name': product, 'product': [], 'demand': demand, 'budget': budget,
  1305. 'order_begin': order_begin, 'order_end': order_end}
  1306. if link not in demand_link:
  1307. demand_link.append(link)
  1308. if len(product_link) > 0:
  1309. attr_dic = {'product_attrs': {'data': product_link, 'header': list(set(headers)), 'header_col': list(set(header_col))}}
  1310. get_product_attrs = True
  1311. else:
  1312. attr_dic = {'product_attrs': {'data': [], 'header': [], 'header_col': []}}
  1313. if len(demand_link) > 0:
  1314. demand_dic = {'demand_info': {'data': demand_link, 'header': headers_demand, 'header_col': header_col}}
  1315. else:
  1316. demand_dic = {'demand_info': {'data': [], 'header': [], 'header_col': []}}
  1317. product_attrs[0] = attr_dic
  1318. if len(product_attrs[1]['demand_info']['data']) == 0:
  1319. product_attrs[1] = demand_dic
  1320. if get_product_attrs:
  1321. break
  1322. # print('predict_by_text: ', product_attrs)
  1323. return product_attrs
  1324. def add_product_attrs(self,channel_dic, product_attrs, list_sentences,list_entitys,list_outlines,product_list,codeName,prem,text,page_time):
  1325. # print(1,product_attrs[1]['demand_info']['data'])
  1326. if channel_dic['docchannel']['docchannel']=="采购意向" and len(product_attrs[1]['demand_info']['data']) == 0:
  1327. product_attrs = self.predict_without_table(product_attrs, list_sentences,list_entitys,codeName,prem,text,page_time)
  1328. # print(2,product_attrs[1]['demand_info']['data'])
  1329. if len(product_attrs[0]['product_attrs']['data']) == 0:
  1330. product_attrs = self.predict_by_text(product_attrs,text,list_outlines,product_list,page_time)
  1331. # print(3,product_attrs[1]['demand_info']['data'])
  1332. if len(product_attrs[1]['demand_info']['data'])>0:
  1333. for d in product_attrs[1]['demand_info']['data']:
  1334. for product in set(prem[0]['product']):
  1335. if product in d['project_name'] and product not in d['product']:
  1336. d['product'].append(product) #把产品在项目名称中的添加进需求要素中