# -*- coding: utf-8 -*- """金额规则预测器。 按 ARCHITECTURE.md Phase 5 拆分建议,从 ``interface/predictor.py`` 迁出以下 金额相关类: - ``MoneyGrade`` — 金额概率分级(原 predictor.py 第 2954-3024 行) - ``DepositPaymentWay`` — 保证金支付方式提取(原 5551-5581 行) - ``TotalUnitMoney`` — 总价单价提取(原 6203-6231 行) ``interface/predictor.py`` 仍 re-export 以上全部名称,老 import 不受影响。 """ from __future__ import absolute_import import re from BiddingKG.dl.money.re_money_total_unit import extract_total_money, extract_unit_money __all__ = [ "MoneyGrade", "DepositPaymentWay", "TotalUnitMoney", ] class MoneyGrade(): def __init__(self): self.tenderee_money_left_9 = "(?P最高(投标)?限价)|控制(价|金额)|拦标价" self.tenderee_money_left_8 = "(?P预算|限价|起始|起拍|底价|标底)" self.tenderer_money_left_9 = "(?P(中标|成交|合同|总报价))" self.tenderer_money_left_8 = "(?P(投标|总价))" self.pattern_list = [self.tenderee_money_left_8, self.tenderer_money_left_8, self.tenderee_money_left_9, self.tenderer_money_left_9] def predict(self, list_sentences, list_entitys, span=10, min_prob=0.7): sentences = sorted(list_sentences[0], key=lambda x:x.sentence_index) role2id = {"tenderee": 0, "tenderer": 1} for entity in list_entitys[0]: if entity.entity_type in ['money'] and entity.label in [0, 1] and entity.values[entity.label]> 0.6: text = sentences[entity.sentence_index].sentence_text in_att = sentences[entity.sentence_index].in_attachment b = entity.wordOffset_begin e = entity.wordOffset_end context = text[max(0, b - span):b] front_long = text[max(0, b - span*3):b] behind = text[e:e+span] not_found = 1 if entity.label == 0 and re.search('招标(控制价|金额|预算)小于$', context): # 修复预算提错。 784107197 招标控制价小于200万的标段招标文件不收取费用 entity.values[entity.label] = 0.49 continue for pattern in self.pattern_list: ser = re.search(pattern, context) if ser: groupdict = pattern.split('>')[0].replace('(?P<', '') _role, _direct, _prob = groupdict.split('_') if re.search('单价|不含税', context[-8:]) or re.search('(最低|风险)控制价', context) or entity.notes == '总投资' or re.match('[至到]($|¥)?\d+', behind):# or float(entity.entity_text)<100: # 753504698 最高限价单价 概率降低 _prob = 6 _label = role2id.get(_role) if _label != entity.label: continue _prob = int(_prob) * 0.1 if in_att: _prob = max(0.5, _prob - 0.2) entity.values[_label] = _prob + entity.values[_label] / 20 not_found = 0 if _label == 0 and float(entity.entity_text)<100 and entity.values[_label] > 0.6: # 20250624 小金额预算概率降低 634252534 包合计才是真正的预算 entity.values[_label] = 0.6 break if not_found and entity.values[entity.label] > min_prob: if re.search('单价', context[-8:]) or re.search('(最低|风险)控制价', context) or float(entity.entity_text)<100: _prob = 0.6 elif in_att: _prob = max(0.5, min_prob - 0.1) else: _prob = min_prob # _prob = min_prob - 0.1 if in_att else min_prob entity.values[entity.label] = _prob + entity.values[entity.label] / 20 # print('找不到规则修改金额概率:', entity.entity_text, entity.label, entity.values) if not_found and re.search('[\d,.万]+元?至¥?$', front_long) and entity.values[entity.label] > 0.5: entity.values[entity.label] = 0.8 # if entity.entity_type in ['money'] and entity.label in [0, 1] and 0.5<=entity.values[entity.label]<0.75 and float(entity.entity_text)<100: # 20241011 低概率小金额改为其他金额 # 20241128 小金额可能为单价,放单价存放 # entity.label = 2 elif entity.entity_type in ['money']: text = sentences[entity.sentence_index].sentence_text in_att = sentences[entity.sentence_index].in_attachment b = entity.wordOffset_begin e = entity.wordOffset_end context = text[max(0, b - span*4):b] if re.search('(?