# -*- coding: utf-8 -*- """上下文窗口与金额标准化纯函数。 按 ARCHITECTURE.md Phase 3 拆分建议,从 ``common/Utils.py`` 迁出。 类型:CORE(业务纯函数,人工主导)。 原位置:``common/Utils.py`` 中以下函数: 上下文窗口: - ``spanWindow`` / ``get_context`` / ``findAllIndex`` / ``find_index`` 金额标准化: - ``getUnifyMoney`` / ``getDigitsDic`` / ``getMultipleFactor`` - ``partMoney`` / ``uniform_num`` / ``uniform_package_name`` - ``money_process`` / ``get_money_entity`` 其他纯函数: - ``combine`` / ``fitDataByRule`` / ``clean_company`` - ``cut_repeat_name`` / ``is_all_winner`` / ``is_deposit_project`` - ``find_package``(含 ``package_number_pattern`` / ``filter_package_pattern``) 本文件自包含,不依赖 ``common/Utils.py``,避免循环 import。 ``common/Utils.py`` 仍 re-export 以上全部名称,老 import 不受影响。 按 ARCHITECTURE.md §4.3 依赖方向约束: common 可被各层引用,但不反向依赖 pipeline / predictors / assembly / rules。 """ from __future__ import absolute_import import re from decimal import Decimal __all__ = [ # 上下文窗口 "spanWindow", "get_context", "findAllIndex", "find_index", # 金额标准化 "getUnifyMoney", "getDigitsDic", "getMultipleFactor", "partMoney", "uniform_num", "uniform_package_name", "money_process", "get_money_entity", # 其他纯函数 "combine", "fitDataByRule", "clean_company", "cut_repeat_name", "is_all_winner", "is_deposit_project", "find_package", # 模块级正则 "package_number_pattern", "filter_package_pattern", ] # ============================================================ # 上下文窗口 # ============================================================ def find_index(list_tofind, text): ''' @summary: 查找所有词汇在字符串中第一次出现的位置 @param: list_tofind:待查找词汇 text:字符串 @return: list,每个词汇第一次出现的位置 ''' result = [] for item in list_tofind: index = text.find(item) if index >= 0: result.append(index) else: result.append(-1) return result def findAllIndex(substr, wholestr): ''' @summary: 找到字符串的子串的所有begin_index @param: substr:子字符串 wholestr:子串所在完整字符串 @return: list,字符串的子串的所有begin_index ''' copystr = wholestr result = [] indexappend = 0 while True: index = copystr.find(substr) if index < 0: break else: result.append(indexappend + index) indexappend += index + len(substr) copystr = copystr[index + len(substr):] return result def spanWindow(tokens, begin_index, end_index, size, center_include=False, word_flag=False, use_text=False, text=None): ''' @summary:取得某个实体的上下文词汇 @param: tokens:句子分词list begin_index:实体的开始index end_index:实体的结束index size:左右两边各取多少个词 center_include:是否包含实体 word_flag:词/字,默认是词 @return: list,实体的上下文词汇 ''' if use_text: assert text is not None length_tokens = len(tokens) if begin_index > size: begin = begin_index - size else: begin = 0 if end_index + size < length_tokens: end = end_index + size + 1 else: end = length_tokens result = [] if not word_flag: result.append(tokens[begin:begin_index]) if center_include: if use_text: result.append(text) else: result.append(tokens[begin_index:end_index + 1]) result.append(tokens[end_index + 1:end]) else: result.append("".join(tokens[begin:begin_index])) if center_include: if use_text: result.append(text) else: result.append("".join(tokens[begin_index:end_index + 1])) result.append("".join(tokens[end_index + 1:end])) return result def get_context(sentence_text, begin_index, end_index, size=20, center_include=False): ''' 返回实体上下文信息 :param sentence_text: 句子文本 :param begin_index: 实体字开始位置 :param end_index: 实体字结束位置 :param size: 字偏移量 :param center_include: :return: ''' result = [] begin = begin_index - size if begin_index > size else 0 end = end_index + size result.append(sentence_text[begin: begin_index]) if center_include: result.append(sentence_text[begin_index: end_index]) result.append(sentence_text[end_index: end]) return result # ============================================================ # 金额标准化 # ============================================================ def getDigitsDic(unit): ''' @summary:拿到中文对应的数字 ''' DigitsDic = {"零": 0, "壹": 1, "贰": 2, "叁": 3, "肆": 4, "伍": 5, "陆": 6, "柒": 7, "捌": 8, "玖": 9, "〇": 0, "一": 1, "二": 2, "三": 3, "四": 4, "五": 5, "六": 6, "七": 7, "八": 8, "九": 9} return DigitsDic.get(unit) def getMultipleFactor(unit): ''' @summary:拿到单位对应的值 ''' MultipleFactor = {"兆": Decimal(1000000000000), "亿": Decimal(100000000), "万": Decimal(10000), "仟": Decimal(1000), "千": Decimal(1000), "佰": Decimal(100), "百": Decimal(100), "拾": Decimal(10), "十": Decimal(10), "元": Decimal(1), "圆": Decimal(1), "角": round(Decimal(0.1), 1), "分": round(Decimal(0.01), 2)} return MultipleFactor.get(unit) def getUnifyMoney(money): ''' @summary:将中文金额字符串转换为数字金额 @param: money:中文金额字符串 @return: decimal,数据金额 ''' MAX_MONEY = 1000000000000 MAX_NUM = 12 # 去掉逗号 money = re.sub("[,,]", "", money) money = re.sub("[^0-9.零壹贰叁肆伍陆柒捌玖拾佰仟萬億圆十百千万亿元角分]", "", money) result = Decimal(0) chnDigits = ["零", "壹", "贰", "叁", "肆", "伍", "陆", "柒", "捌", "玖"] chnFactorUnits = ["兆", "亿", "万", "仟", '千', "佰", '百', "拾", '十', "圆", "元", "角", "分"] LowMoneypattern = re.compile("^[\d,]+(\.\d+)?$") BigMoneypattern = re.compile("^零?(?P[%s])$" % ("".join(chnDigits))) try: if re.search(LowMoneypattern, money) is not None: return Decimal(money) elif re.search(BigMoneypattern, money) is not None: return getDigitsDic(re.search(BigMoneypattern, money).group("BigMoney")) for factorUnit in chnFactorUnits: if re.search(re.compile(".*%s.*" % factorUnit), money) is not None: subMoneys = re.split(re.compile("%s(?!.*%s.*)" % (factorUnit, factorUnit)), money) if re.search(re.compile("^(\d+)(\.\d+)?$"), subMoneys[0]) is not None: if MAX_MONEY / getMultipleFactor(factorUnit) < Decimal(subMoneys[0]): return Decimal(0) result += Decimal(subMoneys[0]) * (getMultipleFactor(factorUnit)) elif len(subMoneys[0]) == 1: if re.search(re.compile("^[%s]$" % ("".join(chnDigits))), subMoneys[0]) is not None: result += Decimal(getDigitsDic(subMoneys[0])) * (getMultipleFactor(factorUnit)) # subMoneys[0]中无金额单位,不可再拆分 elif subMoneys[0] == "": result += 0 elif re.search(re.compile("[%s]" % ("".join(chnFactorUnits))), subMoneys[0]) is None: result += Decimal(getUnifyMoney(subMoneys[0])) * (getMultipleFactor(factorUnit)) else: result += Decimal(getUnifyMoney(subMoneys[0])) * (getMultipleFactor(factorUnit)) if len(subMoneys) > 1: if re.search(re.compile("^(\d+(,)?)+(\.\d+)?[百千万亿]?\s?(元)?$"), subMoneys[1]) is not None: result += Decimal(subMoneys[1]) elif len(subMoneys[1]) == 1: if re.search(re.compile("^[%s]$" % ("".join(chnDigits))), subMoneys[1]) is not None: result += Decimal(getDigitsDic(subMoneys[1])) else: result += Decimal(getUnifyMoney(subMoneys[1])) break except Exception: return Decimal(0) return result def partMoney(entity_text, input2_shape=[7]): ''' @summary:对金额分段 @param: entity_text:数值金额 input2_shape:分类数 @return: array,分段之后的独热编码 ''' import numpy as np money = float(entity_text) parts = np.zeros(input2_shape) if money < 100: parts[0] = 1 elif money < 1000: parts[1] = 1 elif money < 10000: parts[2] = 1 elif money < 100000: parts[3] = 1 elif money < 1000000: parts[4] = 1 elif money < 10000000: parts[5] = 1 else: parts[6] = 1 return parts def uniform_num(num): d1 = {'一': '1', '二': '2', '三': '3', '四': '4', '五': '5', '六': '6', '七': '7', '八': '8', '九': '9', '十': '10'} d3 = {'Ⅰ': '1', 'Ⅱ': '2', 'Ⅲ': '3', 'Ⅳ': '4', 'Ⅴ': '5', 'Ⅵ': '6', 'Ⅶ': '7'} if num.isdigit(): if re.search('^0[\d]$', num): num = num[1:] return num elif re.search('^[一二三四五六七八九十]+$', num): _digit = re.search('^[一二三四五六七八九十]+$', num).group(0) if len(_digit) == 1: num = d1[_digit] elif len(_digit) == 2 and _digit[0] == '十': num = '1' + d1[_digit[1]] elif len(_digit) == 2 and _digit[1] == '十': num = d1[_digit[0]] + '0' elif len(_digit) == 3 and _digit[1] == '十': num = d1[_digit[0]] + d1[_digit[2]] elif re.search('[ⅠⅡⅢⅣⅤⅥⅦ]', num): num = re.search('[ⅠⅡⅢⅣⅤⅥⅦ]', num).group(0) num = d3[num] return num def uniform_package_name(package_name): ''' 统一规范化包号。数值类型统一为阿拉伯数字,字母统一为大写,包含施工监理等抽到前面, 例 A包监理一标段 统一为 监理A1 ; 包Ⅱ 统一为 2 :param package_name: 字符串类型 包号 :return: ''' package_name_raw = package_name package_name = re.sub('pdf|doc|docs|xlsx|rar|\d{4}年', ' ', package_name) package_name = package_name.replace('标段(包)', '标段').replace('№', '') package_name = re.sub('\[|【', '', package_name) kw = re.search('(施工|监理|监测|勘察|设计|劳务)', package_name) name = "" if kw: name += kw.group(0) if re.search('^[a-zA-Z0-9-]{5,}$', package_name): _digit = re.search('^[a-zA-Z0-9-]{5,}$', package_name).group(0).upper() name += _digit elif re.search('(?P[a-zA-Z])包[:)]?第?(?P([0-9]{1,4}|[一二三四五六七八九十]{1,4}|[ⅠⅡⅢⅣⅤⅥⅦ]{1,4}))标段?', package_name): ser = re.search('(?P[a-zA-Z])包[:)]?第?(?P([0-9]{1,4}|[一二三四五六七八九十]{1,4}|[ⅠⅡⅢⅣⅤⅥⅦ]{1,4}))标段?', package_name) _char = ser.groupdict().get('eng') if _char: _char = _char.upper() _digit = ser.groupdict().get('num') _digit = uniform_num(_digit) name += _char.upper() + _digit elif re.search('第?(?P[0-9a-zA-Z-]{1,4})?(?P([0-9]{1,4}|[一二三四五六七八九十]{1,4}|[ⅠⅡⅢⅣⅤⅥⅦ]{1,4}))(标[段号的包项]?|合同[包段]|([分子]?[包标]))', package_name): ser = re.search('第?(?P[0-9a-zA-Z-]{1,4})?(?P([0-9]{1,4}|[一二三四五六七八九十]{1,4}|[ⅠⅡⅢⅣⅤⅥⅦ]{1,4}))(标[段号的包项]?|合同[包段]|([分子]?[包标]))', package_name) _char = ser.groupdict().get('eng') if _char: _char = _char.upper() _digit = ser.groupdict().get('num') _digit = uniform_num(_digit) if _char: name += _char.upper() name += _digit elif re.search('(标[段号的包项]?|项目|子项目?|采购包(?|([分子]?包|包[组件号]))编?号?[::]?(?P[0-9a-zA-Z-]{1,4})?(?P([0-9]{1,4}|[一二三四五六七八九十]{1,4}|[ⅠⅡⅢⅣⅤⅥⅦ]{1,4}))', package_name): ser = re.search('(标[段号的包项]?|项目|子项目?|采购包(?|([分子]?包|包[组件号]))编?号?[::]?(?P[0-9a-zA-Z-]{1,4})?(?P([0-9]{1,4}|[一二三四五六七八九十]{1,4}|[ⅠⅡⅢⅣⅤⅥⅦ]{1,4}))', package_name) _char = ser.groupdict().get('eng') if _char: _char = _char.upper() _digit = ser.groupdict().get('num') _digit = uniform_num(_digit) if _char: name += _char.upper() name += _digit elif re.search('(标[段号的包项]|([分子]?包|包[组件号]))编?号?[::]?(?P[a-zA-Z-]{1,5})', package_name): _digit = re.search('(标[段号的包项]|([分子]?包|包[组件号]))编?号?[::]?(?P[a-zA-Z-]{1,5})', package_name).group('eng').upper() name += _digit elif re.search('(?P[a-zA-Z]{1,4})(标[段号的包项]|([分子]?[包标]|包[组件号]))', package_name): _digit = re.search('(?P[a-zA-Z]{1,4})(标[段号的包项]|([分子]?[包标]|包[组件号]))', package_name).group('eng').upper() name += _digit elif re.search('^([0-9]{1,4}|[一二三四五六七八九十]{1,4}|[ⅠⅡⅢⅣⅤⅥⅦ]{1,4})$', package_name): _digit = re.search('^([0-9]{1,4}|[一二三四五六七八九十]{1,4}|[ⅠⅡⅢⅣⅤⅥⅦ]{1,4})$', package_name).group(0) _digit = uniform_num(_digit) name += _digit elif re.search('^[a-zA-Z0-9-]+$', package_name): _char = re.search('^[a-zA-Z0-9-]+$', package_name).group(0) name += _char.upper() if name == "": return package_name_raw else: if name.isdigit(): name = str(int(name)) return name def money_process(money_text, header): ''' 输入金额文本及金额列表头,返回统一数字化金额及金额单位 :param money_text:金额字符串 :param header:金额列表头,用于提取单位 :return: ''' money = 0 money_unit = "" moneys, _ = get_money_entity('%s:%s' % (header, money_text)) if len(moneys) == 1: money = float(moneys[0][0]) money_unit = moneys[0][3] elif len(moneys) == 2 and moneys[0][0] == moneys[1][0]: money = float(moneys[0][0]) money_unit = moneys[0][3] return (money, money_unit) def get_money_entity(sentence_text, found_yeji=0, in_attachment=False): money_list = [] # 使用正则识别金额 entity_type = "money" list_money_pattern = {"cn": "(()(?P百分之)?(?P[零壹贰叁肆伍陆柒捌玖拾佰仟萬億圆十百千万亿元角分]{3,})())", "key_word": "((?P(?:[¥¥]+,?|(中标|成交|合同|承租|投资|服务|起始))?(金?额|价格?)|价格|预算(金额)?|(监理|设计|勘察)(服务)?费|[单报标限总造]价款?|金额|租金|标的基本情况|CNY|成交结果|资金|(控制|拦标)价|投资|成本)(/折扣率|/投标费率(%))?(\d:|\d=\d[-+×]\d:)?(?:[,,\[(\(]*\s*(人民币|单位:?)?/?(?P[万亿]?(?:[美日欧]元|元(/(M2|[\u4e00-\u9fa5]{1,3}))?)?(?P[台个只吨]*))\s*(/?费率)?(人民币)?[\])\)]?)\s*[,,::]*(RMB|USD|EUR|JPY|CNY)?[::]?(\s*[^壹贰叁肆伍陆柒捌玖拾佰仟萬億分万元编号时间日期计采a-zA-Z]{,8}?))((可填写下浮率、折扣率或费率):?)?(第[123一二三]名[::])?(\d+(\*\d+%)+=)?(?P\d+,\d+\.\d{2,6}|\d{1,3}([,,]\d{3,})+(\.\d+)?|\d+(\.\d+)?[百千]{,1})(?P(E-?\d+))?(?:[(\(]?(?P[%%‰折])*\s*,?((金额)?单位[::])?(?P[万亿]?(?:[美日欧]元|元)?(?P[台只吨斤棵株页亩方条天年月日]*))\s*[)\)]?))", "front_m": "((?P(?:[(\(]?\s*(?P[万亿]?(?:[美日欧]元|元))\s*[)\)]?)\s*[,,::]*(\s*[^壹贰叁肆伍陆柒捌玖拾佰仟萬億分万元编号时间日期计采a-zA-Z金额价格]{,2}?))(?P\d{1,3}([,,]\d{3})+(\.\d+)?|\d+(\.\d+)?(?:,?)[百千]*)(?P(E-?\d+))?())", "behind_m": "(()()(?P\d{1,3}([,,]\d{3})+(\.\d+)?|\d+(\.\d+)?(?:,?)[百千]*)(?P(E-?\d+))?(人民币)?[\((]?(?P[万亿]?(?:[美日欧]元|元)(?P[台个只吨斤棵株页亩方条米]*))[\))]?)"} pattern_money = re.compile("%s|%s|%s|%s" % ( list_money_pattern["cn"], list_money_pattern["key_word"], list_money_pattern["behind_m"], list_money_pattern["front_m"])) sentence_text = re.sub(r"(金额|价格|限价)[:为]?(\d+),(\d+)", r"\1\2\3", sentence_text) sentence_text = sentence_text.replace('总金额**亿元', 'xxxxxxx') ser = re.search('((收费标准|计算[方公]?式):|\w{3,5}\s*=)+\s*[中标投标成交金额招标人预算价格万元\s()()\[\]【】\d\.%%‰\+\-*×/]{20,}[,。]?', sentence_text) if ser: sentence_text = sentence_text.replace(ser.group(0), ' ' * len(ser.group(0))) all_match = re.finditer(pattern_money, sentence_text) for _match in all_match: if re.search('^元/1\d{10},$', _match.group(0)): continue elif re.search('元),?\d$', _match.group(0)) and re.match('[,、]', sentence_text[_match.end():]): continue if len(_match.group()) > 0: notes = '' unit = "" entity_text = "" start_index = "" end_index = "" text_beforeMoney = "" filter = "" filter_unit = False notSure = False science = "" if re.search('业绩(公示|汇总|及|报告|\w{,2}(内容|情况|信息)|[^\w])', sentence_text[:_match.span()[0]]): found_yeji += 1 break for k, v in _match.groupdict().items(): if v != "" and v is not None: if k == 'text_key_word': notSure = True if k.split("_")[0] == "money": entity_text = v if entity_text.endswith(',00'): entity_text = entity_text[:-3] if k.split("_")[0] == "unit": if 'behind' in k or unit == "": unit = v if k.split("_")[0] == "text": text_beforeMoney = v if k.split("_")[0] == "filter": filter = v if re.search("filter_unit", k) is not None: filter_unit = True if k.split("_")[0] == 'science': science = v if filter != "": continue if len(entity_text) > 30 or len(re.sub('[E-]', '', science)) > 2: continue start_index, end_index = _match.span() start_index += len(text_beforeMoney) if re.search('电话|手机|联系|方式|编号|编码|日期|数字|时间', text_beforeMoney): continue elif re.search('^1[3-9]\d{9}$', entity_text) and re.search(':\w{1,3}$', text_beforeMoney): continue elif re.search('^\d(.\d{1,2})?$', entity_text) and re.search('\d$', _match.group(0)) and re.search('^[、.]', sentence_text[_match.end():]): continue if unit == "": if (re.search('(¥|¥|RMB|CNY)[::]?$', text_beforeMoney) or re.search('[零壹贰叁肆伍陆柒捌玖拾佰仟萬億圆十百千万亿元角分]{3,}', entity_text)): if entity_text.endswith('万元'): unit = '万元' entity_text = entity_text[:-2] else: unit = '元' elif re.search('USD[::]?$', text_beforeMoney): unit = '美元' elif re.search('EUR[::]?$', text_beforeMoney): unit = '欧元' elif re.search('JPY[::]?$', text_beforeMoney): unit = '日元' elif re.search('^[-—]+[\d,.]+万元', sentence_text[end_index:]): unit = '万元' elif re.search('^,?(价格币种:\w{2,3},)?价格单位:万元', sentence_text[end_index:]): unit = '万元' elif re.search('万元', sentence_text[max(0, start_index - 10):start_index]): unit = '万元' elif re.search('([单报标限总造]价款?|金额|租金|(中标|成交|合同|承租|投资|控制|拦标))?[价额]|价格|预算(金额)?|(监理|设计|勘察)(服务)?费|成本|报价(单位))(小写)?[::为]*-?$', text_beforeMoney.strip()) and re.search('^0|1[3|4|5|6|7|8|9]\d{9}', entity_text) == None: if re.search('^[\d,,.]+$', entity_text) and float(re.sub('[,,]', '', entity_text)) < 500 and re.search('万元', sentence_text): unit = '万元' elif re.search('^[\d,,.]+$', entity_text) and float(re.sub('[,,]', '', entity_text)) < 100 and re.search('单位:%|费率|下浮率|[%%‰折]|优惠率', sentence_text): continue elif re.search('^\d{1,3}\.\d{4,6}$', entity_text) and re.search('0000$', entity_text) == None: unit = '万元' else: unit = '元' elif re.search('(^\d{,3}(,?\d{3})+(\.\d{2,7},?)$)|(^\d{,3}(,\d{3})+,?$)', entity_text): unit = '元' else: continue elif unit == '万元': if end_index < len(sentence_text) and sentence_text[end_index] == '元' and re.search('\d$', entity_text): unit = '元' elif re.search('^[5-9]\d{6,}\.\d{2}$', entity_text): unit = '元' if unit.find("万") >= 0 and entity_text.find("万") >= 0: unit = "元" if re.search('.*万元万元', entity_text): entity_text = entity_text.replace('万元万元', '万元') else: if filter_unit: continue entity_text = re.sub("[^0-9.零壹贰叁肆伍陆柒捌玖拾佰仟萬億圆十百千万亿元角分]", "", entity_text) if re.search('总投资|投资总额|总预算|总概算|(投资|招标|资金|存放|操作|融资)规模|批复概算|投资额|总规模|工程造价|总金额', sentence_text[max(0, _match.span()[0] - 10):_match.span()[1]]): notes = '总投资' elif re.search('投资|概算|建安费|其他费用|基本预备费', sentence_text[max(0, _match.span()[0] - 8):_match.span()[1]]): notes = '投资' elif (re.search('保证金', sentence_text[max(0, _match.span()[0] - 5):_match.span()[1]]) or re.search('保证金的?(缴纳)?(金额|金\?|额|\?)?[\((]*(万?元|为?人民币|大写|调整|变更|已?修改|更改|更正)?[\))]*[::为]', sentence_text[max(0, _match.span()[0] - 10):_match.span()[1]]) or re.search('保证金由[\d.,]+.{,3}(变更|修改|更改|更正|调整?)为', sentence_text[max(0, _match.span()[0] - 15):_match.span()[1]])): notes = '保证金' elif re.search('成本(警戒|预警)(线|价|值)[^0-9元]{,10}', sentence_text[max(0, _match.span()[0] - 10):_match.span()[0]]): notes = '成本警戒线' elif re.search('(监理|设计|勘察)(服务)?费(报价)?[约为:]|服务金额', sentence_text[_match.span()[0]:_match.span()[1]]) and re.search('缴纳', sentence_text[max(0, _match.span()[0] - 5):_match.span()[1]]) == None: notes = '招标或中标金额' elif re.search('单价|总金额', sentence_text[_match.span()[0]:_match.span()[1]]): notes = '单价' elif re.search('元[/每]', sentence_text[_match.start():_match.end() + 2]): notes = '单价' elif re.search('单价', sentence_text[max(0, _match.start() - 3):_match.start()]) and re.search('单价:\d+,', sentence_text[:_match.start()]) == None: notes = '单价' elif re.search('[零壹贰叁肆伍陆柒捌玖拾佰仟萬億圆]', entity_text) != None: notes = '大写' if entity_text[0] == "拾": entity_text = "壹" + entity_text if len(unit) > 0: if unit.find('万') >= 0 and len(entity_text.split('.')[0]) >= 8: entity_text = str( getUnifyMoney(entity_text) * getMultipleFactor(re.sub("[美日欧]", "", unit)[0]) / 10000) unit = '元' else: entity_text = str(getUnifyMoney(entity_text) * getMultipleFactor(re.sub("[美日欧]", "", unit)[0])) else: if entity_text.find('万') >= 0 and entity_text.split('.')[0].isdigit() and len( entity_text.split('.')[0]) >= 8: entity_text = str(getUnifyMoney(entity_text) / 10000) else: entity_text = str(getUnifyMoney(entity_text)) if science and re.search('^E-?\d+$', science): entity_text = str(Decimal(entity_text + science)) if Decimal(entity_text + science) > 100 and Decimal( entity_text + science) < 10000000000 else entity_text if float(entity_text) > 100000000000: continue if notSure and unit == "" and float(entity_text) > 100 * 10000: continue if re.search('[%%‰折]|费率|下浮率', text_beforeMoney) and float(entity_text) < 1000: continue if notes == '单价' and float(entity_text) > 1000000 and re.search('单价((万元))?:', sentence_text[max(0, _match.start() - 3):_match.end()]) == None: notes = "" if float(entity_text) < 100 and notes != '单价': notes = '单价' money_list.append((entity_text, start_index, end_index, unit, notes)) return money_list, found_yeji # ============================================================ # 其他纯函数 # ============================================================ def combine(list1, list2): ''' @summary:将两个list中的字符串两两拼接 @param: list1:字符串list list2:字符串list @return:拼接结果list ''' result = [] for item1 in list1: for item2 in list2: result.append(str(item1) + str(item2)) return result def fitDataByRule(data): # 根据规则补全编号或名称两边的符号 symbol_dict = {"(": ")", "(": ")", "[": "]", "【": "】", ")": "(", ")": "(", "]": "[", "】": "【"} leftSymbol_pattern = re.compile("[\((\[【]") rightSymbol_pattern = re.compile("[\))\]】]") leftfinds = re.findall(leftSymbol_pattern, data) rightfinds = re.findall(rightSymbol_pattern, data) result = data if len(leftfinds) + len(rightfinds) == 0: return data elif len(leftfinds) == len(rightfinds): return data elif abs(len(leftfinds) - len(rightfinds)) == 1: if len(leftfinds) > len(rightfinds): if symbol_dict.get(data[0]) is not None: result = data[1:] else: result = data + symbol_dict.get(leftfinds[0]) else: if symbol_dict.get(data[-1]) is not None: result = data[:-1] else: result = symbol_dict.get(rightfinds[0]) + data result = re.sub("[。]", "", result) return result def clean_company(entity_text): ''' 清洗公司名称 :param entity_text: :return: ''' entity_text = re.sub('\s', '', entity_text) if re.search('^(\d{4}年)?[\-\d月日份]*\w{2,3}分公司$|^\w{,6}某(部|医院)$|空间布局$', entity_text): return '' elif re.match('xx|XX', entity_text): return '' elif re.match('\.?(rar|zip|pdf|df|doc|docx|xls|xlsx|jpg|png)', entity_text): entity_text = re.sub('\.?(rar|zip|pdf|df|doc|docx|xls|xlsx|jpg|png)', '', entity_text) elif re.match('(\d+)|\d+\.|\s| ', entity_text): entity_text = re.sub('(\d+)|\d+\.|\s| ', '', entity_text) elif re.match( '((\d{4}[年-])[\-\d:\s元月日份]*|\d{1,2}月[\d日.-]*(日?常?计划)?|\d{1,2}[.-]?|[A-Za-z](包|标段?)?|[a-zA-Z0-9]+-[a-zA-Z0-9-]*|[a-zA-Z]{1,2}|[①②③④⑤⑥⑦⑧⑨⑩]|\s|title\=|【[a-zA-Z0-9]+】|[^\w])[\u4e00-\u9fa5]+', entity_text): filter = re.match( '((\d{4}[年-])[\-\d:\s元月日份]*|\d{1,2}月[\d日.-]*(日?常?计划)?|\d{1,2}[.-]?|[A-Za-z](包|标段?)?|[a-zA-Z0-9]+-[a-zA-Z0-9-]*|[a-zA-Z]{1,2}|[①②③④⑤⑥⑦⑧⑨⑩]|\s|title\=|【[a-zA-Z0-9]+】|[^\w])[\u4e00-\u9fa5]+', entity_text).group(1) entity_text = entity_text.replace(filter, '') elif re.search('\]|\[|\]|[【】{}「?:∶〔·.\'#~_ΓΙεⅠ\丨]', entity_text): entity_text = re.sub('\]|\[|\]|[【】「?:∶〔·.\'#~_ΓΙεⅠ\丨]', '', entity_text) if len(re.sub('(项目|分|有限)?公司|集团|制造部|中心|医院|学校|大学|中学|小学|幼儿园', '', entity_text)) < 2: return '' return entity_text def cut_repeat_name(s): ''' 公司连续重复名称去重 :param s: :return: ''' if len(s) >= 8: n = s.count(s[-4:]) id = s.find(s[-4:]) + 4 sub_s = s[:id] if n >= 2 and s == sub_s * n: s = sub_s return s def is_all_winner(title): ''' 是否提取所有投标人作为中标人,存管类不分排名都作中标人;入围类按排名,无排名都做中标人 :param title: 标题 :return: ''' if re.search('(资金|公款|存款)?竞争性存[放款]|(资金|公款|存款)存放|存放银行|存款服务|国库现金管理', title): return 1 elif re.search('招募|入围|框架(协议)?采购|(单位|商|机构)入库|入库供应商|集中采购', title): return 2 return False def is_deposit_project(title, name, requirement): ''' 通过正则判断项目是否为银行存款类项目 :param title: 标题 :param name: 项目名称 :param requirement: 采购内容 :return: ''' if re.search('(资金|公款|存款)?竞争性存[放款]|(资金|公款|存款)((.{2,10}))?存放|存放银行|存款(服务|业务|项目)|国库现金管理|存款账户开户|(管理|存款|合作)(定点|专户)?银行|贷款合作银行|资金监管账户|开户银行项目|专户开户银行|银行专户选择|定期存[款放]|专项债券?专用账户', title + name + requirement): return True return False package_number_pattern = re.compile( '((施工|监理|监测|勘察|设计|劳务)(标段)?:?第?([一二三四五六七八九十]{1,3}|[ⅠⅡⅢⅣⅤⅥⅦ]{1,3}|[a-zA-Z0-9]{1,9}\-?[a-zA-Z0-9-]{,9})[分子]?(标[段包项]?|包[组件标]?|合同[包段]))\ |(([a-zA-Z]包[:()]?)?第?([一二三四五六七八九十]{1,3}|[ⅠⅡⅢⅣⅤⅥⅦ]{1,3}|[a-zA-Z0-9]{1,9}\-?[a-zA-Z0-9-]{,9})[分子]?(标[段包项]?|合同[包段]))\ |(([,;。、:(]|第)?([一二三四五六七八九十]{1,3}|[ⅠⅡⅢⅣⅤⅥⅦ]{1,3}|[a-zA-Z0-9]{1,9}\-?[a-zA-Z0-9-]{,9})[分子]?(标[段包项]?|包[组件标]?|合同[包段]))\ |((标[段包项]|标段(包)|包[组件标]|[标分子(]包)(\[|【)?:?([一二三四五六七八九十]{1,3}|[ⅠⅡⅢⅣⅤⅥⅦ]{1,3}|[a-zA-Z0-9]{1,9}\-?[a-zA-Z0-9-]{,9}))\ |([,;。、:(]|^)(标的?|(招标|采购)?项目|子项目?|采购包()(\[|【)?:?([一二三四五六七八九十]+|[0-9]{1,9})\ |((([标分子(]|合同|项目|采购)包|[,。]标的|子项目|[分子]标|标[段包项]|包[组件标]?)编?号[::]?[a-zA-Z0-9一二三四五六七八九十ⅠⅡⅢⅣⅤⅥⅦ]{1,9}[a-zA-Z0-9一二三四五六七八九十ⅠⅡⅢⅣⅤⅥⅦ-]{0,9})\ |[,;。、:(]?(合同|分|子)?包:?([一二三四五六七八九十]{1,3}|[ⅠⅡⅢⅣⅤⅥⅦ]{1,3}|[a-zA-Z0-9]{1,9}\-?[a-zA-Z0-9-]{,9})') filter_package_pattern = 'CA标|(每个?|所有|相关|个|各|不分)[分子]?(标[段包项]?|包[组件标]?|合同包)|(质量|责任)三包|包[/每]|标段(划分|范围)|(承|压缩|软|皮|书|挂)包\ |标[识注签贴配]|[商油]标号|第X包|第[一二三四五六七八九十]+至[一二三四五六七八九十]+(标[段包项]?|包[组件标]?|合同[包段])\ |\.(docx|doc|pdf|xlsx|xls|jpg)|[一二三四五]次|五金|\d+[年月]|[\d.,]+万?元|\d+\.\d+' def find_package(content): ''' 通过正则找包和标段号 :param content: :return: ''' packages = [] content = content.replace('号,', '号:').replace(':', ':').replace('(', '(').replace(')', ')') content = re.sub('[一二三四五六七八九十\d](标[段包项]|包[组件标])编号', ' 标段编号', content) content = re.sub('打包|标段名称|标段(包)?名称|承,?包|共\d+包|标[厅室]', ' ', content) for it in re.finditer(filter_package_pattern, content): content = content.replace(it.group(0), ' ' * len(it.group(0))) for iter in re.finditer(package_number_pattern, content): if re.search('(业绩|信誉要求):|业绩(如下)?\d*[、:]', content[:iter.start()]): continue if re.match('\d', iter.group(0)) and re.search('\d\.$', content[:iter.start()]): continue if re.search('[承每书/]包|XX|xx', iter.group(0)) or re.search('\d包[/每]\w|一包[0-9一二三四五六七八九十]+', content[ iter.start():iter.end() + 3]) or re.search( '[a-zA-Z0-9一二三四五六七八九十ⅠⅡⅢⅣⅤⅥⅦ-]{6,}', iter.group(0)): continue elif iter.end() + 2 < len(content) and re.search('标的物|包装|划分|标(书|准|志|记|识|签|贴|帜|本|底|价|量)', content[iter.start():iter.end() + 2]): continue elif re.search('同一(标段?|包)', content[max(0, iter.start() - 2):iter.end()]): continue elif re.search('三包', content[max(0, iter.start() - 2):iter.end()]) and re.search('第三包', content[max(0, iter.start() - 2):iter.end()]) == None: continue elif re.search('[1-9]\d{2,}$|\d{4,}|^[1-9]\d{2,}|合同包[A-Za-z]{2,}', iter.group(0)): continue elif re.search('单位:包|1包\d|[张箱]|数量:', content[max(0, iter.start() - 3): iter.end() + 2]): continue elif re.search('(\d+|一)包', iter.group(0)) and re.search('(品牌|规格|参数)\w{,2}:', content[:iter.start()]) and re.search('标[段包]|包号', content[:iter.start()]) == None: continue elif iter.group(0) == '劳务分包': continue elif re.search('^包\d$', iter.group(0)) and re.search('[个只台件套次]', content[iter.end():iter.end() + 5]): continue packages.append(iter) return packages