| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188 |
- # -*- coding: utf-8 -*-
- """Predictor 共享状态与辅助函数。
- 按 ARCHITECTURE.md Phase 5 拆分建议,从 ``interface/predictor.py`` 迁出
- 模块级共享变量和辅助函数,供 ``predictors/`` 下各模块引用。
- 原位置:``interface/predictor.py`` 模块级代码:
- - ``sess_config`` — TF session 配置(当前为 None)
- - ``agency_set`` — 代理机构集合(pickle 加载)
- - ``header_set`` — 表头集合(pickle 加载)
- - ``is_agency`` / ``get_td_companys`` / ``get_role`` — 角色识别辅助函数
- ``interface/predictor.py`` 仍 re-export 以上全部名称,老 import 不受影响。
- """
- from __future__ import absolute_import
- import os
- import re
- import pickle
- # 显式 import,不使用 `from common.Utils import *`
- from BiddingKG.dl.common.logging import log
- from BiddingKG.dl.common.nerUtils import getNers
- from BiddingKG.dl.common.context_utils import clean_company
- __all__ = [
- "INTERFACE_DIR",
- "sess_config",
- "agency_set",
- "header_set",
- "is_agency",
- "get_td_companys",
- "get_role",
- ]
- #: ``interface/`` 目录绝对路径,用于定位模型文件、pickle 等。
- #: ``predictors/`` 与 ``interface/`` 同级,所以向上一级再进 ``interface/``。
- INTERFACE_DIR = os.path.normpath(
- os.path.join(os.path.dirname(os.path.abspath(__file__)), "..", "interface")
- )
- # ============================================================
- # TF session 配置(原 predictor.py 第 40-45 行)
- # ============================================================
- # 原 predictor.py 中 sess_config 先用 tf.ConfigProto 创建再被覆盖为 None。
- # 这里直接设为 None,保持与当前运行时行为一致。
- sess_config = None
- # ============================================================
- # 代理机构集合 / 表头集合(原 predictor.py 第 47-52 行)
- # ============================================================
- _agency_set_path = os.path.join(INTERFACE_DIR, "agency_set.pkl")
- with open(_agency_set_path, "rb") as _f:
- agency_set = pickle.load(_f)
- _header_set_path = os.path.join(INTERFACE_DIR, "header_set.pkl")
- with open(_header_set_path, "rb") as _f:
- header_set = pickle.load(_f)
- # ============================================================
- # 角色识别辅助函数(原 predictor.py 第 54-149 行)
- # ============================================================
- def is_agency(entity_text):
- if re.search(
- "(招投?标|采购|代理|咨询|管理|物资|事务所?|顾问|监理|拍卖)[()\w]{,4}(有限)?(责任)?公司|(采购|招投?标|交易|代理|咨询)[()\w]{,4}(中心|服务所)|法院$",
- entity_text,
- ) or entity_text in agency_set:
- return True
- return False
- def get_td_companys(text, nlp_enterprise):
- """获取字符串 text 角色实体(表格场景)。
- :param text: 待获取实体字符串
- :param nlp_enterprise: 公告中的角色实体列表
- :return: (leader, joint) 主报名人和联合体
- """
- text = re.sub(
- "主报名人:|联合报名人:|联合体:|联合体(成员|单位)[12345一二三四五]?:|(联合体)?成员单位[12345一二三四五]?:|特殊普通合伙:|[((【]([主成]|联合体)[))】]|((联合体)?(牵头|成员)(方|人|单位))",
- ",",
- text,
- )
- text = re.sub("\s", "", text) # 修复 370835008 表格中实体中间有\n
- text = re.sub(
- "[一二三四五六七八九十]+标段[::]|标段[一二三四五六七八九十]+[::]|第[一二三四五六七八九十]+名[::]",
- "",
- text,
- ) # 2024/4/22 修复 372839375 三标段:宁夏一山科技有限公司
- text = re.sub("1[3-9]\d{9}|\d{3}-\d{8}|\d{4}-\d{7}", "", text) # 2024/4/23 去除电话
- text = re.sub(r"([\w()]{5,20}有限)公?$", r"\1公司", text)
- joint = ""
- leader = ""
- if text in nlp_enterprise:
- leader = text
- elif re.sub("(个体工商户)$", "", text) in nlp_enterprise or re.match(
- "[\w()]{6,25}(个体工商户)$", text
- ):
- leader = text
- elif re.match(
- "^\w{3,25}(海关|殡仪馆|店|村委会|纪念馆|监狱|管教所|修养所|社区|农场|林场|羊场|猪场|石场|养殖场|饲养场|经营部|经销部|商贸部|经销处|商行)$",
- text,
- ):
- leader = text
- elif len(text) < 4:
- leader = ""
- elif len(nlp_enterprise) > 0:
- roles = []
- try:
- for it in re.finditer("|".join(nlp_enterprise), text):
- roles.append(it.group(0))
- except Exception as e:
- print("get_td_companys 异常:", e)
- if roles and len("".join(roles)) * 2 > len(text):
- leader = roles[0]
- if len(roles) > 1:
- joint = ",".join(set(roles))
- if leader == "" and len(text) > 4:
- ners = getNers([text], useselffool=True)
- roles = []
- if ners:
- for ner in ners[0]:
- entity_text = ner[3]
- if text[ner[1]:] == "(个体工商户)": # 修复 肇州县一胜建材经销处(个体工商户)这种类型
- entity_text = ner[3] + "(个体工商户)"
- if ner[2] in ["org", "company"]:
- roles.append(entity_text)
- elif ner[2] in ["location"] and re.search(
- "^\w{3,10}(海关|殡仪馆|店|村委会|纪念馆|监狱|管教所|修养所|社区|农场|林场|羊场|猪场|石场|经营部|经销处)$",
- ner[3],
- ):
- roles.append(entity_text)
- if roles and len("".join(roles)) > len(text) * 0.8:
- roles = [clean_company(company) for company in roles]
- roles = [it for it in roles if len(it) > 3]
- if roles:
- leader = roles[0]
- if len(roles) > 1:
- joint = ",".join(set(roles))
- if leader:
- leader = clean_company(leader)
- return leader, joint
- def get_role(text, nlp_enterprise):
- """获取字符串 text 角色实体。
- :param text: 待获取实体字符串
- :param nlp_enterprise: 公告中的角色实体列表
- :return: 角色实体文本
- """
- text = re.sub(
- "主报名人:|联合报名人:|联合体:|联合体(成员|单位)[12345一二三四五]?:|(联合体)?成员单位[12345一二三四五]?:|特殊普通合伙:|[((][主成][))]",
- ",",
- text,
- )
- text = re.sub("\s", "", text) # 修复 370835008 表格中实体中间有\n
- text = re.sub(
- "[一二三四五六七八九十]+标段[::]|标段[一二三四五六七八九十]+[::]|第[一二三四五六七八九十]+名[::]",
- "",
- text,
- ) # 2024/4/22 修复 372839375 三标段:宁夏一山科技有限公司
- text = re.sub("1[3-9]\d{9}|\d{3}-\d{8}|\d{4}-\d{7}", "", text) # 2024/4/23 去除电话
- if text in nlp_enterprise:
- return text
- if len(text) > 50 or len(text) < 4:
- return ""
- ners = getNers([text], useselffool=True)
- roles = []
- if ners:
- for ner in ners[0]:
- entity_text = ner[3]
- if text[ner[1]:] == "(个体工商户)": # 修复 肇州县一胜建材经销处(个体工商户)这种类型
- entity_text = ner[3] + "(个体工商户)"
- if ner[2] in ["org", "company"]:
- roles.append(entity_text)
- elif ner[2] in ["location"] and re.search(
- "^\w{3,10}(海关|殡仪馆|店|村委会|纪念馆|监狱|管教所|修养所|社区|农场|林场|羊场|猪场|石场|经营部|经销处)$",
- ner[3],
- ):
- roles.append(entity_text)
- if roles and len("".join(roles)) > len(text) * 0.8:
- entity = clean_company(roles[0])
- return entity
- else:
- return ""
|