# -*- coding: utf-8 -*- from abc import ABC, abstractmethod from typing import Dict, Optional, Any import re # from lxml import etree def table2list(table): """ 解析HTML表格,处理rowspan/colspan合并单元格,返回完整的二维列表 Args: table: lxml.etree._Element 表格对象 Returns: 二维列表表示的表格数据 """ def _check_cell_validity(i: int, j: int) -> bool: """检查单元格(i, j)是否可以放入_output""" if i >= len(_output): return True if j >= len(_output[i]): return True if _output[i][j] == "#$#": return True return False def _insert(i: int, j: int, height: int, width: int, val: str): """将值val插入到以(i, j)为起点、跨height行width列的矩形区域""" for ii in range(i, i + height): for jj in range(j, j + width): _insert_cell(ii, jj, val) def _insert_cell(i: int, j: int, val: str): """在特定位置(i, j)插入值,自动扩展_output矩阵""" while i >= len(_output): _output.append([]) while j >= len(_output[i]): _output[i].append("#$#") if _output[i][j] == "#$#": _output[i][j] = val # 初始化输出矩阵 _output = [] row_ind = 0 col_ind = 0 # 使用lxml.etree解析HTML try: # parser = etree.HTMLParser(remove_blank_text=True, remove_comments=True) # tree = etree.fromstring(html_content, parser) # # # 获取第一个表格(使用XPath) # tables = tree.xpath('//table') # if not tables: # return [] # table = tables[0] # 获取所有行 rows = table.xpath('./tr | ./tbody/tr') for row in rows: # 记录最小row_span,确定需要跳过多少行 smallest_row_span = 1 # 获取所有单元格(td和th) cells = row.xpath('./td | ./th') for cell in cells: # 处理rowspan row_span_attr = cell.get('rowspan') if row_span_attr and row_span_attr.isdigit(): row_span = int(row_span_attr) if row_span == 0: # 修复rowspan为0的情况 row_span = 1 else: row_span = 1 # 更新最小row_span smallest_row_span = min(smallest_row_span, row_span) # 处理colspan col_span_attr = cell.get('colspan') if col_span_attr and col_span_attr.isdigit(): col_span = int(col_span_attr) if col_span > 20: # 限制过大的colspan col_span = 20 elif col_span == 0: # 修复colspan为0的情况 col_span = 1 else: col_span = 1 # 找到合适的列索引 while True: if _check_cell_validity(row_ind, col_ind): break col_ind += 1 # 提取单元格文本 text = ''.join(cell.itertext()).strip() # 获取所有文本内容 # 处理省略号情况:如果有title属性且文本以...结尾,使用title内容 title_attr = cell.get('title') if (title_attr and text.replace(' ', '').endswith('...') and title_attr.replace(' ', '').startswith(text.replace(' ', '')[:-3])): text = title_attr text = re.sub(r'\s+', '', text) # 合并多余空格 # 插入值到_output _insert(row_ind, col_ind, row_span, col_span, text.replace('(', '(').replace(')', ')')) # 更新列索引 col_ind += col_span # 更新行索引 row_ind += smallest_row_span col_ind = 0 # except etree.ParseError as e: # print(f"HTML解析错误: {e}") # return [] except Exception as e: print(f"处理表格时发生错误: {e}") return [] return _output class AbstractTemplate(ABC): """ 抽象模板基类(遵循文档中“模板调用”的统一接口类设计) 定义所有模板必须实现的通用方法,约束模板结构与调用逻辑 """ def __init__(self, template_id: str, priority: int = 1): """ 初始化模板核心属性(对应文档“模板存在方式”的字段定义) :param template_id: 模板唯一标识ID :param priority: 模板优先级(数值越大优先级越高,用于结果融合) """ self.template_id = template_id self.priority = priority @abstractmethod def check_call_timing(self, preprocessed_data: Dict[str, Any]) -> bool: """ 抽象方法:判断模板调用时机(文档核心要求,避免无效模板遍历) 需子类实现具体维度的调用条件判断 :param preprocessed_data: 要素提取预处理后的结果(含输入参数) :return: True=满足调用条件,False=不满足 """ pass @abstractmethod def extract(self, preprocessed_data: Dict[str, Any]) -> Dict[str, Any]: """ 抽象方法:执行模板提取逻辑(文档核心功能) 需子类实现具体的提取规则(关键词定位/分隔符拆分/正则等) :param preprocessed_data: 要素提取预处理后的结果(需包含input_params指定的参数) :return: 提取结果字典(键为要素名,值为提取内容;空字典表示提取失败) """ pass def get_template_info(self) -> Dict[str, Any]: """ 通用方法:获取模板基本信息(映射文档“模板存在方式”的表格结构) :return: 模板信息字典 """ return { "id": self.template_id, # "维度": self.dimension, # "代码路径": self.code_path, # "输入参数": self.input_params, # "输出格式": self.output_format, # "隶属id": self.parent_id, "优先级": self.priority }