import re from typing import List, Dict,Tuple import re from typing import List, Tuple def extract_repairproject_tables_html1(content: str) -> Tuple[List[str], str]: """ 提取所有维修表格,每个表格从特定表头开始,到下一个表头出现之前结束。 支持两种表头格式(含或不含 rowspan/colspan属性)。 :param content: 原始 HTML/文本内容 :return: (tables: List[str], cleaned_content: str) - tables: 所有提取到的表格块 - cleaned_content: 仅保留第一个表格块,其余已删除 """ # 定义表头的文本核心部分(不包含属性) base_header = '' # 转义特殊字符,但为 rowspan/colspan 留出匹配空间 # 我们需要匹配:
维修项目编号组成编码名称维修项目维修间隔期 等情况 # 这里使用一个更灵活的正则:匹配 ' 字符(非贪婪),直到 > # 这种写法比硬编码两种情况更健壮,能兼容属性顺序变化或额外空格 escaped_base = re.escape('') # 构建针对每个单元格的正则片段 # 匹配 "维修项目编号" 前的 cell_pattern = r']*>' header_pattern = ( escaped_base + cell_pattern + re.escape('维修项目编号') + r'' + cell_pattern + re.escape('组成编码') + r'' + cell_pattern + re.escape('名称') + r'' + cell_pattern + re.escape('维修项目') + r'' + cell_pattern + re.escape('维修间隔期') + r'' ) # 编译正则,用于查找表格开始位置 pattern = re.compile(header_pattern, re.DOTALL) matches = list(pattern.finditer(content)) tables = [] if not matches: return tables, content # 提取所有表格块 for i, match in enumerate(matches): start_pos = match.start() # 确定结束位置:下一个表格的开始,或者文本末尾 if i < len(matches) - 1: end_pos = matches[i + 1].start() else: end_pos = len(content) table_content = content[start_pos:end_pos] tables.append(table_content) # 构建 cleaned_content:只保留第一个表格块 # 即截断到第二个表格开始的位置(如果有的话) if len(matches) > 1: cleaned_content = content[:matches[1].start()] else: cleaned_content = content # 如果只有一个表格,原内容即为清理后内容(因为后面没有表格了) return tables, cleaned_content #提取所有维修项目表的html源码 def extract_repairproject_tables_html(content: str) -> tuple[list[str], str]: """ 提取所有维修表格,每个表格从 start_literal 开始,到下一个 start_literal 出现之前结束。 :param content: 原始 HTML/文本内容 :return: (tables: List[str], cleaned_content: str) - tables: 所有提取到的表格块(包含 start_literal 开头的内容) - cleaned_content: 原文中仅保留第一个表格块,其余已删除 """ # 构建起始模式 start_literal = ( '
' '' '' '' '' '' ) or ( '
维修项目编号组成编码名称维修项目维修间隔期
' '' '' '' '' '' ) # 使用 finditer 查找所有匹配的起始位置 pattern = re.compile(re.escape(start_literal), re.DOTALL) matches = list(pattern.finditer(content)) tables = [] if not matches: # 没有找到表格,返回空列表和原内容 return tables, content # 提取所有表格 for i, match in enumerate(matches): start_pos = match.start() # 如果不是最后一个表格,结束位置是下一个表格的开始位置 if i < len(matches) - 1: end_pos = matches[i + 1].start() else: # 如果是最后一个表格,结束位置是文本末尾 end_pos = len(content) table_content = content[start_pos:end_pos] tables.append(table_content) # 构建 cleaned_content:只保留第一个表格 first_table_end = len(content) if len(matches) > 1: first_table_end = matches[1].start() cleaned_content = content[:first_table_end] return tables, cleaned_content def extract_operation_tables_html(content: str) -> Tuple[List[str], str]: """ 提取所有以“项目编号 + 操作项目”为表头的操作表格。 每个表格的范围:从匹配的 start_pattern 开始,到下一个 start_pattern 之前(或文档结尾)。 同时生成切片:「全文开头前两行 + 前100字上下文 + 表格内容」。 cleaned_content 中仅保留第一个表格(原始形式),其余删除。 """ # 提取全文开头的前两行(保留换行符) lines = content.splitlines(keepends=True) first_two_lines = ''.join(lines[:2]) if len(lines) >= 2 else ''.join(lines) # 起始模式(修复版) start_pattern = r''' ]*> \s* ]*> \s* ]*> \s* 项目编号 \s* \s* ]*> \s* 操作项目 \s* \s* \s* ]*> ''' # 查找所有起始位置 matches = list(re.finditer(start_pattern, content, flags=re.DOTALL | re.VERBOSE)) if not matches: return [], content # 构建每个表格的完整范围:[start_i, start_{i+1}) 或 [start_i, end) table_ranges = [] for i in range(len(matches)): start_pos = matches[i].start() if i + 1 < len(matches): end_pos = matches[i + 1].start() else: end_pos = len(content) table_ranges.append((start_pos, end_pos)) # 提取每个表格的 HTML 内容 tables_html = [content[start:end] for start, end in table_ranges] # 构建 slices: 前两行 + 前100字 + 表格HTML slices = [] for (start, _), html in zip(table_ranges, tables_html): prefix_start = max(0, start - 100) local_prefix = content[prefix_start:start] full_slice = first_two_lines + local_prefix + html # 修复标题格式(根据你的需求) full_slice = full_slice.replace("# 操作步骤..", "操作步骤..") full_slice = full_slice.replace("操作步骤..", "# 操作步骤") slices.append(full_slice) # 构建 cleaned_content:只保留第一个表格,其余删除 cleaned = content if len(table_ranges) > 1: # 从后往前删除第2个及之后的表格 for start, end in reversed(table_ranges[1:]): cleaned = cleaned[:start] + cleaned[end:] return slices, cleaned def extract_cases(md_content): """ 从 Markdown 内容中提取所有以 '# 数字.' 开头的案例。 每个案例从 '# X.XXX' 开始,到下一个 '# Y.YYY' 或末尾结束。 """ # 正则表达式匹配以 "# 数字." 开头的标题 pattern = r'#\s*(\d+)\.\s*(.*?)\n(.*?)(?=(?:#\s*\d+\.\s*.*?))|$' # 使用 re.DOTALL 让 . 匹配换行符 matches = re.findall(pattern, md_content, re.DOTALL | re.MULTILINE) cases = [] for match in matches: case_number = match[0] # 如 "1", "2" title_text = match[1] # 如 "XXX 型号 A 设备 A 故障" content = match[2] # 该案例的具体内容 # 组装完整案例 case_text = f"# {case_number}.{title_text}\n{content}" cases.append(case_text) return cases def extract_fault_sections(text): """ 从文本中提取以“故障名称:”开头的段落,去除最后一行。 """ pattern = r'故障名称:[\s\S]*?(?=故障名称:|$)' matches = re.findall(pattern, text, re.IGNORECASE) results = [] for match in matches: match = match.strip() if match: # 按行分割,去除最后一行,再合并 lines = match.split('\n') if len(lines) > 1: # 去除最后一行 modified_content = '\n'.join(lines[:-1]) results.append(modified_content.strip()) # 如果只有一行,可以根据需要决定是否保留 return results def extract_by_unit(text): """ 从文本中提取以“单位:”开头的段落,去除最后一行。 """ pattern = r'(?=单位:)[\s\S]*?(?=单位:|$)' matches = re.findall(pattern, text, re.IGNORECASE) results = [] for match in matches: match = match.strip() if match: # 按行分割,去除最后一行,再合并 lines = match.split('\n') if len(lines) > 1: # 去除最后一行 modified_content = '\n'.join(lines[:-1]) results.append(modified_content.strip()) return results def extract_tables(content: str, start_len: int = 300,end_len:int = 50) -> Tuple[List[str], List[str], str]: """ 提取所有 HTML 表格及其前后上下文(各最多 context_len 个字符),并从原文中移除表格及上下文。 :param content: 原始 HTML/文本内容 :param context_len: 上下文长度(默认 100 字符) :return: (tables: List[str], contexts: List[str], cleaned_content: str) 其中 contexts[i] 是 tables[i] 对应的 "<前文> +
组成编码名称维修项目维修间隔期
... + <后文>" 字符串 """ # 构造正则:捕获最多 context_len 个任意字符(非贪婪)+ table 标签 + 最多 context_len 个任意字符 pattern = rf'(.{{0,{start_len}}}?)(]*>.*?
)(.{{0,{end_len}}})' matches = re.finditer(pattern, content, flags=re.DOTALL | re.IGNORECASE) tables = [] contexts = [] cleaned_content = content # 从后往前替换,避免位置偏移问题 for match in reversed(list(matches)): pre_context = match.group(1) table = match.group(2) post_context = match.group(3) full_context = pre_context + table + post_context contexts.append(full_context) tables.append(table) # 从原文中删除整个上下文片段(包括前后文和表格) cleaned_content = cleaned_content[:match.start()] + cleaned_content[match.end():] # 因为是从后往前删的,所以 tables 和 contexts 顺序是反的,需要反转回来 tables.reverse() contexts.reverse() return tables, contexts, cleaned_content def extract_guzhang_tables(text: str, context_length: int = 100) -> List[str]: header1 = '' header2 = '
序号故障现象故障原因维修项目
' header1_esc = re.escape(header1) header2_esc = re.escape(header2) # 匹配:前文(最多 context_length) + 表头 + 表格内容 + 结尾 pattern = re.compile( rf'(.{{0,{context_length}}}?)(?:{header1_esc}|{header2_esc})([\s\S]*?
序号故障现象故障原因维修项目
)', re.DOTALL ) results = [] for match in pattern.finditer(text): before = match.group(1) # 前文 body_and_end = match.group(2) # 表格内容 + 结尾 # 判断是哪个表头(从原文中取) pos = match.start(2) # 表头开始位置 if text[pos:pos + len(header1)] == header1: table_code = header1 + body_and_end else: table_code = header2 + body_and_end result_str = before + table_code results.append(result_str) return results def extract_content_slices(text, keywords, context_len=100): # 转义关键词用于正则 escaped = [re.escape(kw) for kw in keywords] pattern = r'^\s*#?\s*(' + '|'.join(escaped) + ')' compiled = re.compile(pattern, re.MULTILINE) # 注意:这里加了 MULTILINE 以确保 ^ 能匹配每行开头 lines = text.splitlines(keepends=True) # keepends=True 保留换行符,用于精确计算位置 if not lines: return [] contents = [] current_content = [] in_section = False section_start_idx = 0 # 记录当前段落在原文中的字符索引位置 # 遍历每一行 for line in lines: if compiled.match(line): # 遇到新关键词 if in_section and current_content: # 1. 拼接当前内容 content_str = ''.join(current_content) # 2. 获取前 context_len 个字符的上下文 # 确保不越界 start_context = max(0, section_start_idx - context_len) prefix_context = text[start_context:section_start_idx] # 3. 组合结果:上下文 + 内容 final_slice = prefix_context + content_str contents.append(final_slice) # 重置,准备下一段 current_content = [] # 开启新段落 in_section = True # 更新当前段落的起始位置(即这一行在原文中的位置) # 这里简化处理:通过查找来定位(因为直接计算索引在复杂文本中容易出错) # find 从 section_start_idx 开始找,避免重复匹配 temp_pos = text.find(line, section_start_idx) if temp_pos != -1: section_start_idx = temp_pos elif in_section: current_content.append(line) # 处理最后一个段落 if in_section and current_content: content_str = ''.join(current_content) start_context = max(0, section_start_idx - context_len) prefix_context = text[start_context:section_start_idx] final_slice = prefix_context + content_str contents.append(final_slice) return contents