| 维修项目编号 | 组成编码 | 名称 | 维修项目 | 维修间隔期 | '
# 转义特殊字符,但为 rowspan/colspan 留出匹配空间
# 我们需要匹配: 或 | 或 | 等情况
# 这里使用一个更灵活的正则:匹配 | ' 字符(非贪婪),直到 >
# 这种写法比硬编码两种情况更健壮,能兼容属性顺序变化或额外空格
escaped_base = re.escape('')
# 构建针对每个单元格的正则片段
# 匹配 "维修项目编号" 前的
cell_pattern = r'| ]*>'
header_pattern = (
escaped_base +
cell_pattern + re.escape('维修项目编号') + r' | ' +
cell_pattern + re.escape('组成编码') + r'' +
cell_pattern + re.escape('名称') + r'' +
cell_pattern + re.escape('维修项目') + r'' +
cell_pattern + re.escape('维修间隔期') + r''
)
# 编译正则,用于查找表格开始位置
pattern = re.compile(header_pattern, re.DOTALL)
matches = list(pattern.finditer(content))
tables = []
if not matches:
return tables, content
# 提取所有表格块
for i, match in enumerate(matches):
start_pos = match.start()
# 确定结束位置:下一个表格的开始,或者文本末尾
if i < len(matches) - 1:
end_pos = matches[i + 1].start()
else:
end_pos = len(content)
table_content = content[start_pos:end_pos]
tables.append(table_content)
# 构建 cleaned_content:只保留第一个表格块
# 即截断到第二个表格开始的位置(如果有的话)
if len(matches) > 1:
cleaned_content = content[:matches[1].start()]
else:
cleaned_content = content # 如果只有一个表格,原内容即为清理后内容(因为后面没有表格了)
return tables, cleaned_content
#提取所有维修项目表的html源码
def extract_repairproject_tables_html(content: str) -> tuple[list[str], str]:
"""
提取所有维修表格,每个表格从 start_literal 开始,到下一个 start_literal 出现之前结束。
:param content: 原始 HTML/文本内容
:return: (tables: List[str], cleaned_content: str)
- tables: 所有提取到的表格块(包含 start_literal 开头的内容)
- cleaned_content: 原文中仅保留第一个表格块,其余已删除
"""
# 构建起始模式
start_literal = (
''
'| 维修项目编号 | '
'组成编码 | '
'名称 | '
'维修项目 | '
'维修间隔期 | '
) or (
''
'| '
' | 组成编码 | '
'名称 | '
'维修项目 | '
'维修间隔期 | '
)
# 使用 finditer 查找所有匹配的起始位置
pattern = re.compile(re.escape(start_literal), re.DOTALL)
matches = list(pattern.finditer(content))
tables = []
if not matches:
# 没有找到表格,返回空列表和原内容
return tables, content
# 提取所有表格
for i, match in enumerate(matches):
start_pos = match.start()
# 如果不是最后一个表格,结束位置是下一个表格的开始位置
if i < len(matches) - 1:
end_pos = matches[i + 1].start()
else:
# 如果是最后一个表格,结束位置是文本末尾
end_pos = len(content)
table_content = content[start_pos:end_pos]
tables.append(table_content)
# 构建 cleaned_content:只保留第一个表格
first_table_end = len(content)
if len(matches) > 1:
first_table_end = matches[1].start()
cleaned_content = content[:first_table_end]
return tables, cleaned_content
def extract_operation_tables_html(content: str) -> Tuple[List[str], str]:
"""
提取所有以“项目编号 + 操作项目”为表头的操作表格。
每个表格的范围:从匹配的 start_pattern 开始,到下一个 start_pattern 之前(或文档结尾)。
同时生成切片:「全文开头前两行 + 前100字上下文 + 表格内容」。
cleaned_content 中仅保留第一个表格(原始形式),其余删除。
"""
# 提取全文开头的前两行(保留换行符)
lines = content.splitlines(keepends=True)
first_two_lines = ''.join(lines[:2]) if len(lines) >= 2 else ''.join(lines)
# 起始模式(修复版)
start_pattern = r'''
]*> \s*
]*> \s*
| ]*> \s* 项目编号 \s* | \s*
]*> \s* 操作项目 \s* | \s*
\s*
]*>
'''
# 查找所有起始位置
matches = list(re.finditer(start_pattern, content, flags=re.DOTALL | re.VERBOSE))
if not matches:
return [], content
# 构建每个表格的完整范围:[start_i, start_{i+1}) 或 [start_i, end)
table_ranges = []
for i in range(len(matches)):
start_pos = matches[i].start()
if i + 1 < len(matches):
end_pos = matches[i + 1].start()
else:
end_pos = len(content)
table_ranges.append((start_pos, end_pos))
# 提取每个表格的 HTML 内容
tables_html = [content[start:end] for start, end in table_ranges]
# 构建 slices: 前两行 + 前100字 + 表格HTML
slices = []
for (start, _), html in zip(table_ranges, tables_html):
prefix_start = max(0, start - 100)
local_prefix = content[prefix_start:start]
full_slice = first_two_lines + local_prefix + html
# 修复标题格式(根据你的需求)
full_slice = full_slice.replace("# 操作步骤..", "操作步骤..")
full_slice = full_slice.replace("操作步骤..", "# 操作步骤")
slices.append(full_slice)
# 构建 cleaned_content:只保留第一个表格,其余删除
cleaned = content
if len(table_ranges) > 1:
# 从后往前删除第2个及之后的表格
for start, end in reversed(table_ranges[1:]):
cleaned = cleaned[:start] + cleaned[end:]
return slices, cleaned
def extract_cases(md_content):
"""
从 Markdown 内容中提取所有以 '# 数字.' 开头的案例。
每个案例从 '# X.XXX' 开始,到下一个 '# Y.YYY' 或末尾结束。
"""
# 正则表达式匹配以 "# 数字." 开头的标题
pattern = r'#\s*(\d+)\.\s*(.*?)\n(.*?)(?=(?:#\s*\d+\.\s*.*?))|$'
# 使用 re.DOTALL 让 . 匹配换行符
matches = re.findall(pattern, md_content, re.DOTALL | re.MULTILINE)
cases = []
for match in matches:
case_number = match[0] # 如 "1", "2"
title_text = match[1] # 如 "XXX 型号 A 设备 A 故障"
content = match[2] # 该案例的具体内容
# 组装完整案例
case_text = f"# {case_number}.{title_text}\n{content}"
cases.append(case_text)
return cases
def extract_fault_sections(text):
"""
从文本中提取以“故障名称:”开头的段落,去除最后一行。
"""
pattern = r'故障名称:[\s\S]*?(?=故障名称:|$)'
matches = re.findall(pattern, text, re.IGNORECASE)
results = []
for match in matches:
match = match.strip()
if match:
# 按行分割,去除最后一行,再合并
lines = match.split('\n')
if len(lines) > 1:
# 去除最后一行
modified_content = '\n'.join(lines[:-1])
results.append(modified_content.strip())
# 如果只有一行,可以根据需要决定是否保留
return results
def extract_by_unit(text):
"""
从文本中提取以“单位:”开头的段落,去除最后一行。
"""
pattern = r'(?=单位:)[\s\S]*?(?=单位:|$)'
matches = re.findall(pattern, text, re.IGNORECASE)
results = []
for match in matches:
match = match.strip()
if match:
# 按行分割,去除最后一行,再合并
lines = match.split('\n')
if len(lines) > 1:
# 去除最后一行
modified_content = '\n'.join(lines[:-1])
results.append(modified_content.strip())
return results
def extract_tables(content: str, start_len: int = 300,end_len:int = 50) -> Tuple[List[str], List[str], str]:
"""
提取所有 HTML 表格及其前后上下文(各最多 context_len 个字符),并从原文中移除表格及上下文。
:param content: 原始 HTML/文本内容
:param context_len: 上下文长度(默认 100 字符)
:return: (tables: List[str], contexts: List[str], cleaned_content: str)
其中 contexts[i] 是 tables[i] 对应的 "<前文> + ... + <后文>" 字符串
"""
# 构造正则:捕获最多 context_len 个任意字符(非贪婪)+ table 标签 + 最多 context_len 个任意字符
pattern = rf'(.{{0,{start_len}}}?)()(.{{0,{end_len}}})'
matches = re.finditer(pattern, content, flags=re.DOTALL | re.IGNORECASE)
tables = []
contexts = []
cleaned_content = content
# 从后往前替换,避免位置偏移问题
for match in reversed(list(matches)):
pre_context = match.group(1)
table = match.group(2)
post_context = match.group(3)
full_context = pre_context + table + post_context
contexts.append(full_context)
tables.append(table)
# 从原文中删除整个上下文片段(包括前后文和表格)
cleaned_content = cleaned_content[:match.start()] + cleaned_content[match.end():]
# 因为是从后往前删的,所以 tables 和 contexts 顺序是反的,需要反转回来
tables.reverse()
contexts.reverse()
return tables, contexts, cleaned_content
def extract_guzhang_tables(text: str, context_length: int = 100) -> List[str]:
header1 = '| 序号 | 故障现象 | 故障原因 | 维修项目 | '
header2 = '| 序号 | 故障现象 | 故障原因 | 维修项目 | '
header1_esc = re.escape(header1)
header2_esc = re.escape(header2)
# 匹配:前文(最多 context_length) + 表头 + 表格内容 + 结尾
pattern = re.compile(
rf'(.{{0,{context_length}}}?)(?:{header1_esc}|{header2_esc})([\s\S]*? )',
re.DOTALL
)
results = []
for match in pattern.finditer(text):
before = match.group(1) # 前文
body_and_end = match.group(2) # 表格内容 + 结尾
# 判断是哪个表头(从原文中取)
pos = match.start(2) # 表头开始位置
if text[pos:pos + len(header1)] == header1:
table_code = header1 + body_and_end
else:
table_code = header2 + body_and_end
result_str = before + table_code
results.append(result_str)
return results
def extract_content_slices(text, keywords, context_len=100):
# 转义关键词用于正则
escaped = [re.escape(kw) for kw in keywords]
pattern = r'^\s*#?\s*(' + '|'.join(escaped) + ')'
compiled = re.compile(pattern, re.MULTILINE) # 注意:这里加了 MULTILINE 以确保 ^ 能匹配每行开头
lines = text.splitlines(keepends=True) # keepends=True 保留换行符,用于精确计算位置
if not lines:
return []
contents = []
current_content = []
in_section = False
section_start_idx = 0 # 记录当前段落在原文中的字符索引位置
# 遍历每一行
for line in lines:
if compiled.match(line):
# 遇到新关键词
if in_section and current_content:
# 1. 拼接当前内容
content_str = ''.join(current_content)
# 2. 获取前 context_len 个字符的上下文
# 确保不越界
start_context = max(0, section_start_idx - context_len)
prefix_context = text[start_context:section_start_idx]
# 3. 组合结果:上下文 + 内容
final_slice = prefix_context + content_str
contents.append(final_slice)
# 重置,准备下一段
current_content = []
# 开启新段落
in_section = True
# 更新当前段落的起始位置(即这一行在原文中的位置)
# 这里简化处理:通过查找来定位(因为直接计算索引在复杂文本中容易出错)
# find 从 section_start_idx 开始找,避免重复匹配
temp_pos = text.find(line, section_start_idx)
if temp_pos != -1:
section_start_idx = temp_pos
elif in_section:
current_content.append(line)
# 处理最后一个段落
if in_section and current_content:
content_str = ''.join(current_content)
start_context = max(0, section_start_idx - context_len)
prefix_context = text[start_context:section_start_idx]
final_slice = prefix_context + content_str
contents.append(final_slice)
return contents
|