#!/usr/bin/env python3 """把旧 docx 标签表迁移为中文 Markdown 标签知识卡片。 这个脚本是一次性/阶段性迁移辅助工具,不是运行时分类器。 长期维护入口应该是 skills/label-master/knowledge 下的 Markdown 文件。 """ from __future__ import annotations import argparse import re import zipfile from dataclasses import dataclass from pathlib import Path from typing import Iterable from xml.etree import ElementTree as ET NS = {'w': 'http://schemas.openxmlformats.org/wordprocessingml/2006/main'} @dataclass class LabelRow: domain: str name: str raw_tag: str agent_candidate: str abstraction: str scope: str examples: str semantic_points: str function_def: str boundary: str confusing: str principle: str notes: str def _cell_text(cell: ET.Element) -> str: parts: list[str] = [] for text_node in cell.findall('.//w:t', NS): if text_node.text: parts.append(text_node.text) return re.sub(r'\s+', ' ', ''.join(parts)).strip() def _docx_tables(path: Path) -> list[list[list[str]]]: with zipfile.ZipFile(path) as zf: xml = zf.read('word/document.xml') root = ET.fromstring(xml) tables: list[list[list[str]]] = [] for table in root.findall('.//w:tbl', NS): rows: list[list[str]] = [] for tr in table.findall('./w:tr', NS): row = [_cell_text(tc) for tc in tr.findall('./w:tc', NS)] if any(cell.strip() for cell in row): rows.append(row) if rows: tables.append(rows) return tables def _norm_header(value: str) -> str: return re.sub(r'[\s/()()]+', '', value) def _find_index(headers: list[str], candidates: Iterable[str]) -> int | None: normalized = [_norm_header(h) for h in headers] for candidate in candidates: needle = _norm_header(candidate) for idx, header in enumerate(normalized): if needle and needle in header: return idx return None def _safe_filename(value: str) -> str: value = re.sub(r'[a-f0-9]{24,}$', '', value, flags=re.IGNORECASE) value = re.sub(r'[\\/:*?"<>|]+', '、', value).strip() value = re.sub(r'\s+', '', value) return value[:48] or '未命名标签' def _domain_name(path: Path) -> str: stem = path.stem return stem.removeprefix('Label定义-') def _read_cell(row: list[str], idx: int | None) -> str: if idx is None or idx >= len(row): return '' return row[idx].strip() def _compact_label_name(tag: str, abstraction: str) -> str: """从旧表格的 tag 列中提取更适合人工维护的中文标签名。 部分旧表格把标签名和长描述写进了同一个单元格,例如 “闹钟闹钟/小憩定时闹钟的增删改查...”。这里尽量保留短标签, 不把整段说明变成文件名或 Agent(tag=...)。 """ tag = re.sub(r'[a-f0-9]{24,}$', '', tag.strip(), flags=re.IGNORECASE) abstraction = abstraction.strip() if tag.startswith('应该属于'): return abstraction or tag.removeprefix('应该属于').strip() if not tag: stop_match = re.search(r'(关于|回答|打开|查询|增|删|改|查|的|功能|垂域|/|、|,|。)', abstraction) if stop_match and stop_match.start() >= 2: return abstraction[:stop_match.start()] return abstraction if abstraction and tag.startswith(abstraction) and len(abstraction) <= 12: return abstraction repeated = re.match(r'^(.{2,8})\1', tag) if repeated: return repeated.group(1) if len(tag) <= 24: return tag stop_match = re.search(r'(关于|回答|打开|查询|增|删|改|查|的|功能|垂域|/|、|,|。)', tag) if stop_match and stop_match.start() >= 2: return tag[:stop_match.start()] return abstraction or tag[:12] def _agent_candidate_for_label(name: str, raw_tag: str) -> str: if not name: return '' if raw_tag.startswith('应该属于'): target = raw_tag.removeprefix('应该属于').strip() if target: return f'Agent(tag="{target}")' if raw_tag and len(raw_tag) <= 24: return f'Agent(tag="{name}")' if raw_tag.startswith(name): return f'Agent(tag="{name}")' return '' def _function_candidate(value: str) -> str: value = value.strip() if not value: return '待确认。' # 保留旧表中的函数/参数定义原文,避免在迁移阶段误判最终输出格式。 return value def _looks_like_header(row: list[str]) -> bool: joined = ''.join(row) return 'tag标签' in joined or ('功能抽象' in joined and '功能和范围定义' in joined) def extract_label_rows(source_dir: Path) -> list[LabelRow]: rows: list[LabelRow] = [] for docx_path in sorted(source_dir.glob('Label定义-*.docx')): domain = _domain_name(docx_path) for table in _docx_tables(docx_path): if not table: continue header_pos = next((idx for idx, row in enumerate(table) if _looks_like_header(row)), None) if header_pos is None: continue headers = table[header_pos] idx_abstraction = _find_index(headers, ['功能抽象']) idx_tag = _find_index(headers, ['tag标签']) idx_scope = _find_index(headers, ['功能和范围定义']) idx_examples = _find_index(headers, ['示例query']) idx_semantic = _find_index(headers, ['三级语义功能点举例']) idx_function = _find_index(headers, ['Function及参数定义', 'Function及参数定义飘黄部分未共识']) idx_boundary = _find_index(headers, ['满足边界问题']) idx_confusing = _find_index(headers, ['易混淆tagfunctionAgent', '易混淆tagfunction', '易混淆']) idx_principle = _find_index(headers, ['划分原则']) idx_notes = _find_index(headers, ['问题备注', '未解决']) for row in table[header_pos + 1:]: abstraction = _read_cell(row, idx_abstraction) tag = _read_cell(row, idx_tag) name = _compact_label_name(tag, abstraction) if not name or name in {'/', '-', '待定'}: continue if len(name) > 80 and not tag: continue agent_candidate = _agent_candidate_for_label(name, tag) rows.append( LabelRow( domain=domain, name=name, raw_tag=tag, agent_candidate=agent_candidate, abstraction=abstraction, scope=_read_cell(row, idx_scope), examples=_read_cell(row, idx_examples), semantic_points=_read_cell(row, idx_semantic), function_def=_read_cell(row, idx_function), boundary=_read_cell(row, idx_boundary), confusing=_read_cell(row, idx_confusing), principle=_read_cell(row, idx_principle), notes=_read_cell(row, idx_notes), ) ) return rows def _section(title: str, body: str) -> str: body = body.strip() if not body: body = '待补充。' return f'## {title}\n\n{body}\n' def _table_cell(value: str, *, limit: int = 80) -> str: value = re.sub(r'\s+', ' ', value.strip()).replace('|', '、') if not value: return '待确认' return value[:limit] + ('...' if len(value) > limit else '') def _output_section(row: LabelRow) -> str: raw_tag = row.raw_tag or '待确认。' agent_candidate = row.agent_candidate or '待确认。' function_candidate = _function_candidate(row.function_def) return '\n'.join( [ '## 标注输出', '', '> 注意:旧标签资料中同时存在 tag 标签和 Function 定义。本卡片不直接声明最终训练标签;生成数据前必须结合 `判断维度/标注输出形态.md` 确认当前任务使用 Agent 形式还是 Function 形式。', '', f'- 旧 tag 标签:{raw_tag}', f'- Agent 包装候选(不代表最终):{agent_candidate}', f'- Function 输出候选:{function_candidate}', '- 推荐输出形态:待确认。', '', ] ) def _card_text(row: LabelRow) -> str: title = row.name return '\n'.join( [ f'# {title}', '', _output_section(row), _section('功能抽象', row.abstraction), _section('适用范围', row.scope), _section('典型 Query', row.examples), _section('三级语义功能点', row.semantic_points), _section('Function / Agent 说明', row.function_def), _section('满足边界问题', row.boundary), _section('易混淆标签', row.confusing), _section('划分原则', row.principle), _section('未解决问题', row.notes), ] ).rstrip() + '\n' def write_knowledge(rows: list[LabelRow], output_dir: Path) -> None: label_root = output_dir / '标签' boundary_root = output_dir / '边界' index_root = output_dir / '索引' label_root.mkdir(parents=True, exist_ok=True) boundary_root.mkdir(parents=True, exist_ok=True) index_root.mkdir(parents=True, exist_ok=True) domains: dict[str, list[LabelRow]] = {} for row in rows: domains.setdefault(row.domain, []).append(row) domain_dir = label_root / _safe_filename(row.domain) domain_dir.mkdir(parents=True, exist_ok=True) path = domain_dir / f'{_safe_filename(row.name)}.md' path.write_text(_card_text(row), encoding='utf-8') for domain, domain_rows in sorted(domains.items()): lines = [f'# {domain}边界', ''] for row in domain_rows: if not row.confusing and not row.principle and not row.boundary: continue lines.append(f'## {row.name}') if row.confusing: lines.extend(['', f'- 易混淆:{row.confusing}']) if row.principle: lines.extend(['', f'- 划分原则:{row.principle}']) if row.boundary: lines.extend(['', f'- 满足边界问题:{row.boundary}']) lines.append('') if len(lines) > 2: (boundary_root / f'{_safe_filename(domain)}边界.md').write_text('\n'.join(lines).rstrip() + '\n', encoding='utf-8') overview_lines = ['# 标签迁移索引', ''] label_index_lines = [ '# 标签索引', '', '这个索引用于 Agent 第一阶段快速判断候选标签和输出形态。详细边界仍需读取具体标签卡片和边界文件。', '', '| 领域 | 标签 | 旧 tag 标签 | Agent 形式候选 | Function 形式候选 | 知识卡片 |', '| --- | --- | --- | --- | --- | --- |', ] for domain, domain_rows in sorted(domains.items()): overview_lines.append(f'## {domain}') overview_lines.append('') seen_links: set[str] = set() for row in sorted(domain_rows, key=lambda item: item.name): rel = Path('标签') / _safe_filename(domain) / f'{_safe_filename(row.name)}.md' rel_text = rel.as_posix() if rel_text in seen_links: continue seen_links.add(rel_text) overview_lines.append(f'- [{row.name}]({rel.as_posix()})') rel_from_index = Path('..') / rel function_hint = _table_cell(row.function_def, limit=60) if row.function_def else '待确认' label_index_lines.append( '| {domain} | {name} | {raw_tag} | {agent} | {function} | [{name}]({path}) |'.format( domain=_table_cell(domain, limit=20), name=_table_cell(row.name, limit=24), raw_tag=_table_cell(row.raw_tag, limit=32), agent=_table_cell(row.agent_candidate, limit=36), function=function_hint, path=rel_from_index.as_posix(), ) ) overview_lines.append('') (output_dir / '标签迁移索引.md').write_text('\n'.join(overview_lines).rstrip() + '\n', encoding='utf-8') (index_root / '标签索引.md').write_text('\n'.join(label_index_lines).rstrip() + '\n', encoding='utf-8') def main() -> int: parser = argparse.ArgumentParser(description='迁移 docx 标签定义表为 Markdown 知识卡片') parser.add_argument('--source-dir', default='标签定义', help='旧标签定义目录') parser.add_argument('--output-dir', default='skills/label-master/knowledge', help='知识库输出目录') args = parser.parse_args() source_dir = Path(args.source_dir) output_dir = Path(args.output_dir) rows = extract_label_rows(source_dir) write_knowledge(rows, output_dir) print(f'已迁移 {len(rows)} 条标签知识到 {output_dir}') return 0 if __name__ == '__main__': raise SystemExit(main())