From d2712fdb673e5666a26efe82628feba8d48755fc Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E6=AD=A6=E9=98=B3?= Date: Fri, 8 May 2026 10:40:06 +0800 Subject: [PATCH] Normalize data agent records output path --- src/agent_tool_specs/data_agent.py | 15 ++++++++-- src/agent_tools.py | 36 +++++++++++++++++++---- src/data_agent_records.py | 3 ++ src/skills/bundled/online-mining/SKILL.md | 6 ++-- src/skills/bundled/product-data/SKILL.md | 18 ++++++------ tests/test_data_agent_records.py | 36 +++++++++++++++++++++-- 6 files changed, 93 insertions(+), 21 deletions(-) diff --git a/src/agent_tool_specs/data_agent.py b/src/agent_tool_specs/data_agent.py index 2975d13..56b4377 100644 --- a/src/agent_tool_specs/data_agent.py +++ b/src/agent_tool_specs/data_agent.py @@ -135,7 +135,14 @@ def build_data_agent_tools(handlers: Mapping[str, ToolHandler]) -> list[AgentToo 'turn_mix': {'type': 'string', 'description': 'Single-turn and multi-turn count or ratio.'}, 'coverage': {'type': 'string', 'description': 'Query types, intent boundaries, or error types to cover.'}, 'exclusions': {'type': 'string', 'description': 'Negative examples or boundaries to avoid.'}, - 'output_path': {'type': 'string', 'description': 'Where draft, records, and validation files should be written.'}, + 'output_path': { + 'type': 'string', + 'description': ( + 'Requested records path. The runtime normalizes canonical records to the current ' + 'session output/records.jsonl; use output/records.jsonl and do not add dataset ' + 'subdirectories or random names.' + ), + }, 'notes': {'type': 'string'}, }, 'required': [ @@ -500,7 +507,11 @@ def build_data_agent_tools(handlers: Mapping[str, ToolHandler]) -> list[AgentToo }, 'output_path': { 'type': 'string', - 'description': 'Workspace-relative path to write, usually ending in .jsonl.', + 'description': ( + 'Requested workspace-relative path. For canonical records, the runtime normalizes this ' + 'to the current session output/records.jsonl or output/records.json; do not encode ' + 'dataset names or timestamps in the filename.' + ), }, 'output_format': { 'type': 'string', diff --git a/src/agent_tools.py b/src/agent_tools.py index 4aa5ee6..35dc63d 100644 --- a/src/agent_tools.py +++ b/src/agent_tools.py @@ -1207,7 +1207,11 @@ def _data_agent_prepare_generation_plan_tool(arguments: dict[str, Any], context: turn_mix=_require_string(arguments, 'turn_mix'), coverage=_require_string(arguments, 'coverage'), exclusions=_require_string(arguments, 'exclusions'), - output_path=_data_agent_output_path(_require_string(arguments, 'output_path'), context), + output_path=_data_agent_output_path( + _require_string(arguments, 'output_path'), + context, + canonical_filename='records.jsonl', + ), target_definitions=_optional_target_definitions(arguments.get('target_definitions')), notes=notes, confirmed_goal_id=confirmed_goal_id or None, @@ -1480,10 +1484,15 @@ def _export_dataset_records_tool(arguments: dict[str, Any], context: ToolExecuti raise ToolExecutionError('overwrite must be a boolean') try: records = records_from_tool_argument(arguments['records']) + canonical_filename = 'records.json' if output_format == 'json' else 'records.jsonl' payload = export_dataset_records( records, root=str(context.root), - output_path=_data_agent_output_path(_require_string(arguments, 'output_path'), context), + output_path=_data_agent_output_path( + _require_string(arguments, 'output_path'), + context, + canonical_filename=canonical_filename, + ), output_format=output_format, require_validation_ok=require_validation_ok, overwrite=overwrite, @@ -1493,15 +1502,32 @@ def _export_dataset_records_tool(arguments: dict[str, Any], context: ToolExecuti return json.dumps(payload, ensure_ascii=False, indent=2) -def _data_agent_output_path(output_path: str, context: ToolExecutionContext) -> str: - """把数据 Agent 产物默认路由到当前用户/会话的 output 目录。""" +def _data_agent_output_path( + output_path: str, + context: ToolExecutionContext, + *, + canonical_filename: str | None = None, +) -> str: + """把数据 Agent 产物默认路由到当前用户/会话的 output 目录。 + + canonical_filename 用于强制收口最终 records 文件名,避免模型按数据集名 + 自行创建子目录或临时文件名,导致产物位置不稳定。 + """ raw = output_path.strip() path = Path(raw).expanduser() + + output_root = _data_agent_session_output_root(context) + if canonical_filename: + resolved = (output_root / canonical_filename).resolve(strict=False) + try: + return resolved.relative_to(context.root).as_posix() + except ValueError: + return resolved.as_posix() + if path.is_absolute() or path.parts[:1] == ('.port_sessions',): return raw - output_root = _data_agent_session_output_root(context) parts = path.parts suffix = Path(*parts[1:]) if parts and parts[0] in {'output', 'outputs'} else path resolved = (output_root / suffix).resolve(strict=False) diff --git a/src/data_agent_records.py b/src/data_agent_records.py index 053df5f..86d6fa8 100644 --- a/src/data_agent_records.py +++ b/src/data_agent_records.py @@ -549,6 +549,9 @@ def _dedupe_output_path(root: str, output_path: str, plan_id: str) -> str: from pathlib import Path path = Path(output_path) + # 会话 output 下的标准 records 文件名需要保持稳定;导出工具默认允许覆盖。 + if path.name in {'records.jsonl', 'records.json'} and 'output' in path.parts: + return output_path resolved = path if path.is_absolute() else Path(root).resolve() / path if not resolved.exists(): return output_path diff --git a/src/skills/bundled/online-mining/SKILL.md b/src/skills/bundled/online-mining/SKILL.md index 6c9a8fe..4a0cdbe 100644 --- a/src/skills/bundled/online-mining/SKILL.md +++ b/src/skills/bundled/online-mining/SKILL.md @@ -8,7 +8,7 @@ allowed_tools: read_file, write_file, edit_file, grep_search, glob_search, ask_u 使用这个 skill 处理“badcase 或标签定义 -> 挖掘策略 -> 候选召回 -> 抽样 review -> 策略迭代 -> mined dataset”的工作流。 -所有线上挖掘产物必须放在当前用户当前会话的 output 目录下。不要把 records 或 review 结果写到项目根目录的 `output/`、`tasks/` 或源码目录。工具会把相对 `output_path` 自动路由到会话 output 目录;展示给用户时以工具返回的实际路径为准。 +所有线上挖掘产物必须放在当前用户当前会话的 output 目录下。不要把 records 或 review 结果写到项目根目录的 `output/`、`tasks/` 或源码目录。canonical records 的稳定输出文件名固定为 `output/records.jsonl`,不要按数据集名创建子目录,不要自定义时间戳、中文专题名或随机文件名。工具会把相对 `output_path` 自动路由到会话 output 目录,并把 records 路径归一到当前会话 `output/records.jsonl`;展示给用户时以工具返回的实际路径为准。 ## 两条分支 @@ -25,7 +25,7 @@ allowed_tools: read_file, write_file, edit_file, grep_search, glob_search, ask_u 3. 用 `data_agent_sample_router_candidates` 抽样展示给用户 review。 4. 根据用户 review 意见形成 include/exclude/uncertain 决策和目标标签。 5. 用 `data_agent_convert_router_candidates_to_records` 把线上候选直接转换为 canonical records。 -6. 如果用户要求落盘,用 `data_agent_export_dataset_records` 导出紧凑 JSONL。 +6. 如果用户要求落盘,用 `data_agent_export_dataset_records` 导出紧凑 JSONL,`output_path` 固定传 `output/records.jsonl`。 这个分支不生成新 query,不调用数据生成计划工具,不调用 dataset draft 归一化工具。 @@ -89,7 +89,7 @@ allowed_tools: read_file, write_file, edit_file, grep_search, glob_search, ask_u - `data_agent_search_router_sessions`:按日期、设备、domain、intent、func、query 关键词/正则、轮次数等条件召回线上 session 候选;输出命中 turn、前文 turn、req_id 和 action_json 解析结果。 - `data_agent_sample_router_candidates`:对召回候选做稳定抽样,并输出 review 需要的基础统计。 - `data_agent_convert_router_candidates_to_records`:把 review 后的线上候选直接转换为 canonical records;用于“线上数据作为样本”的分支。 -- `data_agent_export_dataset_records`:后续把 review 后的线上候选转换为 canonical records 后落盘;默认紧凑 JSONL。 +- `data_agent_export_dataset_records`:后续把 review 后的线上候选转换为 canonical records 后落盘;默认紧凑 JSONL,固定传 `output/records.jsonl`。 TODO:后续规划中的专用工具: diff --git a/src/skills/bundled/product-data/SKILL.md b/src/skills/bundled/product-data/SKILL.md index 8c9c6ed..5fc0a74 100644 --- a/src/skills/bundled/product-data/SKILL.md +++ b/src/skills/bundled/product-data/SKILL.md @@ -55,7 +55,7 @@ source_refs: - **一次确认模式**:用户直接给出手写规则、完整 target 表达,并且明确希望生成数据时,直接调用 `data_agent_prepare_generation_plan`,传 `direct_review=true` 和 `target_definitions`。这一次 review 同时确认目标、数量、轮次和路径;用户回复“确认,开始生成”后即可调用 `data_agent_confirm_generation_plan`。 - **两段确认模式**:用户提供文件、表格、badcase、长文档,或者标签/边界/字段含义有歧义时,先用 `data_agent_prepare_generation_goal` 做目标 review;目标确认后再做 plan review。 -所有数据产物必须放在当前用户当前会话的 output 目录下。不要把 records、draft 或 validation 写到项目根目录的 `output/`、`tasks/` 或其他源码目录。工具会把相对 `output_path` 自动路由到会话 output 目录;展示给用户时以工具返回的实际路径为准。 +所有数据产物必须放在当前用户当前会话的 output 目录下。不要把 records、draft 或 validation 写到项目根目录的 `output/`、`tasks/` 或其他源码目录。canonical records 的稳定输出文件名固定为 `output/records.jsonl`,不要按数据集名创建子目录,不要自定义时间戳、中文专题名或随机文件名。工具会把相对 `output_path` 自动路由到会话 output 目录,并把 records 路径归一到当前会话 `output/records.jsonl`;展示给用户时以工具返回的实际路径为准。 开始生成前必须确认这些信息: @@ -64,13 +64,13 @@ source_refs: - 生成数量:总条数,以及单轮/多轮数量或比例。 - 覆盖范围:需要覆盖哪些 query 类型、意图边界或错误类型。 - 负例/排除项:哪些表达不要生成,或哪些边界容易误判。 -- 落盘路径:draft、canonical records、validation result 写到哪里。 +- 落盘路径:canonical records 固定使用 `output/records.jsonl`,由工具路由到当前会话 output 目录。 如果任一信息缺失,不要生成数据,不要调用 `data_agent_prepare_generation_plan`,不要调用 `data_agent_normalize_dataset_draft`,不要调用 `data_agent_validate_dataset_records`,只向用户提出需要确认的问题。 两段确认模式下,信息完整后,调用 `data_agent_prepare_generation_goal` 创建 pending goal。这个工具会暂停本轮,必须把返回的 `generation_goal` 展示给用户 review。用户确认 goal 之后,调用 `data_agent_confirm_generation_goal` 获取 `confirmed_goal_id`,再调用 `data_agent_prepare_generation_plan` 创建 pending plan。创建 plan 后也会暂停本轮,必须等待用户 review。 -一次确认模式下,不要先创建 generation goal;直接创建 pending plan,并在 plan 里包含 `target_definitions`、`total_count`、`turn_mix`、`coverage`、`exclusions`、`output_path`。不要让用户先确认目标再确认计划。 +一次确认模式下,不要先创建 generation goal;直接创建 pending plan,并在 plan 里包含 `target_definitions`、`total_count`、`turn_mix`、`coverage`、`exclusions`、`output_path`。`output_path` 固定传 `output/records.jsonl`。不要让用户先确认目标再确认计划。 用户确认后,拿到 `confirmed_plan_id`,才能生成 dataset draft text,并继续调用工具。 @@ -131,7 +131,7 @@ review 展示必须简短清晰,不要重复解释工具和流程。每次 rev 11. 生成 dataset draft text v1。 12. 调用 `data_agent_normalize_dataset_draft`,必须传入 `confirmed_plan_id`。 13. 调用 `data_agent_validate_dataset_records`。 -14. 如果用户要求落盘 canonical records,调用 `data_agent_export_dataset_records`,默认导出紧凑 JSONL,不要用 `write_file` 手写 JSON。 +14. 如果用户要求落盘 canonical records,调用 `data_agent_export_dataset_records`,`output_path` 固定传 `output/records.jsonl`,默认导出紧凑 JSONL,不要用 `write_file` 手写 JSON。 15. 本阶段默认只推进到 canonical metadata records;除非用户另行要求,不做最终训练/评测格式导出。 ## 数据生成输出格式 @@ -174,7 +174,7 @@ notes: 可选,说明覆盖的问题或边界 `data_agent_normalize_dataset_draft` 会把 dataset draft text 转成 canonical records,并统一补齐 `record_id`、`source`、`timestamp`、`context`、`target_type` 等机械字段。 -canonical records 落盘必须使用 `data_agent_export_dataset_records`,默认格式是紧凑 JSONL:一行一个 canonical record,不带外层数组,不手写缩进 JSON。只有用户明确要求兼容旧文件时,才使用 `output_format="json"` 导出紧凑 JSON 数组。 +canonical records 落盘必须使用 `data_agent_export_dataset_records`,默认格式是紧凑 JSONL:一行一个 canonical record,不带外层数组,不手写缩进 JSON。默认 `output_path` 固定传 `output/records.jsonl`;不要传 `output/<数据集名>/records.jsonl`,不要传 `tasks/...`,不要自定义文件名。只有用户明确要求兼容旧文件时,才使用 `output_format="json"` 导出紧凑 JSON 数组,此时工具会归一为 `output/records.json`。 当前 canonical record v1 工作格式: @@ -258,7 +258,7 @@ canonical records 落盘必须使用 `data_agent_export_dataset_records`,默 "source_refs": [] } ], - "plan_hint": "建议先生成 50 条单轮,输出到 tasks/<数据集名称>/records.jsonl;具体数量、轮次和路径在 generation plan 中确认。", + "plan_hint": "建议先生成 50 条单轮,输出到 output/records.jsonl;具体数量和轮次在 generation plan 中确认。", "coverage": "需要覆盖的 query 语义、功能点、错误类型", "exclusions": "不要生成或需要排除的表达", "open_questions": [], @@ -266,7 +266,7 @@ canonical records 落盘必须使用 `data_agent_export_dataset_records`,默 } ``` -`generation_goal` 只确认“做什么数据、为什么做、标签边界是什么”。不要在 goal 中维护结构化的 `total_count`、`turn_mix` 或 `output_path`;这些字段属于后续 `generation_plan`。如果需要在 goal review 阶段提示执行方向,只写一句 `plan_hint`,例如“建议先生成 50 条单轮,输出到 tasks/.../records.jsonl;具体数量、轮次和路径在 generation plan 中确认”。 +`generation_goal` 只确认“做什么数据、为什么做、标签边界是什么”。不要在 goal 中维护结构化的 `total_count`、`turn_mix` 或 `output_path`;这些字段属于后续 `generation_plan`。如果需要在 goal review 阶段提示执行方向,只写一句 `plan_hint`,例如“建议先生成 50 条单轮,输出到 output/records.jsonl;具体数量和轮次在 generation plan 中确认”。 如果未来增加 `data_agent_validate_generation_goal`,它只做结构校验和缺失字段提示,不做语义判断,不替代用户 review,也不替代 `data_agent_prepare_generation_plan`。 @@ -307,13 +307,13 @@ canonical records 落盘必须使用 `data_agent_export_dataset_records`,默 - `data_agent_confirm_generation_plan`:用户明确确认当前计划版本后使用,获取 `confirmed_plan_id`。 - `data_agent_normalize_dataset_draft`:用户确认计划后,把 dataset draft text v1 转成 canonical records;必须传入 `confirmed_plan_id`。 - `data_agent_validate_dataset_records`:对 canonical records 做结构、标签、时间戳和多轮上下文校验。 -- `data_agent_export_dataset_records`:校验 canonical records 并落盘;默认写紧凑 JSONL,一行一条,不要再用 `write_file` 手写 records 文件。 +- `data_agent_export_dataset_records`:校验 canonical records 并落盘;默认写紧凑 JSONL,一行一条,固定传 `output/records.jsonl`,不要再用 `write_file` 手写 records 文件。 ## 约束 - 不要静默解决产品或标签歧义。 - 如果 `ask_user_question` 不可用,使用普通回复向用户提问并停止,不要自己替用户确认。 - canonical records 通过校验前,不要生成最终导出格式。 -- canonical records 需要落盘时,必须用 `data_agent_export_dataset_records`;不要自己拼接 JSON/JSONL。 +- canonical records 需要落盘时,必须用 `data_agent_export_dataset_records`;不要自己拼接 JSON/JSONL;不要创建数据集子目录或自定义 records 文件名。 - 除非用户明确要求,否则不要把“修改标签定义”和“生成数据”混在一起做。 - 不要因为用户说“生成一些数据”就跳过边界总结和 generation plan review。 diff --git a/tests/test_data_agent_records.py b/tests/test_data_agent_records.py index e8f5970..db1b91f 100644 --- a/tests/test_data_agent_records.py +++ b/tests/test_data_agent_records.py @@ -571,16 +571,48 @@ target: Agent(tag="life_service") export_result = execute_tool( registry, 'data_agent_export_dataset_records', - {'records': records, 'output_path': 'output/demo/records.jsonl'}, + {'records': records, 'output_path': 'output/地图和生活边界数据/自定义名字.jsonl'}, context, ) self.assertTrue(export_result.ok, export_result.content) export_payload = json.loads(export_result.content) - expected = '.port_sessions/accounts/alice/sessions/s1/output/demo/records.jsonl' + expected = '.port_sessions/accounts/alice/sessions/s1/output/records.jsonl' self.assertEqual(plan_payload['plan']['output_path'], expected) self.assertEqual(export_payload['output_path'], expected) + def test_data_agent_session_records_path_stays_stable_when_file_exists(self) -> None: + with tempfile.TemporaryDirectory() as tmp_dir: + root = Path(tmp_dir) + scratchpad = root / '.port_sessions' / 'accounts' / 'alice' / 'sessions' / 's1' / 'scratchpad' + output_root = scratchpad.parent / 'output' + output_root.mkdir(parents=True) + (output_root / 'records.jsonl').write_text('{}\n', encoding='utf-8') + context = build_tool_context(AgentRuntimeConfig(cwd=root), scratchpad_directory=scratchpad) + + plan_result = execute_tool( + default_tool_registry(), + 'data_agent_prepare_generation_plan', + { + 'direct_review': True, + 'dataset_label': '地图和生活边界数据', + 'target': 'Agent(tag="life_service")', + 'total_count': 1, + 'turn_mix': '1 条单轮', + 'coverage': '附近吃喝玩乐', + 'exclusions': '不要生成导航路线类 query', + 'output_path': 'output/任意子目录/任意名字.jsonl', + }, + context, + ) + self.assertTrue(plan_result.ok, plan_result.content) + plan_payload = json.loads(plan_result.content) + + self.assertEqual( + plan_payload['plan']['output_path'], + '.port_sessions/accounts/alice/sessions/s1/output/records.jsonl', + ) + if __name__ == '__main__': unittest.main()