Normalize data agent records output path

This commit is contained in:
武阳
2026-05-08 10:40:06 +08:00
parent 9e94648ae6
commit d2712fdb67
6 changed files with 93 additions and 21 deletions
+13 -2
View File
@@ -135,7 +135,14 @@ def build_data_agent_tools(handlers: Mapping[str, ToolHandler]) -> list[AgentToo
'turn_mix': {'type': 'string', 'description': 'Single-turn and multi-turn count or ratio.'},
'coverage': {'type': 'string', 'description': 'Query types, intent boundaries, or error types to cover.'},
'exclusions': {'type': 'string', 'description': 'Negative examples or boundaries to avoid.'},
'output_path': {'type': 'string', 'description': 'Where draft, records, and validation files should be written.'},
'output_path': {
'type': 'string',
'description': (
'Requested records path. The runtime normalizes canonical records to the current '
'session output/records.jsonl; use output/records.jsonl and do not add dataset '
'subdirectories or random names.'
),
},
'notes': {'type': 'string'},
},
'required': [
@@ -500,7 +507,11 @@ def build_data_agent_tools(handlers: Mapping[str, ToolHandler]) -> list[AgentToo
},
'output_path': {
'type': 'string',
'description': 'Workspace-relative path to write, usually ending in .jsonl.',
'description': (
'Requested workspace-relative path. For canonical records, the runtime normalizes this '
'to the current session output/records.jsonl or output/records.json; do not encode '
'dataset names or timestamps in the filename.'
),
},
'output_format': {
'type': 'string',
+31 -5
View File
@@ -1207,7 +1207,11 @@ def _data_agent_prepare_generation_plan_tool(arguments: dict[str, Any], context:
turn_mix=_require_string(arguments, 'turn_mix'),
coverage=_require_string(arguments, 'coverage'),
exclusions=_require_string(arguments, 'exclusions'),
output_path=_data_agent_output_path(_require_string(arguments, 'output_path'), context),
output_path=_data_agent_output_path(
_require_string(arguments, 'output_path'),
context,
canonical_filename='records.jsonl',
),
target_definitions=_optional_target_definitions(arguments.get('target_definitions')),
notes=notes,
confirmed_goal_id=confirmed_goal_id or None,
@@ -1480,10 +1484,15 @@ def _export_dataset_records_tool(arguments: dict[str, Any], context: ToolExecuti
raise ToolExecutionError('overwrite must be a boolean')
try:
records = records_from_tool_argument(arguments['records'])
canonical_filename = 'records.json' if output_format == 'json' else 'records.jsonl'
payload = export_dataset_records(
records,
root=str(context.root),
output_path=_data_agent_output_path(_require_string(arguments, 'output_path'), context),
output_path=_data_agent_output_path(
_require_string(arguments, 'output_path'),
context,
canonical_filename=canonical_filename,
),
output_format=output_format,
require_validation_ok=require_validation_ok,
overwrite=overwrite,
@@ -1493,15 +1502,32 @@ def _export_dataset_records_tool(arguments: dict[str, Any], context: ToolExecuti
return json.dumps(payload, ensure_ascii=False, indent=2)
def _data_agent_output_path(output_path: str, context: ToolExecutionContext) -> str:
"""把数据 Agent 产物默认路由到当前用户/会话的 output 目录。"""
def _data_agent_output_path(
output_path: str,
context: ToolExecutionContext,
*,
canonical_filename: str | None = None,
) -> str:
"""把数据 Agent 产物默认路由到当前用户/会话的 output 目录。
canonical_filename 用于强制收口最终 records 文件名,避免模型按数据集名
自行创建子目录或临时文件名,导致产物位置不稳定。
"""
raw = output_path.strip()
path = Path(raw).expanduser()
output_root = _data_agent_session_output_root(context)
if canonical_filename:
resolved = (output_root / canonical_filename).resolve(strict=False)
try:
return resolved.relative_to(context.root).as_posix()
except ValueError:
return resolved.as_posix()
if path.is_absolute() or path.parts[:1] == ('.port_sessions',):
return raw
output_root = _data_agent_session_output_root(context)
parts = path.parts
suffix = Path(*parts[1:]) if parts and parts[0] in {'output', 'outputs'} else path
resolved = (output_root / suffix).resolve(strict=False)
+3
View File
@@ -549,6 +549,9 @@ def _dedupe_output_path(root: str, output_path: str, plan_id: str) -> str:
from pathlib import Path
path = Path(output_path)
# 会话 output 下的标准 records 文件名需要保持稳定;导出工具默认允许覆盖。
if path.name in {'records.jsonl', 'records.json'} and 'output' in path.parts:
return output_path
resolved = path if path.is_absolute() else Path(root).resolve() / path
if not resolved.exists():
return output_path
+3 -3
View File
@@ -8,7 +8,7 @@ allowed_tools: read_file, write_file, edit_file, grep_search, glob_search, ask_u
使用这个 skill 处理“badcase 或标签定义 -> 挖掘策略 -> 候选召回 -> 抽样 review -> 策略迭代 -> mined dataset”的工作流。
所有线上挖掘产物必须放在当前用户当前会话的 output 目录下。不要把 records 或 review 结果写到项目根目录的 `output/``tasks/` 或源码目录。工具会把相对 `output_path` 自动路由到会话 output 目录;展示给用户时以工具返回的实际路径为准。
所有线上挖掘产物必须放在当前用户当前会话的 output 目录下。不要把 records 或 review 结果写到项目根目录的 `output/``tasks/` 或源码目录。canonical records 的稳定输出文件名固定为 `output/records.jsonl`,不要按数据集名创建子目录,不要自定义时间戳、中文专题名或随机文件名。工具会把相对 `output_path` 自动路由到会话 output 目录,并把 records 路径归一到当前会话 `output/records.jsonl`;展示给用户时以工具返回的实际路径为准。
## 两条分支
@@ -25,7 +25,7 @@ allowed_tools: read_file, write_file, edit_file, grep_search, glob_search, ask_u
3.`data_agent_sample_router_candidates` 抽样展示给用户 review。
4. 根据用户 review 意见形成 include/exclude/uncertain 决策和目标标签。
5.`data_agent_convert_router_candidates_to_records` 把线上候选直接转换为 canonical records。
6. 如果用户要求落盘,用 `data_agent_export_dataset_records` 导出紧凑 JSONL。
6. 如果用户要求落盘,用 `data_agent_export_dataset_records` 导出紧凑 JSONL`output_path` 固定传 `output/records.jsonl`
这个分支不生成新 query,不调用数据生成计划工具,不调用 dataset draft 归一化工具。
@@ -89,7 +89,7 @@ allowed_tools: read_file, write_file, edit_file, grep_search, glob_search, ask_u
- `data_agent_search_router_sessions`:按日期、设备、domain、intent、func、query 关键词/正则、轮次数等条件召回线上 session 候选;输出命中 turn、前文 turn、req_id 和 action_json 解析结果。
- `data_agent_sample_router_candidates`:对召回候选做稳定抽样,并输出 review 需要的基础统计。
- `data_agent_convert_router_candidates_to_records`:把 review 后的线上候选直接转换为 canonical records;用于“线上数据作为样本”的分支。
- `data_agent_export_dataset_records`:后续把 review 后的线上候选转换为 canonical records 后落盘;默认紧凑 JSONL。
- `data_agent_export_dataset_records`:后续把 review 后的线上候选转换为 canonical records 后落盘;默认紧凑 JSONL,固定传 `output/records.jsonl`
TODO:后续规划中的专用工具:
+9 -9
View File
@@ -55,7 +55,7 @@ source_refs:
- **一次确认模式**:用户直接给出手写规则、完整 target 表达,并且明确希望生成数据时,直接调用 `data_agent_prepare_generation_plan`,传 `direct_review=true``target_definitions`。这一次 review 同时确认目标、数量、轮次和路径;用户回复“确认,开始生成”后即可调用 `data_agent_confirm_generation_plan`
- **两段确认模式**:用户提供文件、表格、badcase、长文档,或者标签/边界/字段含义有歧义时,先用 `data_agent_prepare_generation_goal` 做目标 review;目标确认后再做 plan review。
所有数据产物必须放在当前用户当前会话的 output 目录下。不要把 records、draft 或 validation 写到项目根目录的 `output/``tasks/` 或其他源码目录。工具会把相对 `output_path` 自动路由到会话 output 目录;展示给用户时以工具返回的实际路径为准。
所有数据产物必须放在当前用户当前会话的 output 目录下。不要把 records、draft 或 validation 写到项目根目录的 `output/``tasks/` 或其他源码目录。canonical records 的稳定输出文件名固定为 `output/records.jsonl`,不要按数据集名创建子目录,不要自定义时间戳、中文专题名或随机文件名。工具会把相对 `output_path` 自动路由到会话 output 目录,并把 records 路径归一到当前会话 `output/records.jsonl`;展示给用户时以工具返回的实际路径为准。
开始生成前必须确认这些信息:
@@ -64,13 +64,13 @@ source_refs:
- 生成数量:总条数,以及单轮/多轮数量或比例。
- 覆盖范围:需要覆盖哪些 query 类型、意图边界或错误类型。
- 负例/排除项:哪些表达不要生成,或哪些边界容易误判。
- 落盘路径:draft、canonical records、validation result 写到哪里
- 落盘路径:canonical records 固定使用 `output/records.jsonl`,由工具路由到当前会话 output 目录
如果任一信息缺失,不要生成数据,不要调用 `data_agent_prepare_generation_plan`,不要调用 `data_agent_normalize_dataset_draft`,不要调用 `data_agent_validate_dataset_records`,只向用户提出需要确认的问题。
两段确认模式下,信息完整后,调用 `data_agent_prepare_generation_goal` 创建 pending goal。这个工具会暂停本轮,必须把返回的 `generation_goal` 展示给用户 review。用户确认 goal 之后,调用 `data_agent_confirm_generation_goal` 获取 `confirmed_goal_id`,再调用 `data_agent_prepare_generation_plan` 创建 pending plan。创建 plan 后也会暂停本轮,必须等待用户 review。
一次确认模式下,不要先创建 generation goal;直接创建 pending plan,并在 plan 里包含 `target_definitions``total_count``turn_mix``coverage``exclusions``output_path`。不要让用户先确认目标再确认计划。
一次确认模式下,不要先创建 generation goal;直接创建 pending plan,并在 plan 里包含 `target_definitions``total_count``turn_mix``coverage``exclusions``output_path``output_path` 固定传 `output/records.jsonl`不要让用户先确认目标再确认计划。
用户确认后,拿到 `confirmed_plan_id`,才能生成 dataset draft text,并继续调用工具。
@@ -131,7 +131,7 @@ review 展示必须简短清晰,不要重复解释工具和流程。每次 rev
11. 生成 dataset draft text v1。
12. 调用 `data_agent_normalize_dataset_draft`,必须传入 `confirmed_plan_id`
13. 调用 `data_agent_validate_dataset_records`
14. 如果用户要求落盘 canonical records,调用 `data_agent_export_dataset_records`,默认导出紧凑 JSONL,不要用 `write_file` 手写 JSON。
14. 如果用户要求落盘 canonical records,调用 `data_agent_export_dataset_records``output_path` 固定传 `output/records.jsonl`默认导出紧凑 JSONL,不要用 `write_file` 手写 JSON。
15. 本阶段默认只推进到 canonical metadata records;除非用户另行要求,不做最终训练/评测格式导出。
## 数据生成输出格式
@@ -174,7 +174,7 @@ notes: 可选,说明覆盖的问题或边界
`data_agent_normalize_dataset_draft` 会把 dataset draft text 转成 canonical records,并统一补齐 `record_id``source``timestamp``context``target_type` 等机械字段。
canonical records 落盘必须使用 `data_agent_export_dataset_records`,默认格式是紧凑 JSONL:一行一个 canonical record,不带外层数组,不手写缩进 JSON。只有用户明确要求兼容旧文件时,才使用 `output_format="json"` 导出紧凑 JSON 数组。
canonical records 落盘必须使用 `data_agent_export_dataset_records`,默认格式是紧凑 JSONL:一行一个 canonical record,不带外层数组,不手写缩进 JSON。默认 `output_path` 固定传 `output/records.jsonl`;不要传 `output/<数据集名>/records.jsonl`,不要传 `tasks/...`,不要自定义文件名。只有用户明确要求兼容旧文件时,才使用 `output_format="json"` 导出紧凑 JSON 数组,此时工具会归一为 `output/records.json`
当前 canonical record v1 工作格式:
@@ -258,7 +258,7 @@ canonical records 落盘必须使用 `data_agent_export_dataset_records`,默
"source_refs": []
}
],
"plan_hint": "建议先生成 50 条单轮,输出到 tasks/<数据集名称>/records.jsonl;具体数量轮次和路径在 generation plan 中确认。",
"plan_hint": "建议先生成 50 条单轮,输出到 output/records.jsonl;具体数量轮次在 generation plan 中确认。",
"coverage": "需要覆盖的 query 语义、功能点、错误类型",
"exclusions": "不要生成或需要排除的表达",
"open_questions": [],
@@ -266,7 +266,7 @@ canonical records 落盘必须使用 `data_agent_export_dataset_records`,默
}
```
`generation_goal` 只确认“做什么数据、为什么做、标签边界是什么”。不要在 goal 中维护结构化的 `total_count``turn_mix``output_path`;这些字段属于后续 `generation_plan`。如果需要在 goal review 阶段提示执行方向,只写一句 `plan_hint`,例如“建议先生成 50 条单轮,输出到 tasks/.../records.jsonl;具体数量轮次和路径在 generation plan 中确认”。
`generation_goal` 只确认“做什么数据、为什么做、标签边界是什么”。不要在 goal 中维护结构化的 `total_count``turn_mix``output_path`;这些字段属于后续 `generation_plan`。如果需要在 goal review 阶段提示执行方向,只写一句 `plan_hint`,例如“建议先生成 50 条单轮,输出到 output/records.jsonl;具体数量轮次在 generation plan 中确认”。
如果未来增加 `data_agent_validate_generation_goal`,它只做结构校验和缺失字段提示,不做语义判断,不替代用户 review,也不替代 `data_agent_prepare_generation_plan`
@@ -307,13 +307,13 @@ canonical records 落盘必须使用 `data_agent_export_dataset_records`,默
- `data_agent_confirm_generation_plan`:用户明确确认当前计划版本后使用,获取 `confirmed_plan_id`
- `data_agent_normalize_dataset_draft`:用户确认计划后,把 dataset draft text v1 转成 canonical records;必须传入 `confirmed_plan_id`
- `data_agent_validate_dataset_records`:对 canonical records 做结构、标签、时间戳和多轮上下文校验。
- `data_agent_export_dataset_records`:校验 canonical records 并落盘;默认写紧凑 JSONL,一行一条,不要再用 `write_file` 手写 records 文件。
- `data_agent_export_dataset_records`:校验 canonical records 并落盘;默认写紧凑 JSONL,一行一条,固定传 `output/records.jsonl`不要再用 `write_file` 手写 records 文件。
## 约束
- 不要静默解决产品或标签歧义。
- 如果 `ask_user_question` 不可用,使用普通回复向用户提问并停止,不要自己替用户确认。
- canonical records 通过校验前,不要生成最终导出格式。
- canonical records 需要落盘时,必须用 `data_agent_export_dataset_records`;不要自己拼接 JSON/JSONL。
- canonical records 需要落盘时,必须用 `data_agent_export_dataset_records`;不要自己拼接 JSON/JSONL;不要创建数据集子目录或自定义 records 文件名
- 除非用户明确要求,否则不要把“修改标签定义”和“生成数据”混在一起做。
- 不要因为用户说“生成一些数据”就跳过边界总结和 generation plan review。
+34 -2
View File
@@ -571,16 +571,48 @@ target: Agent(tag="life_service")
export_result = execute_tool(
registry,
'data_agent_export_dataset_records',
{'records': records, 'output_path': 'output/demo/records.jsonl'},
{'records': records, 'output_path': 'output/地图和生活边界数据/自定义名字.jsonl'},
context,
)
self.assertTrue(export_result.ok, export_result.content)
export_payload = json.loads(export_result.content)
expected = '.port_sessions/accounts/alice/sessions/s1/output/demo/records.jsonl'
expected = '.port_sessions/accounts/alice/sessions/s1/output/records.jsonl'
self.assertEqual(plan_payload['plan']['output_path'], expected)
self.assertEqual(export_payload['output_path'], expected)
def test_data_agent_session_records_path_stays_stable_when_file_exists(self) -> None:
with tempfile.TemporaryDirectory() as tmp_dir:
root = Path(tmp_dir)
scratchpad = root / '.port_sessions' / 'accounts' / 'alice' / 'sessions' / 's1' / 'scratchpad'
output_root = scratchpad.parent / 'output'
output_root.mkdir(parents=True)
(output_root / 'records.jsonl').write_text('{}\n', encoding='utf-8')
context = build_tool_context(AgentRuntimeConfig(cwd=root), scratchpad_directory=scratchpad)
plan_result = execute_tool(
default_tool_registry(),
'data_agent_prepare_generation_plan',
{
'direct_review': True,
'dataset_label': '地图和生活边界数据',
'target': 'Agent(tag="life_service")',
'total_count': 1,
'turn_mix': '1 条单轮',
'coverage': '附近吃喝玩乐',
'exclusions': '不要生成导航路线类 query',
'output_path': 'output/任意子目录/任意名字.jsonl',
},
context,
)
self.assertTrue(plan_result.ok, plan_result.content)
plan_payload = json.loads(plan_result.content)
self.assertEqual(
plan_payload['plan']['output_path'],
'.port_sessions/accounts/alice/sessions/s1/output/records.jsonl',
)
if __name__ == '__main__':
unittest.main()