From b04b49a397facc6168baa59af3d329fae3bbb125 Mon Sep 17 00:00:00 2001 From: Edson Date: Thu, 1 Oct 2026 02:33:31 -0400 Subject: [PATCH] fix(dataset): align Anthropic tool results with tool_use order Anthropic pairs tool_result blocks with tool_use blocks by tool_use_id, so a parallel result batch may arrive in a different order from the calls. anthropic_to_messages discarded the ids and emitted positional tool_response messages in arrival order, pairing each result with the wrong call. Align a complete result batch (one result per tool_use, distinct ids) to the tool_use order before the ids are discarded, as normalize_openai_tool_calls does for OpenAI tool_calls (#10174). The blocks are reordered rather than the emitted messages, so images inside the results stay consistent with their placeholders. Incomplete or id-less batches are left as-is. --- swift/dataset/preprocessor/core.py | 21 +++++ tests/general/test_data_preprocess.py | 124 ++++++++++++++++++++++++++ 2 files changed, 145 insertions(+) diff --git a/swift/dataset/preprocessor/core.py b/swift/dataset/preprocessor/core.py index e5b72ce823..93a05d8981 100644 --- a/swift/dataset/preprocessor/core.py +++ b/swift/dataset/preprocessor/core.py @@ -568,6 +568,22 @@ def _anthropic_block_content(cls, content: Any, media: Dict[str, List[Any]]) -> raise ValueError(f'Unsupported Anthropic content block type: {block_type}') return ''.join(parts) + @staticmethod + def _align_anthropic_tool_results(tool_uses: List[Dict[str, Any]], + content: List[Dict[str, Any]]) -> List[Dict[str, Any]]: + # Canonical tool responses are positional, so align a complete result batch + # before the tool_use IDs are discarded. Keep incomplete or ID-less batches as-is. + call_order = {block.get('id'): i for i, block in enumerate(tool_uses)} + end = 0 + while end < len(content) and content[end].get('type') == 'tool_result': + end += 1 + results = content[:end] + if (None not in call_order and len(call_order) == len(tool_uses) == len(results) + and set(call_order) == {result.get('tool_use_id') + for result in results}): + content = sorted(results, key=lambda result: call_order[result['tool_use_id']]) + content[end:] + return content + @classmethod def anthropic_to_messages(cls, messages: List[Dict[str, Any]], @@ -575,11 +591,16 @@ def anthropic_to_messages(cls, """Convert Anthropic content blocks to the SWIFT canonical roles.""" media = media if media is not None else {'images': []} new_messages = [] + tool_uses = [] for message in messages: content = message.get('content') if not isinstance(content, list): new_messages.append(message) + tool_uses = [] continue + if tool_uses: + content = cls._align_anthropic_tool_results(tool_uses, content) + tool_uses = [block for block in content if block.get('type') == 'tool_use'] pending_content = [] message_metadata = {key: message[key] for key in ['loss', 'loss_scale'] if key in message} diff --git a/tests/general/test_data_preprocess.py b/tests/general/test_data_preprocess.py index 19d9813722..1f55ec8e6f 100644 --- a/tests/general/test_data_preprocess.py +++ b/tests/general/test_data_preprocess.py @@ -467,6 +467,130 @@ def test_anthropic_multimodal_content_blocks(self): self.assertEqual(template_inputs.images, result['images']) self.assertEqual(template_inputs.messages[-1]['content'], 'A sunny beach.') + def test_anthropic_parallel_tool_results_out_of_order(self): + # Anthropic pairs `tool_result` blocks with `tool_use` blocks by `tool_use_id`, not by + # position, so results may arrive in a different order from the calls. Canonical + # `tool_response` messages are positional (the IDs are discarded), so the results must + # be aligned to the call order first, as `normalize_openai_tool_calls` does for OpenAI + # `tool_calls` (#10174). + tool_uses = [{ + 'type': 'tool_use', + 'id': 'toolu_beijing', + 'name': 'get_weather', + 'input': { + 'city': 'Beijing' + }, + }, { + 'type': 'tool_use', + 'id': 'toolu_shanghai', + 'name': 'get_weather', + 'input': { + 'city': 'Shanghai' + }, + }] + results = { + 'toolu_beijing': { + 'type': 'tool_result', + 'tool_use_id': 'toolu_beijing', + 'content': 'sunny' + }, + 'toolu_shanghai': { + 'type': 'tool_result', + 'tool_use_id': 'toolu_shanghai', + 'content': 'rainy' + }, + } + expected = [{ + 'role': 'tool_call', + 'content': { + 'name': 'get_weather', + 'arguments': { + 'city': 'Beijing' + } + }, + }, { + 'role': 'tool_call', + 'content': { + 'name': 'get_weather', + 'arguments': { + 'city': 'Shanghai' + } + }, + }, { + 'role': 'tool_response', + 'content': 'sunny' + }, { + 'role': 'tool_response', + 'content': 'rainy' + }] + for order in (['toolu_beijing', 'toolu_shanghai'], ['toolu_shanghai', 'toolu_beijing']): + with self.subTest(result_order=order): + row = { + 'messages': [{ + 'role': 'assistant', + 'content': tool_uses, + }, { + 'role': 'user', + 'content': [results[tool_use_id] for tool_use_id in order], + }] + } + result = AnthropicMessagesPreprocessor().preprocess(row) + self.assertEqual(result['messages'], expected) + + def test_anthropic_out_of_order_tool_results_keep_images_aligned(self): + # Reordering the result blocks (not the emitted messages) keeps `images` consistent + # with the `` placeholders, and a trailing user text block stays after the results. + def image_result(name): + return { + 'type': + 'tool_result', + 'tool_use_id': + f'toolu_{name}', + 'content': [{ + 'type': 'image', + 'source': { + 'type': 'url', + 'url': f'https://example.com/{name}.png' + } + }, { + 'type': 'text', + 'text': name + }], + } + + row = { + 'messages': [{ + 'role': + 'assistant', + 'content': [{ + 'type': 'tool_use', + 'id': f'toolu_{name}', + 'name': 'inspect_image', + 'input': {} + } for name in ['first', 'second']], + }, { + 'role': + 'user', + 'content': [image_result('second'), + image_result('first'), { + 'type': 'text', + 'text': 'Thanks.' + }], + }] + } + result = AnthropicMessagesPreprocessor().preprocess(row) + self.assertEqual(result['messages'][2:], [{ + 'role': 'tool_response', + 'content': 'first', + }, { + 'role': 'tool_response', + 'content': 'second', + }, { + 'role': 'user', + 'content': 'Thanks.', + }]) + self.assertEqual(result['images'], ['https://example.com/first.png', 'https://example.com/second.png']) + if __name__ == '__main__': unittest.main()