import { describe, expect, it } from 'vitest'; import { createTrainingExample } from '../src/training-data.js'; describe('training data normalization', () => { it('exports Chat Completions histories, reasoning, tool calls, and tool results', () => { const example = createTrainingExample({ endpoint: '/oai/v1/chat/completions', responseStatus: 200, requestBody: { model: 'model-a', temperature: 0.2, max_tokens: 200, tool_choice: 'auto', metadata: { dataset: 'test' }, tools: [{ type: 'function', function: { name: 'weather', description: 'Get weather', parameters: { type: 'object' } } }], messages: [ { role: 'system', content: 'Be useful' }, { role: 'user', content: 'Find the weather' }, { role: 'assistant', content: null, tool_calls: [{ id: 'call_1', type: 'function', function: { name: 'weather', arguments: '{"city":"Oslo"}' } }] }, { role: 'tool', tool_call_id: 'call_1', content: '{"temp":12}' }, { role: 'user', content: 'Summarize it' }, ], }, responseBody: JSON.stringify({ id: 'chatcmpl_1', model: 'model-a', usage: { prompt_tokens: 20, completion_tokens: 8 }, choices: [{ finish_reason: 'tool_calls', message: { role: 'assistant', reasoning_content: 'The result says 12 degrees.', content: 'It is 12°C in Oslo.', tool_calls: [{ id: 'call_2', type: 'function', function: { name: 'log', arguments: '{"temp":12}' } }], } }] }), durationMs: 123.5, }); expect(example.messages).toHaveLength(6); expect(example.messages[2].tool_calls[0]).toMatchObject({ id: 'call_1', function: { name: 'weather' } }); expect(example.messages[3]).toEqual({ role: 'tool', tool_call_id: 'call_1', content: '{"temp":12}' }); expect(example.messages.at(-1)).toEqual({ role: 'assistant', content: 'It is 12°C in Oslo.', reasoning_content: 'The result says 12 degrees.', tool_calls: [{ id: 'call_2', type: 'function', function: { name: 'log', arguments: '{"temp":12}' } }], }); expect(example.metadata).toEqual({ schema_version: 1, api: 'openai_chat_completions', endpoint: '/oai/v1/chat/completions', model: 'model-a', stream: false, tools: [{ type: 'function', function: { name: 'weather', description: 'Get weather', parameters: { type: 'object' } } }], tool_choice: 'auto', request_metadata: { dataset: 'test' }, parameters: { temperature: 0.2, max_tokens: 200 }, response_status: 200, duration_ms: 123.5, response: { id: 'chatcmpl_1', model: 'model-a', finish_reason: 'tool_calls', usage: { prompt_tokens: 20, completion_tokens: 8 } }, }); }); it('reconstructs streamed Chat Completions output', () => { const responseBody = [ 'data: {"choices":[{"delta":{"reasoning_content":"check "}}]}', 'data: {"choices":[{"delta":{"content":"Done","tool_calls":[{"index":0,"id":"call_1","function":{"name":"save","arguments":"{\\"ok\\":"}}]}}]}', 'data: {"choices":[{"delta":{"tool_calls":[{"index":0,"function":{"arguments":"true}"}}]},"finish_reason":"tool_calls"}]}', 'data: [DONE]', '', ].join('\n\n'); const example = createTrainingExample({ endpoint: '/oai/v1/chat/completions', responseStatus: 200, stream: true, requestBody: { messages: [{ role: 'user', content: 'Do it' }] }, responseBody, }); expect(example.messages.at(-1)).toEqual({ role: 'assistant', content: 'Done', reasoning_content: 'check ', tool_calls: [{ id: 'call_1', type: 'function', function: { name: 'save', arguments: '{"ok":true}' } }], }); }); it('normalizes Responses input and output items', () => { const example = createTrainingExample({ endpoint: '/oai/v1/responses', responseStatus: 200, requestBody: { model: 'model-r', instructions: 'Use tools', tools: [{ type: 'function', name: 'read', parameters: { type: 'object' } }, { type: 'web_search_preview' }], input: [ { type: 'message', role: 'user', content: [{ type: 'input_text', text: 'Read file' }] }, { type: 'reasoning', summary: [{ type: 'summary_text', text: 'Need the file.' }] }, { type: 'function_call', call_id: 'call_read', name: 'read', arguments: '{"path":"a.txt"}' }, { type: 'function_call_output', call_id: 'call_read', output: 'hello' }, ], }, responseBody: JSON.stringify({ output: [ { type: 'reasoning', summary: [{ type: 'summary_text', text: 'The file says hello.' }] }, { type: 'message', role: 'assistant', content: [{ type: 'output_text', text: 'It says hello.' }] }, { type: 'function_call', call_id: 'call_log', name: 'log', arguments: '{"value":"hello"}' }, ] }), }); expect(example.messages).toEqual([ { role: 'system', content: 'Use tools' }, { role: 'user', content: [{ type: 'text', text: 'Read file' }] }, { role: 'assistant', content: null, reasoning_content: 'Need the file.', tool_calls: [{ id: 'call_read', type: 'function', function: { name: 'read', arguments: '{"path":"a.txt"}' } }] }, { role: 'tool', tool_call_id: 'call_read', content: 'hello' }, { role: 'assistant', content: 'It says hello.', reasoning_content: 'The file says hello.', tool_calls: [{ id: 'call_log', type: 'function', function: { name: 'log', arguments: '{"value":"hello"}' } }] }, ]); expect(example.metadata.tools).toEqual([ { type: 'function', name: 'read', parameters: { type: 'object' } }, { type: 'web_search_preview' }, ]); expect(example.metadata).toMatchObject({ api: 'openai_responses', model: 'model-r', stream: false, response_status: 200 }); }); it('uses the completed Responses streaming event as the assistant output', () => { const responseBody = [ 'event: response.completed', 'data: {"type":"response.completed","response":{"output":[{"type":"reasoning","summary":[{"type":"summary_text","text":"Need a lookup."}]},{"type":"message","role":"assistant","content":[{"type":"output_text","text":"Found it."}]},{"type":"function_call","call_id":"call_1","name":"lookup","arguments":"{\\"q\\":\\"x\\"}"}]}}', '', '', ].join('\n'); const example = createTrainingExample({ endpoint: '/oai/v1/responses', responseStatus: 200, stream: true, requestBody: { input: 'Find x' }, responseBody, }); expect(example.messages.at(-1)).toEqual({ role: 'assistant', content: 'Found it.', reasoning_content: 'Need a lookup.', tool_calls: [{ id: 'call_1', type: 'function', function: { name: 'lookup', arguments: '{"q":"x"}' } }], }); }); it('normalizes Anthropic thinking, tool use, and tool results', () => { const example = createTrainingExample({ endpoint: '/ant/v1/messages', responseStatus: 200, requestBody: { model: 'model-ant', system: 'Be useful', tools: [{ name: 'calculator', description: 'Calculate', input_schema: { type: 'object' } }], messages: [ { role: 'user', content: 'Calculate' }, { role: 'assistant', content: [{ type: 'thinking', thinking: 'Need calculator.' }, { type: 'tool_use', id: 'tool_1', name: 'calculator', input: { expression: '2+2' } }] }, { role: 'user', content: [{ type: 'tool_result', tool_use_id: 'tool_1', content: '4' }] }, ], }, responseBody: JSON.stringify({ content: [ { type: 'thinking', thinking: 'The result is four.' }, { type: 'text', text: 'The answer is 4.' }, { type: 'tool_use', id: 'tool_2', name: 'record', input: { answer: 4 } }, ] }), }); expect(example.messages).toEqual([ { role: 'system', content: 'Be useful' }, { role: 'user', content: 'Calculate' }, { role: 'assistant', content: null, reasoning_content: 'Need calculator.', tool_calls: [{ id: 'tool_1', type: 'function', function: { name: 'calculator', arguments: '{"expression":"2+2"}' } }] }, { role: 'tool', tool_call_id: 'tool_1', content: '4' }, { role: 'assistant', content: 'The answer is 4.', reasoning_content: 'The result is four.', tool_calls: [{ id: 'tool_2', type: 'function', function: { name: 'record', arguments: '{"answer":4}' } }] }, ]); expect(example.metadata.tools).toEqual([ { name: 'calculator', description: 'Calculate', input_schema: { type: 'object' } }, ]); expect(example.metadata).toMatchObject({ api: 'anthropic_messages', model: 'model-ant', stream: false, response_status: 200 }); }); it('reconstructs streamed Anthropic content blocks', () => { const responseBody = [ 'event: content_block_start\ndata: {"type":"content_block_start","index":0,"content_block":{"type":"thinking","thinking":""}}', 'event: content_block_delta\ndata: {"type":"content_block_delta","index":0,"delta":{"type":"thinking_delta","thinking":"Need tool."}}', 'event: content_block_start\ndata: {"type":"content_block_start","index":1,"content_block":{"type":"text","text":""}}', 'event: content_block_delta\ndata: {"type":"content_block_delta","index":1,"delta":{"type":"text_delta","text":"Working"}}', 'event: content_block_start\ndata: {"type":"content_block_start","index":2,"content_block":{"type":"tool_use","id":"tool_1","name":"run","input":{}}}', 'event: content_block_delta\ndata: {"type":"content_block_delta","index":2,"delta":{"type":"input_json_delta","partial_json":"{\\"x\\":1}"}}', '', ].join('\n\n'); const example = createTrainingExample({ endpoint: '/ant/v1/messages', responseStatus: 200, stream: true, requestBody: { messages: [{ role: 'user', content: 'Work' }] }, responseBody, }); expect(example.messages.at(-1)).toEqual({ role: 'assistant', content: 'Working', reasoning_content: 'Need tool.', tool_calls: [{ id: 'tool_1', type: 'function', function: { name: 'run', arguments: '{"x":1}' } }], }); }); it('excludes failed and non-chat requests', () => { expect(createTrainingExample({ endpoint: '/oai/v1/models', responseStatus: 200, requestBody: {} })).toBeNull(); expect(createTrainingExample({ endpoint: '/oai/v1/chat/completions', responseStatus: 500, requestBody: { messages: [] } })).toBeNull(); }); });