adding data collection
This commit is contained in:
@@ -0,0 +1,168 @@
|
||||
import { describe, expect, it } from 'vitest';
|
||||
import { createTrainingExample } from '../src/training-data.js';
|
||||
|
||||
describe('training data normalization', () => {
|
||||
it('exports Chat Completions histories, reasoning, tool calls, and tool results', () => {
|
||||
const example = createTrainingExample({
|
||||
endpoint: '/oai/v1/chat/completions', responseStatus: 200,
|
||||
requestBody: {
|
||||
model: 'model-a', temperature: 0.2, max_tokens: 200, tool_choice: 'auto', metadata: { dataset: 'test' },
|
||||
tools: [{ type: 'function', function: { name: 'weather', description: 'Get weather', parameters: { type: 'object' } } }],
|
||||
messages: [
|
||||
{ role: 'system', content: 'Be useful' },
|
||||
{ role: 'user', content: 'Find the weather' },
|
||||
{ role: 'assistant', content: null, tool_calls: [{ id: 'call_1', type: 'function', function: { name: 'weather', arguments: '{"city":"Oslo"}' } }] },
|
||||
{ role: 'tool', tool_call_id: 'call_1', content: '{"temp":12}' },
|
||||
{ role: 'user', content: 'Summarize it' },
|
||||
],
|
||||
},
|
||||
responseBody: JSON.stringify({ id: 'chatcmpl_1', model: 'model-a', usage: { prompt_tokens: 20, completion_tokens: 8 }, choices: [{ finish_reason: 'tool_calls', message: {
|
||||
role: 'assistant', reasoning_content: 'The result says 12 degrees.', content: 'It is 12°C in Oslo.',
|
||||
tool_calls: [{ id: 'call_2', type: 'function', function: { name: 'log', arguments: '{"temp":12}' } }],
|
||||
} }] }),
|
||||
durationMs: 123.5,
|
||||
});
|
||||
|
||||
expect(example.messages).toHaveLength(6);
|
||||
expect(example.messages[2].tool_calls[0]).toMatchObject({ id: 'call_1', function: { name: 'weather' } });
|
||||
expect(example.messages[3]).toEqual({ role: 'tool', tool_call_id: 'call_1', content: '{"temp":12}' });
|
||||
expect(example.messages.at(-1)).toEqual({
|
||||
role: 'assistant', content: 'It is 12°C in Oslo.', reasoning_content: 'The result says 12 degrees.',
|
||||
tool_calls: [{ id: 'call_2', type: 'function', function: { name: 'log', arguments: '{"temp":12}' } }],
|
||||
});
|
||||
expect(example.metadata).toEqual({
|
||||
schema_version: 1,
|
||||
api: 'openai_chat_completions', endpoint: '/oai/v1/chat/completions', model: 'model-a', stream: false,
|
||||
tools: [{ type: 'function', function: { name: 'weather', description: 'Get weather', parameters: { type: 'object' } } }],
|
||||
tool_choice: 'auto', request_metadata: { dataset: 'test' }, parameters: { temperature: 0.2, max_tokens: 200 }, response_status: 200, duration_ms: 123.5,
|
||||
response: { id: 'chatcmpl_1', model: 'model-a', finish_reason: 'tool_calls', usage: { prompt_tokens: 20, completion_tokens: 8 } },
|
||||
});
|
||||
});
|
||||
|
||||
it('reconstructs streamed Chat Completions output', () => {
|
||||
const responseBody = [
|
||||
'data: {"choices":[{"delta":{"reasoning_content":"check "}}]}',
|
||||
'data: {"choices":[{"delta":{"content":"Done","tool_calls":[{"index":0,"id":"call_1","function":{"name":"save","arguments":"{\\"ok\\":"}}]}}]}',
|
||||
'data: {"choices":[{"delta":{"tool_calls":[{"index":0,"function":{"arguments":"true}"}}]},"finish_reason":"tool_calls"}]}',
|
||||
'data: [DONE]', '',
|
||||
].join('\n\n');
|
||||
const example = createTrainingExample({
|
||||
endpoint: '/oai/v1/chat/completions', responseStatus: 200, stream: true,
|
||||
requestBody: { messages: [{ role: 'user', content: 'Do it' }] }, responseBody,
|
||||
});
|
||||
|
||||
expect(example.messages.at(-1)).toEqual({
|
||||
role: 'assistant', content: 'Done', reasoning_content: 'check ',
|
||||
tool_calls: [{ id: 'call_1', type: 'function', function: { name: 'save', arguments: '{"ok":true}' } }],
|
||||
});
|
||||
});
|
||||
|
||||
it('normalizes Responses input and output items', () => {
|
||||
const example = createTrainingExample({
|
||||
endpoint: '/oai/v1/responses', responseStatus: 200,
|
||||
requestBody: {
|
||||
model: 'model-r', instructions: 'Use tools',
|
||||
tools: [{ type: 'function', name: 'read', parameters: { type: 'object' } }, { type: 'web_search_preview' }],
|
||||
input: [
|
||||
{ type: 'message', role: 'user', content: [{ type: 'input_text', text: 'Read file' }] },
|
||||
{ type: 'reasoning', summary: [{ type: 'summary_text', text: 'Need the file.' }] },
|
||||
{ type: 'function_call', call_id: 'call_read', name: 'read', arguments: '{"path":"a.txt"}' },
|
||||
{ type: 'function_call_output', call_id: 'call_read', output: 'hello' },
|
||||
],
|
||||
},
|
||||
responseBody: JSON.stringify({ output: [
|
||||
{ type: 'reasoning', summary: [{ type: 'summary_text', text: 'The file says hello.' }] },
|
||||
{ type: 'message', role: 'assistant', content: [{ type: 'output_text', text: 'It says hello.' }] },
|
||||
{ type: 'function_call', call_id: 'call_log', name: 'log', arguments: '{"value":"hello"}' },
|
||||
] }),
|
||||
});
|
||||
|
||||
expect(example.messages).toEqual([
|
||||
{ role: 'system', content: 'Use tools' },
|
||||
{ role: 'user', content: [{ type: 'text', text: 'Read file' }] },
|
||||
{ role: 'assistant', content: null, reasoning_content: 'Need the file.', tool_calls: [{ id: 'call_read', type: 'function', function: { name: 'read', arguments: '{"path":"a.txt"}' } }] },
|
||||
{ role: 'tool', tool_call_id: 'call_read', content: 'hello' },
|
||||
{ role: 'assistant', content: 'It says hello.', reasoning_content: 'The file says hello.', tool_calls: [{ id: 'call_log', type: 'function', function: { name: 'log', arguments: '{"value":"hello"}' } }] },
|
||||
]);
|
||||
expect(example.metadata.tools).toEqual([
|
||||
{ type: 'function', name: 'read', parameters: { type: 'object' } },
|
||||
{ type: 'web_search_preview' },
|
||||
]);
|
||||
expect(example.metadata).toMatchObject({ api: 'openai_responses', model: 'model-r', stream: false, response_status: 200 });
|
||||
});
|
||||
|
||||
it('uses the completed Responses streaming event as the assistant output', () => {
|
||||
const responseBody = [
|
||||
'event: response.completed',
|
||||
'data: {"type":"response.completed","response":{"output":[{"type":"reasoning","summary":[{"type":"summary_text","text":"Need a lookup."}]},{"type":"message","role":"assistant","content":[{"type":"output_text","text":"Found it."}]},{"type":"function_call","call_id":"call_1","name":"lookup","arguments":"{\\"q\\":\\"x\\"}"}]}}',
|
||||
'', '',
|
||||
].join('\n');
|
||||
const example = createTrainingExample({
|
||||
endpoint: '/oai/v1/responses', responseStatus: 200, stream: true,
|
||||
requestBody: { input: 'Find x' }, responseBody,
|
||||
});
|
||||
|
||||
expect(example.messages.at(-1)).toEqual({
|
||||
role: 'assistant', content: 'Found it.', reasoning_content: 'Need a lookup.',
|
||||
tool_calls: [{ id: 'call_1', type: 'function', function: { name: 'lookup', arguments: '{"q":"x"}' } }],
|
||||
});
|
||||
});
|
||||
|
||||
it('normalizes Anthropic thinking, tool use, and tool results', () => {
|
||||
const example = createTrainingExample({
|
||||
endpoint: '/ant/v1/messages', responseStatus: 200,
|
||||
requestBody: {
|
||||
model: 'model-ant', system: 'Be useful',
|
||||
tools: [{ name: 'calculator', description: 'Calculate', input_schema: { type: 'object' } }],
|
||||
messages: [
|
||||
{ role: 'user', content: 'Calculate' },
|
||||
{ role: 'assistant', content: [{ type: 'thinking', thinking: 'Need calculator.' }, { type: 'tool_use', id: 'tool_1', name: 'calculator', input: { expression: '2+2' } }] },
|
||||
{ role: 'user', content: [{ type: 'tool_result', tool_use_id: 'tool_1', content: '4' }] },
|
||||
],
|
||||
},
|
||||
responseBody: JSON.stringify({ content: [
|
||||
{ type: 'thinking', thinking: 'The result is four.' },
|
||||
{ type: 'text', text: 'The answer is 4.' },
|
||||
{ type: 'tool_use', id: 'tool_2', name: 'record', input: { answer: 4 } },
|
||||
] }),
|
||||
});
|
||||
|
||||
expect(example.messages).toEqual([
|
||||
{ role: 'system', content: 'Be useful' },
|
||||
{ role: 'user', content: 'Calculate' },
|
||||
{ role: 'assistant', content: null, reasoning_content: 'Need calculator.', tool_calls: [{ id: 'tool_1', type: 'function', function: { name: 'calculator', arguments: '{"expression":"2+2"}' } }] },
|
||||
{ role: 'tool', tool_call_id: 'tool_1', content: '4' },
|
||||
{ role: 'assistant', content: 'The answer is 4.', reasoning_content: 'The result is four.', tool_calls: [{ id: 'tool_2', type: 'function', function: { name: 'record', arguments: '{"answer":4}' } }] },
|
||||
]);
|
||||
expect(example.metadata.tools).toEqual([
|
||||
{ name: 'calculator', description: 'Calculate', input_schema: { type: 'object' } },
|
||||
]);
|
||||
expect(example.metadata).toMatchObject({ api: 'anthropic_messages', model: 'model-ant', stream: false, response_status: 200 });
|
||||
});
|
||||
|
||||
it('reconstructs streamed Anthropic content blocks', () => {
|
||||
const responseBody = [
|
||||
'event: content_block_start\ndata: {"type":"content_block_start","index":0,"content_block":{"type":"thinking","thinking":""}}',
|
||||
'event: content_block_delta\ndata: {"type":"content_block_delta","index":0,"delta":{"type":"thinking_delta","thinking":"Need tool."}}',
|
||||
'event: content_block_start\ndata: {"type":"content_block_start","index":1,"content_block":{"type":"text","text":""}}',
|
||||
'event: content_block_delta\ndata: {"type":"content_block_delta","index":1,"delta":{"type":"text_delta","text":"Working"}}',
|
||||
'event: content_block_start\ndata: {"type":"content_block_start","index":2,"content_block":{"type":"tool_use","id":"tool_1","name":"run","input":{}}}',
|
||||
'event: content_block_delta\ndata: {"type":"content_block_delta","index":2,"delta":{"type":"input_json_delta","partial_json":"{\\"x\\":1}"}}',
|
||||
'',
|
||||
].join('\n\n');
|
||||
const example = createTrainingExample({
|
||||
endpoint: '/ant/v1/messages', responseStatus: 200, stream: true,
|
||||
requestBody: { messages: [{ role: 'user', content: 'Work' }] }, responseBody,
|
||||
});
|
||||
|
||||
expect(example.messages.at(-1)).toEqual({
|
||||
role: 'assistant', content: 'Working', reasoning_content: 'Need tool.',
|
||||
tool_calls: [{ id: 'tool_1', type: 'function', function: { name: 'run', arguments: '{"x":1}' } }],
|
||||
});
|
||||
});
|
||||
|
||||
it('excludes failed and non-chat requests', () => {
|
||||
expect(createTrainingExample({ endpoint: '/oai/v1/models', responseStatus: 200, requestBody: {} })).toBeNull();
|
||||
expect(createTrainingExample({ endpoint: '/oai/v1/chat/completions', responseStatus: 500, requestBody: { messages: [] } })).toBeNull();
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user