Files
opencode-proxy/test/training-data.test.js
T
2026-07-21 13:53:06 -05:00

169 lines
10 KiB
JavaScript

import { describe, expect, it } from 'vitest';
import { createTrainingExample } from '../src/training-data.js';
describe('training data normalization', () => {
it('exports Chat Completions histories, reasoning, tool calls, and tool results', () => {
const example = createTrainingExample({
endpoint: '/oai/v1/chat/completions', responseStatus: 200,
requestBody: {
model: 'model-a', temperature: 0.2, max_tokens: 200, tool_choice: 'auto', metadata: { dataset: 'test' },
tools: [{ type: 'function', function: { name: 'weather', description: 'Get weather', parameters: { type: 'object' } } }],
messages: [
{ role: 'system', content: 'Be useful' },
{ role: 'user', content: 'Find the weather' },
{ role: 'assistant', content: null, tool_calls: [{ id: 'call_1', type: 'function', function: { name: 'weather', arguments: '{"city":"Oslo"}' } }] },
{ role: 'tool', tool_call_id: 'call_1', content: '{"temp":12}' },
{ role: 'user', content: 'Summarize it' },
],
},
responseBody: JSON.stringify({ id: 'chatcmpl_1', model: 'model-a', usage: { prompt_tokens: 20, completion_tokens: 8 }, choices: [{ finish_reason: 'tool_calls', message: {
role: 'assistant', reasoning_content: 'The result says 12 degrees.', content: 'It is 12°C in Oslo.',
tool_calls: [{ id: 'call_2', type: 'function', function: { name: 'log', arguments: '{"temp":12}' } }],
} }] }),
durationMs: 123.5,
});
expect(example.messages).toHaveLength(6);
expect(example.messages[2].tool_calls[0]).toMatchObject({ id: 'call_1', function: { name: 'weather' } });
expect(example.messages[3]).toEqual({ role: 'tool', tool_call_id: 'call_1', content: '{"temp":12}' });
expect(example.messages.at(-1)).toEqual({
role: 'assistant', content: 'It is 12°C in Oslo.', reasoning_content: 'The result says 12 degrees.',
tool_calls: [{ id: 'call_2', type: 'function', function: { name: 'log', arguments: '{"temp":12}' } }],
});
expect(example.metadata).toEqual({
schema_version: 1,
api: 'openai_chat_completions', endpoint: '/oai/v1/chat/completions', model: 'model-a', stream: false,
tools: [{ type: 'function', function: { name: 'weather', description: 'Get weather', parameters: { type: 'object' } } }],
tool_choice: 'auto', request_metadata: { dataset: 'test' }, parameters: { temperature: 0.2, max_tokens: 200 }, response_status: 200, duration_ms: 123.5,
response: { id: 'chatcmpl_1', model: 'model-a', finish_reason: 'tool_calls', usage: { prompt_tokens: 20, completion_tokens: 8 } },
});
});
it('reconstructs streamed Chat Completions output', () => {
const responseBody = [
'data: {"choices":[{"delta":{"reasoning_content":"check "}}]}',
'data: {"choices":[{"delta":{"content":"Done","tool_calls":[{"index":0,"id":"call_1","function":{"name":"save","arguments":"{\\"ok\\":"}}]}}]}',
'data: {"choices":[{"delta":{"tool_calls":[{"index":0,"function":{"arguments":"true}"}}]},"finish_reason":"tool_calls"}]}',
'data: [DONE]', '',
].join('\n\n');
const example = createTrainingExample({
endpoint: '/oai/v1/chat/completions', responseStatus: 200, stream: true,
requestBody: { messages: [{ role: 'user', content: 'Do it' }] }, responseBody,
});
expect(example.messages.at(-1)).toEqual({
role: 'assistant', content: 'Done', reasoning_content: 'check ',
tool_calls: [{ id: 'call_1', type: 'function', function: { name: 'save', arguments: '{"ok":true}' } }],
});
});
it('normalizes Responses input and output items', () => {
const example = createTrainingExample({
endpoint: '/oai/v1/responses', responseStatus: 200,
requestBody: {
model: 'model-r', instructions: 'Use tools',
tools: [{ type: 'function', name: 'read', parameters: { type: 'object' } }, { type: 'web_search_preview' }],
input: [
{ type: 'message', role: 'user', content: [{ type: 'input_text', text: 'Read file' }] },
{ type: 'reasoning', summary: [{ type: 'summary_text', text: 'Need the file.' }] },
{ type: 'function_call', call_id: 'call_read', name: 'read', arguments: '{"path":"a.txt"}' },
{ type: 'function_call_output', call_id: 'call_read', output: 'hello' },
],
},
responseBody: JSON.stringify({ output: [
{ type: 'reasoning', summary: [{ type: 'summary_text', text: 'The file says hello.' }] },
{ type: 'message', role: 'assistant', content: [{ type: 'output_text', text: 'It says hello.' }] },
{ type: 'function_call', call_id: 'call_log', name: 'log', arguments: '{"value":"hello"}' },
] }),
});
expect(example.messages).toEqual([
{ role: 'system', content: 'Use tools' },
{ role: 'user', content: [{ type: 'text', text: 'Read file' }] },
{ role: 'assistant', content: null, reasoning_content: 'Need the file.', tool_calls: [{ id: 'call_read', type: 'function', function: { name: 'read', arguments: '{"path":"a.txt"}' } }] },
{ role: 'tool', tool_call_id: 'call_read', content: 'hello' },
{ role: 'assistant', content: 'It says hello.', reasoning_content: 'The file says hello.', tool_calls: [{ id: 'call_log', type: 'function', function: { name: 'log', arguments: '{"value":"hello"}' } }] },
]);
expect(example.metadata.tools).toEqual([
{ type: 'function', name: 'read', parameters: { type: 'object' } },
{ type: 'web_search_preview' },
]);
expect(example.metadata).toMatchObject({ api: 'openai_responses', model: 'model-r', stream: false, response_status: 200 });
});
it('uses the completed Responses streaming event as the assistant output', () => {
const responseBody = [
'event: response.completed',
'data: {"type":"response.completed","response":{"output":[{"type":"reasoning","summary":[{"type":"summary_text","text":"Need a lookup."}]},{"type":"message","role":"assistant","content":[{"type":"output_text","text":"Found it."}]},{"type":"function_call","call_id":"call_1","name":"lookup","arguments":"{\\"q\\":\\"x\\"}"}]}}',
'', '',
].join('\n');
const example = createTrainingExample({
endpoint: '/oai/v1/responses', responseStatus: 200, stream: true,
requestBody: { input: 'Find x' }, responseBody,
});
expect(example.messages.at(-1)).toEqual({
role: 'assistant', content: 'Found it.', reasoning_content: 'Need a lookup.',
tool_calls: [{ id: 'call_1', type: 'function', function: { name: 'lookup', arguments: '{"q":"x"}' } }],
});
});
it('normalizes Anthropic thinking, tool use, and tool results', () => {
const example = createTrainingExample({
endpoint: '/ant/v1/messages', responseStatus: 200,
requestBody: {
model: 'model-ant', system: 'Be useful',
tools: [{ name: 'calculator', description: 'Calculate', input_schema: { type: 'object' } }],
messages: [
{ role: 'user', content: 'Calculate' },
{ role: 'assistant', content: [{ type: 'thinking', thinking: 'Need calculator.' }, { type: 'tool_use', id: 'tool_1', name: 'calculator', input: { expression: '2+2' } }] },
{ role: 'user', content: [{ type: 'tool_result', tool_use_id: 'tool_1', content: '4' }] },
],
},
responseBody: JSON.stringify({ content: [
{ type: 'thinking', thinking: 'The result is four.' },
{ type: 'text', text: 'The answer is 4.' },
{ type: 'tool_use', id: 'tool_2', name: 'record', input: { answer: 4 } },
] }),
});
expect(example.messages).toEqual([
{ role: 'system', content: 'Be useful' },
{ role: 'user', content: 'Calculate' },
{ role: 'assistant', content: null, reasoning_content: 'Need calculator.', tool_calls: [{ id: 'tool_1', type: 'function', function: { name: 'calculator', arguments: '{"expression":"2+2"}' } }] },
{ role: 'tool', tool_call_id: 'tool_1', content: '4' },
{ role: 'assistant', content: 'The answer is 4.', reasoning_content: 'The result is four.', tool_calls: [{ id: 'tool_2', type: 'function', function: { name: 'record', arguments: '{"answer":4}' } }] },
]);
expect(example.metadata.tools).toEqual([
{ name: 'calculator', description: 'Calculate', input_schema: { type: 'object' } },
]);
expect(example.metadata).toMatchObject({ api: 'anthropic_messages', model: 'model-ant', stream: false, response_status: 200 });
});
it('reconstructs streamed Anthropic content blocks', () => {
const responseBody = [
'event: content_block_start\ndata: {"type":"content_block_start","index":0,"content_block":{"type":"thinking","thinking":""}}',
'event: content_block_delta\ndata: {"type":"content_block_delta","index":0,"delta":{"type":"thinking_delta","thinking":"Need tool."}}',
'event: content_block_start\ndata: {"type":"content_block_start","index":1,"content_block":{"type":"text","text":""}}',
'event: content_block_delta\ndata: {"type":"content_block_delta","index":1,"delta":{"type":"text_delta","text":"Working"}}',
'event: content_block_start\ndata: {"type":"content_block_start","index":2,"content_block":{"type":"tool_use","id":"tool_1","name":"run","input":{}}}',
'event: content_block_delta\ndata: {"type":"content_block_delta","index":2,"delta":{"type":"input_json_delta","partial_json":"{\\"x\\":1}"}}',
'',
].join('\n\n');
const example = createTrainingExample({
endpoint: '/ant/v1/messages', responseStatus: 200, stream: true,
requestBody: { messages: [{ role: 'user', content: 'Work' }] }, responseBody,
});
expect(example.messages.at(-1)).toEqual({
role: 'assistant', content: 'Working', reasoning_content: 'Need tool.',
tool_calls: [{ id: 'tool_1', type: 'function', function: { name: 'run', arguments: '{"x":1}' } }],
});
});
it('excludes failed and non-chat requests', () => {
expect(createTrainingExample({ endpoint: '/oai/v1/models', responseStatus: 200, requestBody: {} })).toBeNull();
expect(createTrainingExample({ endpoint: '/oai/v1/chat/completions', responseStatus: 500, requestBody: { messages: [] } })).toBeNull();
});
});