maybe fix
This commit is contained in:
+20
-5
@@ -128,6 +128,20 @@ function validateContext(body, maxContextTokens) {
|
||||
}
|
||||
}
|
||||
|
||||
function normalizeChatCompletionRequest(body) {
|
||||
if (!body || typeof body !== 'object' || Array.isArray(body) || body.max_completion_tokens === undefined) {
|
||||
return body;
|
||||
}
|
||||
|
||||
// OpenAI renamed max_tokens to max_completion_tokens, but OpenCode Go's
|
||||
// OpenAI-compatible providers do not all accept the newer field. Normalize
|
||||
// it at the provider boundary so the same client request works across
|
||||
// models. If a client supplied both names, keep the explicit legacy value.
|
||||
const { max_completion_tokens: maxCompletionTokens, ...normalized } = body;
|
||||
if (normalized.max_tokens === undefined) normalized.max_tokens = maxCompletionTokens;
|
||||
return normalized;
|
||||
}
|
||||
|
||||
function withContextLimit(payload, maxContextTokens, allowedModels) {
|
||||
return {
|
||||
...payload,
|
||||
@@ -252,10 +266,11 @@ export function createProxyApp({ keys, keyRotator = createKeyRotator(keys), fetc
|
||||
|
||||
app.post('/oai/v1/chat/completions', async (request, response) => {
|
||||
try {
|
||||
validateContext(request.body, maxContextTokens);
|
||||
const upstreamBody = normalizeChatCompletionRequest(request.body);
|
||||
validateContext(upstreamBody, maxContextTokens);
|
||||
const ip = clientIp(request);
|
||||
const unlimited = hasUnlimitedKey(request, unlimitedKeys);
|
||||
if (!unlimited) validateAllowedModel(request.body?.model, allowedModels);
|
||||
if (!unlimited) validateAllowedModel(upstreamBody?.model, allowedModels);
|
||||
if (!unlimited) {
|
||||
const limit = await rateLimiter.check(ip);
|
||||
if (!limit.allowed) return rateLimitError(response, false, limit);
|
||||
@@ -264,15 +279,15 @@ export function createProxyApp({ keys, keyRotator = createKeyRotator(keys), fetc
|
||||
method: 'POST',
|
||||
headers: {
|
||||
'content-type': 'application/json',
|
||||
accept: request.body?.stream ? 'text/event-stream' : 'application/json',
|
||||
accept: upstreamBody?.stream ? 'text/event-stream' : 'application/json',
|
||||
},
|
||||
body: JSON.stringify(request.body),
|
||||
body: JSON.stringify(upstreamBody),
|
||||
});
|
||||
|
||||
if (!upstream.ok) return forwardError(upstream, response);
|
||||
copyResponseHeaders(upstream, response);
|
||||
|
||||
if (request.body?.stream && upstream.body) {
|
||||
if (upstreamBody?.stream && upstream.body) {
|
||||
response.status(upstream.status);
|
||||
let streamBuffer = '';
|
||||
let recordedCost = false;
|
||||
|
||||
@@ -69,12 +69,39 @@ describe('proxy', () => {
|
||||
expect(JSON.parse(fetchImpl.mock.calls[0][1].body)).toEqual(body);
|
||||
});
|
||||
|
||||
it('normalizes max_completion_tokens for OpenCode Go providers', async () => {
|
||||
const fetchImpl = vi.fn().mockResolvedValue(response('{"id":"completion-1"}'));
|
||||
const app = createProxyApp({ keys: ['key'], fetchImpl });
|
||||
|
||||
await request(app).post('/oai/v1/chat/completions').send({
|
||||
model: 'kimi-k3', messages: [{ role: 'user', content: 'Hi' }],
|
||||
max_completion_tokens: 4096, stream: true, stream_options: { include_usage: true },
|
||||
}).expect(200);
|
||||
|
||||
expect(JSON.parse(fetchImpl.mock.calls[0][1].body)).toEqual({
|
||||
model: 'kimi-k3', messages: [{ role: 'user', content: 'Hi' }],
|
||||
max_tokens: 4096, stream: true, stream_options: { include_usage: true },
|
||||
});
|
||||
});
|
||||
|
||||
it('prefers max_tokens when both token limit field names are supplied', async () => {
|
||||
const fetchImpl = vi.fn().mockResolvedValue(response('{"id":"completion-1"}'));
|
||||
const app = createProxyApp({ keys: ['key'], fetchImpl });
|
||||
|
||||
await request(app).post('/oai/v1/chat/completions').send({
|
||||
model: 'm', max_tokens: 2048, max_completion_tokens: 4096,
|
||||
}).expect(200);
|
||||
|
||||
expect(JSON.parse(fetchImpl.mock.calls[0][1].body)).toEqual({ model: 'm', max_tokens: 2048 });
|
||||
});
|
||||
|
||||
it('enforces the configured context and IP limits and records cost', async () => {
|
||||
const fetchImpl = vi.fn()
|
||||
.mockResolvedValueOnce(response(JSON.stringify({ id: 'c', cost: 0 })))
|
||||
.mockResolvedValue(response(JSON.stringify({ id: 'c', cost: 10 })));
|
||||
const app = createProxyApp({ keys: ['key'], fetchImpl, maxContextTokens: 100 });
|
||||
await request(app).post('/oai/v1/chat/completions').set('X-Real-IP', '198.51.100.1').send({ model: 'm', max_tokens: 101 }).expect(400);
|
||||
await request(app).post('/oai/v1/chat/completions').set('X-Real-IP', '198.51.100.1').send({ model: 'm', max_completion_tokens: 101 }).expect(400);
|
||||
await request(app).post('/oai/v1/chat/completions').set('X-Real-IP', '198.51.100.1').send({ model: 'm', max_tokens: 1 }).expect(200);
|
||||
await request(app).post('/oai/v1/chat/completions').set('X-Real-IP', '198.51.100.1').send({ model: 'm', max_tokens: 1 }).expect(200);
|
||||
await request(app).post('/oai/v1/chat/completions').set('X-Real-IP', '198.51.100.1').send({ model: 'm', max_tokens: 1 }).expect(429);
|
||||
|
||||
Reference in New Issue
Block a user