maybe fix
This commit is contained in:
+20
-5
@@ -128,6 +128,20 @@ function validateContext(body, maxContextTokens) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
function normalizeChatCompletionRequest(body) {
|
||||||
|
if (!body || typeof body !== 'object' || Array.isArray(body) || body.max_completion_tokens === undefined) {
|
||||||
|
return body;
|
||||||
|
}
|
||||||
|
|
||||||
|
// OpenAI renamed max_tokens to max_completion_tokens, but OpenCode Go's
|
||||||
|
// OpenAI-compatible providers do not all accept the newer field. Normalize
|
||||||
|
// it at the provider boundary so the same client request works across
|
||||||
|
// models. If a client supplied both names, keep the explicit legacy value.
|
||||||
|
const { max_completion_tokens: maxCompletionTokens, ...normalized } = body;
|
||||||
|
if (normalized.max_tokens === undefined) normalized.max_tokens = maxCompletionTokens;
|
||||||
|
return normalized;
|
||||||
|
}
|
||||||
|
|
||||||
function withContextLimit(payload, maxContextTokens, allowedModels) {
|
function withContextLimit(payload, maxContextTokens, allowedModels) {
|
||||||
return {
|
return {
|
||||||
...payload,
|
...payload,
|
||||||
@@ -252,10 +266,11 @@ export function createProxyApp({ keys, keyRotator = createKeyRotator(keys), fetc
|
|||||||
|
|
||||||
app.post('/oai/v1/chat/completions', async (request, response) => {
|
app.post('/oai/v1/chat/completions', async (request, response) => {
|
||||||
try {
|
try {
|
||||||
validateContext(request.body, maxContextTokens);
|
const upstreamBody = normalizeChatCompletionRequest(request.body);
|
||||||
|
validateContext(upstreamBody, maxContextTokens);
|
||||||
const ip = clientIp(request);
|
const ip = clientIp(request);
|
||||||
const unlimited = hasUnlimitedKey(request, unlimitedKeys);
|
const unlimited = hasUnlimitedKey(request, unlimitedKeys);
|
||||||
if (!unlimited) validateAllowedModel(request.body?.model, allowedModels);
|
if (!unlimited) validateAllowedModel(upstreamBody?.model, allowedModels);
|
||||||
if (!unlimited) {
|
if (!unlimited) {
|
||||||
const limit = await rateLimiter.check(ip);
|
const limit = await rateLimiter.check(ip);
|
||||||
if (!limit.allowed) return rateLimitError(response, false, limit);
|
if (!limit.allowed) return rateLimitError(response, false, limit);
|
||||||
@@ -264,15 +279,15 @@ export function createProxyApp({ keys, keyRotator = createKeyRotator(keys), fetc
|
|||||||
method: 'POST',
|
method: 'POST',
|
||||||
headers: {
|
headers: {
|
||||||
'content-type': 'application/json',
|
'content-type': 'application/json',
|
||||||
accept: request.body?.stream ? 'text/event-stream' : 'application/json',
|
accept: upstreamBody?.stream ? 'text/event-stream' : 'application/json',
|
||||||
},
|
},
|
||||||
body: JSON.stringify(request.body),
|
body: JSON.stringify(upstreamBody),
|
||||||
});
|
});
|
||||||
|
|
||||||
if (!upstream.ok) return forwardError(upstream, response);
|
if (!upstream.ok) return forwardError(upstream, response);
|
||||||
copyResponseHeaders(upstream, response);
|
copyResponseHeaders(upstream, response);
|
||||||
|
|
||||||
if (request.body?.stream && upstream.body) {
|
if (upstreamBody?.stream && upstream.body) {
|
||||||
response.status(upstream.status);
|
response.status(upstream.status);
|
||||||
let streamBuffer = '';
|
let streamBuffer = '';
|
||||||
let recordedCost = false;
|
let recordedCost = false;
|
||||||
|
|||||||
@@ -69,12 +69,39 @@ describe('proxy', () => {
|
|||||||
expect(JSON.parse(fetchImpl.mock.calls[0][1].body)).toEqual(body);
|
expect(JSON.parse(fetchImpl.mock.calls[0][1].body)).toEqual(body);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
it('normalizes max_completion_tokens for OpenCode Go providers', async () => {
|
||||||
|
const fetchImpl = vi.fn().mockResolvedValue(response('{"id":"completion-1"}'));
|
||||||
|
const app = createProxyApp({ keys: ['key'], fetchImpl });
|
||||||
|
|
||||||
|
await request(app).post('/oai/v1/chat/completions').send({
|
||||||
|
model: 'kimi-k3', messages: [{ role: 'user', content: 'Hi' }],
|
||||||
|
max_completion_tokens: 4096, stream: true, stream_options: { include_usage: true },
|
||||||
|
}).expect(200);
|
||||||
|
|
||||||
|
expect(JSON.parse(fetchImpl.mock.calls[0][1].body)).toEqual({
|
||||||
|
model: 'kimi-k3', messages: [{ role: 'user', content: 'Hi' }],
|
||||||
|
max_tokens: 4096, stream: true, stream_options: { include_usage: true },
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
it('prefers max_tokens when both token limit field names are supplied', async () => {
|
||||||
|
const fetchImpl = vi.fn().mockResolvedValue(response('{"id":"completion-1"}'));
|
||||||
|
const app = createProxyApp({ keys: ['key'], fetchImpl });
|
||||||
|
|
||||||
|
await request(app).post('/oai/v1/chat/completions').send({
|
||||||
|
model: 'm', max_tokens: 2048, max_completion_tokens: 4096,
|
||||||
|
}).expect(200);
|
||||||
|
|
||||||
|
expect(JSON.parse(fetchImpl.mock.calls[0][1].body)).toEqual({ model: 'm', max_tokens: 2048 });
|
||||||
|
});
|
||||||
|
|
||||||
it('enforces the configured context and IP limits and records cost', async () => {
|
it('enforces the configured context and IP limits and records cost', async () => {
|
||||||
const fetchImpl = vi.fn()
|
const fetchImpl = vi.fn()
|
||||||
.mockResolvedValueOnce(response(JSON.stringify({ id: 'c', cost: 0 })))
|
.mockResolvedValueOnce(response(JSON.stringify({ id: 'c', cost: 0 })))
|
||||||
.mockResolvedValue(response(JSON.stringify({ id: 'c', cost: 10 })));
|
.mockResolvedValue(response(JSON.stringify({ id: 'c', cost: 10 })));
|
||||||
const app = createProxyApp({ keys: ['key'], fetchImpl, maxContextTokens: 100 });
|
const app = createProxyApp({ keys: ['key'], fetchImpl, maxContextTokens: 100 });
|
||||||
await request(app).post('/oai/v1/chat/completions').set('X-Real-IP', '198.51.100.1').send({ model: 'm', max_tokens: 101 }).expect(400);
|
await request(app).post('/oai/v1/chat/completions').set('X-Real-IP', '198.51.100.1').send({ model: 'm', max_tokens: 101 }).expect(400);
|
||||||
|
await request(app).post('/oai/v1/chat/completions').set('X-Real-IP', '198.51.100.1').send({ model: 'm', max_completion_tokens: 101 }).expect(400);
|
||||||
await request(app).post('/oai/v1/chat/completions').set('X-Real-IP', '198.51.100.1').send({ model: 'm', max_tokens: 1 }).expect(200);
|
await request(app).post('/oai/v1/chat/completions').set('X-Real-IP', '198.51.100.1').send({ model: 'm', max_tokens: 1 }).expect(200);
|
||||||
await request(app).post('/oai/v1/chat/completions').set('X-Real-IP', '198.51.100.1').send({ model: 'm', max_tokens: 1 }).expect(200);
|
await request(app).post('/oai/v1/chat/completions').set('X-Real-IP', '198.51.100.1').send({ model: 'm', max_tokens: 1 }).expect(200);
|
||||||
await request(app).post('/oai/v1/chat/completions').set('X-Real-IP', '198.51.100.1').send({ model: 'm', max_tokens: 1 }).expect(429);
|
await request(app).post('/oai/v1/chat/completions').set('X-Real-IP', '198.51.100.1').send({ model: 'm', max_tokens: 1 }).expect(429);
|
||||||
|
|||||||
Reference in New Issue
Block a user