feat: consume server model reasoning capabilities
This commit is contained in:
@@ -2,6 +2,8 @@ import { mkdtemp, readFile, rm, writeFile } from 'node:fs/promises';
|
||||
import { tmpdir } from 'node:os';
|
||||
import path from 'node:path';
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest';
|
||||
import { streamSimple } from '@earendil-works/pi-ai/api/openai-completions';
|
||||
import type { Model } from '@earendil-works/pi-ai';
|
||||
import type { ProviderAccount } from '@electron/shared/providers/types';
|
||||
import {
|
||||
buildPiProviderCatalog,
|
||||
@@ -269,11 +271,11 @@ describe('Pi Provider catalog', () => {
|
||||
requiresReasoningContentOnAssistantMessages: true,
|
||||
},
|
||||
thinkingLevelMap: {
|
||||
off: null,
|
||||
minimal: null,
|
||||
low: null,
|
||||
low: 'low',
|
||||
medium: null,
|
||||
high: 'high',
|
||||
max: 'max',
|
||||
},
|
||||
});
|
||||
expect(written).toMatchObject({
|
||||
@@ -283,6 +285,210 @@ describe('Pi Provider catalog', () => {
|
||||
});
|
||||
});
|
||||
|
||||
it('uses server reasoning capabilities while retaining local model metadata', () => {
|
||||
const provider = account({
|
||||
id: 'niancode-user-models',
|
||||
vendorId: 'custom',
|
||||
apiProtocol: 'openai-completions',
|
||||
baseUrl: 'http://127.0.0.1:54321/api/ai-proxy/v1',
|
||||
model: 'qwen3.8-max',
|
||||
metadata: {
|
||||
worksSquareCredentialMode: 'works_square_ai_gateway_proxy',
|
||||
customModels: ['qwen3.8-max'],
|
||||
worksSquareModelCapabilities: {
|
||||
'qwen3.8-max': {
|
||||
reasoningEfforts: ['low'],
|
||||
reasoningCanDisable: false,
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
const descriptor = buildPiProviderCatalog({ accounts: [provider] }).descriptors[0]!.models[0]!;
|
||||
|
||||
expect(descriptor).toMatchObject({
|
||||
id: 'qwen3.8-max',
|
||||
input: ['text', 'image'],
|
||||
reasoning: true,
|
||||
contextWindow: 1_000_000,
|
||||
maxOutputTokens: 131_072,
|
||||
thinkingLevelMap: {
|
||||
off: null,
|
||||
minimal: null,
|
||||
low: 'low',
|
||||
medium: null,
|
||||
high: null,
|
||||
max: null,
|
||||
},
|
||||
compat: {
|
||||
thinkingFormat: 'qwen',
|
||||
supportsDeveloperRole: false,
|
||||
supportsReasoningEffort: true,
|
||||
supportsStore: false,
|
||||
},
|
||||
});
|
||||
});
|
||||
|
||||
it('enables reasoning-effort serialization for a server model without a local profile', () => {
|
||||
const provider = account({
|
||||
id: 'niancode-user-models',
|
||||
vendorId: 'custom',
|
||||
apiProtocol: 'openai-completions',
|
||||
baseUrl: 'http://127.0.0.1:54321/api/ai-proxy/v1',
|
||||
model: 'future-reasoning-model',
|
||||
metadata: {
|
||||
worksSquareCredentialMode: 'works_square_ai_gateway_proxy',
|
||||
customModels: ['future-reasoning-model'],
|
||||
worksSquareModelCapabilities: {
|
||||
'future-reasoning-model': {
|
||||
reasoningEfforts: ['low', 'high', 'max'],
|
||||
reasoningCanDisable: true,
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
const descriptor = buildPiProviderCatalog({ accounts: [provider] }).descriptors[0]!.models[0]!;
|
||||
|
||||
expect(descriptor).toMatchObject({
|
||||
id: 'future-reasoning-model',
|
||||
reasoning: true,
|
||||
compat: {
|
||||
supportsDeveloperRole: false,
|
||||
supportsReasoningEffort: true,
|
||||
},
|
||||
thinkingLevelMap: {
|
||||
minimal: null,
|
||||
low: 'low',
|
||||
medium: null,
|
||||
high: 'high',
|
||||
max: 'max',
|
||||
},
|
||||
});
|
||||
expect(descriptor.thinkingLevelMap).not.toHaveProperty('off');
|
||||
});
|
||||
|
||||
it('lets an explicit empty server effort list override a locally reasoning model', () => {
|
||||
const provider = account({
|
||||
id: 'niancode-user-models',
|
||||
vendorId: 'custom',
|
||||
apiProtocol: 'openai-completions',
|
||||
baseUrl: 'http://127.0.0.1:54321/api/ai-proxy/v1',
|
||||
model: 'deepseek-v4-pro',
|
||||
metadata: {
|
||||
worksSquareCredentialMode: 'works_square_ai_gateway_proxy',
|
||||
customModels: ['deepseek-v4-pro'],
|
||||
worksSquareModelCapabilities: {
|
||||
'deepseek-v4-pro': {
|
||||
reasoningEfforts: [],
|
||||
reasoningCanDisable: false,
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
const descriptor = buildPiProviderCatalog({ accounts: [provider] }).descriptors[0]!.models[0]!;
|
||||
|
||||
expect(descriptor.reasoning).toBe(false);
|
||||
expect(descriptor.thinkingLevelMap).toEqual({
|
||||
off: null,
|
||||
minimal: null,
|
||||
low: null,
|
||||
medium: null,
|
||||
high: null,
|
||||
max: null,
|
||||
});
|
||||
expect(descriptor.compat).toMatchObject({
|
||||
thinkingFormat: 'deepseek',
|
||||
supportsReasoningEffort: true,
|
||||
requiresReasoningContentOnAssistantMessages: true,
|
||||
});
|
||||
});
|
||||
|
||||
it('serializes DeepSeek off without sending a reasoning effort', async () => {
|
||||
const payloads: unknown[] = [];
|
||||
const model: Model<'openai-completions'> = {
|
||||
id: 'deepseek-v4-pro',
|
||||
name: 'DeepSeek V4 Pro',
|
||||
api: 'openai-completions',
|
||||
provider: 'deepseek',
|
||||
baseUrl: 'https://gateway.test/v1',
|
||||
reasoning: true,
|
||||
input: ['text'],
|
||||
contextWindow: 1_000_000,
|
||||
maxTokens: 384_000,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
thinkingLevelMap: {
|
||||
minimal: null,
|
||||
low: 'low',
|
||||
medium: null,
|
||||
high: 'high',
|
||||
max: 'max',
|
||||
},
|
||||
compat: {
|
||||
thinkingFormat: 'deepseek',
|
||||
supportsReasoningEffort: true,
|
||||
},
|
||||
};
|
||||
|
||||
const stream = streamSimple(model, { messages: [] }, {
|
||||
apiKey: 'test-key',
|
||||
reasoning: 'off',
|
||||
onPayload(payload) {
|
||||
payloads.push(payload);
|
||||
throw new Error('stop before network');
|
||||
},
|
||||
});
|
||||
const result = await stream.result();
|
||||
|
||||
expect(result.stopReason).toBe('error');
|
||||
expect(payloads[0]).toMatchObject({ thinking: { type: 'disabled' } });
|
||||
expect(payloads[0]).not.toHaveProperty('reasoning_effort');
|
||||
});
|
||||
|
||||
it.each(['low', 'high', 'max'] as const)('serializes DeepSeek %s with its reasoning effort', async (level) => {
|
||||
const payloads: unknown[] = [];
|
||||
const model: Model<'openai-completions'> = {
|
||||
id: 'deepseek-v4-pro',
|
||||
name: 'DeepSeek V4 Pro',
|
||||
api: 'openai-completions',
|
||||
provider: 'deepseek',
|
||||
baseUrl: 'https://gateway.test/v1',
|
||||
reasoning: true,
|
||||
input: ['text'],
|
||||
contextWindow: 1_000_000,
|
||||
maxTokens: 384_000,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
thinkingLevelMap: {
|
||||
minimal: null,
|
||||
low: 'low',
|
||||
medium: null,
|
||||
high: 'high',
|
||||
max: 'max',
|
||||
},
|
||||
compat: {
|
||||
thinkingFormat: 'deepseek',
|
||||
supportsReasoningEffort: true,
|
||||
},
|
||||
};
|
||||
|
||||
const stream = streamSimple(model, { messages: [] }, {
|
||||
apiKey: 'test-key',
|
||||
reasoning: level,
|
||||
onPayload(payload) {
|
||||
payloads.push(payload);
|
||||
throw new Error('stop before network');
|
||||
},
|
||||
});
|
||||
const result = await stream.result();
|
||||
|
||||
expect(result.stopReason).toBe('error');
|
||||
expect(payloads[0]).toMatchObject({
|
||||
thinking: { type: 'enabled' },
|
||||
reasoning_effort: level,
|
||||
});
|
||||
});
|
||||
|
||||
it('does not replace an existing catalog when model selection is unavailable', async () => {
|
||||
const root = await mkdtemp(path.join(tmpdir(), 'makelore-pi-provider-'));
|
||||
temporaryRoots.push(root);
|
||||
|
||||
Reference in New Issue
Block a user