Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions api/utils/tokens.spec.js
Original file line number Diff line number Diff line change
Expand Up @@ -333,6 +333,9 @@ describe('getModelMaxTokens', () => {
expect(getModelMaxTokens('gemini-3.1-pro-preview-customtools', EModelEndpoint.google)).toBe(
maxTokensMap[EModelEndpoint.google]['gemini-3.1'],
);
expect(getModelMaxTokens('gemini-3.5-flash', EModelEndpoint.google)).toBe(
maxTokensMap[EModelEndpoint.google]['gemini-3.5-flash'],
);
expect(getModelMaxTokens('gemini-2.5-pro', EModelEndpoint.google)).toBe(
maxTokensMap[EModelEndpoint.google]['gemini-2.5-pro'],
);
Expand Down
144 changes: 144 additions & 0 deletions packages/api/src/endpoints/google/llm.spec.ts
Original file line number Diff line number Diff line change
Expand Up @@ -628,6 +628,150 @@ describe('getGoogleConfig', () => {
});
});

it('should default Gemini 3.5 Flash to medium thinkingLevel', () => {
const credentials = {
[AuthKeys.GOOGLE_API_KEY]: 'test-api-key',
};

const result = getGoogleConfig(credentials, {
modelOptions: {
model: 'gemini-3.5-flash',
},
});

expect((result.llmConfig as Record<string, unknown>).thinkingConfig).toMatchObject({
includeThoughts: true,
thinkingLevel: 'MEDIUM',
});
});

it('should preserve explicit Gemini 3.5 Flash thinkingLevel', () => {
const credentials = {
[AuthKeys.GOOGLE_API_KEY]: 'test-api-key',
};

const result = getGoogleConfig(credentials, {
modelOptions: {
model: 'gemini-3.5-flash',
thinkingLevel: ThinkingLevel.low,
},
});

expect((result.llmConfig as Record<string, unknown>).thinkingConfig).toMatchObject({
includeThoughts: true,
thinkingLevel: 'LOW',
});
});

it('should apply Gemini 3.5 Flash overrides to versioned aliases', () => {
const credentials = {
[AuthKeys.GOOGLE_API_KEY]: 'test-api-key',
};

const result = getGoogleConfig(credentials, {
modelOptions: {
model: 'google/gemini-3.5-flash-latest',
temperature: 0.7,
},
});

expect(result.llmConfig).not.toHaveProperty('temperature');
expect((result.llmConfig as Record<string, unknown>).thinkingConfig).toMatchObject({
includeThoughts: true,
thinkingLevel: 'MEDIUM',
});
});

it('should remove legacy sampling params for Gemini 3.5 Flash', () => {
const credentials = {
[AuthKeys.GOOGLE_API_KEY]: 'test-api-key',
};

const modelOptions = {
model: 'gemini-3.5-flash',
temperature: 0.7,
topP: 0.9,
topK: 40,
top_p: 0.9,
top_k: 40,
thinking_budget: 5000,
} as unknown as t.GoogleParameters;

const result = getGoogleConfig(credentials, {
modelOptions,
defaultParams: {
temperature: 0.5,
topP: 0.8,
topK: 20,
},
addParams: {
temperature: 0.2,
topP: 0.6,
topK: 10,
},
});

expect(result.llmConfig).not.toHaveProperty('temperature');
expect(result.llmConfig).not.toHaveProperty('topP');
expect(result.llmConfig).not.toHaveProperty('topK');
expect(result.llmConfig).not.toHaveProperty('top_p');
expect(result.llmConfig).not.toHaveProperty('top_k');
expect(result.llmConfig).not.toHaveProperty('thinking_budget');
});

it('should respect dropParams for Gemini 3.5 Flash thinkingConfig', () => {
const credentials = {
[AuthKeys.GOOGLE_API_KEY]: 'test-api-key',
};

const result = getGoogleConfig(credentials, {
modelOptions: {
model: 'gemini-3.5-flash',
},
dropParams: ['thinkingConfig'],
});

expect(result.llmConfig).not.toHaveProperty('thinkingConfig');
});

it('should respect dropParams for Gemini 3.5 Flash includeThoughts', () => {
const credentials = {
[AuthKeys.GOOGLE_SERVICE_KEY]: {
project_id: 'test-project',
},
};

const result = getGoogleConfig(credentials, {
modelOptions: {
model: 'gemini-3.5-flash',
},
dropParams: ['includeThoughts'],
});

expect(result.llmConfig).not.toHaveProperty('includeThoughts');
expect((result.llmConfig as Record<string, unknown>).thinkingConfig).toMatchObject({
thinkingLevel: 'MEDIUM',
});
expect((result.llmConfig as Record<string, unknown>).thinkingConfig).not.toHaveProperty(
'includeThoughts',
);
});

it('should remove empty Gemini 3.5 Flash thinkingConfig when all fields are dropped', () => {
const credentials = {
[AuthKeys.GOOGLE_API_KEY]: 'test-api-key',
};

const result = getGoogleConfig(credentials, {
modelOptions: {
model: 'gemini-3.5-flash',
},
dropParams: ['includeThoughts', 'thinkingLevel'],
});

expect(result.llmConfig).not.toHaveProperty('thinkingConfig');
});

it('should omit thinkingLevel when unset (empty string) for Gemini 3', () => {
const credentials = {
[AuthKeys.GOOGLE_API_KEY]: 'test-api-key',
Expand Down
89 changes: 88 additions & 1 deletion packages/api/src/endpoints/google/llm.ts
Original file line number Diff line number Diff line change
Expand Up @@ -7,10 +7,22 @@ import { isEnabled } from '~/utils';

type GoogleThinkingLevel = 'THINKING_LEVEL_UNSPECIFIED' | 'MINIMAL' | 'LOW' | 'MEDIUM' | 'HIGH';
type GoogleThinkingConfig = {
includeThoughts: boolean;
includeThoughts?: boolean;
thinkingLevel?: GoogleThinkingLevel;
};

const GEMINI_3_5_FLASH = 'gemini-3.5-flash';
const GEMINI_3_5_FLASH_DEFAULT_THINKING_LEVEL: GoogleThinkingLevel = 'MEDIUM';
const gemini35FlashLegacyParams = [
'temperature',
'topP',
'topK',
'top_p',
'top_k',
'thinkingBudget',
'thinking_budget',
] as const;

const googleThinkingLevels = new Set<GoogleThinkingLevel>([
'THINKING_LEVEL_UNSPECIFIED',
'MINIMAL',
Expand Down Expand Up @@ -116,6 +128,12 @@ function normalizeGoogleThinkingLevel(value: unknown): GoogleThinkingLevel | und
return normalized;
}

function isGemini35Flash(model: string) {
const normalized = model.toLowerCase();
const modelId = normalized.split('/').pop() ?? normalized;
return modelId === GEMINI_3_5_FLASH || modelId.startsWith(`${GEMINI_3_5_FLASH}-`);
}

function getVertexMultiRegionEndpoint(location: string): string | undefined {
return vertexMultiRegionEndpoints.get(location);
}
Expand All @@ -128,6 +146,68 @@ function sanitizeModelOptions(modelOptions: Partial<t.GoogleParameters> | undefi
return sanitizedOptions;
}

function applyGemini35FlashOverrides({
config,
provider,
thinking,
dropParams,
}: {
config: GoogleClientOptions | VertexAIClientOptions;
provider: Providers;
thinking: boolean;
dropParams?: string[];
}) {
const mutableConfig = config as Record<string, unknown>;
const model = mutableConfig.model;
if (typeof model !== 'string' || !isGemini35Flash(model)) {
return;
}

gemini35FlashLegacyParams.forEach((param) => {
delete mutableConfig[param];
});
Comment thread
danny-avila marked this conversation as resolved.

if (!thinking) {
return;
}

const droppedParams = new Set(dropParams ?? []);
if (droppedParams.has('thinkingConfig')) {
return;
}

const shouldDropIncludeThoughts = droppedParams.has('includeThoughts');
const shouldDropThinkingLevel = droppedParams.has('thinkingLevel');
const configWithThinking = config as { thinkingConfig?: GoogleThinkingConfig };
const thinkingConfig: GoogleThinkingConfig = { ...(configWithThinking.thinkingConfig ?? {}) };

if (shouldDropIncludeThoughts) {
delete thinkingConfig.includeThoughts;
}

if (shouldDropThinkingLevel) {
delete thinkingConfig.thinkingLevel;
}

if (!shouldDropIncludeThoughts && thinkingConfig.includeThoughts == null) {
thinkingConfig.includeThoughts = true;
}

if (!shouldDropThinkingLevel && !thinkingConfig.thinkingLevel) {
thinkingConfig.thinkingLevel = GEMINI_3_5_FLASH_DEFAULT_THINKING_LEVEL;
}

if (Object.keys(thinkingConfig).length > 0) {
configWithThinking.thinkingConfig = thinkingConfig;
} else {
delete configWithThinking.thinkingConfig;
}
Comment thread
danny-avila marked this conversation as resolved.

if (provider === Providers.VERTEXAI && !shouldDropIncludeThoughts) {
(config as VertexAIClientOptions).includeThoughts = true;
}
}

function isAllowedVertexEndpoint(endpoint: string): boolean {
if (!/^[a-z0-9][a-z0-9.-]*[a-z0-9]$/.test(endpoint)) {
return false;
Expand Down Expand Up @@ -453,6 +533,13 @@ export function getGoogleConfig(
});
}

applyGemini35FlashOverrides({
config: llmConfig,
provider,
thinking,
dropParams: Array.isArray(options.dropParams) ? options.dropParams : undefined,
});

if (provider === Providers.VERTEXAI && shouldSyncVertexEndpoint && !hasCustomVertexEndpoint) {
applyVertexMultiRegionEndpoint(llmConfig as VertexAIClientOptions & { endpoint?: string });
}
Expand Down
1 change: 1 addition & 0 deletions packages/api/src/utils/tokens.ts
Original file line number Diff line number Diff line change
Expand Up @@ -117,6 +117,7 @@ const googleModels = {
'gemini-3-pro-image': 1000000,
'gemini-3.1': 1000000,
'gemini-3.1-flash-lite': 1000000,
'gemini-3.5-flash': 1048576,
};

const anthropicModels = {
Expand Down
2 changes: 2 additions & 0 deletions packages/data-provider/src/config.ts
Original file line number Diff line number Diff line change
Expand Up @@ -1595,6 +1595,8 @@ export const defaultModels = {
[EModelEndpoint.assistants]: [...sharedOpenAIModels, 'chatgpt-4o-latest'],
[EModelEndpoint.agents]: sharedOpenAIModels, // TODO: Add agent models (agentsModels)
[EModelEndpoint.google]: [
// Gemini 3.5 Models
'gemini-3.5-flash',
// Gemini 3.1 Models
'gemini-3.1-pro-preview',
'gemini-3.1-pro-preview-customtools',
Expand Down
2 changes: 1 addition & 1 deletion packages/data-provider/src/schemas.ts
Original file line number Diff line number Diff line change
Expand Up @@ -372,7 +372,7 @@ export const googleSettings = {
},
maxOutputTokens: {
min: 1 as const,
max: 64000 as const,
max: 65536 as const,
step: 1 as const,
default: 8192 as const,
},
Expand Down
18 changes: 18 additions & 0 deletions packages/data-schemas/src/methods/tx.spec.ts
Original file line number Diff line number Diff line change
Expand Up @@ -1497,6 +1497,7 @@ describe('Google Model Tests', () => {
'gemini-3.1-pro-preview',
'gemini-3.1-pro-preview-customtools',
'gemini-3.1-flash-lite-preview',
'gemini-3.5-flash',
'gemini-2.5-pro',
'gemini-2.5-flash',
'gemini-2.5-flash-lite',
Expand Down Expand Up @@ -1544,6 +1545,7 @@ describe('Google Model Tests', () => {
'gemini-3.1-pro-preview': 'gemini-3.1',
'gemini-3.1-pro-preview-customtools': 'gemini-3.1',
'gemini-3.1-flash-lite-preview': 'gemini-3.1-flash-lite',
'gemini-3.5-flash': 'gemini-3.5-flash',
'gemini-2.5-pro': 'gemini-2.5-pro',
'gemini-2.5-flash': 'gemini-2.5-flash',
'gemini-2.5-flash-lite': 'gemini-2.5-flash-lite',
Expand Down Expand Up @@ -1645,6 +1647,22 @@ describe('Google Model Tests', () => {
cacheTokenValues['gemini-3.1-flash-lite'].read,
);
});

it('should return correct rates for Gemini 3.5 Flash', () => {
const model = 'gemini-3.5-flash';
expect(getMultiplier({ model, tokenType: 'prompt', endpoint: EModelEndpoint.google })).toBe(
tokenValues['gemini-3.5-flash'].prompt,
);
expect(getMultiplier({ model, tokenType: 'completion', endpoint: EModelEndpoint.google })).toBe(
tokenValues['gemini-3.5-flash'].completion,
);
expect(getCacheMultiplier({ model, cacheType: 'write' })).toBe(
cacheTokenValues['gemini-3.5-flash'].write,
);
expect(getCacheMultiplier({ model, cacheType: 'read' })).toBe(
cacheTokenValues['gemini-3.5-flash'].read,
);
});
});

describe('Gemini 3.1 Premium Token Pricing', () => {
Expand Down
3 changes: 3 additions & 0 deletions packages/data-schemas/src/methods/tx.ts
Original file line number Diff line number Diff line change
Expand Up @@ -186,6 +186,7 @@ export const tokenValues: Record<string, { prompt: number; completion: number }>
'gemini-3-pro-image': { prompt: 2, completion: 120 },
'gemini-3.1': { prompt: 2, completion: 12 },
'gemini-3.1-flash-lite': { prompt: 0.25, completion: 1.5 },
'gemini-3.5-flash': { prompt: 1.5, completion: 9 },
'gemini-pro-vision': { prompt: 0.5, completion: 1.5 },
grok: { prompt: 2.0, completion: 10.0 },
'grok-beta': { prompt: 5.0, completion: 15.0 },
Expand Down Expand Up @@ -329,6 +330,8 @@ export const cacheTokenValues: Record<string, { write: number; read: number }> =
'gemini-3.1': { write: 2, read: 0.2 },
// Gemini 3.1 Flash-Lite - cache write: $0.25/1M, cache read: $0.025/1M
'gemini-3.1-flash-lite': { write: 0.25, read: 0.025 },
// Gemini 3.5 Flash - cache write: $1.50/1M, cache read: $0.15/1M
'gemini-3.5-flash': { write: 1.5, read: 0.15 },
};

/**
Expand Down
Loading