diff --git a/api/app/clients/OpenAIClient.js b/api/app/clients/OpenAIClient.js index ca0c8d8424..d1b908efbb 100644 --- a/api/app/clients/OpenAIClient.js +++ b/api/app/clients/OpenAIClient.js @@ -131,7 +131,8 @@ class OpenAIClient extends BaseClient { const { isChatGptModel } = this; this.isUnofficialChatGptModel = model.startsWith('text-chat') || model.startsWith('text-davinci-002-render'); - this.maxContextTokens = getModelMaxTokens(model) ?? 4095; // 1 less than maximum + this.maxContextTokens = + getModelMaxTokens(model, this.options.endpointType ?? this.options.endpoint) ?? 4095; // 1 less than maximum if (this.shouldSummarize) { this.maxContextTokens = Math.floor(this.maxContextTokens / 2); diff --git a/api/server/services/Config/loadCustomConfig.js b/api/server/services/Config/loadCustomConfig.js index c17d3283b4..769e4fb1e7 100644 --- a/api/server/services/Config/loadCustomConfig.js +++ b/api/server/services/Config/loadCustomConfig.js @@ -25,7 +25,8 @@ async function loadCustomConfig() { logger.error(`Invalid custom config file at ${configPath}`, result.error); return null; } else { - logger.info('Loaded custom config file'); + logger.info('Loaded custom config file:'); + logger.info(JSON.stringify(customConfig, null, 2)); } if (customConfig.cache) { diff --git a/api/utils/tokens.js b/api/utils/tokens.js index 3c95cd96a2..fb6e363d8b 100644 --- a/api/utils/tokens.js +++ b/api/utils/tokens.js @@ -57,28 +57,32 @@ const openAIModels = { 'mistral-': 31990, // -10 from max }; +const googleModels = { + /* Max I/O is combined so we subtract the amount from max response tokens for actual total */ + gemini: 32750, // -10 from max + 'text-bison-32k': 32758, // -10 from max + 'chat-bison-32k': 32758, // -10 from max + 'code-bison-32k': 32758, // -10 from max + 'codechat-bison-32k': 32758, + /* Codey, -5 from max: 6144 */ + 'code-': 6139, + 'codechat-': 6139, + /* PaLM2, -5 from max: 8192 */ + 'text-': 8187, + 'chat-': 8187, +}; + +const anthropicModels = { + 'claude-2.1': 200000, + 'claude-': 100000, +}; + // Order is important here: by model series and context size (gpt-4 then gpt-3, ascending) const maxTokensMap = { [EModelEndpoint.openAI]: openAIModels, - [EModelEndpoint.custom]: openAIModels, - [EModelEndpoint.google]: { - /* Max I/O is combined so we subtract the amount from max response tokens for actual total */ - gemini: 32750, // -10 from max - 'text-bison-32k': 32758, // -10 from max - 'chat-bison-32k': 32758, // -10 from max - 'code-bison-32k': 32758, // -10 from max - 'codechat-bison-32k': 32758, - /* Codey, -5 from max: 6144 */ - 'code-': 6139, - 'codechat-': 6139, - /* PaLM2, -5 from max: 8192 */ - 'text-': 8187, - 'chat-': 8187, - }, - [EModelEndpoint.anthropic]: { - 'claude-2.1': 200000, - 'claude-': 100000, - }, + [EModelEndpoint.custom]: { ...openAIModels, ...googleModels, ...anthropicModels }, + [EModelEndpoint.google]: googleModels, + [EModelEndpoint.anthropic]: anthropicModels, }; /**