From c9ca4dfa14e5464e053b0df0e0978f6965f5083d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?F=C3=A9lix=20Malfait?= Date: Mon, 23 Mar 2026 13:07:53 +0100 Subject: [PATCH] fix: run AI catalog sync as standalone script to avoid DB dependency (#18853) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ## Summary - The `ci-ai-catalog-sync` cron workflow was failing because the `ai:sync-models-dev` NestJS command bootstraps the full app, which tries to connect to PostgreSQL — unavailable in CI. - Converted the sync logic to a standalone `ts-node` script (`scripts/ai-sync-models-dev.ts`) that runs without NestJS, eliminating the database dependency. - Removed the `Build twenty-server` step from the workflow since it's no longer needed, making the job faster. ## Test plan - [x] Verified the standalone script runs successfully locally via `npx nx run twenty-server:ts-node-no-deps-transpile-only -- ./scripts/ai-sync-models-dev.ts` - [x] Verified `--dry-run` flag works correctly - [x] Verified the output `ai-providers.json` is correctly written with valid JSON (135 models across 5 providers) - [x] Verified the script passes linting with zero errors - [ ] CI should pass without requiring a database service Fixes: https://github.com/twentyhq/twenty/actions/runs/23424202182/job/68135439740 Made with [Cursor](https://cursor.com) --- .github/workflows/ci-ai-catalog-sync.yaml | 5 +- .../scripts/ai-sync-models-dev.ts | 343 +++++++++++++++++ .../commands/ai-sync-models-dev.command.ts | 360 ------------------ .../commands/database-command.module.ts | 2 - 4 files changed, 344 insertions(+), 366 deletions(-) create mode 100644 packages/twenty-server/scripts/ai-sync-models-dev.ts delete mode 100644 packages/twenty-server/src/database/commands/ai-sync-models-dev.command.ts diff --git a/.github/workflows/ci-ai-catalog-sync.yaml b/.github/workflows/ci-ai-catalog-sync.yaml index 19b4fe9ac69..881f964b2c0 100644 --- a/.github/workflows/ci-ai-catalog-sync.yaml +++ b/.github/workflows/ci-ai-catalog-sync.yaml @@ -27,11 +27,8 @@ jobs: - name: Build dependencies run: npx nx build twenty-shared - - name: Build twenty-server - run: npx nx build twenty-server - - name: Run catalog sync - run: npx nx run twenty-server:command-no-deps ai:sync-models-dev + run: npx nx run twenty-server:ts-node-no-deps-transpile-only -- ./scripts/ai-sync-models-dev.ts - name: Check for changes id: changes diff --git a/packages/twenty-server/scripts/ai-sync-models-dev.ts b/packages/twenty-server/scripts/ai-sync-models-dev.ts new file mode 100644 index 00000000000..aaa9cc82b7f --- /dev/null +++ b/packages/twenty-server/scripts/ai-sync-models-dev.ts @@ -0,0 +1,343 @@ +import * as fs from 'fs'; +import * as path from 'path'; + +import { + type AiSdkPackage, + NATIVE_AI_SDK_PROVIDER_IDS, +} from 'twenty-shared/ai'; + +import { MODELS_DEV_API_URL } from 'src/engine/metadata-modules/ai/ai-models/constants/models-dev.const'; +import { type ModelFamily } from 'src/engine/metadata-modules/ai/ai-models/types/model-family.enum'; +import { type ModelsDevData } from 'src/engine/metadata-modules/ai/ai-models/types/models-dev-data.type'; +import { inferModelFamily } from 'src/engine/metadata-modules/ai/ai-models/utils/infer-model-family.util'; + +const EXCLUDED_MODEL_PREFIXES = [ + 'text-embedding', + 'embedding', + 'dall-e', + 'tts-', + 'whisper', + 'moderation', + 'davinci', + 'babbage', + 'ada', + 'curie', + 'text-search', + 'text-similarity', + 'code-search', + 'text-davinci', + 'text-curie', + 'text-babbage', + 'text-ada', + 'ft:', + 'canary', +]; + +const EXCLUDED_MODEL_SUFFIXES = ['-audio-preview', '-realtime-preview']; + +const LONG_CONTEXT_THRESHOLD_TOKENS = 200000; + +const PROVIDER_LABELS: Record = { + openai: 'OpenAI', + anthropic: 'Anthropic', + google: 'Google', + mistral: 'Mistral', + xai: 'xAI', +}; + +const API_KEY_TEMPLATES: Record = { + openai: '{{OPENAI_API_KEY}}', + anthropic: '{{ANTHROPIC_API_KEY}}', + google: '{{GOOGLE_API_KEY}}', + mistral: '{{MISTRAL_API_KEY}}', + xai: '{{XAI_API_KEY}}', +}; + +type LongContextCostEntry = { + inputCostPerMillionTokens: number; + outputCostPerMillionTokens: number; + cachedInputCostPerMillionTokens?: number; + cacheCreationCostPerMillionTokens?: number; + thresholdTokens: number; +}; + +type GeneratedModel = { + name: string; + label: string; + description?: string; + modelFamily?: ModelFamily; + inputCostPerMillionTokens?: number; + outputCostPerMillionTokens?: number; + cachedInputCostPerMillionTokens?: number; + cacheCreationCostPerMillionTokens?: number; + longContextCost?: LongContextCostEntry; + contextWindowTokens?: number; + maxOutputTokens?: number; + modalities?: string[]; + supportsReasoning?: boolean; + isDeprecated?: boolean; +}; + +type GeneratedProvider = { + npm: AiSdkPackage; + label: string; + apiKey: string; + models: GeneratedModel[]; +}; + +const isLanguageModel = (modelId: string): boolean => { + const lowerId = modelId.toLowerCase(); + + for (const prefix of EXCLUDED_MODEL_PREFIXES) { + if (lowerId.startsWith(prefix)) return false; + } + + for (const suffix of EXCLUDED_MODEL_SUFFIXES) { + if (lowerId.endsWith(suffix)) return false; + } + + return true; +}; + +const meetsInclusionCriteria = ( + modelData: Record, +): boolean => { + if (modelData.status === 'beta') return false; + if (modelData.tool_call !== true) return false; + + const cost = modelData.cost as + | { input?: number; output?: number } + | undefined; + + if (cost?.input === undefined) return false; + + const limit = modelData.limit as + | { context?: number; output?: number } + | undefined; + + if (limit?.context === undefined) return false; + + return true; +}; + +const extractCost = ( + modelData: Record, + model: GeneratedModel, +): void => { + const cost = modelData.cost as Record | undefined; + + if (!cost) return; + + if (typeof cost.input === 'number') { + model.inputCostPerMillionTokens = cost.input; + } + if (typeof cost.output === 'number') { + model.outputCostPerMillionTokens = cost.output; + } + if (typeof cost.cache_read === 'number') { + model.cachedInputCostPerMillionTokens = cost.cache_read; + } + if (typeof cost.cache_write === 'number') { + model.cacheCreationCostPerMillionTokens = cost.cache_write; + } + + const longCtx = cost.context_over_200k as + | Record + | undefined; + + if (longCtx && typeof longCtx.input === 'number') { + model.longContextCost = { + inputCostPerMillionTokens: longCtx.input as number, + outputCostPerMillionTokens: (longCtx.output as number) ?? 0, + thresholdTokens: LONG_CONTEXT_THRESHOLD_TOKENS, + }; + if (typeof longCtx.cache_read === 'number') { + model.longContextCost.cachedInputCostPerMillionTokens = + longCtx.cache_read; + } + if (typeof longCtx.cache_write === 'number') { + model.longContextCost.cacheCreationCostPerMillionTokens = + longCtx.cache_write; + } + } +}; + +const extractLimits = ( + modelData: Record, + model: GeneratedModel, +): void => { + const limit = modelData.limit as Record | undefined; + + if (!limit) return; + + if (typeof limit.context === 'number') { + model.contextWindowTokens = limit.context; + } + if (typeof limit.output === 'number') { + model.maxOutputTokens = limit.output; + } +}; + +const extractModalities = ( + modelData: Record, + model: GeneratedModel, +): void => { + const modalities = modelData.modalities as { input?: string[] } | undefined; + + if (!modalities?.input) return; + + const relevant = modalities.input.filter((modality) => modality !== 'text'); + + if (relevant.length > 0) { + model.modalities = relevant; + } +}; + +const buildModelsForProvider = ( + providerName: string, + modelsDevModels: Record>, +): GeneratedModel[] => { + const qualifying: GeneratedModel[] = []; + + for (const [modelId, modelData] of Object.entries(modelsDevModels)) { + if (!isLanguageModel(modelId)) continue; + if (!meetsInclusionCriteria(modelData)) continue; + + const family = inferModelFamily(providerName, modelId); + + const model: GeneratedModel = { + name: modelId, + label: modelData.name ?? modelId, + modelFamily: family, + }; + + extractCost(modelData, model); + extractLimits(modelData, model); + extractModalities(modelData, model); + + if (modelData.reasoning === true) { + model.supportsReasoning = true; + } + + if (modelData.status === 'deprecated') { + model.isDeprecated = true; + } + + qualifying.push(model); + } + + return qualifying; +}; + +const generateCatalog = ( + data: ModelsDevData, +): Record => { + const result: Record = {}; + + for (const providerName of NATIVE_AI_SDK_PROVIDER_IDS) { + const providerData = data[providerName]; + + if (!providerData) { + // oxlint-disable-next-line no-console + console.warn(`Provider "${providerName}" not found in models.dev`); + continue; + } + + const models = buildModelsForProvider(providerName, providerData.models); + + if (models.length === 0) { + // oxlint-disable-next-line no-console + console.warn(`No qualifying models for "${providerName}", skipping`); + continue; + } + + result[providerName] = { + npm: `@ai-sdk/${providerName}`, + label: PROVIDER_LABELS[providerName] ?? providerName, + apiKey: API_KEY_TEMPLATES[providerName] ?? '', + models, + }; + } + + return result; +}; + +const printSummary = (catalog: Record): void => { + // oxlint-disable-next-line no-console + console.log('=== Generation Summary ==='); + + let totalModels = 0; + let deprecatedCount = 0; + + for (const [providerName, provider] of Object.entries(catalog)) { + const deprecatedModelCount = provider.models.filter( + (model) => model.isDeprecated, + ).length; + const active = provider.models.length - deprecatedModelCount; + + // oxlint-disable-next-line no-console + console.log( + ` ${providerName}: ${provider.models.length} models (${active} active, ${deprecatedModelCount} deprecated)`, + ); + totalModels += provider.models.length; + deprecatedCount += deprecatedModelCount; + } + + // oxlint-disable-next-line no-console + console.log( + `Total: ${totalModels} models (${totalModels - deprecatedCount} active, ${deprecatedCount} deprecated)`, + ); +}; + +const main = async (): Promise => { + const dryRun = process.argv.includes('--dry-run'); + + // oxlint-disable-next-line no-console + console.log('Fetching models.dev API...'); + + const response = await fetch(MODELS_DEV_API_URL); + + if (!response.ok) { + throw new Error( + `Failed to fetch: ${response.status} ${response.statusText}`, + ); + } + + const data: ModelsDevData = await response.json(); + + // oxlint-disable-next-line no-console + console.log(`Fetched ${Object.keys(data).length} providers from models.dev`); + + const generated = generateCatalog(data); + const json = JSON.stringify(generated, null, 2) + '\n'; + + printSummary(generated); + + if (dryRun) { + // oxlint-disable-next-line no-console + console.log('[DRY RUN] Would write ai-providers.json'); + + return; + } + + const outputPath = path.resolve( + __dirname, + '..', + 'src', + 'engine', + 'metadata-modules', + 'ai', + 'ai-models', + 'ai-providers.json', + ); + + fs.writeFileSync(outputPath, json, 'utf-8'); + // oxlint-disable-next-line no-console + console.log(`Wrote ${outputPath}`); +}; + +main().catch((error) => { + // oxlint-disable-next-line no-console + console.error('AI catalog sync failed:', error); + process.exit(1); +}); diff --git a/packages/twenty-server/src/database/commands/ai-sync-models-dev.command.ts b/packages/twenty-server/src/database/commands/ai-sync-models-dev.command.ts deleted file mode 100644 index af13ecff1ee..00000000000 --- a/packages/twenty-server/src/database/commands/ai-sync-models-dev.command.ts +++ /dev/null @@ -1,360 +0,0 @@ -import * as fs from 'fs'; -import * as path from 'path'; - -import { Logger } from '@nestjs/common'; - -import { Command, CommandRunner, Option } from 'nest-commander'; - -import { - type AiSdkPackage, - NATIVE_AI_SDK_PROVIDER_IDS, -} from 'twenty-shared/ai'; - -import { MODELS_DEV_API_URL } from 'src/engine/metadata-modules/ai/ai-models/constants/models-dev.const'; -import { type ModelFamily } from 'src/engine/metadata-modules/ai/ai-models/types/model-family.enum'; -import { type ModelsDevData } from 'src/engine/metadata-modules/ai/ai-models/types/models-dev-data.type'; -import { inferModelFamily } from 'src/engine/metadata-modules/ai/ai-models/utils/infer-model-family.util'; - -const EXCLUDED_MODEL_PREFIXES = [ - 'text-embedding', - 'embedding', - 'dall-e', - 'tts-', - 'whisper', - 'moderation', - 'davinci', - 'babbage', - 'ada', - 'curie', - 'text-search', - 'text-similarity', - 'code-search', - 'text-davinci', - 'text-curie', - 'text-babbage', - 'text-ada', - 'ft:', - 'canary', -]; - -const EXCLUDED_MODEL_SUFFIXES = ['-audio-preview', '-realtime-preview']; - -const LONG_CONTEXT_THRESHOLD_TOKENS = 200000; - -type ProviderLabels = Record; - -const PROVIDER_LABELS: ProviderLabels = { - openai: 'OpenAI', - anthropic: 'Anthropic', - google: 'Google', - mistral: 'Mistral', - xai: 'xAI', -}; - -const API_KEY_TEMPLATES: Record = { - openai: '{{OPENAI_API_KEY}}', - anthropic: '{{ANTHROPIC_API_KEY}}', - google: '{{GOOGLE_API_KEY}}', - mistral: '{{MISTRAL_API_KEY}}', - xai: '{{XAI_API_KEY}}', -}; - -type LongContextCostEntry = { - inputCostPerMillionTokens: number; - outputCostPerMillionTokens: number; - cachedInputCostPerMillionTokens?: number; - cacheCreationCostPerMillionTokens?: number; - thresholdTokens: number; -}; - -type GeneratedModel = { - name: string; - label: string; - description?: string; - modelFamily?: ModelFamily; - inputCostPerMillionTokens?: number; - outputCostPerMillionTokens?: number; - cachedInputCostPerMillionTokens?: number; - cacheCreationCostPerMillionTokens?: number; - longContextCost?: LongContextCostEntry; - contextWindowTokens?: number; - maxOutputTokens?: number; - modalities?: string[]; - supportsReasoning?: boolean; - isDeprecated?: boolean; -}; - -type GeneratedProvider = { - npm: AiSdkPackage; - label: string; - apiKey: string; - models: GeneratedModel[]; -}; - -type CommandOptions = { dryRun?: boolean }; - -@Command({ - name: 'ai:sync-models-dev', - description: - 'Generate ai-providers.json from models.dev API data with objective inclusion/deprecation criteria', -}) -export class AiSyncModelsDevCommand extends CommandRunner { - private readonly logger = new Logger(AiSyncModelsDevCommand.name); - - @Option({ - flags: '-d, --dry-run', - description: 'Print what would change without writing ai-providers.json', - required: false, - }) - parseDryRun(): boolean { - return true; - } - - async run(_args: string[], options?: CommandOptions): Promise { - const dryRun = options?.dryRun ?? false; - - this.logger.log('Fetching models.dev API...'); - - const response = await fetch(MODELS_DEV_API_URL); - - if (!response.ok) { - this.logger.error( - `Failed to fetch: ${response.status} ${response.statusText}`, - ); - - return; - } - - const data: ModelsDevData = await response.json(); - - this.logger.log( - `Fetched ${Object.keys(data).length} providers from models.dev`, - ); - - const generated = this.generateCatalog(data); - const json = JSON.stringify(generated, null, 2) + '\n'; - - this.printSummary(generated); - - if (dryRun) { - this.logger.log('[DRY RUN] Would write ai-providers.json'); - - return; - } - - const outputPath = path.resolve( - process.cwd(), - 'src', - 'engine', - 'metadata-modules', - 'ai', - 'ai-models', - 'ai-providers.json', - ); - - fs.writeFileSync(outputPath, json, 'utf-8'); - this.logger.log(`Wrote ${outputPath}`); - } - - private generateCatalog( - data: ModelsDevData, - ): Record { - const result: Record = {}; - - for (const providerName of NATIVE_AI_SDK_PROVIDER_IDS) { - const providerData = data[providerName]; - - if (!providerData) { - this.logger.warn(`Provider "${providerName}" not found in models.dev`); - continue; - } - - const models = this.buildModelsForProvider( - providerName, - providerData.models, - ); - - if (models.length === 0) { - this.logger.warn( - `No qualifying models for "${providerName}", skipping`, - ); - continue; - } - - result[providerName] = { - npm: `@ai-sdk/${providerName}`, - label: PROVIDER_LABELS[providerName] ?? providerName, - apiKey: API_KEY_TEMPLATES[providerName] ?? '', - models, - }; - } - - return result; - } - - private buildModelsForProvider( - providerName: string, - modelsDevModels: Record>, - ): GeneratedModel[] { - const qualifying: GeneratedModel[] = []; - - for (const [modelId, modelData] of Object.entries(modelsDevModels)) { - if (!this.isLanguageModel(modelId)) continue; - if (!this.meetsInclusionCriteria(modelData)) continue; - - const family = inferModelFamily(providerName, modelId); - - const model: GeneratedModel = { - name: modelId, - label: modelData.name ?? modelId, - modelFamily: family, - }; - - this.extractCost(modelData, model); - this.extractLimits(modelData, model); - this.extractModalities(modelData, model); - - if (modelData.reasoning === true) { - model.supportsReasoning = true; - } - - if (modelData.status === 'deprecated') { - model.isDeprecated = true; - } - - qualifying.push(model); - } - - return qualifying; - } - - private extractCost( - modelData: Record, - model: GeneratedModel, - ): void { - const cost = modelData.cost as Record | undefined; - - if (!cost) return; - - if (typeof cost.input === 'number') { - model.inputCostPerMillionTokens = cost.input; - } - if (typeof cost.output === 'number') { - model.outputCostPerMillionTokens = cost.output; - } - if (typeof cost.cache_read === 'number') { - model.cachedInputCostPerMillionTokens = cost.cache_read; - } - if (typeof cost.cache_write === 'number') { - model.cacheCreationCostPerMillionTokens = cost.cache_write; - } - - const longCtx = cost.context_over_200k as - | Record - | undefined; - - if (longCtx && typeof longCtx.input === 'number') { - model.longContextCost = { - inputCostPerMillionTokens: longCtx.input as number, - outputCostPerMillionTokens: (longCtx.output as number) ?? 0, - thresholdTokens: LONG_CONTEXT_THRESHOLD_TOKENS, - }; - if (typeof longCtx.cache_read === 'number') { - model.longContextCost.cachedInputCostPerMillionTokens = - longCtx.cache_read; - } - if (typeof longCtx.cache_write === 'number') { - model.longContextCost.cacheCreationCostPerMillionTokens = - longCtx.cache_write; - } - } - } - - private extractLimits( - modelData: Record, - model: GeneratedModel, - ): void { - const limit = modelData.limit as Record | undefined; - - if (!limit) return; - - if (typeof limit.context === 'number') { - model.contextWindowTokens = limit.context; - } - if (typeof limit.output === 'number') { - model.maxOutputTokens = limit.output; - } - } - - private extractModalities( - modelData: Record, - model: GeneratedModel, - ): void { - const modalities = modelData.modalities as { input?: string[] } | undefined; - - if (!modalities?.input) return; - - const relevant = modalities.input.filter((modality) => modality !== 'text'); - - if (relevant.length > 0) { - model.modalities = relevant; - } - } - - private isLanguageModel(modelId: string): boolean { - const lowerId = modelId.toLowerCase(); - - for (const prefix of EXCLUDED_MODEL_PREFIXES) { - if (lowerId.startsWith(prefix)) return false; - } - - for (const suffix of EXCLUDED_MODEL_SUFFIXES) { - if (lowerId.endsWith(suffix)) return false; - } - - return true; - } - - private meetsInclusionCriteria(modelData: Record): boolean { - if (modelData.status === 'beta') return false; - if (modelData.tool_call !== true) return false; - - const cost = modelData.cost as - | { input?: number; output?: number } - | undefined; - - if (cost?.input === undefined) return false; - - const limit = modelData.limit as - | { context?: number; output?: number } - | undefined; - - if (limit?.context === undefined) return false; - - return true; - } - - private printSummary(catalog: Record): void { - this.logger.log('=== Generation Summary ==='); - - let totalModels = 0; - let deprecatedCount = 0; - - for (const [providerName, provider] of Object.entries(catalog)) { - const deprecatedModelCount = provider.models.filter( - (model) => model.isDeprecated, - ).length; - const active = provider.models.length - deprecatedModelCount; - - this.logger.log( - ` ${providerName}: ${provider.models.length} models (${active} active, ${deprecatedModelCount} deprecated)`, - ); - totalModels += provider.models.length; - deprecatedCount += deprecatedModelCount; - } - - this.logger.log( - `Total: ${totalModels} models (${totalModels - deprecatedCount} active, ${deprecatedCount} deprecated)`, - ); - } -} diff --git a/packages/twenty-server/src/database/commands/database-command.module.ts b/packages/twenty-server/src/database/commands/database-command.module.ts index 2dea8b6a311..f2dd7cc8595 100644 --- a/packages/twenty-server/src/database/commands/database-command.module.ts +++ b/packages/twenty-server/src/database/commands/database-command.module.ts @@ -1,7 +1,6 @@ import { Module } from '@nestjs/common'; import { TypeOrmModule } from '@nestjs/typeorm'; -import { AiSyncModelsDevCommand } from 'src/database/commands/ai-sync-models-dev.command'; import { CronRegisterAllCommand } from 'src/database/commands/cron-register-all.command'; import { DataSeedWorkspaceCommand } from 'src/database/commands/data-seed-dev-workspace.command'; import { ListOrphanedWorkspaceEntitiesCommand } from 'src/database/commands/list-and-delete-orphaned-workspace-entities.command'; @@ -71,7 +70,6 @@ import { AutomatedTriggerModule } from 'src/modules/workflow/workflow-trigger/au StaleRegistrationCleanupModule, ], providers: [ - AiSyncModelsDevCommand, DataSeedWorkspaceCommand, ConfirmationQuestion, CronRegisterAllCommand,