Rename copilot-fast/copilot-base to copilot-utility-small/copilot-utility

Introduces dedicated endpoint resolver classes (CopilotUtilitySmallChatEndpoint,
CopilotUtilityChatEndpoint) in place of the ModelAliasRegistry indirection, and
republishes the resolved utility endpoints under their new family ids
('copilot-utility-small', 'copilot-utility') as LanguageModelChatInformation
entries so that workbench callers using selectLanguageModels({ vendor: 'copilot',
id: 'copilot-utility-small' }) keep working.

- Renames the internal family identifiers everywhere they're consumed:
  callers, tests, and workbench code in src/vs/workbench/contrib/chat/.
- Drops src/platform/endpoint/common/modelAliasRegistry.ts.
- CopilotUtilitySmallChatEndpoint.resolve tries a small primary model and
  falls back to a second small model if the primary is unavailable.
- CopilotUtilityChatEndpoint.resolve returns the API-marked default base
  model (is_chat_fallback === true), preserving existing behavior.
- Updates _copilotBaseModel field and 'copilot-base' literal in
  ModelMetadataFetcher to _copilotUtilityModel / 'copilot-utility'.

No user-facing behavior change. The setting-driven 'disabled'/'default'/BYOK
selector support will be layered on top in a follow-up.

Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com>

remove comment

remove code smell

remove more code smell
This commit is contained in:
vritant24
2026-05-14 10:37:23 -07:00
co-authored by Copilot
parent be263d4ccd
commit ef061ccb0f
60 changed files with 224 additions and 172 deletions
@@ -277,7 +277,7 @@ Learn more about [GitHub Copilot](https://docs.github.com/copilot/using-github-c
private async switchToBaseModel(request: vscode.ChatRequest, stream: vscode.ChatResponseStream): Promise<ChatRequest> {
const endpoint = await this.endpointProvider.getChatEndpoint(request);
const baseEndpoint = await this.endpointProvider.getChatEndpoint('copilot-base');
const baseEndpoint = await this.endpointProvider.getChatEndpoint('copilot-utility');
// If it has a 0x multipler, it's free so don't switch them. If it's BYOK, it's free so don't switch them.
if (endpoint.multiplier === 0 || request.model.vendor !== 'copilot' || endpoint.multiplier === undefined) {
return request;
@@ -12,9 +12,8 @@ import { IBlockedExtensionService } from '../../../platform/chat/common/blockedE
import { ChatFetchResponseType, ChatLocation, getErrorDetailsFromChatFetchError } from '../../../platform/chat/common/commonTypes';
import { getTextPart } from '../../../platform/chat/common/globalStringUtils';
import { EmbeddingType, getWellKnownEmbeddingTypeInfo, IEmbeddingsComputer } from '../../../platform/embeddings/common/embeddingsComputer';
import { IEndpointProvider } from '../../../platform/endpoint/common/endpointProvider';
import { ChatEndpointFamily, IEndpointProvider } from '../../../platform/endpoint/common/endpointProvider';
import { CustomDataPartMimeTypes } from '../../../platform/endpoint/common/endpointTypes';
import { ModelAliasRegistry } from '../../../platform/endpoint/common/modelAliasRegistry';
import { encodeStatefulMarker } from '../../../platform/endpoint/common/statefulMarkerContainer';
import { isAnthropicFamily, isGeminiFamily } from '../../../platform/endpoint/common/chatModelCapabilities';
import { AutoChatEndpoint } from '../../../platform/endpoint/node/autoChatEndpoint';
@@ -168,6 +167,13 @@ export class LanguageModelAccess extends Disposable implements IExtensionContrib
private readonly _onDidChange = this._register(new Emitter<void>());
private _currentModels: vscode.LanguageModelChatInformation[] = []; // Store current models for reference
private _chatEndpoints: IChatEndpoint[] = [];
/**
* Maps utility family aliases (e.g. `copilot-utility-small`,
* `copilot-utility`) to the resolved endpoint they were last published
* under. Lets {@link _getEndpointForModel} route alias lookups to the
* underlying endpoint without re-resolving the user setting.
*/
private _utilityAliasEndpoints: Map<string, IChatEndpoint> = new Map();
private _lmWrapper: CopilotLanguageModelWrapper;
private _promptBaseCountCache: LanguageModelAccessPromptBaseCountCache;
@@ -330,30 +336,57 @@ export class LanguageModelAccess extends Disposable implements IExtensionContrib
};
models.push(model);
// Register aliases for this model
const aliases = ModelAliasRegistry.getAliases(model.id);
for (const alias of aliases) {
models.push({
...model,
id: alias,
family: alias,
isUserSelectable: false,
});
}
}
this._currentModels = models;
this._chatEndpoints = chatEndpoints;
await this._registerUtilityAliasModels(models);
return models;
}
private async _registerUtilityAliasModels(models: vscode.LanguageModelChatInformation[]): Promise<void> {
this._utilityAliasEndpoints.clear();
const aliasFamilies: ChatEndpointFamily[] = ['copilot-utility-small', 'copilot-utility'];
for (const family of aliasFamilies) {
let endpoint: IChatEndpoint | undefined;
try {
endpoint = await this._endpointProvider.getChatEndpoint(family);
} catch (err) {
this._logService.warn(`[LanguageModelAccess] Failed to resolve utility alias '${family}': ${err}`);
continue;
}
if (!endpoint) {
continue;
}
// Only republish endpoints that are surfaced by this provider
// (vendor `copilot`). BYOK selectors are already published by
// their own provider under a different vendor.
const base = models.find(m => m.id === endpoint.model);
if (!base) {
continue;
}
this._utilityAliasEndpoints.set(family, endpoint);
models.push({
...base,
id: family,
family,
isUserSelectable: false,
isDefault: false,
});
}
}
private async _getEndpointForModel(model: vscode.LanguageModelChatInformation) {
if (model.id === AutoChatEndpoint.pseudoModelId) {
const allEndpoints = await this._endpointProvider.getAllChatEndpoints();
return await this._automodeService.resolveAutoModeEndpoint(undefined, allEndpoints);
}
return this._chatEndpoints.find(e => e.model === ModelAliasRegistry.resolveAlias(model.id));
const aliasEndpoint = this._utilityAliasEndpoints.get(model.id);
if (aliasEndpoint) {
return aliasEndpoint;
}
return this._chatEndpoints.find(e => e.model === model.id);
}
private async _provideLanguageModelChatResponse(
@@ -158,7 +158,7 @@ class TerminalQuickFixGenerator {
}
}
const endpoint = await this._endpointProvider.getChatEndpoint('copilot-fast');
const endpoint = await this._endpointProvider.getChatEndpoint('copilot-utility-small');
const promptRenderer = PromptRenderer.create(this._instantiationService, endpoint, TerminalQuickFixPrompt, {
commandLine: commandMatchResult.commandLine,
@@ -216,7 +216,7 @@ class TerminalQuickFixGenerator {
}
private async _generateTerminalQuickFixFileContext(commandMatchResult: vscode.TerminalCommandMatchResult, token: CancellationToken) {
const endpoint = await this._endpointProvider.getChatEndpoint('copilot-fast');
const endpoint = await this._endpointProvider.getChatEndpoint('copilot-utility-small');
const promptRenderer = PromptRenderer.create(this._instantiationService, endpoint, TerminalQuickFixFileContextPrompt, {
commandLine: commandMatchResult.commandLine,
@@ -36,7 +36,7 @@ suite('CopilotLanguageModelWrapper', () => {
let endpoint: IChatEndpoint;
setup(async () => {
createAccessor();
endpoint = await accessor.get(IEndpointProvider).getChatEndpoint('copilot-base');
endpoint = await accessor.get(IEndpointProvider).getChatEndpoint('copilot-utility');
wrapper = instaService.createInstance(CopilotLanguageModelWrapper);
});
@@ -66,7 +66,7 @@ suite('CopilotLanguageModelWrapper', () => {
let endpoint: IChatEndpoint;
setup(async () => {
createAccessor();
endpoint = await accessor.get(IEndpointProvider).getChatEndpoint('copilot-base');
endpoint = await accessor.get(IEndpointProvider).getChatEndpoint('copilot-utility');
wrapper = instaService.createInstance(CopilotLanguageModelWrapper);
});
const runTest = async (messages: vscode.LanguageModelChatMessage[], tools?: vscode.LanguageModelChatTool[]) => {
@@ -97,7 +97,7 @@ suite('CopilotLanguageModelWrapper', () => {
setup(async () => {
createAccessor();
fetcher = accessor.get(IChatMLFetcher) as MockChatMLFetcher;
endpoint = await accessor.get(IEndpointProvider).getChatEndpoint('copilot-base');
endpoint = await accessor.get(IEndpointProvider).getChatEndpoint('copilot-utility');
wrapper = instaService.createInstance(CopilotLanguageModelWrapper);
});
@@ -94,7 +94,7 @@ export class InlineChatProgressMessages {
}
try {
const endpoint = await this._endpointProvider.getChatEndpoint('copilot-fast');
const endpoint = await this._endpointProvider.getChatEndpoint('copilot-utility-small');
const selectedCode = documentContext.selection.isEmpty
? undefined
@@ -187,7 +187,7 @@ export class InlineChatProgressMessages {
private async _fetchMessages(scenario: ProgressMessageScenario): Promise<void> {
try {
const endpoint = await this._endpointProvider.getChatEndpoint('copilot-fast');
const endpoint = await this._endpointProvider.getChatEndpoint('copilot-utility-small');
const props: ProgressMessagesPromptProps = { scenario, count: MESSAGES_PER_FETCH };
const { messages: promptMessages } = await renderPromptElement(
@@ -20,7 +20,7 @@ abstract class NewWorkspaceContentGenerator {
) { }
public async generate(promptArgs: NewWorkspaceContentsPromptProps, token: CancellationToken): Promise<string> {
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-fast');
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-utility-small');
const promptRenderer = PromptRenderer.create(this.instantiationService, endpoint, this.promptType, promptArgs);
const prompt = await promptRenderer.render();
@@ -55,7 +55,7 @@ export class SearchKeywordsIntent implements IIntent {
async invoke(invocationContext: IIntentInvocationContext): Promise<IIntentInvocation> {
const location = invocationContext.location;
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-fast');
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-utility-small');
return this.instantiationService.createInstance(SearchKeywordsIntentInvocation, this, location, endpoint);
}
}
@@ -55,7 +55,7 @@ export class SearchPanelIntent implements IIntent {
async invoke(invocationContext: IIntentInvocationContext): Promise<IIntentInvocation> {
const location = invocationContext.location;
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-base');
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-utility');
return this.instantiationService.createInstance(SearchIntentInvocation, this, location, endpoint);
}
}
@@ -32,7 +32,7 @@ export class UserQueryParser {
) { }
public async parse(query: string): Promise<ParsedUserQuery | null> {
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-fast');
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-utility-small');
const promptRenderer = PromptRenderer.create(
this.instantiationService,
endpoint,
@@ -53,7 +53,7 @@ export class McpToolCallingLoop extends ToolCallingLoop<IMcpToolCallingLoopOptio
}
private async getEndpoint() {
return await this.endpointProvider.getChatEndpoint('copilot-fast');
return await this.endpointProvider.getChatEndpoint('copilot-utility-small');
}
protected async buildPrompt(buildPromptContext: IBuildPromptContext, progress: Progress<ChatResponseReferencePart | ChatResponseProgressPart>, token: CancellationToken): Promise<IBuildPromptResult> {
@@ -51,7 +51,7 @@ export class DebugCommandToConfigConverter implements IDebugCommandToConfigConve
public async convert(cwd: string, args: readonly string[], token: CancellationToken): Promise<IDebugConfigResult> {
const relCwd = getPathRelativeToWorkspaceFolder(cwd, this.workspace);
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-base');
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-utility');
const promptRenderer = PromptRenderer.create(
this.instantiationService,
endpoint,
@@ -30,7 +30,7 @@ export class LanguageToolsProvider {
}
public async getToolsForLanguages(languages: string[], token: CancellationToken) {
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-base');
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-utility');
const promptRenderer = PromptRenderer.create(
this.instantiationService,
endpoint,
@@ -59,7 +59,7 @@ export class CodebaseToolCallingLoop extends ToolCallingLoop<ICodebaseToolCallin
private async getEndpoint(request: ChatRequest) {
let endpoint = await this.endpointProvider.getChatEndpoint(this.options.request);
if (!endpoint.supportsToolCalls) {
endpoint = await this.endpointProvider.getChatEndpoint('copilot-base');
endpoint = await this.endpointProvider.getChatEndpoint('copilot-utility');
}
return endpoint;
}
@@ -49,7 +49,7 @@ export class DevContainerConfigGenerator {
const startTime = Date.now();
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-base');
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-utility');
const charLimit = Math.floor((endpoint.modelMaxPromptTokens * 4) / 3);
const processedFilenames = this.processFilenames(filenames, charLimit);
@@ -54,7 +54,7 @@ export class FeedbackGenerator {
};
}
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-base');
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-utility');
const prompts: RenderPromptResult[] = [];
const batches = [filteredInput];
@@ -41,7 +41,7 @@ export class GitBranchNameGenerator {
const sessionResource = context.sessionResource;
const parentChatSessionId = sessionResource ? sessionResourceToId(URI.from(sessionResource)) : undefined;
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-fast');
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-utility-small');
const normalizedCommand = firstRequest.command?.trim().replace(/^\/+/, '') ?? '';
const command = normalizedCommand ? `/${normalizedCommand} ` : '';
const userRequest = `${command}${firstRequest.prompt}`;
@@ -33,7 +33,7 @@ export class GitCommitMessageGenerator {
async generateGitCommitMessage(repositoryName: string, branchName: string, changes: Diff[], recentCommitMessages: RecentCommitMessages, attemptCount: number, token: CancellationToken): Promise<string | undefined> {
const startTime = Date.now();
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-fast');
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-utility-small');
const promptRenderer = PromptRenderer.create(this.instantiationService, endpoint, GitCommitMessagePrompt, { repositoryName, branchName, changes, recentCommitMessages });
const prompt = await promptRenderer.render(undefined, undefined);
@@ -81,7 +81,7 @@ export class GitHubPullRequestTitleAndDescriptionGenerator implements TitleAndDe
const template: string | undefined = context.template;
const compareBranch: string | undefined = context.compareBranch;
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-fast');
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-utility-small');
const charLimit = Math.floor((endpoint.modelMaxPromptTokens * 4) / 3);
const prompt = await this.createPRTitleAndDescriptionPrompt(commitMessages, patches, issues, template, compareBranch, charLimit);
@@ -190,7 +190,7 @@ export class GitHubPullRequestTitleAndDescriptionGenerator implements TitleAndDe
}
}
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-fast');
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-utility-small');
const promptRenderer = PromptRenderer.create(this.instantiationService, endpoint, GitHubPullRequestPrompt, { commitMessages, issues, patches, template, compareBranch });
return promptRenderer.render(undefined, undefined);
}
@@ -187,7 +187,7 @@ export class IntentDetector implements ChatParticipantDetectionProvider {
return undefined;
}
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-fast');
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-utility-small');
const preferredIntent = await this.getPreferredIntent(location, documentContext, history, messageText);
@@ -260,7 +260,7 @@ export class IntentDetector implements ChatParticipantDetectionProvider {
history: Turn[] = [],
document?: TextDocumentSnapshot
) {
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-fast');
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-utility-small');
const { messages: currentSelection } = await renderPromptElement(this.instantiationService, endpoint, CurrentSelection, { document });
const { messages: conversationHistory } = await renderPromptElement(this.instantiationService, endpoint, ConversationHistory, { history, priority: 1000 }, undefined, undefined).catch(() => ({ messages: [] }));
@@ -172,13 +172,13 @@ export class PromptCategorizerService implements IPromptCategorizerService {
// Gather context signals (outside try block for telemetry access)
const currentLanguage = this.tabsAndEditorsService.activeTextEditor?.document.languageId;
// Use 10 second timeout - classification should be fast with copilot-fast model
// Use 10 second timeout - classification should be fast with copilot-utility-small model
const CATEGORIZATION_TIMEOUT_MS = 10_000;
const cts = new CancellationTokenSource();
const timeoutHandle = setTimeout(() => cts.cancel(), CATEGORIZATION_TIMEOUT_MS);
try {
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-fast');
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-utility-small');
const { messages } = await renderPromptElement(
this.instantiationService,
@@ -40,7 +40,7 @@ export class ChatSummarizerProvider implements vscode.ChatSummarizer {
return '';
}
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-fast');
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-utility-small');
const promptContext: IBuildPromptContext = {
requestId: 'chat-summary',
query: '',
@@ -41,7 +41,7 @@ export class ChatTitleProvider implements vscode.ChatTitleProvider {
const sessionResource = context.sessionResource;
const parentChatSessionId = sessionResource ? sessionResourceToId(URI.from(sessionResource)) : undefined;
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-fast');
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-utility-small');
const { messages } = await renderPromptElement(this.instantiationService, endpoint, TitlePrompt, { userRequest: firstRequest.prompt });
const capturingToken = new CapturingToken(
@@ -9,7 +9,7 @@ import { IConfigurationService } from '../../../platform/configuration/common/co
import { ChatEndpointFamily, EmbeddingsEndpointFamily, IChatModelInformation, ICompletionModelInformation, IEmbeddingModelInformation, IEndpointProvider } from '../../../platform/endpoint/common/endpointProvider';
import { AutoChatEndpoint } from '../../../platform/endpoint/node/autoChatEndpoint';
import { IAutomodeService } from '../../../platform/endpoint/node/automodeService';
import { CopilotChatEndpoint } from '../../../platform/endpoint/node/copilotChatEndpoint';
import { CopilotChatEndpoint, CopilotUtilityChatEndpoint, CopilotUtilitySmallChatEndpoint } from '../../../platform/endpoint/node/copilotChatEndpoint';
import { EmbeddingEndpoint } from '../../../platform/endpoint/node/embeddingsEndpoint';
import { IModelMetadataFetcher, ModelMetadataFetcher } from '../../../platform/endpoint/node/modelMetadataFetcher';
import { ExtensionContributedChatEndpoint } from '../../../platform/endpoint/vscode-node/extChatEndpoint';
@@ -66,14 +66,13 @@ export class ProductionEndpointProvider extends Disposable implements IEndpointP
this._logService.trace(`Resolving chat model`);
if (typeof requestOrFamilyOrModel === 'string') {
const modelMetadata = await this._modelFetcher.getChatModelFromFamily(requestOrFamilyOrModel);
return this.getOrCreateChatEndpointInstance(modelMetadata!);
return this._resolveUtilityFamily(requestOrFamilyOrModel);
}
const model = 'model' in requestOrFamilyOrModel ? requestOrFamilyOrModel.model : requestOrFamilyOrModel;
if (!model) {
return this.getChatEndpoint('copilot-base');
return this.getChatEndpoint('copilot-utility');
}
if (model.vendor !== 'copilot') {
@@ -85,13 +84,30 @@ export class ProductionEndpointProvider extends Disposable implements IEndpointP
const allEndpoints = await this.getAllChatEndpoints();
return this._autoModeService.resolveAutoModeEndpoint(requestOrFamilyOrModel as ChatRequest, allEndpoints);
} catch {
return this.getChatEndpoint('copilot-base');
return this.getChatEndpoint('copilot-utility');
}
}
const modelMetadata = await this._modelFetcher.getChatModelFromApiModel(model);
// If we fail to resolve a model since this is panel we give copilot base. This really should never happen as the picker is powered by the same service.
return modelMetadata ? this.getOrCreateChatEndpointInstance(modelMetadata) : this.getChatEndpoint('copilot-base');
// If we fail to resolve a model since this is panel we give copilot utility. This really should never happen as the picker is powered by the same service.
return modelMetadata ? this.getOrCreateChatEndpointInstance(modelMetadata) : this.getChatEndpoint('copilot-utility');
}
/**
* Resolves an internal utility family (`copilot-utility-small` /
* `copilot-utility`) to a concrete `CopilotChatEndpoint`. The model
* selection for each family lives in the corresponding resolver
* class so callers don't need to know which CAPI family backs each
* purpose.
*/
private _resolveUtilityFamily(family: ChatEndpointFamily): Promise<IChatEndpoint> {
if (family === 'copilot-utility-small') {
return CopilotUtilitySmallChatEndpoint.resolve(this._modelFetcher, this._instantiationService);
} else if (family === 'copilot-utility') {
return CopilotUtilityChatEndpoint.resolve(this._modelFetcher, this._instantiationService);
} else {
throw new Error(`Unrecognized chat endpoint family ${family}`);
}
}
async getEmbeddingsEndpoint(family?: EmbeddingsEndpointFamily): Promise<IEmbeddingsEndpoint> {
@@ -41,11 +41,11 @@ export class ScenarioAutomationEndpointProviderImpl extends ProductionEndpointPr
try {
return await super.getChatEndpoint(requestOrFamilyOrModel);
} catch (error) {
// In scenario automation, some model families (e.g. copilot-fast → gpt-4o-mini) may
// not be available via the capi proxy. Fall back to copilot-base.
// In scenario automation, some model families (e.g. copilot-utility-small → gpt-4o-mini) may
// not be available via the capi proxy. Fall back to copilot-utility.
if (typeof requestOrFamilyOrModel === 'string') {
this._logService.trace(`ScenarioAutomation: failed to resolve model family '${requestOrFamilyOrModel}', falling back to copilot-base`);
return super.getChatEndpoint('copilot-base');
this._logService.trace(`ScenarioAutomation: failed to resolve model family '${requestOrFamilyOrModel}', falling back to copilot-utility`);
return super.getChatEndpoint('copilot-utility');
}
throw error;
}
@@ -77,7 +77,7 @@ export class SettingsEditorSearchServiceImpl implements ISettingsEditorSearchSer
return;
}
const endpointName: ChatEndpointFamily = 'copilot-base';
const endpointName: ChatEndpointFamily = 'copilot-utility';
const endpoint = await this.endpointProvider.getChatEndpoint(endpointName);
const generator = this.instantiationService.createInstance(SettingsEditorSearchResultsSelector);
const llmSearchSuggestions = await generator.selectTopSearchResults(endpoint, query, embeddingSettings, token);
@@ -232,7 +232,7 @@ export class BackgroundTodoProcessor {
// ── First-pass fast path / progressive backoff ─────────────
// No todos exist yet for this session. We want to fire early so
// even pure-exploration sessions get a plan as soon as there is
// something to track — but not re-invoke copilot-fast on every
// something to track — but not re-invoke copilot-utility-small on every
// INITIAL_SUBSTANTIVE_THRESHOLD reads when the model keeps no-op'ing.
//
// After each no-op the required threshold doubles (exponential
@@ -503,7 +503,7 @@ export class BackgroundTodoProcessor {
this._consecutiveInitialNoops = 0;
} else if (!this._hasCreatedTodos) {
// noop on the initial branch — back off so exploration-heavy sessions
// don't re-invoke copilot-fast every INITIAL_SUBSTANTIVE_THRESHOLD reads.
// don't re-invoke copilot-utility-small every INITIAL_SUBSTANTIVE_THRESHOLD reads.
this._consecutiveInitialNoops++;
}
this._logService?.debug(`[BackgroundTodo] pass #${passNum} completed: outcome=${result.outcome}, durationMs=${result.durationMs ?? '?'}, model=${result.model ?? '?'}, promptTokens=${result.promptTokens ?? '?'}, completionTokens=${result.completionTokens ?? '?'}`);
@@ -544,7 +544,7 @@ export class BackgroundTodoProcessor {
}
/**
* The actual background work: render the todo prompt against copilot-fast,
* The actual background work: render the todo prompt against copilot-utility-small,
* parse tool calls, and invoke the todo tool.
*/
private static async _doExecute(
@@ -562,10 +562,10 @@ export class BackgroundTodoProcessor {
try {
fastEndpoint = await context.instantiationService.invokeFunction(async (accessor) => {
const ep = accessor.get(IEndpointProvider);
return ep.getChatEndpoint('copilot-fast');
return ep.getChatEndpoint('copilot-utility-small');
});
} catch (err) {
context.logService.warn(`[BackgroundTodo] copilot-fast endpoint unavailable, skipping pass: ${err}`);
context.logService.warn(`[BackgroundTodo] copilot-utility-small endpoint unavailable, skipping pass: ${err}`);
BackgroundTodoProcessor._sendTelemetry(context.telemetryService, 'skipped', conversationId, associatedRequestId, Date.now() - startTime);
return { outcome: 'noop' };
}
@@ -670,7 +670,7 @@ export class BackgroundTodoProcessor {
// propagate as errors so the delta is NOT marked processed — a later pass
// can retry with fresh or coalesced activity.
if (response.type !== ChatFetchResponseType.Success) {
context.logService.warn(`[BackgroundTodo] copilot-fast returned non-success response: ${response.type}`);
context.logService.warn(`[BackgroundTodo] copilot-utility-small returned non-success response: ${response.type}`);
BackgroundTodoProcessor._sendTelemetry(context.telemetryService, 'modelError', conversationId, associatedRequestId, durationMs);
throw new Error(`Background todo model request failed: ${response.type}`);
}
@@ -213,7 +213,7 @@ export async function renderPromptElementJSON<P extends BasePromptElementProps>(
// todo@lramos15: We should pass in endpoint provider rather than doing invoke function, but this was easier
const endpoint = await instantiationService.invokeFunction(async (accessor) => {
const endpointProvider = accessor.get(IEndpointProvider);
return await endpointProvider.getChatEndpoint('copilot-base');
return await endpointProvider.getChatEndpoint('copilot-utility');
});
const hydratedInstaService = instantiationService.createChild(new ServiceCollection([IPromptEndpoint, endpoint]));
const renderer = new PromptRendererForJSON(ctor as any, props, tokenOptions, endpoint, hydratedInstaService);
@@ -324,7 +324,7 @@ export class CodeMapper {
}
// continue with "slow rewrite endpoint" when fast rewriting was not possible
// use copilot base as fallback
const chatEndpoint = await this.endpointProvider.getChatEndpoint('copilot-base');
const chatEndpoint = await this.endpointProvider.getChatEndpoint('copilot-utility');
// Only attempt a full file rewrite if the original document fits into 3/4 of the max output token limit, leaving space for the model to add code. The limit is currently a flat 4K tokens from CAPI across all our models.
// If there are multiple input documents, pick the longest one to base the limit on
@@ -436,7 +436,7 @@ export class ChatToolReferences extends PromptElement<ChatToolCallProps, void> {
continue;
}
const toolArgsEndpoint = await this.endpointProvider.getChatEndpoint('copilot-fast');
const toolArgsEndpoint = await this.endpointProvider.getChatEndpoint('copilot-utility-small');
const internalToolArgs = toolReference.input ?? {};
const toolArgs = await this.fetchToolArgs(tool, toolArgsEndpoint);
@@ -79,7 +79,7 @@ export class NewWorkspacePrompt extends PromptElement<NewWorkspacePromptProps, N
}
progress?.report(new ChatResponseProgressPart(l10n.t('Determining user intent...')));
const endpoint = await this.endPointProvider.getChatEndpoint('copilot-fast');
const endpoint = await this.endPointProvider.getChatEndpoint('copilot-utility-small');
const { messages } = await buildNewWorkspaceMetaPrompt(this.instantiationService, endpoint, this.props.promptContext);
if (token.isCancellationRequested) {
@@ -234,7 +234,7 @@ export class StartDebuggingPrompt extends PromptElement<StartDebuggingPromptProp
}
private async queryModelForRequestedFiles(debuggerType: string | undefined, progress: vscode.Progress<vscode.ChatResponseProgressPart> | undefined, token: vscode.CancellationToken) {
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-fast');
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-utility-small');
const promptRenderer = this.props.input.type === StartDebuggingType.CommandLine
? PromptRenderer.create(
this.instantiationService,
@@ -295,7 +295,7 @@ export class StartDebuggingPrompt extends PromptElement<StartDebuggingPromptProp
}
private async getDebuggerType(progress: vscode.Progress<vscode.ChatResponseProgressPart> | undefined, token: vscode.CancellationToken): Promise<string | undefined> {
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-fast');
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-utility-small');
const promptRenderer = PromptRenderer.create(
this.instantiationService,
@@ -240,7 +240,7 @@ describe('ChatToolCalls (toolCalling.tsx)', () => {
const accessor = testingServiceCollection.createTestingAccessor();
const instantiationService = accessor.get(IInstantiationService);
const endpointProvider = accessor.get(IEndpointProvider);
const endpoint = await endpointProvider.getChatEndpoint('copilot-base');
const endpoint = await endpointProvider.getChatEndpoint('copilot-utility');
const round: IToolCallRound = {
id: 'round-1',
@@ -314,7 +314,7 @@ describe('ChatToolCalls (toolCalling.tsx)', () => {
const accessor = testingServiceCollection.createTestingAccessor();
const instantiationService = accessor.get(IInstantiationService);
const endpointProvider = accessor.get(IEndpointProvider);
const endpoint = await endpointProvider.getChatEndpoint('copilot-base');
const endpoint = await endpointProvider.getChatEndpoint('copilot-utility');
const round: IToolCallRound = {
id: 'round-1',
@@ -400,7 +400,7 @@ describe('ChatToolCalls (toolCalling.tsx)', () => {
const accessor = testingServiceCollection.createTestingAccessor();
const instantiationService = accessor.get(IInstantiationService);
const endpointProvider = accessor.get(IEndpointProvider);
const endpoint = await endpointProvider.getChatEndpoint('copilot-base');
const endpoint = await endpointProvider.getChatEndpoint('copilot-utility');
const round: IToolCallRound = {
id: 'round-1',
@@ -472,7 +472,7 @@ describe('ChatToolCalls (toolCalling.tsx)', () => {
const accessor = testingServiceCollection.createTestingAccessor();
const instantiationService = accessor.get(IInstantiationService);
const endpointProvider = accessor.get(IEndpointProvider);
const endpoint = await endpointProvider.getChatEndpoint('copilot-base');
const endpoint = await endpointProvider.getChatEndpoint('copilot-utility');
const imageData = new Uint8Array(1024);
const toolCallResults: Record<string, vscode.LanguageModelToolResult> = {
@@ -536,7 +536,7 @@ describe('ChatToolCalls (toolCalling.tsx)', () => {
const accessor = testingServiceCollection.createTestingAccessor();
const instantiationService = accessor.get(IInstantiationService);
const endpointProvider = accessor.get(IEndpointProvider);
const endpoint = await endpointProvider.getChatEndpoint('copilot-base');
const endpoint = await endpointProvider.getChatEndpoint('copilot-utility');
// Disable image uploads so images go through the base64 path where the budget applies
const configService = accessor.get(IConfigurationService);
@@ -598,7 +598,7 @@ describe('ChatToolCalls (toolCalling.tsx)', () => {
const accessor = testingServiceCollection.createTestingAccessor();
const instantiationService = accessor.get(IInstantiationService);
const endpointProvider = accessor.get(IEndpointProvider);
const endpoint = await endpointProvider.getChatEndpoint('copilot-base');
const endpoint = await endpointProvider.getChatEndpoint('copilot-utility');
const telemetryService = accessor.get(ITelemetryService);
const configService = accessor.get(IConfigurationService);
@@ -67,7 +67,7 @@ export class VscodePrompt extends PromptElement<VscodePromptProps, VscodePromptS
progress?.report(new ChatResponseProgressPart(l10n.t('Refining question to improve search accuracy.')));
let userQuery: string = this.props.promptContext.query;
const endpoint = await this.endPointProvider.getChatEndpoint('copilot-fast');
const endpoint = await this.endPointProvider.getChatEndpoint('copilot-utility-small');
const renderer = PromptRenderer.create(this.instantiationService, endpoint, VscodeMetaPrompt, this.props.promptContext);
const { messages } = await renderer.render();
if (token.isCancellationRequested) {
@@ -56,7 +56,7 @@ export class RenameSuggestionsPrompt extends PromptElement<Props, State> {
const isDefinitionBeingRenamed = defState.k === 'found' && defState.definitions.some(def => def.excerptRange.contains(this.props.range));
if (!isDefinitionBeingRenamed) {
const endpointInfo = await this.endpointProvider.getChatEndpoint('copilot-fast');
const endpointInfo = await this.endpointProvider.getChatEndpoint('copilot-utility-small');
const documentContext: IDocumentContext = {
document,
fileIndentInfo: undefined,
@@ -107,7 +107,7 @@ export class RenameSuggestionsProvider implements vscode.NewSymbolNamesProvider
if (token.isCancellationRequested) {
cancellationReason = ProvideCallCancellationReason.AfterEnablementCheck;
} else {
const endpoint = await this._endpointProvider.getChatEndpoint('copilot-fast');
const endpoint = await this._endpointProvider.getChatEndpoint('copilot-utility-small');
expectedDelayBeforeFetch = this.delayBeforeFetchMs;
if (token.isCancellationRequested) {
@@ -27,7 +27,14 @@ class FakeModelMetadataFetcher implements IModelMetadataFetcher {
async getChatModelFromApiModel(model: LanguageModelChat): Promise<IChatModelInformation | undefined> {
return undefined;
}
async getChatModelFromFamily(modelId: string): Promise<IChatModelInformation> {
async getCopilotUtilityModel(): Promise<IChatModelInformation> {
return this._fakeChatModel('copilot-utility');
}
async getChatModelFromCapiFamily(family: string): Promise<IChatModelInformation> {
return this._fakeChatModel(family);
}
private _fakeChatModel(modelId: string): IChatModelInformation {
return {
id: modelId,
vendor: 'fake-vendor',
@@ -39,7 +39,7 @@ export class AIEvaluationService implements IAIEvaluationService {
}
async evaluate(response: string, criteria: string, token: CancellationToken): Promise<EvaluationResult> {
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-fast');
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-utility-small');
const promptRenderer = PromptRenderer.create(this.instantiationService, endpoint, EvaluationPrompt, {
response, criteria
});
@@ -139,7 +139,7 @@ class WorkspaceMutation implements IWorkspaceMutation {
const originalText = document?.getText();
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-fast');
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-utility-small');
const promptRenderer = PromptRenderer.create(this.instantiationService, endpoint, WorkspaceMutationFilePrompt, {
file,
document,
@@ -174,7 +174,7 @@ class WorkspaceMutation implements IWorkspaceMutation {
}
private async getFileDescriptions() {
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-fast');
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-utility-small');
const promptRenderer = PromptRenderer.create(this.instantiationService, endpoint, WorkspaceMutationInstructionsPrompt, {
fileTreeStr: this.opts.fileTree,
query: this.opts.query,
@@ -25,7 +25,7 @@ import { TOOLS_AND_GROUPS_LIMIT } from './virtualToolsConstants';
import { describeBulkToolGroups } from './virtualToolSummarizer';
import { ISummarizedToolCategory, ISummarizedToolCategoryUpdatable, IToolCategorization, IToolGroupingCache } from './virtualToolTypes';
const CATEGORIZATION_ENDPOINT = 'copilot-fast';
const CATEGORIZATION_ENDPOINT = 'copilot-utility-small';
const SUMMARY_PREFIX = 'Call this tool when you need access to a new category of tools. The category of tools is described as follows:\n\n';
const SUMMARY_SUFFIX = '\n\nBe sure to call this tool if you need a capability related to the above.';
@@ -459,7 +459,7 @@ export abstract class AbstractReplaceStringTool<T extends { explanation: string
newString,
},
eol,
await this.endpointProvider.getChatEndpoint('copilot-fast'),
await this.endpointProvider.getChatEndpoint('copilot-utility-small'),
token
);
if (healed.params.oldString === healed.params.newString) {
@@ -521,7 +521,7 @@ export class ApplyPatchTool implements ICopilotTool<IApplyPatchToolParams> {
* and do another turn.
*/
private async healCommit(patch: string, docs: DocText, explanation: string, token: CancellationToken) {
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-fast');
const endpoint = await this.endpointProvider.getChatEndpoint('copilot-utility-small');
const prompt = await PromptRenderer.create(
this.instantiationService,
endpoint,
@@ -48,7 +48,7 @@ export class NewNotebookTool implements ICopilotTool<IBuildPromptContext> {
let outcome: 'failedToCreatePlanningEndpoint' | 'failedToRenderPlanningPrompt' | 'failedToMakePlanningRequest' | 'failedToRenderNewNotebookPrompt' = 'failedToCreatePlanningEndpoint';
try {
// Get the endpoint
const planningEndpoint = await this.endpointProvider.getChatEndpoint(options.model || 'copilot-base');
const planningEndpoint = await this.endpointProvider.getChatEndpoint(options.model || 'copilot-utility');
const originalCreateNotebookQuery = `Create notebook: ${this._input?.query ?? options.input.query}`;
const mockContext: IBuildPromptContext = {
query: originalCreateNotebookQuery,
@@ -35,7 +35,7 @@ suite('FindTextInFilesResult', () => {
}
};
const endpoint = await services.get(IEndpointProvider).getChatEndpoint('copilot-base');
const endpoint = await services.get(IEndpointProvider).getChatEndpoint('copilot-utility');
const renderer = PromptRenderer.create(services.get(IInstantiationService), endpoint, clz, {});
const r = await renderer.render();
@@ -23,7 +23,7 @@ export async function renderElementToString(accessor: ServicesAccessor, element:
}
};
const endpoint = await accessor.get(IEndpointProvider).getChatEndpoint('copilot-base');
const endpoint = await accessor.get(IEndpointProvider).getChatEndpoint('copilot-utility');
// eslint-disable-next-line local/code-no-accessor-after-await
const renderer = PromptRenderer.create(accessor.get(IInstantiationService), endpoint, clz, {});
@@ -81,7 +81,7 @@ export class SemanticSearchTextSearchProvider implements vscode.AITextSearchProv
) { }
private async getEndpoint() {
this._endpoint = this._endpoint ?? await this._endpointProvider.getChatEndpoint('copilot-fast');
this._endpoint = this._endpoint ?? await this._endpointProvider.getChatEndpoint('copilot-utility-small');
return this._endpoint;
}
@@ -672,7 +672,7 @@ export namespace ConfigKey {
/** Maximum number of tool calls the execution subagent can make */
export const ExecutionSubagentToolCallLimit = defineSetting<number>('chat.executionSubagent.toolCallLimit', ConfigType.ExperimentBased, 10);
/** When enabled, the main agent's manage_todo_list tool is disabled and a background copilot-fast model maintains the todo list instead. */
/** When enabled, the main agent's manage_todo_list tool is disabled and a background copilot-utility-small model maintains the todo list instead. */
export const BackgroundTodoAgentEnabled = defineSetting<boolean>('chat.agent.backgroundTodoAgent.enabled', ConfigType.ExperimentBased, false);
export const InlineEditsTriggerOnEditorChangeAfterSeconds = defineAndMigrateExpSetting<number | undefined>('chat.advanced.inlineEdits.triggerOnEditorChangeAfterSeconds', 'chat.inlineEdits.triggerOnEditorChangeAfterSeconds', 10);
@@ -142,7 +142,7 @@ export function isCompletionModelInformation(model: IModelAPIResponse): model is
return model.capabilities.type === 'completion';
}
export type ChatEndpointFamily = 'copilot-base' | 'copilot-fast';
export type ChatEndpointFamily = 'copilot-utility' | 'copilot-utility-small';
export type EmbeddingsEndpointFamily = 'text3small' | 'metis';
export interface IEndpointProvider {
@@ -1,50 +0,0 @@
/*---------------------------------------------------------------------------------------------
* Copyright (c) Microsoft Corporation. All rights reserved.
* Licensed under the MIT License. See License.txt in the project root for license information.
*--------------------------------------------------------------------------------------------*/
export class ModelAliasRegistry {
private readonly _aliasToModelId = new Map<string, string>();
private readonly _modelIdToAliases = new Map<string, string[]>();
private static readonly _instance = new ModelAliasRegistry();
private constructor() { }
private static _updateAliasesForModelId(modelId: string): void {
const aliases: string[] = [];
for (const [alias, mappedModelId] of this._instance._aliasToModelId.entries()) {
if (mappedModelId === modelId) {
aliases.push(alias);
}
}
if (aliases.length > 0) {
this._instance._modelIdToAliases.set(modelId, aliases);
} else {
this._instance._modelIdToAliases.delete(modelId);
}
}
static registerAlias(alias: string, modelId: string): void {
this._instance._aliasToModelId.set(alias, modelId);
this._updateAliasesForModelId(modelId);
}
static deregisterAlias(alias: string): void {
const modelId = this._instance._aliasToModelId.get(alias);
this._instance._aliasToModelId.delete(alias);
if (modelId) {
this._updateAliasesForModelId(modelId);
}
}
static resolveAlias(alias: string): string {
return this._instance._aliasToModelId.get(alias) ?? alias;
}
static getAliases(modelId: string): string[] {
return this._instance._modelIdToAliases.get(modelId) ?? [];
}
}
ModelAliasRegistry.registerAlias('copilot-fast', 'gpt-4o-mini');
@@ -6,10 +6,11 @@
import { IInstantiationService } from '../../../util/vs/platform/instantiation/common/instantiation';
import { IAuthenticationService } from '../../authentication/common/authentication';
import { IChatMLFetcher } from '../../chat/common/chatMLFetcher';
import { IConfigurationService } from '../../configuration/common/configurationService';
import { CHAT_MODEL, IConfigurationService } from '../../configuration/common/configurationService';
import { IEnvService } from '../../env/common/envService';
import { ILogService } from '../../log/common/logService';
import { IFetcherService } from '../../networking/common/fetcherService';
import { IChatEndpoint } from '../../networking/common/networking';
import { RawMessageConversionCallback } from '../../networking/common/openai';
import { IChatWebSocketManager } from '../../networking/node/chatWebSocketManager';
import { IExperimentationService } from '../../telemetry/common/nullExperimentationService';
@@ -19,6 +20,7 @@ import { ICAPIClientService } from '../common/capiClient';
import { IDomainService } from '../common/domainService';
import { IChatModelInformation } from '../common/endpointProvider';
import { ChatEndpoint } from './chatEndpoint';
import { IModelMetadataFetcher } from './modelMetadataFetcher';
export class CopilotChatEndpoint extends ChatEndpoint {
constructor(
@@ -59,3 +61,34 @@ export class CopilotChatEndpoint extends ChatEndpoint {
};
}
}
/**
* Resolves the built-in Copilot model used for the `copilot-utility-small`
* internal family (formerly `copilot-fast`). This is the small/fast model
* that powers background utility flows like commit message generation,
* prompt categorization, inline-chat progress messages, etc.
*
* The CAPI `/models` response does not flag a model as "fast" / "small",
* so the family is selected client-side. Today that's `gpt-4o-mini`.
*/
export class CopilotUtilitySmallChatEndpoint {
static readonly capiFamily: string = CHAT_MODEL.GPT4OMINI;
static async resolve(modelFetcher: IModelMetadataFetcher, instantiationService: IInstantiationService): Promise<IChatEndpoint> {
const modelMetadata = await modelFetcher.getChatModelFromCapiFamily(CopilotUtilitySmallChatEndpoint.capiFamily);
return instantiationService.createInstance(CopilotChatEndpoint, modelMetadata);
}
}
/**
* Resolves the built-in Copilot model used for the `copilot-utility`
* internal family (formerly `copilot-base`). This is the API-marked
* default base model — whichever model the CAPI `/models` response
* flags with `is_chat_fallback === true`.
*/
export class CopilotUtilityChatEndpoint {
static async resolve(modelFetcher: IModelMetadataFetcher, instantiationService: IInstantiationService): Promise<IChatEndpoint> {
const modelMetadata = await modelFetcher.getCopilotUtilityModel();
return instantiationService.createInstance(CopilotChatEndpoint, modelMetadata);
}
}
@@ -19,8 +19,7 @@ import { ILogService } from '../../log/common/logService';
import { getRequest } from '../../networking/common/networking';
import { IRequestLogger } from '../../requestLogger/common/requestLogger';
import { IExperimentationService } from '../../telemetry/common/nullExperimentationService';
import { ChatEndpointFamily, IChatModelInformation, ICompletionModelInformation, IEmbeddingModelInformation, IModelAPIResponse, isChatModelInformation, isCompletionModelInformation, isEmbeddingModelInformation } from '../common/endpointProvider';
import { ModelAliasRegistry } from '../common/modelAliasRegistry';
import { IChatModelInformation, ICompletionModelInformation, IEmbeddingModelInformation, IModelAPIResponse, isChatModelInformation, isCompletionModelInformation, isEmbeddingModelInformation } from '../common/endpointProvider';
export interface IModelMetadataFetcher {
@@ -41,10 +40,20 @@ export interface IModelMetadataFetcher {
getAllChatModels(): Promise<IChatModelInformation[]>;
/**
* Retrieves a chat model by its family name
* @param family The family of the model to fetch
* Retrieves the API-marked default Copilot utility model — the model
* the CAPI `/models` response flags with `is_chat_fallback === true`.
* Used to back the `copilot-utility` internal endpoint family.
*/
getChatModelFromFamily(family: ChatEndpointFamily): Promise<IChatModelInformation>;
getCopilotUtilityModel(): Promise<IChatModelInformation>;
/**
* Retrieves a chat model by its CAPI family identifier (e.g.
* `gpt-4o-mini`, `claude-sonnet-4`). The family must match
* `IChatModelCapabilities.family` of a model returned from the
* `/models` endpoint.
* @param family The CAPI family identifier of the model to fetch
*/
getChatModelFromCapiFamily(family: string): Promise<IChatModelInformation>;
/**
* Retrieves a chat model by its id
@@ -71,7 +80,7 @@ export class ModelMetadataFetcher extends Disposable implements IModelMetadataFe
private _familyMap: Map<string, IModelAPIResponse[]> = new Map();
private _completionsFamilyMap: Map<string, IModelAPIResponse[]> = new Map();
private _copilotBaseModel: IModelAPIResponse | undefined;
private _copilotUtilityModel: IModelAPIResponse | undefined;
private _lastFetchTime: number = 0;
private readonly _taskSingler = new TaskSingler<IModelAPIResponse | undefined | void>();
private _lastFetchError: any;
@@ -161,18 +170,20 @@ export class ModelMetadataFetcher extends Disposable implements IModelMetadataFe
return resolvedModel;
}
public async getChatModelFromFamily(family: ChatEndpointFamily): Promise<IChatModelInformation> {
public async getCopilotUtilityModel(): Promise<IChatModelInformation> {
await this._taskSingler.getOrCreate(ModelMetadataFetcher.ALL_MODEL_KEY, this._fetchModels.bind(this));
let resolvedModel: IModelAPIResponse | undefined;
family = ModelAliasRegistry.resolveAlias(family) as ChatEndpointFamily;
if (family === 'copilot-base') {
resolvedModel = this._copilotBaseModel;
} else {
resolvedModel = this._familyMap.get(family)?.[0];
}
const resolvedModel = this._copilotUtilityModel;
if (!resolvedModel || !isChatModelInformation(resolvedModel)) {
throw new Error(await this._getErrorMessage(`Unable to resolve chat model with family selection: ${family}`));
throw new Error(await this._getErrorMessage('Unable to resolve Copilot utility chat model (server did not mark a chat fallback model)'));
}
return resolvedModel;
}
public async getChatModelFromCapiFamily(family: string): Promise<IChatModelInformation> {
await this._taskSingler.getOrCreate(ModelMetadataFetcher.ALL_MODEL_KEY, this._fetchModels.bind(this));
const resolvedModel = this._familyMap.get(family)?.[0];
if (!resolvedModel || !isChatModelInformation(resolvedModel)) {
throw new Error(await this._getErrorMessage(`Unable to resolve chat model with CAPI family selection: ${family}`));
}
return resolvedModel;
}
@@ -267,9 +278,9 @@ export class ModelMetadataFetcher extends Disposable implements IModelMetadataFe
for (let model of data) {
model = await this._hydrateResolvedModel(model);
const isCompletionModel = isCompletionModelInformation(model);
// The base model is whatever model is deemed "fallback" by the server
// The utility model is whatever model is deemed "fallback" by the server
if (model.is_chat_fallback && !isCompletionModel) {
this._copilotBaseModel = model;
this._copilotUtilityModel = model;
}
const family = model.capabilities.family;
const familyMap = isCompletionModel ? this._completionsFamilyMap : this._familyMap;
@@ -105,7 +105,7 @@ describe('CopilotChatEndpoint - Reasoning Properties', () => {
beforeEach(() => {
mockServices = createMockServices();
modelMetadata = {
id: 'copilot-base',
id: 'copilot-utility',
vendor: 'Copilot',
name: 'Copilot Base',
version: '1.0',
@@ -187,11 +187,13 @@ export class TestEndpointProvider implements IEndpointProvider {
}
return Array.from(this._chatEndpoints.values());
}
async getChatEndpoint(requestOrFamilyOrModel: LanguageModelChat | ChatRequest | ChatEndpointFamily): Promise<IChatEndpoint> {
getChatEndpoint(requestOrModel: LanguageModelChat | ChatRequest): Promise<IChatEndpoint>;
getChatEndpoint(family: ChatEndpointFamily): Promise<IChatEndpoint | undefined>;
async getChatEndpoint(requestOrFamilyOrModel: LanguageModelChat | ChatRequest | ChatEndpointFamily): Promise<IChatEndpoint | undefined> {
if (typeof requestOrFamilyOrModel !== 'string') {
requestOrFamilyOrModel = 'copilot-base';
requestOrFamilyOrModel = 'copilot-utility';
}
if (requestOrFamilyOrModel === 'copilot-base') {
if (requestOrFamilyOrModel === 'copilot-utility') {
return await this.getChatEndpointInfo(this.gpt4ModelToRunAgainst ?? CHAT_MODEL.GPT41, await this._modelLabChatModelMetadata, await this._prodChatModelMetadata);
} else {
return await this.getChatEndpointInfo(this.gpt4oMiniModelToRunAgainst ?? CHAT_MODEL.GPT4OMINI, await this._modelLabChatModelMetadata, await this._prodChatModelMetadata);
@@ -274,7 +274,7 @@ export interface IChatEndpointTokenPricing {
export interface IChatEndpoint extends IEndpoint {
readonly maxOutputTokens: number;
/** The model ID- this may change and will be `copilot-base` for the base model. Use `family` to switch behavior based on model type. */
/** The model ID- this may change and will be `copilot-utility` for the utility (fallback) model. Use `family` to switch behavior based on model type. */
readonly model: string;
readonly modelProvider: string;
readonly apiType?: string;
@@ -37,7 +37,7 @@ ssuite.skip({ title: 'vscode', subtitle: 'metaprompt', location: 'panel' }, asyn
const accessor = testingServiceCollection.createTestingAccessor();
const instantiationService = accessor.get(IInstantiationService);
const endpoint = await accessor.get(IEndpointProvider).getChatEndpoint('copilot-base');
const endpoint = await accessor.get(IEndpointProvider).getChatEndpoint('copilot-utility');
const vscodePrompt = instantiationService.createInstance(VscodePrompt, {
promptContext: {
chatVariables: new ChatVariablesCollection([]),
@@ -18,7 +18,7 @@ import { isValidPythonFile } from '../simulation/diagnosticProviders/python';
ssuite({ title: 'newNotebook', subtitle: 'prompt', location: 'panel' }, () => {
stest({ description: 'generate code cell', language: 'python' }, async (testingServiceCollection) => {
const accessor = testingServiceCollection.createTestingAccessor();
const endpoint = await accessor.get(IEndpointProvider).getChatEndpoint('copilot-base');
const endpoint = await accessor.get(IEndpointProvider).getChatEndpoint('copilot-utility');
const topic = 'Creating Random Arrays with Numpy';
const sections: INotebookSection[] = [
{
@@ -57,7 +57,7 @@ ssuite({ title: 'newNotebook', subtitle: 'prompt', location: 'panel' }, () => {
stest({ description: 'Generate code cell (numpy)', language: 'python' }, async (testingServiceCollection) => {
const accessor = testingServiceCollection.createTestingAccessor();
const endpoint = await accessor.get(IEndpointProvider).getChatEndpoint('copilot-base');
const endpoint = await accessor.get(IEndpointProvider).getChatEndpoint('copilot-utility');
const topic = 'A Jupyter notebook that creates a structured array with NumPy.';
const sections: INotebookSection[] = [
{
@@ -95,7 +95,7 @@ ssuite({ title: 'newNotebook', subtitle: 'prompt', location: 'panel' }, () => {
stest({ description: 'Generate code cell (seaborn + pandas)', language: 'python' }, async (testingServiceCollection) => {
const accessor = testingServiceCollection.createTestingAccessor();
const endpoint = await accessor.get(IEndpointProvider).getChatEndpoint('copilot-base');
const endpoint = await accessor.get(IEndpointProvider).getChatEndpoint('copilot-utility');
const topic = 'A Jupyter notebook that loads planets data from Seaborn and performs aggregation in Pandas.';
const sections: INotebookSection[] = [
{ title: 'Import Required Libraries', content: 'Import the necessary libraries, including Pandas and Seaborn.' },
@@ -224,7 +224,7 @@ ssuite({ title: 'settingsEditorSearchResultsSelector', location: 'external' }, (
},
];
const endpointProvider = accessor.get(IEndpointProvider);
const endpoint = await endpointProvider.getChatEndpoint('copilot-base');
const endpoint = await endpointProvider.getChatEndpoint('copilot-utility');
const results = await selector.selectTopSearchResults(endpoint, 'Hide search bar at top of window', settingsList, CancellationToken.None);
assert.ok(results.length > 0, 'No settings were selected');
assert.ok(results.some(result => result === 'window.commandCenter'), 'Expected setting "window.commandCenter" was not found');
@@ -392,7 +392,7 @@ ssuite({ title: 'settingsEditorSearchResultsSelector', location: 'external' }, (
];
const endpointProvider = accessor.get(IEndpointProvider);
const endpoint = await endpointProvider.getChatEndpoint('copilot-base');
const endpoint = await endpointProvider.getChatEndpoint('copilot-utility');
const results = await selector.selectTopSearchResults(endpoint, 'agentmode', settingsList, CancellationToken.None);
assert.ok(results.length > 0, 'No settings were selected');
assert.ok(results.some(result => result === 'chat.agent.enabled'), 'Expected setting "chat.agent.enabled" was not found');
@@ -1043,7 +1043,7 @@ configurationRegistry.registerConfiguration({
[ChatConfiguration.ToolRiskAssessmentModel]: {
type: 'string',
description: nls.localize('chat.tools.riskAssessment.model', "The language model id used to generate tool risk assessments. Should be a small, fast model."),
default: 'copilot-fast',
default: 'copilot-utility-small',
tags: ['experimental', 'advanced'],
experiment: {
mode: 'auto'
@@ -237,7 +237,7 @@ export class ChatEditingExplanationModelManager extends Disposable implements IC
try {
// Select a model for understanding all changes together
const models = await this._languageModelsService.selectLanguageModels({ vendor: 'copilot', id: 'copilot-fast' });
const models = await this._languageModelsService.selectLanguageModels({ vendor: 'copilot', id: 'copilot-utility-small' });
if (!models.length) {
for (const fileData of fileChanges) {
this._updateUriStatePartial(fileData.uri, {
@@ -109,7 +109,7 @@ export class ChatToolRiskAssessmentService implements IChatToolRiskAssessmentSer
}
private async _invokeModel(tool: IToolData, parameters: unknown, token: CancellationToken): Promise<IToolRiskAssessment | undefined> {
const modelId = this._configurationService.getValue<string>(ChatConfiguration.ToolRiskAssessmentModel) || 'copilot-fast';
const modelId = this._configurationService.getValue<string>(ChatConfiguration.ToolRiskAssessmentModel) || 'copilot-utility-small';
const models = await this._languageModelsService.selectLanguageModels({ vendor: 'copilot', id: modelId });
if (!models.length || token.isCancellationRequested) {
@@ -1162,7 +1162,7 @@ export class ChatThinkingContentPart extends ChatCollapsibleContentPart implemen
const timeout = setTimeout(() => cts.cancel(), 5000);
try {
const models = await this.languageModelsService.selectLanguageModels({ vendor: 'copilot', id: 'copilot-fast' });
const models = await this.languageModelsService.selectLanguageModels({ vendor: 'copilot', id: 'copilot-utility-small' });
if (!models.length) {
this.setFallbackTitle();
return;