Skip to content

Commit f598f9b

Browse files
committed
v3.7.0: dynamic model fetching + ADR-001 tiered tool-payload reduction
1 parent e11d8fc commit f598f9b

12 files changed

Lines changed: 1171 additions & 660 deletions

File tree

‎CHANGELOG.md‎

Lines changed: 9 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1,5 +1,14 @@
11
# Changelog
22

3+
## [3.7.0](https://github.com/abbalochdev/opencode-for-copilot/compare/v3.6.0...v3.7.0) (2026-07-26)
4+
5+
### Features
6+
7+
* **models:** dynamic model fetching — model catalogue extracted from `consts.ts` into `provider/opencode-models.ts` (METADATA_OVERLAY as single source of truth); models are fetched asynchronously from the OpenCode Go/Zen APIs with a 5-minute cache and static fallback on network failure; unknown model IDs from the API get auto-generated defaults so they appear in the picker before a new extension release
8+
* **tools:** implement tiered tool-payload reduction for free-tier models (ADR-001) — deterministic stable sort + soft-cap trimming (`preferredToolLimit: 32`) for eligible request kinds (`main-agent`, `background`), preventing HTTP 500 errors when free models receive more tools than their server-side limit
9+
* **tools:** add reactive retry on HTTP 500 — when a request with >8 tools fails, retry once with half the tools before propagating the error (mirrors Claude Code's `hasAttemptedReactiveCompact` pattern)
10+
* **error:** improve HTTP 500 error message to include tool count when the request carried many tools, helping users diagnose payload-size issues
11+
312
## [3.6.0](https://github.com/abbalochdev/opencode-for-copilot/compare/v3.5.0...v3.6.0) (2026-07-21)
413

514
### Features

‎package.json‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -3,7 +3,7 @@
33
"displayName": "OpenCode for Copilot",
44
"description": "Use OpenCode Go & Zen coding models (GLM, Kimi, DeepSeek, Claude, Grok, Qwen, free models) in the Copilot Chat model picker. Vision, thinking mode, agent tools — zero config, BYOK.",
55
"icon": "resources/icon.png",
6-
"version": "3.6.0",
6+
"version": "3.7.0",
77
"packageManager": "pnpm@11.7.0",
88
"publisher": "abbalochdev",
99
"license": "MIT",

‎src/client/core.ts‎

Lines changed: 47 additions & 24 deletions
Original file line numberDiff line numberDiff line change
@@ -10,7 +10,7 @@ import type {
1010
StreamCallbacks,
1111
} from '../types';
1212
import { convertToAnthropicRequest, parseAnthropicStream } from './anthropic';
13-
import { createHttpError, formatRequestError, normalizeRequestError } from './error';
13+
import { createHttpError, formatRequestError, GLMRequestError, normalizeRequestError } from './error';
1414

1515
/**
1616
* Lightweight SSE-streaming GLM API client.
@@ -29,16 +29,59 @@ export class GLMClient {
2929
/**
3030
* Stream a chat completion from the GLM API.
3131
* Parses SSE chunks and dispatches callbacks for content, thinking, and tool calls.
32+
*
33+
* Tier 3 (reactive retry): if the first attempt fails with HTTP 500 and the
34+
* request carries more than 8 tools, retry once with half the tools — mirroring
35+
* Claude Code's `hasAttemptedReactiveCompact` pattern. This covers the case
36+
* where a free-tier model's actual tool limit is lower than advertised.
3237
*/
3338
async streamChatCompletion(
3439
request: GLMRequest,
3540
callbacks: StreamCallbacks,
3641
cancellationToken?: CancellationToken,
3742
): Promise<void> {
38-
if (this.protocol === 'anthropic') {
39-
return this.streamAnthropicCompletion(request, callbacks, cancellationToken);
43+
const dispatch = (req: GLMRequest) =>
44+
this.protocol === 'anthropic'
45+
? this.streamAnthropicCompletion(req, callbacks, cancellationToken)
46+
: this.streamOpenAIChatCompletion(req, callbacks, cancellationToken);
47+
48+
try {
49+
await dispatch(request);
50+
} catch (error) {
51+
if (isAbortError(error) && cancellationToken?.isCancellationRequested) {
52+
return;
53+
}
54+
if (
55+
error instanceof GLMRequestError
56+
&& error.status === 500
57+
&& (request.tools?.length ?? 0) > 8
58+
) {
59+
const halvedTools = request.tools!.slice(0, Math.ceil(request.tools!.length / 2));
60+
logger.warn(
61+
`Reactive retry: HTTP 500 with ${request.tools!.length} tools, retrying with ${halvedTools.length} tools`,
62+
);
63+
const retriedRequest = { ...request, tools: halvedTools };
64+
try {
65+
await dispatch(retriedRequest);
66+
} catch (retryError) {
67+
// Retry exhausted — propagate the retry error, not the original.
68+
const normalizedRetry = normalizeRequestError(retryError, {
69+
baseUrl: this.baseUrl,
70+
request: retriedRequest,
71+
});
72+
logger.error('GLM reactive retry failed:', formatRequestError(normalizedRetry));
73+
callbacks.onError(normalizedRetry);
74+
}
75+
return;
76+
}
77+
// Not retryable — propagate as before.
78+
const normalizedError = normalizeRequestError(error, {
79+
baseUrl: this.baseUrl,
80+
request,
81+
});
82+
logger.error('GLM request failed:', formatRequestError(normalizedError));
83+
callbacks.onError(normalizedError);
4084
}
41-
return this.streamOpenAIChatCompletion(request, callbacks, cancellationToken);
4285
}
4386

4487
/**
@@ -232,16 +275,6 @@ export class GLMClient {
232275
pendingToolCalls.clear();
233276
reportFinalUsage(callbacks, latestUsage);
234277
callbacks.onDone();
235-
} catch (error) {
236-
if (isAbortError(error) && cancellationToken?.isCancellationRequested) {
237-
return;
238-
}
239-
const normalizedError = normalizeRequestError(error, {
240-
baseUrl: this.baseUrl,
241-
request,
242-
});
243-
logger.error('GLM request failed:', formatRequestError(normalizedError));
244-
callbacks.onError(normalizedError);
245278
} finally {
246279
// Release the response stream lock on every exit path (`[DONE]`
247280
// early-return, cancellation, normal completion and errors). On the
@@ -313,16 +346,6 @@ export class GLMClient {
313346
}
314347
});
315348
}
316-
} catch (error) {
317-
if (isAbortError(error) && cancellationToken?.isCancellationRequested) {
318-
return;
319-
}
320-
const normalizedError = normalizeRequestError(error, {
321-
baseUrl: this.baseUrl,
322-
request,
323-
});
324-
logger.error('GLM Anthropic request failed:', formatRequestError(normalizedError));
325-
callbacks.onError(normalizedError);
326349
} finally {
327350
cancelListener?.dispose();
328351
controller.abort();

‎src/client/error/index.ts‎

Lines changed: 19 additions & 12 deletions
Original file line numberDiff line numberDiff line change
@@ -2,18 +2,18 @@ import { isOfficialGLMBaseUrl } from '../../endpoint';
22
import { t } from '../../i18n';
33
import { safeStringify } from '../../json';
44
import {
5-
API_PROVIDER_HTTP_ERROR_LINKS,
6-
GLM_BUSINESS_ERROR_CODES,
7-
MAX_DIAGNOSTIC_FIELD_LENGTH,
5+
API_PROVIDER_HTTP_ERROR_LINKS,
6+
GLM_BUSINESS_ERROR_CODES,
7+
MAX_DIAGNOSTIC_FIELD_LENGTH,
88
} from '../consts';
99
import type {
10-
ApiProviderId,
11-
ErrorActionLink,
12-
ErrorActionUrls,
13-
GLMRequestErrorKind,
14-
HttpErrorLinkDefinition,
15-
HttpErrorLinkStatusKey,
16-
RequestErrorContext,
10+
ApiProviderId,
11+
ErrorActionLink,
12+
ErrorActionUrls,
13+
GLMRequestErrorKind,
14+
HttpErrorLinkDefinition,
15+
HttpErrorLinkStatusKey,
16+
RequestErrorContext,
1717
} from '../types';
1818
import { getNetworkErrorCauseInfo, getNetworkErrorCode, getNetworkErrorMessage } from './network';
1919
export type { ErrorActionUrls, GLMRequestErrorKind } from '../types';
@@ -82,6 +82,7 @@ export async function createHttpError(
8282
baseUrl,
8383
businessCode,
8484
serverMessage,
85+
toolCount: context.request.tools?.length,
8586
});
8687

8788
return new GLMRequestError({
@@ -179,8 +180,9 @@ function getHttpErrorMessage(params: {
179180
baseUrl: string;
180181
businessCode?: string;
181182
serverMessage?: string;
183+
toolCount?: number;
182184
}): string {
183-
const { status, baseUrl, businessCode, serverMessage } = params;
185+
const { status, baseUrl, businessCode, serverMessage, toolCount } = params;
184186
const isOfficialGlm = isOfficialGLMBaseUrl(baseUrl);
185187

186188
// 1) 已知业务错误码 → 使用官方错误表对应的精确文案。GLM 的服务端消息通常
@@ -209,7 +211,12 @@ function getHttpErrorMessage(params: {
209211
}
210212

211213
// 4) 兜底:仅有 HTTP 状态码时,使用原有的状态码泛化文案。
212-
return getHttpStatusMessage(status, baseUrl);
214+
// For 500 errors with many tools, suggest the payload may be too large.
215+
const baseMessage = getHttpStatusMessage(status, baseUrl);
216+
if (status === 500 && toolCount !== undefined && toolCount > 8) {
217+
return `${baseMessage} — the request included ${toolCount} tools, which may exceed this model's limit`;
218+
}
219+
return baseMessage;
213220
}
214221

215222
/**

‎src/config.ts‎

Lines changed: 41 additions & 13 deletions
Original file line numberDiff line numberDiff line change
@@ -1,20 +1,23 @@
11
import vscode from 'vscode';
22
import { CONFIG_SECTION, MODELS } from './consts';
33
import {
4-
deriveEndpointPreset,
5-
normalizeBaseUrl,
6-
resolveApiKeyUrl,
7-
resolveEndpointApiKeyUrl,
8-
resolveEndpointBaseUrl,
9-
resolveEndpointProtocol,
4+
deriveEndpointPreset,
5+
normalizeBaseUrl,
6+
resolveApiKeyUrl,
7+
resolveEndpointApiKeyUrl,
8+
resolveEndpointBaseUrl,
9+
resolveEndpointProtocol,
1010
} from './endpoint';
11+
import {
12+
getDynamicModels
13+
} from './provider/opencode-models';
1114
import type {
12-
ApiMode,
13-
ApiProtocol,
14-
ApiRegion,
15-
CustomModelConfig,
16-
EndpointPreset,
17-
ModelDefinition,
15+
ApiMode,
16+
ApiProtocol,
17+
ApiRegion,
18+
CustomModelConfig,
19+
EndpointPreset,
20+
ModelDefinition,
1821
} from './types';
1922

2023
export type DebugMode = 'minimal' | 'metadata' | 'verbose';
@@ -187,14 +190,39 @@ export function getCustomModels(): ModelDefinition[] {
187190
return [...byId.values()];
188191
}
189192

193+
/**
194+
* Dynamic model list override. When set by `refreshDynamicModels()`, this is
195+
* used instead of the static `MODELS` array. This lets us serve live model
196+
* lists from the OpenCode API while keeping the static array as a fallback.
197+
*/
198+
let dynamicModelsOverride: readonly ModelDefinition[] | undefined;
199+
200+
/**
201+
* Synchronous model list — used by the model picker, request handler, and tests.
202+
* Returns dynamic models if available, otherwise falls back to static MODELS.
203+
*/
190204
export function listProviderModels(): ModelDefinition[] {
191-
const byId = new Map(MODELS.map((model) => [model.id, model]));
205+
const source = dynamicModelsOverride ?? MODELS;
206+
const byId = new Map(source.map((model) => [model.id, model]));
192207
for (const model of getCustomModels()) {
193208
byId.set(model.id, model);
194209
}
195210
return [...byId.values()];
196211
}
197212

213+
/**
214+
* Asynchronously refresh the model list from the OpenCode API.
215+
* Updates `dynamicModelsOverride` and invalidates the cache so the next call
216+
* to `listProviderModels()` returns fresh data.
217+
*
218+
* On network failure the existing list (or static fallback) stays in place.
219+
*/
220+
export async function refreshDynamicModels(): Promise<void> {
221+
const customModels = getCustomModels();
222+
const fallback = dynamicModelsOverride ?? MODELS;
223+
dynamicModelsOverride = await getDynamicModels(customModels, fallback);
224+
}
225+
198226
export function findModelDefinition(modelId: string): ModelDefinition | undefined {
199227
return listProviderModels().find((model) => model.id === modelId);
200228
}

0 commit comments

Comments
 (0)