Compare commits

...
2 changed files with 86 additions and 3 deletions
+6 -3
View File
@@ -33,7 +33,8 @@ const patterns = [
/context[_ ]length[_ ]exceeded/i,
/context length is only \d+ tokens/i,
/input length.*exceeds.*context length/i,
/prompt too long; exceeded (?:max )?context length/i,
// Z.ai code 1261 arrives as `Prompt too long` or `Prompt 超长`.
/prompt (?:too long|超长)/i,
/too large for model with \d+ maximum context length/i,
/prompt has [\d,]+ tokens?, but the configured context size is [\d,]+ tokens?/i,
/model_context_window_exceeded/i,
@@ -145,13 +146,15 @@ const CONTENT_POLICY_CODES = new Set([
const GATEWAY_CODE_LABEL = /^[^:\n]+: \[([A-Za-z0-9_.-]+)\]/
const RATE_LIMIT_TEXT = /rate increased too quickly|rate[-_\s]?limit|too[_\s]?many[_\s]?requests/i
// Only consulted on 429, where throttles and account caps share a status.
const QUOTA_TEXT = /insufficient[-_\s]?quota|quota[-_\s]?exceeded|budget exceeded|usage limit/i
// Z.ai reports balance, plan expiry, plan limits, plan model access, and fair-use restrictions on 429.
const QUOTA_TEXT =
/insufficient[-_\s]?(?:quota|balance)|quota[-_\s]?exceeded|budget exceeded|usage limit|limit exhausted|package has expired|plan does not yet include|fair usage policy/i
// Policy rejections without a dedicated code, matched against the provider's own
// explanation only. OpenAI reuses `invalid_prompt` for usage-policy rejections while
// Bedrock Mantle reuses it for schema validation; Anthropic reports blocked output
// under `invalid_request_error`.
const CONTENT_POLICY_TEXT =
/violating our usage policy|blocked by content filtering policy|content[-_\s]?policy|rejected as a result of our safety system/i
/violating our usage policy|blocked by content filtering policy|content[-_\s]?policy|rejected as a result of our safety system|detected potentially unsafe or sensitive content/i
const SERVER_ERROR_TEXT =
/\b(?:try again|(?:please |you can )?retry (?:the |this |your )?request|try (?:the |this |your )?request again|(?:currently |temporarily )?at capacity|overloaded|temporarily unavailable|service[-_\s]?unavailable|(?:server|internal)[-_\s]?error|server (?:is )?busy|provider returned (?:an )?error|resource[-_\s]?exhausted|upstream (?:connect|connection|request)|request buffer limit while retrying upstream)\b/i
+80
View File
@@ -282,6 +282,86 @@ describe("provider error classification", () => {
).toEqual(Array(6).fill("QuotaExceeded"))
})
test("classifies Z.ai plan and balance limits as quota rather than throttling", () => {
const zai = (code: string, message: string) => ({ error: { code, message } })
const cases = [
zai("1113", "Insufficient balance or no resource package. Please recharge."),
zai("1308", "Usage limit reached for 5 hours. Your limit will reset at 2026-10-01 00:00:00"),
zai(
"1309",
"Your GLM Coding Plan package has expired and is temporarily unavailable. You can resume using it after renewing the subscription on the official website.",
),
zai("1310", "Weekly/Monthly Limit Exhausted. Your limit will reset at 2026-10-01 00:00:00"),
zai("1311", "Your current subscription plan does not yet include access to glm-5"),
zai("1314", "Your enterprise package has expired. Please contact your enterprise administrator."),
zai(
"1313",
"Your account's current usage pattern does not comply with the Fair Usage Policy, and your request frequency has been limited. For details, please refer to the Subscription Service Agreement. To restore access, please submit a request.",
),
// Z.ai's Anthropic-compatible endpoint wraps the code and request ID into the message.
{
type: "error",
error: {
type: "rate_limit_error",
code: "1309",
message:
"[1309][Your GLM Coding Plan package has expired and is temporarily unavailable. You can resume using it after renewing the subscription on the official website.][20260929132151e73af01340d54b58]",
},
},
]
expect(
cases.map(
(body) =>
classifyProviderFailure({ message: body.error.message, status: 429, rawBody: JSON.stringify(body) })._tag,
),
).toEqual(Array(cases.length).fill("QuotaExceeded"))
})
test("classifies Z.ai prompt length rejections as context overflow", () => {
const cases = [
{ error: { code: "1261", message: "Prompt 超长" } },
{ error: { code: "1261", message: "Prompt too long" } },
{
type: "error",
error: { type: "invalid_request_error", code: "1261", message: "[1261][Prompt too long][2026092913]" },
},
]
expect(
cases.map((body) => {
const reason = classifyProviderFailure({
message: body.error.message,
status: 400,
rawBody: JSON.stringify(body),
})
return reason._tag === "InvalidRequest" ? reason.classification : reason._tag
}),
).toEqual(["context-overflow", "context-overflow", "context-overflow"])
})
test("classifies Z.ai sensitive content rejections as content policy", () => {
const message =
"System detected potentially unsafe or sensitive content in input or generation. Please avoid using prompts that may generate sensitive content. Thank you for your cooperation."
expect(
classifyProviderFailure({
message,
status: 400,
rawBody: JSON.stringify({ error: { code: "1301", message } }),
})._tag,
).toBe("ContentPolicy")
})
test("keeps Z.ai throttling and overload retryable", () => {
expect(
[
{ error: { code: "1302", message: "Rate limit reached for requests" } },
{ error: { code: "1305", message: "The service may be temporarily overloaded, please try again later" } },
].map(
(body) =>
classifyProviderFailure({ message: body.error.message, status: 429, rawBody: JSON.stringify(body) })._tag,
),
).toEqual(["RateLimit", "RateLimit"])
})
test("does not let substituted server codes make a 4xx retryable", () => {
const openai = { error: { type: "server_error", message: "Upstream request failed: Model is unavailable." } }
const anthropic = {