Files
sub2api/backend/internal/handler/failover_loop.go
T
erio 6d0e056244 feat(fingerprint): Claude Code CLI fingerprint mimicry suite (v0.1.114.1)
Squash-merge feat/fingerprint-mimic into release/custom-0.1.114.

Functional changes:
- TLS ClientHello mimicry: 17→52 cipher suites, ALPN http/1.1 only,
  X25519MLKEM768, JA3/JA4 aligned with official CLI baseline
- HTTP layer: User-Agent 2.1.112 sdk-cli, 3 new anthropic-beta tokens,
  Stainless headers synced
- Sidecar traffic: usage poll + count_tokens injection + startup probe
  (three fire-and-forget channels with jittered intervals)
- Sticky session UUID: per-account metadata.user_id reuse via Redis,
  WithSessionHash propagation through gateway
- 429 no-switch: short-circuit on rate-limit instead of cross-account fail-over
- Admin UI: per-account "randomize fingerprint" button + ConfirmDialog +
  i18n (zh/en, 9 keys)

Engineering hardening (fixes carried over from hai/snapshot):
- safe.Go/Run package: panic-recovering goroutine helper with slog
- redis.Nil treated as cache miss in identity cache
- SOCKS5 dialer uses ContextDialer for ctx cancellation propagation
- Token refresh selects on stopCh during retry backoff
- Sidecar goroutines wrapped with safe.Go
- .gitattributes forces LF for *.json (fixes baseline file CRLF drift)

Refactoring (zero behavior change):
- sidecar probe constants centralized
- identity_service rewrite preflight extracted into helpers
- capture_fingerprint tool split from 894 lines into 9 files
- gofmt sweep on all hai-imported files

Tests:
- safe package: panic recovery + attrs propagation
- tlsfingerprint: SOCKS5 dialer + capture parity
- identity sticky-session preflight paths (early-return + cache-hit)
- sidecar probe: jittered interval + dry-run
- token refresh: graceful shutdown during backoff

Beta-validated on 0.1.112.21 (commit 565da163 of feat/fingerprint-mimic).
Bumps VERSION 0.1.112.20 -> 0.1.114.1.
2026-04-18 16:07:36 +08:00

194 lines
6.8 KiB
Go

package handler
import (
"context"
"net/http"
"time"
"github.com/Wei-Shaw/sub2api/internal/pkg/logger"
"github.com/Wei-Shaw/sub2api/internal/service"
"go.uber.org/zap"
)
// TempUnscheduler 用于 HandleFailoverError 中同账号重试耗尽后的临时封禁。
// GatewayService 隐式实现此接口。
type TempUnscheduler interface {
TempUnscheduleRetryableError(ctx context.Context, accountID int64, failoverErr *service.UpstreamFailoverError)
}
// FailoverAction 表示 failover 错误处理后的下一步动作
type FailoverAction int
const (
// FailoverContinue 继续循环(同账号重试或切换账号,调用方统一 continue)
FailoverContinue FailoverAction = iota
// FailoverExhausted 切换次数耗尽(调用方应返回错误响应)
FailoverExhausted
// FailoverCanceled context 已取消(调用方应直接 return)
FailoverCanceled
)
const (
// maxSameAccountRetries 同账号重试次数上限(针对 RetryableOnSameAccount 错误)
maxSameAccountRetries = 3
// sameAccountRetryDelay 同账号重试间隔
sameAccountRetryDelay = 500 * time.Millisecond
// singleAccountBackoffDelay 单账号分组 503 退避重试固定延时。
// Service 层在 SingleAccountRetry 模式下已做充分原地重试(最多 3 次、总等待 30s),
// Handler 层只需短暂间隔后重新进入 Service 层即可。
singleAccountBackoffDelay = 2 * time.Second
)
// FailoverState 跨循环迭代共享的 failover 状态
type FailoverState struct {
SwitchCount int
MaxSwitches int
FailedAccountIDs map[int64]struct{}
SameAccountRetryCount map[int64]int
LastFailoverErr *service.UpstreamFailoverError
ForceCacheBilling bool
hasBoundSession bool
}
// NewFailoverState 创建 failover 状态
func NewFailoverState(maxSwitches int, hasBoundSession bool) *FailoverState {
return &FailoverState{
MaxSwitches: maxSwitches,
FailedAccountIDs: make(map[int64]struct{}),
SameAccountRetryCount: make(map[int64]int),
hasBoundSession: hasBoundSession,
}
}
// HandleFailoverError 处理 UpstreamFailoverError,返回下一步动作。
// 包含:缓存计费判断、同账号重试、临时封禁、切换计数、Antigravity 延时。
func (s *FailoverState) HandleFailoverError(
ctx context.Context,
gatewayService TempUnscheduler,
accountID int64,
platform string,
failoverErr *service.UpstreamFailoverError,
) FailoverAction {
s.LastFailoverErr = failoverErr
// Anti-fingerprinting (2026-04-15):
// Surface 429 (rate limit) directly to the client instead of auto-swapping
// accounts. Real Claude Code errors out on 429; a relay that retries the
// same request on a second account within ~200ms is a textbook rate-limit
// evasion signal. We mark the error as "exhausted" so the handler returns
// the upstream 429 verbatim.
//
// Note: RetryableOnSameAccount errors (e.g. Google intermittent 400 /
// empty response) are unrelated to rate limiting and still retry on the
// same account below. Only the *account switch* path is short-circuited.
if failoverErr != nil && failoverErr.StatusCode == http.StatusTooManyRequests && !failoverErr.RetryableOnSameAccount {
s.FailedAccountIDs[accountID] = struct{}{}
logger.FromContext(ctx).Warn("gateway.failover_429_no_switch",
zap.Int64("account_id", accountID),
zap.Int("upstream_status", failoverErr.StatusCode),
)
return FailoverExhausted
}
// 缓存计费判断
if needForceCacheBilling(s.hasBoundSession, failoverErr) {
s.ForceCacheBilling = true
}
// 同账号重试:对 RetryableOnSameAccount 的临时性错误,先在同一账号上重试
if failoverErr.RetryableOnSameAccount && s.SameAccountRetryCount[accountID] < maxSameAccountRetries {
s.SameAccountRetryCount[accountID]++
logger.FromContext(ctx).Warn("gateway.failover_same_account_retry",
zap.Int64("account_id", accountID),
zap.Int("upstream_status", failoverErr.StatusCode),
zap.Int("same_account_retry_count", s.SameAccountRetryCount[accountID]),
zap.Int("same_account_retry_max", maxSameAccountRetries),
)
if !sleepWithContext(ctx, sameAccountRetryDelay) {
return FailoverCanceled
}
return FailoverContinue
}
// 同账号重试用尽,执行临时封禁
if failoverErr.RetryableOnSameAccount {
gatewayService.TempUnscheduleRetryableError(ctx, accountID, failoverErr)
}
// 加入失败列表
s.FailedAccountIDs[accountID] = struct{}{}
// 检查是否耗尽
if s.SwitchCount >= s.MaxSwitches {
return FailoverExhausted
}
// 递增切换计数
s.SwitchCount++
logger.FromContext(ctx).Warn("gateway.failover_switch_account",
zap.Int64("account_id", accountID),
zap.Int("upstream_status", failoverErr.StatusCode),
zap.Int("switch_count", s.SwitchCount),
zap.Int("max_switches", s.MaxSwitches),
)
// Antigravity 平台换号线性递增延时
if platform == service.PlatformAntigravity {
delay := time.Duration(s.SwitchCount-1) * time.Second
if !sleepWithContext(ctx, delay) {
return FailoverCanceled
}
}
return FailoverContinue
}
// HandleSelectionExhausted 处理选号失败(所有候选账号都在排除列表中)时的退避重试决策。
// 针对 Antigravity 单账号分组的 503 (MODEL_CAPACITY_EXHAUSTED) 场景:
// 清除排除列表、等待退避后重新选号。
//
// 返回 FailoverContinue 时,调用方应设置 SingleAccountRetry context 并 continue。
// 返回 FailoverExhausted 时,调用方应返回错误响应。
// 返回 FailoverCanceled 时,调用方应直接 return。
func (s *FailoverState) HandleSelectionExhausted(ctx context.Context) FailoverAction {
if s.LastFailoverErr != nil &&
s.LastFailoverErr.StatusCode == http.StatusServiceUnavailable &&
s.SwitchCount <= s.MaxSwitches {
logger.FromContext(ctx).Warn("gateway.failover_single_account_backoff",
zap.Duration("backoff_delay", singleAccountBackoffDelay),
zap.Int("switch_count", s.SwitchCount),
zap.Int("max_switches", s.MaxSwitches),
)
if !sleepWithContext(ctx, singleAccountBackoffDelay) {
return FailoverCanceled
}
logger.FromContext(ctx).Warn("gateway.failover_single_account_retry",
zap.Int("switch_count", s.SwitchCount),
zap.Int("max_switches", s.MaxSwitches),
)
s.FailedAccountIDs = make(map[int64]struct{})
return FailoverContinue
}
return FailoverExhausted
}
// needForceCacheBilling 判断 failover 时是否需要强制缓存计费。
// 粘性会话切换账号、或上游明确标记时,将 input_tokens 转为 cache_read 计费。
func needForceCacheBilling(hasBoundSession bool, failoverErr *service.UpstreamFailoverError) bool {
return hasBoundSession || (failoverErr != nil && failoverErr.ForceCacheBilling)
}
// sleepWithContext 等待指定时长,返回 false 表示 context 已取消。
func sleepWithContext(ctx context.Context, d time.Duration) bool {
if d <= 0 {
return true
}
select {
case <-ctx.Done():
return false
case <-time.After(d):
return true
}
}