fix(dispatcher,mcp-go): 配置拉取改为后台重试,根治启动竞态
此前 dispatcher(chat)/mcp-go(embedding) 启动时一次性请求控制面配置,3s 扑空即 降级,且只能干等热更新广播——若消费方早于 gateway 启动,会全程降级(LLM 跑桩、 RAG 无向量),必须手动重启才恢复。 改为:先订阅热更新,再后台 RequestConfigWithRetry(重试至拿到配置,容忍 gateway 晚启)。新增 shared/bus.RequestConfigWithRetry + dispatcher Subscriber 包装。 验收:故意先起 dispatcher/mcp-go、后起 gateway,二者自动重试拿到 chat/embedding 配置,无需手动重启;make test-go 全绿。 Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -284,6 +284,26 @@ func (b *Bus) RequestConfig(ctx context.Context, kind string) (*contract.ModelCo
|
||||
return &cfg, nil
|
||||
}
|
||||
|
||||
// RequestConfigWithRetry 后台重试拉取某 kind 的初始配置,直到成功或重试耗尽。
|
||||
// 容忍消费方(dispatcher/mcp-go)早于控制面(gateway)启动——一次性请求扑空后不再干等热更新。
|
||||
// 拿到即调 apply 并返回;ctx 取消或重试上限到则放弃(此后仍可由热更新广播兜底)。
|
||||
func (b *Bus) RequestConfigWithRetry(ctx context.Context, kind string, apply func(*contract.ModelConfig)) {
|
||||
for i := 0; i < 60; i++ {
|
||||
cctx, cancel := context.WithTimeout(ctx, 3*time.Second)
|
||||
cfg, _ := b.RequestConfig(cctx, kind)
|
||||
cancel()
|
||||
if cfg != nil {
|
||||
apply(cfg)
|
||||
return
|
||||
}
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-time.After(2 * time.Second):
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ServeConfig 让控制面响应某 kind 的配置请求;provide 返回当前激活配置(可为 nil)。
|
||||
func (b *Bus) ServeConfig(kind string, provide func() *contract.ModelConfig) (unsub func() error, err error) {
|
||||
sub, err := b.nc.Subscribe(contract.ConfigGetSubject(kind), func(m *nats.Msg) {
|
||||
|
||||
Reference in New Issue
Block a user