mirror of
https://gitee.com/mateos/mateclaw.git
synced 2026-09-15 11:58:34 +08:00
fix(channel): gate WS/long-polling channels on a distributed leader lease (#85)
This commit is contained in:
parent
b7f69dbefe
commit
3f1f83bbfe
@ -199,6 +199,34 @@ public interface ChannelAdapter {
|
|||||||
return getChannelType();
|
return getChannelType();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Whether this adapter must run on exactly one node in a multi-instance
|
||||||
|
* deployment.
|
||||||
|
*
|
||||||
|
* <p>Return {@code true} when the underlying transport rejects multiple
|
||||||
|
* concurrent connections from the same credentials — e.g. a bot WebSocket
|
||||||
|
* gateway that enforces a per-app connection cap, or a long-polling
|
||||||
|
* endpoint where multiple consumers would steal updates from each other.
|
||||||
|
* The channel manager will gate {@link #start()} on a distributed lease
|
||||||
|
* so only one node connects at a time, and failover to another node when
|
||||||
|
* the lease holder dies.
|
||||||
|
*
|
||||||
|
* <p>Webhook-based channels (DingTalk, WeCom, Slack, …) should leave this
|
||||||
|
* at the default {@code false}: inbound HTTP traffic is fanned out by the
|
||||||
|
* load balancer, so every node may safely subscribe.
|
||||||
|
*
|
||||||
|
* <p>Scope: this hook is honored by the framework for <b>DB-backed
|
||||||
|
* channels</b> registered via {@code ChannelManager.startChannel}. For
|
||||||
|
* plugin-registered channels the framework can only gate the initial
|
||||||
|
* register attempt — there is no follower retry, no hot-swap, and no
|
||||||
|
* disable-detection (plugins have a register/unregister lifecycle, not
|
||||||
|
* a DB-driven one). Plugin authors needing full single-leader semantics
|
||||||
|
* should depend on {@code ChannelLeaderElection} directly.
|
||||||
|
*/
|
||||||
|
default boolean requiresSingleLeader() {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* RFC-024 Change 2:本 adapter 认为"多久没活动就视作 stale 需要重启"的阈值。
|
* RFC-024 Change 2:本 adapter 认为"多久没活动就视作 stale 需要重启"的阈值。
|
||||||
*
|
*
|
||||||
|
|||||||
@ -10,6 +10,8 @@ import org.springframework.stereotype.Component;
|
|||||||
import vip.mate.channel.dingtalk.DingTalkChannelAdapter;
|
import vip.mate.channel.dingtalk.DingTalkChannelAdapter;
|
||||||
import vip.mate.channel.discord.DiscordChannelAdapter;
|
import vip.mate.channel.discord.DiscordChannelAdapter;
|
||||||
import vip.mate.channel.feishu.FeishuChannelAdapter;
|
import vip.mate.channel.feishu.FeishuChannelAdapter;
|
||||||
|
import vip.mate.channel.leader.ChannelLeaderElection;
|
||||||
|
import vip.mate.channel.leader.LeaderLease;
|
||||||
import vip.mate.channel.model.ChannelEntity;
|
import vip.mate.channel.model.ChannelEntity;
|
||||||
import vip.mate.channel.qq.QQChannelAdapter;
|
import vip.mate.channel.qq.QQChannelAdapter;
|
||||||
import vip.mate.channel.service.ChannelService;
|
import vip.mate.channel.service.ChannelService;
|
||||||
@ -17,7 +19,9 @@ import vip.mate.channel.telegram.TelegramChannelAdapter;
|
|||||||
import vip.mate.channel.web.WebChannelAdapter;
|
import vip.mate.channel.web.WebChannelAdapter;
|
||||||
import vip.mate.channel.wecom.WeComChannelAdapter;
|
import vip.mate.channel.wecom.WeComChannelAdapter;
|
||||||
import vip.mate.channel.weixin.WeixinChannelAdapter;
|
import vip.mate.channel.weixin.WeixinChannelAdapter;
|
||||||
|
import vip.mate.exception.MateClawException;
|
||||||
|
|
||||||
|
import java.time.LocalDateTime;
|
||||||
import java.util.*;
|
import java.util.*;
|
||||||
import java.util.concurrent.*;
|
import java.util.concurrent.*;
|
||||||
import java.util.concurrent.locks.ReadWriteLock;
|
import java.util.concurrent.locks.ReadWriteLock;
|
||||||
@ -71,12 +75,66 @@ public class ChannelManager {
|
|||||||
*/
|
*/
|
||||||
private final vip.mate.channel.wecom.WeComKeepaliveScheduler weComKeepaliveScheduler;
|
private final vip.mate.channel.wecom.WeComKeepaliveScheduler weComKeepaliveScheduler;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Distributed leader election. Channels whose adapter reports
|
||||||
|
* {@link ChannelAdapter#requiresSingleLeader()} are gated on a lease so
|
||||||
|
* only one node opens the upstream WebSocket / long-poll at a time.
|
||||||
|
*/
|
||||||
|
private final ChannelLeaderElection leaderElection;
|
||||||
|
|
||||||
/** 运行中的渠道适配器:channelId -> adapter */
|
/** 运行中的渠道适配器:channelId -> adapter */
|
||||||
private final Map<Long, ChannelAdapter> activeAdapters = new HashMap<>();
|
private final Map<Long, ChannelAdapter> activeAdapters = new HashMap<>();
|
||||||
|
|
||||||
/** 插件注册的渠道适配器:pluginName -> adapter */
|
/** 插件注册的渠道适配器:pluginName -> adapter */
|
||||||
private final Map<String, ChannelAdapter> pluginChannels = new ConcurrentHashMap<>();
|
private final Map<String, ChannelAdapter> pluginChannels = new ConcurrentHashMap<>();
|
||||||
|
|
||||||
|
/** Held leadership leases for plugin channels: pluginName -> lease */
|
||||||
|
private final Map<String, LeaderLease> pluginLeases = new ConcurrentHashMap<>();
|
||||||
|
|
||||||
|
/** Lease-extension futures for plugin channels: pluginName -> heartbeat */
|
||||||
|
private final Map<String, ScheduledFuture<?>> pluginHeartbeatFutures = new ConcurrentHashMap<>();
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Serializes register / unregister / shutdown / heartbeat-loss cleanup
|
||||||
|
* for plugin channels. The three plugin maps each use
|
||||||
|
* {@code ConcurrentHashMap}, which makes individual put/remove atomic
|
||||||
|
* — but not the multi-map sequences these paths run (snapshot, clear,
|
||||||
|
* stop adapter, release lease). Without this lock, a concurrent
|
||||||
|
* {@code registerPluginChannel} could insert into {@code pluginChannels}
|
||||||
|
* after {@code stopAll} snapshot-copies it but before
|
||||||
|
* {@code pluginChannels.clear()}, leaking an unstopped adapter and a
|
||||||
|
* never-released lease.
|
||||||
|
*/
|
||||||
|
private final Object pluginLifecycleLock = new Object();
|
||||||
|
|
||||||
|
/** Held leadership leases for leader-required channels: channelId -> lease */
|
||||||
|
private final Map<Long, LeaderLease> activeLeases = new HashMap<>();
|
||||||
|
|
||||||
|
/** Lease-extension futures: channelId -> heartbeat */
|
||||||
|
private final Map<Long, ScheduledFuture<?>> heartbeatFutures = new HashMap<>();
|
||||||
|
|
||||||
|
/** Follower retry futures: channelId -> retry */
|
||||||
|
private final Map<Long, ScheduledFuture<?>> followerRetryFutures = new HashMap<>();
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Reconcile futures for <b>non</b>-leader-required active adapters
|
||||||
|
* (e.g. webhook-mode Feishu, polling-disabled Telegram). Without these,
|
||||||
|
* cross-node admin actions on a follower would never reach the running
|
||||||
|
* adapter on another node — only leaders run reconciliation through
|
||||||
|
* their heartbeat, so a webhook-running node would otherwise be deaf
|
||||||
|
* to disable / delete / config / mode-flip changes processed elsewhere.
|
||||||
|
*/
|
||||||
|
private final Map<Long, ScheduledFuture<?>> reconcileFutures = new HashMap<>();
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Last {@code update_time} the leader observed for each owned channel.
|
||||||
|
* Compared against the DB row on every heartbeat — a newer value means
|
||||||
|
* another node has applied a config change, and we (the connection
|
||||||
|
* holder) need to apply it locally too. Otherwise the leader keeps
|
||||||
|
* running with stale credentials/mode after admin edits.
|
||||||
|
*/
|
||||||
|
private final Map<Long, LocalDateTime> lastSeenChannelUpdateTime = new HashMap<>();
|
||||||
|
|
||||||
/** 读写锁:读操作(getAdapter 等)用读锁,写操作(start/stop/replace)用写锁 */
|
/** 读写锁:读操作(getAdapter 等)用读锁,写操作(start/stop/replace)用写锁 */
|
||||||
private final ReadWriteLock adapterLock = new ReentrantReadWriteLock();
|
private final ReadWriteLock adapterLock = new ReentrantReadWriteLock();
|
||||||
|
|
||||||
@ -87,9 +145,34 @@ public class ChannelManager {
|
|||||||
return t;
|
return t;
|
||||||
});
|
});
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Heartbeat + follower-retry scheduler. A small pool is enough — both
|
||||||
|
* tasks are short-lived (lock extend or a single startChannel attempt).
|
||||||
|
*/
|
||||||
|
private final ScheduledExecutorService leaderScheduler = Executors.newScheduledThreadPool(2, r -> {
|
||||||
|
Thread t = new Thread(r, "channel-leader");
|
||||||
|
t.setDaemon(true);
|
||||||
|
return t;
|
||||||
|
});
|
||||||
|
|
||||||
/** 旧 Adapter stop() 超时时间(秒) */
|
/** 旧 Adapter stop() 超时时间(秒) */
|
||||||
private static final int STOP_TIMEOUT_SECONDS = 5;
|
private static final int STOP_TIMEOUT_SECONDS = 5;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Lease heartbeat cadence. Must be well under
|
||||||
|
* {@link ChannelLeaderElection#LOCK_AT_MOST_FOR} (60s) so a single
|
||||||
|
* missed tick doesn't lose leadership; 20s gives us three chances per
|
||||||
|
* window.
|
||||||
|
*/
|
||||||
|
private static final long HEARTBEAT_INTERVAL_SECONDS = 20L;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* How often a follower retries to acquire leadership. Picked so a
|
||||||
|
* leader dying gets failed-over within (lease window + this interval)
|
||||||
|
* = ~90s in the worst case, without hammering the DB.
|
||||||
|
*/
|
||||||
|
private static final long FOLLOWER_RETRY_INTERVAL_SECONDS = 30L;
|
||||||
|
|
||||||
/** 支持的渠道类型 */
|
/** 支持的渠道类型 */
|
||||||
private static final Set<String> SUPPORTED_TYPES = Set.of(
|
private static final Set<String> SUPPORTED_TYPES = Set.of(
|
||||||
"web", "dingtalk", "feishu", "telegram", "discord", "wecom", "qq", "weixin", "slack", "webchat"
|
"web", "dingtalk", "feishu", "telegram", "discord", "wecom", "qq", "weixin", "slack", "webchat"
|
||||||
@ -122,6 +205,7 @@ public class ChannelManager {
|
|||||||
public void destroy() {
|
public void destroy() {
|
||||||
log.info("Shutting down ChannelManager, stopping {} active channels...", activeAdapters.size());
|
log.info("Shutting down ChannelManager, stopping {} active channels...", activeAdapters.size());
|
||||||
stopAll();
|
stopAll();
|
||||||
|
leaderScheduler.shutdownNow();
|
||||||
stopExecutor.shutdownNow();
|
stopExecutor.shutdownNow();
|
||||||
messageRouter.shutdown();
|
messageRouter.shutdown();
|
||||||
}
|
}
|
||||||
@ -140,9 +224,25 @@ public class ChannelManager {
|
|||||||
}
|
}
|
||||||
|
|
||||||
ChannelAdapter adapter = createAdapter(channel);
|
ChannelAdapter adapter = createAdapter(channel);
|
||||||
adapter.start();
|
if (adapter.requiresSingleLeader()) {
|
||||||
activeAdapters.put(channel.getId(), adapter);
|
attemptLeaderStart(channel, adapter);
|
||||||
log.info("Channel started: {} (type={}, id={})", channel.getName(), channel.getChannelType(), channel.getId());
|
} else {
|
||||||
|
adapter.start();
|
||||||
|
activeAdapters.put(channel.getId(), adapter);
|
||||||
|
lastSeenChannelUpdateTime.put(channel.getId(), channel.getUpdateTime());
|
||||||
|
// A follower retry may still be scheduled if this channel was
|
||||||
|
// previously in leader-required mode and just flipped to a
|
||||||
|
// non-leader transport (e.g. Feishu websocket → webhook).
|
||||||
|
// Cancel it so we don't tick forever on a no-op startChannel.
|
||||||
|
cancelFollowerRetryLocked(channel.getId());
|
||||||
|
// Schedule cross-node reconciliation for this non-leader adapter:
|
||||||
|
// followers driven by their retry tick re-read the channel, but
|
||||||
|
// a running non-leader has neither a heartbeat nor a retry, so
|
||||||
|
// without this ticker it would never notice admin actions
|
||||||
|
// processed on a different node.
|
||||||
|
scheduleReconcileLocked(channel.getId(), channel.getName());
|
||||||
|
log.info("Channel started: {} (type={}, id={})", channel.getName(), channel.getChannelType(), channel.getId());
|
||||||
|
}
|
||||||
} finally {
|
} finally {
|
||||||
adapterLock.writeLock().unlock();
|
adapterLock.writeLock().unlock();
|
||||||
}
|
}
|
||||||
@ -153,9 +253,15 @@ public class ChannelManager {
|
|||||||
*/
|
*/
|
||||||
public void stopChannel(Long channelId) {
|
public void stopChannel(Long channelId) {
|
||||||
ChannelAdapter oldAdapter;
|
ChannelAdapter oldAdapter;
|
||||||
|
LeaderLease lease;
|
||||||
adapterLock.writeLock().lock();
|
adapterLock.writeLock().lock();
|
||||||
try {
|
try {
|
||||||
oldAdapter = activeAdapters.remove(channelId);
|
oldAdapter = activeAdapters.remove(channelId);
|
||||||
|
lease = activeLeases.remove(channelId);
|
||||||
|
lastSeenChannelUpdateTime.remove(channelId);
|
||||||
|
cancelHeartbeatLocked(channelId);
|
||||||
|
cancelFollowerRetryLocked(channelId);
|
||||||
|
cancelReconcileLocked(channelId);
|
||||||
} finally {
|
} finally {
|
||||||
adapterLock.writeLock().unlock();
|
adapterLock.writeLock().unlock();
|
||||||
}
|
}
|
||||||
@ -163,6 +269,371 @@ public class ChannelManager {
|
|||||||
if (oldAdapter != null) {
|
if (oldAdapter != null) {
|
||||||
stopAdapterSafely(oldAdapter, "stopChannel");
|
stopAdapterSafely(oldAdapter, "stopChannel");
|
||||||
}
|
}
|
||||||
|
if (lease != null) {
|
||||||
|
try {
|
||||||
|
lease.release();
|
||||||
|
log.info("[leader] Released lease for channel id={}", channelId);
|
||||||
|
} catch (Exception e) {
|
||||||
|
log.warn("[leader] Failed to release lease for channel id={}: {}", channelId, e.getMessage());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ==================== Leader election ====================
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Try to become leader for {@code channel} and start its adapter on
|
||||||
|
* success. On failure (another node holds the lease) schedule a
|
||||||
|
* follower retry so we'll take over when the current leader dies.
|
||||||
|
*
|
||||||
|
* <p>Caller must hold the adapter write lock.
|
||||||
|
*/
|
||||||
|
private void attemptLeaderStart(ChannelEntity channel, ChannelAdapter adapter) {
|
||||||
|
String key = channel.getChannelType() + ":" + channel.getId();
|
||||||
|
Optional<LeaderLease> maybeLease = leaderElection.tryAcquire(key);
|
||||||
|
if (maybeLease.isEmpty()) {
|
||||||
|
log.info("[leader] Channel {} (id={}, type={}) is owned by another node — entering follower mode",
|
||||||
|
channel.getName(), channel.getId(), channel.getChannelType());
|
||||||
|
scheduleFollowerRetryLocked(channel.getId());
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
LeaderLease lease = maybeLease.get();
|
||||||
|
try {
|
||||||
|
adapter.start();
|
||||||
|
activeAdapters.put(channel.getId(), adapter);
|
||||||
|
activeLeases.put(channel.getId(), lease);
|
||||||
|
lastSeenChannelUpdateTime.put(channel.getId(), channel.getUpdateTime());
|
||||||
|
scheduleHeartbeatLocked(channel.getId(), channel.getName());
|
||||||
|
cancelFollowerRetryLocked(channel.getId());
|
||||||
|
log.info("[leader] Channel started as leader: {} (type={}, id={})",
|
||||||
|
channel.getName(), channel.getChannelType(), channel.getId());
|
||||||
|
} catch (Exception e) {
|
||||||
|
log.error("[leader] Adapter start failed after acquiring lease for channel {}: {} — releasing lease",
|
||||||
|
channel.getName(), e.getMessage(), e);
|
||||||
|
lease.release();
|
||||||
|
throw e instanceof RuntimeException re ? re
|
||||||
|
: new RuntimeException("Channel start failed: " + e.getMessage(), e);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Schedule periodic heartbeat to extend the lease. Caller must hold
|
||||||
|
* the adapter write lock.
|
||||||
|
*/
|
||||||
|
private void scheduleHeartbeatLocked(Long channelId, String channelName) {
|
||||||
|
cancelHeartbeatLocked(channelId);
|
||||||
|
ScheduledFuture<?> f = leaderScheduler.scheduleAtFixedRate(
|
||||||
|
() -> heartbeat(channelId, channelName),
|
||||||
|
HEARTBEAT_INTERVAL_SECONDS, HEARTBEAT_INTERVAL_SECONDS, TimeUnit.SECONDS);
|
||||||
|
heartbeatFutures.put(channelId, f);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* One heartbeat tick. Extends the lease, then reconciles the local
|
||||||
|
* adapter against the current DB row — the leader is the only node
|
||||||
|
* holding the upstream connection, so it's the only one that can
|
||||||
|
* apply admin actions (disable / config change / delete) issued on
|
||||||
|
* a different node. Without this, e.g. a {@code /toggle?enabled=false}
|
||||||
|
* processed by a follower would never reach the actual connection.
|
||||||
|
*/
|
||||||
|
private void heartbeat(Long channelId, String channelName) {
|
||||||
|
LeaderLease lease;
|
||||||
|
adapterLock.readLock().lock();
|
||||||
|
try {
|
||||||
|
lease = activeLeases.get(channelId);
|
||||||
|
} finally {
|
||||||
|
adapterLock.readLock().unlock();
|
||||||
|
}
|
||||||
|
if (lease == null) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
boolean stillOurs = lease.extend(ChannelLeaderElection.LOCK_AT_MOST_FOR);
|
||||||
|
if (!stillOurs) {
|
||||||
|
log.warn("[leader] Lost leadership for channel {} (id={}) — stopping local adapter and re-entering election",
|
||||||
|
channelName, channelId);
|
||||||
|
handleLeadershipLoss(channelId);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
reconcileChannel(channelId, channelName);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Re-read the channel from the DB and apply any admin-side changes
|
||||||
|
* (disable, delete, config update) the leader hasn't seen yet because
|
||||||
|
* the API call was processed by a different node.
|
||||||
|
*
|
||||||
|
* <p>Package-private for unit testing — callers should rely on the
|
||||||
|
* heartbeat scheduler invoking this on its tick.
|
||||||
|
*/
|
||||||
|
void reconcileChannel(Long channelId, String channelName) {
|
||||||
|
ChannelEntity current;
|
||||||
|
try {
|
||||||
|
current = channelService.getChannel(channelId);
|
||||||
|
} catch (MateClawException e) {
|
||||||
|
if (e.getMsgKey() != null && e.getMsgKey().startsWith("err.channel.not_found")) {
|
||||||
|
log.info("[reconcile] Channel id={} no longer exists — stopping local adapter and releasing lease",
|
||||||
|
channelId);
|
||||||
|
stopChannel(channelId);
|
||||||
|
} else {
|
||||||
|
log.debug("[reconcile] Channel lookup failed for id={}: {}", channelId, e.getMessage());
|
||||||
|
}
|
||||||
|
return;
|
||||||
|
} catch (Exception e) {
|
||||||
|
// Transient DB issue; skip this tick and try again on the next heartbeat.
|
||||||
|
log.debug("[reconcile] Channel lookup failed for id={}: {}", channelId, e.getMessage());
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!Boolean.TRUE.equals(current.getEnabled())) {
|
||||||
|
log.info("[reconcile] Channel {} (id={}) is now disabled — stopping local adapter",
|
||||||
|
channelName, channelId);
|
||||||
|
stopChannel(channelId);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
LocalDateTime previousSeen;
|
||||||
|
adapterLock.readLock().lock();
|
||||||
|
try {
|
||||||
|
previousSeen = lastSeenChannelUpdateTime.get(channelId);
|
||||||
|
} finally {
|
||||||
|
adapterLock.readLock().unlock();
|
||||||
|
}
|
||||||
|
LocalDateTime currentUpdateTime = current.getUpdateTime();
|
||||||
|
if (currentUpdateTime != null && previousSeen != null
|
||||||
|
&& currentUpdateTime.isAfter(previousSeen)) {
|
||||||
|
log.info("[reconcile] Channel {} (id={}) config changed ({} → {})",
|
||||||
|
channelName, channelId, previousSeen, currentUpdateTime);
|
||||||
|
applyConfigChange(channelId, current);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Apply a detected config change to a locally-running channel.
|
||||||
|
*
|
||||||
|
* <p>The fast path is the in-place swap that preserves the lease —
|
||||||
|
* but it is only valid when we are the current leader (we already
|
||||||
|
* hold {@code activeLeases[channelId]}). Without that gate, a
|
||||||
|
* non-leader node observing a {@code webhook → websocket} flip
|
||||||
|
* would call {@code newAdapter.start()} directly inside the swap
|
||||||
|
* and open a duplicate upstream connection, defeating the leader
|
||||||
|
* election. Every other transition — including
|
||||||
|
* {@code non-leader → leader-required}, {@code leader-required →
|
||||||
|
* non-leader}, and plain non-leader config updates — must go
|
||||||
|
* through {@code stopChannel} + {@code startChannel} so the lease
|
||||||
|
* is correctly released or acquired and follower retry is
|
||||||
|
* scheduled when election is lost.
|
||||||
|
*
|
||||||
|
* <p>Package-private for unit testing — see {@link #reconcileChannel}.
|
||||||
|
*/
|
||||||
|
void applyConfigChange(Long channelId, ChannelEntity newChannel) {
|
||||||
|
ChannelAdapter probe = createAdapter(newChannel);
|
||||||
|
boolean newRequiresLeader = probe.requiresSingleLeader();
|
||||||
|
boolean weHaveLease;
|
||||||
|
adapterLock.readLock().lock();
|
||||||
|
try {
|
||||||
|
weHaveLease = activeLeases.containsKey(channelId);
|
||||||
|
} finally {
|
||||||
|
adapterLock.readLock().unlock();
|
||||||
|
}
|
||||||
|
|
||||||
|
if (newRequiresLeader && weHaveLease) {
|
||||||
|
// Case A: same leader-required mode and we are the current leader
|
||||||
|
// — preserve the lease across the adapter swap.
|
||||||
|
swapAdapterPreservingLease(channelId, newChannel);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
// All other cases: tear down local state and route through
|
||||||
|
// startChannel so leader election runs, lease is released, or both.
|
||||||
|
log.info("[reconcile] Channel {} (id={}) config change (newRequiresLeader={}, weHaveLease={}) — stop+start",
|
||||||
|
newChannel.getName(), channelId, newRequiresLeader, weHaveLease);
|
||||||
|
stopChannel(channelId);
|
||||||
|
try {
|
||||||
|
startChannel(newChannel);
|
||||||
|
} catch (Exception e) {
|
||||||
|
log.error("[reconcile] Restart after config change failed for channel {} (id={}): {}",
|
||||||
|
newChannel.getName(), channelId, e.getMessage(), e);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Swap to a freshly-built adapter using the new config, while keeping
|
||||||
|
* the leadership lease and heartbeat in place. The lease is only
|
||||||
|
* released if the new adapter fails to start, in which case we fall
|
||||||
|
* back to follower mode so another node can try.
|
||||||
|
*/
|
||||||
|
private void swapAdapterPreservingLease(Long channelId, ChannelEntity newChannel) {
|
||||||
|
ChannelAdapter oldAdapter;
|
||||||
|
adapterLock.writeLock().lock();
|
||||||
|
try {
|
||||||
|
oldAdapter = activeAdapters.remove(channelId);
|
||||||
|
} finally {
|
||||||
|
adapterLock.writeLock().unlock();
|
||||||
|
}
|
||||||
|
if (oldAdapter != null) {
|
||||||
|
stopAdapterSafely(oldAdapter, "reconcile-swap");
|
||||||
|
}
|
||||||
|
|
||||||
|
ChannelAdapter newAdapter = createAdapter(newChannel);
|
||||||
|
boolean started = false;
|
||||||
|
Exception startError = null;
|
||||||
|
try {
|
||||||
|
newAdapter.start();
|
||||||
|
started = true;
|
||||||
|
} catch (Exception e) {
|
||||||
|
startError = e;
|
||||||
|
}
|
||||||
|
|
||||||
|
LeaderLease leaseToRelease = null;
|
||||||
|
adapterLock.writeLock().lock();
|
||||||
|
try {
|
||||||
|
if (started) {
|
||||||
|
activeAdapters.put(channelId, newAdapter);
|
||||||
|
lastSeenChannelUpdateTime.put(channelId, newChannel.getUpdateTime());
|
||||||
|
} else {
|
||||||
|
leaseToRelease = activeLeases.remove(channelId);
|
||||||
|
lastSeenChannelUpdateTime.remove(channelId);
|
||||||
|
cancelHeartbeatLocked(channelId);
|
||||||
|
scheduleFollowerRetryLocked(channelId);
|
||||||
|
}
|
||||||
|
} finally {
|
||||||
|
adapterLock.writeLock().unlock();
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!started) {
|
||||||
|
log.error("[reconcile] New adapter start failed for channel {} (id={}): {} — released lease, entering follower mode",
|
||||||
|
newChannel.getName(), channelId,
|
||||||
|
startError != null ? startError.getMessage() : "unknown");
|
||||||
|
if (leaseToRelease != null) {
|
||||||
|
leaseToRelease.release();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Drop the local adapter (without releasing the already-lost lease)
|
||||||
|
* and start follower retry so we'll attempt to reclaim leadership
|
||||||
|
* once the current owner stops renewing.
|
||||||
|
*/
|
||||||
|
private void handleLeadershipLoss(Long channelId) {
|
||||||
|
ChannelAdapter local;
|
||||||
|
adapterLock.writeLock().lock();
|
||||||
|
try {
|
||||||
|
local = activeAdapters.remove(channelId);
|
||||||
|
activeLeases.remove(channelId); // already lost; do not call release()
|
||||||
|
cancelHeartbeatLocked(channelId);
|
||||||
|
scheduleFollowerRetryLocked(channelId);
|
||||||
|
} finally {
|
||||||
|
adapterLock.writeLock().unlock();
|
||||||
|
}
|
||||||
|
if (local != null) {
|
||||||
|
stopAdapterSafely(local, "leadership-loss");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Schedule periodic follower retry. Caller must hold the adapter
|
||||||
|
* write lock.
|
||||||
|
*/
|
||||||
|
private void scheduleFollowerRetryLocked(Long channelId) {
|
||||||
|
if (followerRetryFutures.containsKey(channelId)) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
ScheduledFuture<?> f = leaderScheduler.scheduleAtFixedRate(
|
||||||
|
() -> followerRetry(channelId),
|
||||||
|
FOLLOWER_RETRY_INTERVAL_SECONDS, FOLLOWER_RETRY_INTERVAL_SECONDS, TimeUnit.SECONDS);
|
||||||
|
followerRetryFutures.put(channelId, f);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* One follower-retry tick. Re-reads the channel from the DB (it may
|
||||||
|
* have been disabled or deleted) and attempts to start it. The retry
|
||||||
|
* cancels itself once we successfully become leader.
|
||||||
|
*/
|
||||||
|
/** Package-private for unit testing — see {@link #reconcileChannel}. */
|
||||||
|
void followerRetry(Long channelId) {
|
||||||
|
ChannelEntity current;
|
||||||
|
try {
|
||||||
|
current = channelService.getChannel(channelId);
|
||||||
|
} catch (MateClawException e) {
|
||||||
|
// Channel was deleted on another node — cancel the retry so we
|
||||||
|
// don't leak a scheduled task forever. Other exception codes
|
||||||
|
// (e.g. transient DB errors) fall through to the generic catch
|
||||||
|
// and let the retry continue.
|
||||||
|
if (e.getMsgKey() != null && e.getMsgKey().startsWith("err.channel.not_found")) {
|
||||||
|
log.info("[leader] Follower retry: channel id={} no longer exists — cancelling retry", channelId);
|
||||||
|
adapterLock.writeLock().lock();
|
||||||
|
try {
|
||||||
|
cancelFollowerRetryLocked(channelId);
|
||||||
|
} finally {
|
||||||
|
adapterLock.writeLock().unlock();
|
||||||
|
}
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
log.debug("[leader] Follower retry: lookup failed for channel id={}: {}", channelId, e.getMessage());
|
||||||
|
return;
|
||||||
|
} catch (Exception e) {
|
||||||
|
log.debug("[leader] Follower retry: lookup failed for channel id={}: {}", channelId, e.getMessage());
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (!Boolean.TRUE.equals(current.getEnabled())) {
|
||||||
|
adapterLock.writeLock().lock();
|
||||||
|
try {
|
||||||
|
cancelFollowerRetryLocked(channelId);
|
||||||
|
} finally {
|
||||||
|
adapterLock.writeLock().unlock();
|
||||||
|
}
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
startChannel(current);
|
||||||
|
} catch (Exception e) {
|
||||||
|
log.debug("[leader] Follower retry: startChannel failed for id={}: {}", channelId, e.getMessage());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Package-private for unit testing. */
|
||||||
|
boolean hasFollowerRetry(Long channelId) {
|
||||||
|
adapterLock.readLock().lock();
|
||||||
|
try {
|
||||||
|
return followerRetryFutures.containsKey(channelId);
|
||||||
|
} finally {
|
||||||
|
adapterLock.readLock().unlock();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private void cancelHeartbeatLocked(Long channelId) {
|
||||||
|
ScheduledFuture<?> f = heartbeatFutures.remove(channelId);
|
||||||
|
if (f != null) {
|
||||||
|
f.cancel(false);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private void cancelFollowerRetryLocked(Long channelId) {
|
||||||
|
ScheduledFuture<?> f = followerRetryFutures.remove(channelId);
|
||||||
|
if (f != null) {
|
||||||
|
f.cancel(false);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Schedule the cross-node reconcile ticker for a non-leader active
|
||||||
|
* adapter. Caller must hold the adapter write lock.
|
||||||
|
*/
|
||||||
|
private void scheduleReconcileLocked(Long channelId, String channelName) {
|
||||||
|
cancelReconcileLocked(channelId);
|
||||||
|
ScheduledFuture<?> f = leaderScheduler.scheduleAtFixedRate(
|
||||||
|
() -> reconcileChannel(channelId, channelName),
|
||||||
|
FOLLOWER_RETRY_INTERVAL_SECONDS, FOLLOWER_RETRY_INTERVAL_SECONDS, TimeUnit.SECONDS);
|
||||||
|
reconcileFutures.put(channelId, f);
|
||||||
|
}
|
||||||
|
|
||||||
|
private void cancelReconcileLocked(Long channelId) {
|
||||||
|
ScheduledFuture<?> f = reconcileFutures.remove(channelId);
|
||||||
|
if (f != null) {
|
||||||
|
f.cancel(false);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@ -186,6 +657,50 @@ public class ChannelManager {
|
|||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Leader-required channels: if we don't already own the local adapter
|
||||||
|
// (i.e. we are a follower, or this is a brand-new channel), there is
|
||||||
|
// nothing to hot-swap. Fall through to startChannel which handles
|
||||||
|
// lease acquisition and follower retry. Hot-swap (which briefly opens
|
||||||
|
// a second upstream connection) is also avoided here so we don't
|
||||||
|
// double-occupy the bot's connection quota during a restart.
|
||||||
|
boolean weHaveAdapter;
|
||||||
|
boolean weHaveLease;
|
||||||
|
adapterLock.readLock().lock();
|
||||||
|
try {
|
||||||
|
weHaveAdapter = activeAdapters.containsKey(channelId);
|
||||||
|
weHaveLease = activeLeases.containsKey(channelId);
|
||||||
|
} finally {
|
||||||
|
adapterLock.readLock().unlock();
|
||||||
|
}
|
||||||
|
ChannelAdapter probe = createAdapter(channel);
|
||||||
|
if (probe.requiresSingleLeader()) {
|
||||||
|
log.info("[hot-swap] Channel {} requires single-leader; stop+start instead of hot-swap (weHaveAdapter={}, weHaveLease={})",
|
||||||
|
channel.getName(), weHaveAdapter, weHaveLease);
|
||||||
|
stopChannel(channelId);
|
||||||
|
startChannel(channel);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
// Mode flip: we currently hold a lease but the new config is no
|
||||||
|
// longer leader-required (e.g. Feishu WS → webhook). The lease,
|
||||||
|
// heartbeat, and lastSeenUpdateTime must all be torn down before
|
||||||
|
// the new non-leader adapter starts — the in-place hot-swap path
|
||||||
|
// below would leave them behind until the next heartbeat tick
|
||||||
|
// noticed and re-restarted, causing a redundant restart and a
|
||||||
|
// window where this node is silently holding a lease nobody else
|
||||||
|
// can grab. Stop+start handles all the cleanup in one shot.
|
||||||
|
if (weHaveLease) {
|
||||||
|
log.info("[hot-swap] Channel {} flipping leader-required → non-leader; stop+start to release lease",
|
||||||
|
channel.getName());
|
||||||
|
stopChannel(channelId);
|
||||||
|
startChannel(channel);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (!weHaveAdapter) {
|
||||||
|
log.info("[hot-swap] No local adapter for channel {}, delegating to startChannel", channel.getName());
|
||||||
|
startChannel(channel);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
log.info("[hot-swap] Starting hot-swap for channel: {} (type={}, id={})",
|
log.info("[hot-swap] Starting hot-swap for channel: {} (type={}, id={})",
|
||||||
channel.getName(), channel.getChannelType(), channelId);
|
channel.getName(), channel.getChannelType(), channelId);
|
||||||
|
|
||||||
@ -207,6 +722,10 @@ public class ChannelManager {
|
|||||||
adapterLock.writeLock().lock();
|
adapterLock.writeLock().lock();
|
||||||
try {
|
try {
|
||||||
oldAdapter = activeAdapters.put(channelId, newAdapter);
|
oldAdapter = activeAdapters.put(channelId, newAdapter);
|
||||||
|
// Mark the version we've now applied so the reconcile ticker
|
||||||
|
// (running for non-leader adapters) doesn't immediately fire a
|
||||||
|
// redundant swap on its next tick.
|
||||||
|
lastSeenChannelUpdateTime.put(channelId, channel.getUpdateTime());
|
||||||
log.info("[hot-swap] Adapter reference swapped for channel: {} (old={})",
|
log.info("[hot-swap] Adapter reference swapped for channel: {} (old={})",
|
||||||
channel.getName(), oldAdapter != null ? "present" : "none");
|
channel.getName(), oldAdapter != null ? "present" : "none");
|
||||||
} finally {
|
} finally {
|
||||||
@ -228,17 +747,65 @@ public class ChannelManager {
|
|||||||
*/
|
*/
|
||||||
public void stopAll() {
|
public void stopAll() {
|
||||||
List<ChannelAdapter> adaptersToStop;
|
List<ChannelAdapter> adaptersToStop;
|
||||||
|
List<LeaderLease> leasesToRelease;
|
||||||
|
List<ChannelAdapter> pluginAdaptersToStop;
|
||||||
|
List<LeaderLease> pluginLeasesToRelease;
|
||||||
adapterLock.writeLock().lock();
|
adapterLock.writeLock().lock();
|
||||||
try {
|
try {
|
||||||
adaptersToStop = new ArrayList<>(activeAdapters.values());
|
adaptersToStop = new ArrayList<>(activeAdapters.values());
|
||||||
|
leasesToRelease = new ArrayList<>(activeLeases.values());
|
||||||
activeAdapters.clear();
|
activeAdapters.clear();
|
||||||
|
activeLeases.clear();
|
||||||
|
lastSeenChannelUpdateTime.clear();
|
||||||
|
heartbeatFutures.values().forEach(f -> f.cancel(false));
|
||||||
|
heartbeatFutures.clear();
|
||||||
|
followerRetryFutures.values().forEach(f -> f.cancel(false));
|
||||||
|
followerRetryFutures.clear();
|
||||||
|
reconcileFutures.values().forEach(f -> f.cancel(false));
|
||||||
|
reconcileFutures.clear();
|
||||||
} finally {
|
} finally {
|
||||||
adapterLock.writeLock().unlock();
|
adapterLock.writeLock().unlock();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Plugin channels live on a different map (keyed by pluginName), but
|
||||||
|
// shutdown must release their leases + cancel their heartbeats just
|
||||||
|
// like DB-backed ones. Without this, a plugin-supplied single-leader
|
||||||
|
// adapter would skip graceful release on @PreDestroy and its lease
|
||||||
|
// would stay locked until the lockAtMostFor window expired.
|
||||||
|
//
|
||||||
|
// Holding pluginLifecycleLock makes the snapshot+clear atomic
|
||||||
|
// against concurrent register / unregister / heartbeat-loss
|
||||||
|
// cleanup, so we don't leak an adapter that was registered after
|
||||||
|
// the snapshot but before the clear.
|
||||||
|
synchronized (pluginLifecycleLock) {
|
||||||
|
pluginAdaptersToStop = new ArrayList<>(pluginChannels.values());
|
||||||
|
pluginChannels.clear();
|
||||||
|
pluginLeasesToRelease = new ArrayList<>(pluginLeases.values());
|
||||||
|
pluginLeases.clear();
|
||||||
|
pluginHeartbeatFutures.values().forEach(f -> f.cancel(false));
|
||||||
|
pluginHeartbeatFutures.clear();
|
||||||
|
}
|
||||||
|
|
||||||
for (ChannelAdapter adapter : adaptersToStop) {
|
for (ChannelAdapter adapter : adaptersToStop) {
|
||||||
stopAdapterSafely(adapter, "stopAll");
|
stopAdapterSafely(adapter, "stopAll");
|
||||||
}
|
}
|
||||||
|
for (ChannelAdapter adapter : pluginAdaptersToStop) {
|
||||||
|
stopAdapterSafely(adapter, "stopAll-plugin");
|
||||||
|
}
|
||||||
|
for (LeaderLease lease : leasesToRelease) {
|
||||||
|
try {
|
||||||
|
lease.release();
|
||||||
|
} catch (Exception e) {
|
||||||
|
log.warn("[leader] stopAll: failed to release lease '{}': {}", lease.getName(), e.getMessage());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for (LeaderLease lease : pluginLeasesToRelease) {
|
||||||
|
try {
|
||||||
|
lease.release();
|
||||||
|
} catch (Exception e) {
|
||||||
|
log.warn("[leader] stopAll: failed to release plugin lease '{}': {}", lease.getName(), e.getMessage());
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// ==================== 查询(读锁保护) ====================
|
// ==================== 查询(读锁保护) ====================
|
||||||
@ -398,16 +965,42 @@ public class ChannelManager {
|
|||||||
/**
|
/**
|
||||||
* Register a channel adapter from a plugin.
|
* Register a channel adapter from a plugin.
|
||||||
*
|
*
|
||||||
|
* <p>If the adapter reports {@link ChannelAdapter#requiresSingleLeader()},
|
||||||
|
* the framework gates the local register on a distributed lease keyed by
|
||||||
|
* {@code plugin:{pluginName}}. When another node already owns the lease
|
||||||
|
* this node skips registration (its plugin instance is loaded but inert
|
||||||
|
* locally) — see the scope note on {@link ChannelAdapter#requiresSingleLeader()}.
|
||||||
|
*
|
||||||
* @param pluginName the plugin name (used as key for unregistration)
|
* @param pluginName the plugin name (used as key for unregistration)
|
||||||
* @param adapter the channel adapter
|
* @param adapter the channel adapter
|
||||||
*/
|
*/
|
||||||
public void registerPluginChannel(String pluginName, ChannelAdapter adapter) {
|
public void registerPluginChannel(String pluginName, ChannelAdapter adapter) {
|
||||||
try {
|
synchronized (pluginLifecycleLock) {
|
||||||
adapter.start();
|
LeaderLease lease = null;
|
||||||
pluginChannels.put(pluginName, adapter);
|
if (adapter.requiresSingleLeader()) {
|
||||||
log.info("Plugin channel registered: {} (type={})", pluginName, adapter.getChannelType());
|
Optional<LeaderLease> maybeLease = leaderElection.tryAcquire("plugin:" + pluginName);
|
||||||
} catch (Exception e) {
|
if (maybeLease.isEmpty()) {
|
||||||
log.error("Failed to start plugin channel {}: {}", pluginName, e.getMessage(), e);
|
log.info("[leader] Plugin channel {} (type={}) is owned by another node — skipping local registration",
|
||||||
|
pluginName, adapter.getChannelType());
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
lease = maybeLease.get();
|
||||||
|
}
|
||||||
|
|
||||||
|
try {
|
||||||
|
adapter.start();
|
||||||
|
pluginChannels.put(pluginName, adapter);
|
||||||
|
if (lease != null) {
|
||||||
|
pluginLeases.put(pluginName, lease);
|
||||||
|
schedulePluginHeartbeatLocked(pluginName);
|
||||||
|
}
|
||||||
|
log.info("Plugin channel registered: {} (type={})", pluginName, adapter.getChannelType());
|
||||||
|
} catch (Exception e) {
|
||||||
|
log.error("Failed to start plugin channel {}: {}", pluginName, e.getMessage(), e);
|
||||||
|
if (lease != null) {
|
||||||
|
lease.release();
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@ -415,11 +1008,78 @@ public class ChannelManager {
|
|||||||
* Unregister a plugin channel.
|
* Unregister a plugin channel.
|
||||||
*/
|
*/
|
||||||
public void unregisterPluginChannel(String pluginName) {
|
public void unregisterPluginChannel(String pluginName) {
|
||||||
ChannelAdapter adapter = pluginChannels.remove(pluginName);
|
ChannelAdapter adapter;
|
||||||
|
ScheduledFuture<?> heartbeat;
|
||||||
|
LeaderLease lease;
|
||||||
|
synchronized (pluginLifecycleLock) {
|
||||||
|
adapter = pluginChannels.remove(pluginName);
|
||||||
|
heartbeat = pluginHeartbeatFutures.remove(pluginName);
|
||||||
|
lease = pluginLeases.remove(pluginName);
|
||||||
|
}
|
||||||
|
if (heartbeat != null) {
|
||||||
|
heartbeat.cancel(false);
|
||||||
|
}
|
||||||
if (adapter != null) {
|
if (adapter != null) {
|
||||||
stopAdapterSafely(adapter, "unregisterPluginChannel");
|
stopAdapterSafely(adapter, "unregisterPluginChannel");
|
||||||
log.info("Plugin channel unregistered: {}", pluginName);
|
log.info("Plugin channel unregistered: {}", pluginName);
|
||||||
}
|
}
|
||||||
|
if (lease != null) {
|
||||||
|
try {
|
||||||
|
lease.release();
|
||||||
|
} catch (Exception e) {
|
||||||
|
log.warn("[leader] Failed to release plugin lease '{}': {}", pluginName, e.getMessage());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Caller must hold {@link #pluginLifecycleLock}. */
|
||||||
|
private void schedulePluginHeartbeatLocked(String pluginName) {
|
||||||
|
ScheduledFuture<?> existing = pluginHeartbeatFutures.remove(pluginName);
|
||||||
|
if (existing != null) {
|
||||||
|
existing.cancel(false);
|
||||||
|
}
|
||||||
|
ScheduledFuture<?> f = leaderScheduler.scheduleAtFixedRate(
|
||||||
|
() -> pluginHeartbeatTick(pluginName),
|
||||||
|
HEARTBEAT_INTERVAL_SECONDS, HEARTBEAT_INTERVAL_SECONDS, TimeUnit.SECONDS);
|
||||||
|
pluginHeartbeatFutures.put(pluginName, f);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* One plugin heartbeat tick. The tick runs on the scheduler thread, so
|
||||||
|
* it can race with {@code registerPluginChannel} / {@code unregisterPluginChannel}
|
||||||
|
* / {@code stopAll}. Serializing the loss-handler on
|
||||||
|
* {@link #pluginLifecycleLock} keeps the three maps consistent (no
|
||||||
|
* "stop ran twice" or "lease released after register re-acquired").
|
||||||
|
*/
|
||||||
|
private void pluginHeartbeatTick(String pluginName) {
|
||||||
|
LeaderLease lease = pluginLeases.get(pluginName);
|
||||||
|
if (lease == null) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
boolean stillOurs = lease.extend(ChannelLeaderElection.LOCK_AT_MOST_FOR);
|
||||||
|
if (stillOurs) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
ChannelAdapter local;
|
||||||
|
ScheduledFuture<?> self;
|
||||||
|
synchronized (pluginLifecycleLock) {
|
||||||
|
// Re-check under the lock — a concurrent unregister may have
|
||||||
|
// already torn everything down between the failed extend and
|
||||||
|
// our entering the locked region.
|
||||||
|
LeaderLease currentLease = pluginLeases.remove(pluginName);
|
||||||
|
if (currentLease == null) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
local = pluginChannels.remove(pluginName);
|
||||||
|
self = pluginHeartbeatFutures.remove(pluginName);
|
||||||
|
}
|
||||||
|
log.warn("[leader] Lost plugin lease '{}' — unregistering local adapter", pluginName);
|
||||||
|
if (self != null) {
|
||||||
|
self.cancel(false);
|
||||||
|
}
|
||||||
|
if (local != null) {
|
||||||
|
stopAdapterSafely(local, "plugin-leadership-loss");
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// ==================== 内部方法 ====================
|
// ==================== 内部方法 ====================
|
||||||
@ -471,8 +1131,11 @@ public class ChannelManager {
|
|||||||
/**
|
/**
|
||||||
* 根据渠道实体创建对应的适配器实例
|
* 根据渠道实体创建对应的适配器实例
|
||||||
* 采用渠道注册表模式,根据类型创建对应适配器
|
* 采用渠道注册表模式,根据类型创建对应适配器
|
||||||
|
*
|
||||||
|
* <p>Package-private + non-final so unit tests can substitute a stub
|
||||||
|
* adapter without spinning up real WebSocket / HTTP clients.
|
||||||
*/
|
*/
|
||||||
private ChannelAdapter createAdapter(ChannelEntity channel) {
|
ChannelAdapter createAdapter(ChannelEntity channel) {
|
||||||
String type = channel.getChannelType();
|
String type = channel.getChannelType();
|
||||||
return switch (type) {
|
return switch (type) {
|
||||||
case "web" -> new WebChannelAdapter(channel, messageRouter, objectMapper);
|
case "web" -> new WebChannelAdapter(channel, messageRouter, objectMapper);
|
||||||
|
|||||||
@ -373,6 +373,18 @@ public class DiscordChannelAdapter extends AbstractChannelAdapter {
|
|||||||
return CHANNEL_TYPE;
|
return CHANNEL_TYPE;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Discord enforces a single Gateway session per bot token (any duplicate
|
||||||
|
* {@code IDENTIFY} closes the previous shard) and there is no active
|
||||||
|
* webhook fallback in this adapter — the legacy webhook endpoint is
|
||||||
|
* a no-op. The leader gate ensures only one node holds the Gateway
|
||||||
|
* connection at a time.
|
||||||
|
*/
|
||||||
|
@Override
|
||||||
|
public boolean requiresSingleLeader() {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
// ==================== Webhook 兼容(保留接口,不再使用) ====================
|
// ==================== Webhook 兼容(保留接口,不再使用) ====================
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|||||||
@ -1435,4 +1435,19 @@ public class FeishuChannelAdapter extends AbstractChannelAdapter {
|
|||||||
public String getChannelType() {
|
public String getChannelType() {
|
||||||
return CHANNEL_TYPE;
|
return CHANNEL_TYPE;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* WebSocket mode opens a long-lived connection to Lark's gateway, which
|
||||||
|
* caps concurrent connections per bot app (~2). In a multi-instance
|
||||||
|
* deployment every node would race for that quota and reconnect-loop on
|
||||||
|
* {@code 1000040350: the number of connections exceeded the limit}.
|
||||||
|
* The leader gate ensures only one node holds the connection at a time.
|
||||||
|
*
|
||||||
|
* <p>Webhook mode is exempt: callbacks are HTTP-fanned by the load
|
||||||
|
* balancer, so all nodes can safely subscribe.
|
||||||
|
*/
|
||||||
|
@Override
|
||||||
|
public boolean requiresSingleLeader() {
|
||||||
|
return "websocket".equals(getConfigString("connection_mode", "websocket"));
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@ -0,0 +1,84 @@
|
|||||||
|
package vip.mate.channel.leader;
|
||||||
|
|
||||||
|
import lombok.RequiredArgsConstructor;
|
||||||
|
import lombok.extern.slf4j.Slf4j;
|
||||||
|
import net.javacrumbs.shedlock.core.LockConfiguration;
|
||||||
|
import net.javacrumbs.shedlock.core.LockProvider;
|
||||||
|
import net.javacrumbs.shedlock.core.SimpleLock;
|
||||||
|
import org.springframework.stereotype.Component;
|
||||||
|
|
||||||
|
import java.time.Duration;
|
||||||
|
import java.time.Instant;
|
||||||
|
import java.util.Optional;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Distributed leader election for channel adapters whose upstream
|
||||||
|
* service rejects multiple concurrent connections from the same bot
|
||||||
|
* credentials.
|
||||||
|
*
|
||||||
|
* <p>Typical examples are WebSocket-mode IM channels: Feishu's Lark
|
||||||
|
* SDK enforces a per-app connection cap (~2) and QQ's bot gateway
|
||||||
|
* rejects duplicate {@code IDENTIFY} sessions. Without coordination,
|
||||||
|
* every node of a multi-instance deployment trips that cap on startup
|
||||||
|
* and reconnects in a tight loop.
|
||||||
|
*
|
||||||
|
* <p>Backed by ShedLock's {@link LockProvider} (already wired for cron
|
||||||
|
* coordination), so single-node deployments incur no extra
|
||||||
|
* infrastructure — the lock is acquired trivially on the only node.
|
||||||
|
*
|
||||||
|
* <p>Semantics:
|
||||||
|
* <ul>
|
||||||
|
* <li>{@link #tryAcquire(String)} returns the lease, or empty if
|
||||||
|
* another node already holds it.</li>
|
||||||
|
* <li>The owning node must call {@link LeaderLease#extend(Duration)}
|
||||||
|
* on a fixed cadence (heartbeat) shorter than the
|
||||||
|
* {@code lockAtMostFor} window, or the lease expires and another
|
||||||
|
* node may claim it.</li>
|
||||||
|
* <li>If a node dies without releasing, the lease auto-expires after
|
||||||
|
* {@code lockAtMostFor}, after which any waiting follower can
|
||||||
|
* become leader on its next retry tick.</li>
|
||||||
|
* </ul>
|
||||||
|
*/
|
||||||
|
@Slf4j
|
||||||
|
@Component
|
||||||
|
@RequiredArgsConstructor
|
||||||
|
public class ChannelLeaderElection {
|
||||||
|
|
||||||
|
/**
|
||||||
|
* How long the lock is held without a renewal. A failed node's lease
|
||||||
|
* stays locked for this long before another node can take over —
|
||||||
|
* so longer values increase failover latency, shorter values increase
|
||||||
|
* the risk of false handover during a GC pause or DB hiccup.
|
||||||
|
*/
|
||||||
|
public static final Duration LOCK_AT_MOST_FOR = Duration.ofSeconds(60);
|
||||||
|
|
||||||
|
private final LockProvider lockProvider;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Attempt to acquire leadership for the given key.
|
||||||
|
*
|
||||||
|
* @param key a stable identifier for the resource (e.g.
|
||||||
|
* {@code "feishu:42"}). Used verbatim as the underlying
|
||||||
|
* lock name (prefixed by this class to avoid collisions
|
||||||
|
* with other lock users).
|
||||||
|
* @return an empty optional if another node already holds the lease,
|
||||||
|
* otherwise a {@link LeaderLease} that the caller is
|
||||||
|
* responsible for periodically extending and finally
|
||||||
|
* releasing.
|
||||||
|
*/
|
||||||
|
public Optional<LeaderLease> tryAcquire(String key) {
|
||||||
|
String lockName = "channel-leader:" + key;
|
||||||
|
LockConfiguration config = new LockConfiguration(
|
||||||
|
Instant.now(),
|
||||||
|
lockName,
|
||||||
|
LOCK_AT_MOST_FOR,
|
||||||
|
Duration.ZERO);
|
||||||
|
Optional<SimpleLock> lock = lockProvider.lock(config);
|
||||||
|
if (lock.isEmpty()) {
|
||||||
|
log.debug("[leader] Lock '{}' is held by another node", lockName);
|
||||||
|
return Optional.empty();
|
||||||
|
}
|
||||||
|
log.info("[leader] Acquired lease '{}'", lockName);
|
||||||
|
return Optional.of(new LeaderLease(lockName, lock.get()));
|
||||||
|
}
|
||||||
|
}
|
||||||
@ -0,0 +1,77 @@
|
|||||||
|
package vip.mate.channel.leader;
|
||||||
|
|
||||||
|
import lombok.extern.slf4j.Slf4j;
|
||||||
|
import net.javacrumbs.shedlock.core.SimpleLock;
|
||||||
|
|
||||||
|
import java.time.Duration;
|
||||||
|
import java.util.Optional;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* A held leadership lease on a single resource (typically a channel id).
|
||||||
|
*
|
||||||
|
* <p>Wraps a ShedLock {@link SimpleLock} so callers don't depend on the
|
||||||
|
* underlying lock provider. The lease must be periodically extended via
|
||||||
|
* {@link #extend(Duration)} or it expires automatically, at which point
|
||||||
|
* another node can claim leadership.
|
||||||
|
*
|
||||||
|
* <p>Threading: a single lease instance is not safe for concurrent
|
||||||
|
* {@link #extend(Duration)} / {@link #release()} calls. The owning
|
||||||
|
* scheduler is expected to serialize them.
|
||||||
|
*/
|
||||||
|
@Slf4j
|
||||||
|
public class LeaderLease {
|
||||||
|
|
||||||
|
private final String name;
|
||||||
|
private volatile SimpleLock current;
|
||||||
|
private volatile boolean released;
|
||||||
|
|
||||||
|
LeaderLease(String name, SimpleLock initial) {
|
||||||
|
this.name = name;
|
||||||
|
this.current = initial;
|
||||||
|
}
|
||||||
|
|
||||||
|
public String getName() {
|
||||||
|
return name;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Try to extend this lease for another {@code lockAtMostFor} window.
|
||||||
|
*
|
||||||
|
* @return true if the lease is still ours; false if it has been lost
|
||||||
|
* (e.g. the previous window expired before extend ran and
|
||||||
|
* another node acquired the lock — the caller should treat
|
||||||
|
* this as a leadership loss and stop the protected resource).
|
||||||
|
*/
|
||||||
|
public boolean extend(Duration lockAtMostFor) {
|
||||||
|
if (released) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
Optional<SimpleLock> next = current.extend(lockAtMostFor, Duration.ZERO);
|
||||||
|
if (next.isPresent()) {
|
||||||
|
current = next.get();
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
log.warn("[leader] Lease '{}' extend returned empty — lock lost", name);
|
||||||
|
return false;
|
||||||
|
} catch (Exception e) {
|
||||||
|
log.warn("[leader] Lease '{}' extend threw: {}", name, e.getMessage());
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Release this lease. Idempotent — calling release twice is safe.
|
||||||
|
*/
|
||||||
|
public void release() {
|
||||||
|
if (released) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
released = true;
|
||||||
|
try {
|
||||||
|
current.unlock();
|
||||||
|
} catch (Exception e) {
|
||||||
|
log.warn("[leader] Lease '{}' unlock threw: {}", name, e.getMessage());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@ -199,6 +199,17 @@ public class QQChannelAdapter extends AbstractChannelAdapter {
|
|||||||
return CHANNEL_TYPE;
|
return CHANNEL_TYPE;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The QQ bot gateway rejects duplicate {@code IDENTIFY} sessions for
|
||||||
|
* the same app credentials. Multiple nodes connecting simultaneously
|
||||||
|
* trip the connection cap and reconnect-loop forever. The leader gate
|
||||||
|
* ensures only one node holds the WebSocket at a time.
|
||||||
|
*/
|
||||||
|
@Override
|
||||||
|
public boolean requiresSingleLeader() {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
// ==================== Access Token 管理 ====================
|
// ==================== Access Token 管理 ====================
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|||||||
@ -165,8 +165,13 @@ public class TelegramChannelAdapter extends AbstractChannelAdapter {
|
|||||||
* - connection_mode=polling → Polling
|
* - connection_mode=polling → Polling
|
||||||
* - connection_mode 缺失 + webhook_url 非空 → Webhook(兼容旧配置)
|
* - connection_mode 缺失 + webhook_url 非空 → Webhook(兼容旧配置)
|
||||||
* - 其余 → Polling
|
* - 其余 → Polling
|
||||||
|
*
|
||||||
|
* <p>Package-private so {@link #requiresSingleLeader()} can mirror the
|
||||||
|
* same predicate — the two answers must stay in lockstep, otherwise a
|
||||||
|
* single change to mode detection here would silently mis-classify the
|
||||||
|
* channel for multi-node coordination.
|
||||||
*/
|
*/
|
||||||
private boolean resolveWebhookMode() {
|
boolean resolveWebhookMode() {
|
||||||
String connectionMode = getConfigString("connection_mode");
|
String connectionMode = getConfigString("connection_mode");
|
||||||
String webhookUrl = getConfigString("webhook_url");
|
String webhookUrl = getConfigString("webhook_url");
|
||||||
boolean hasWebhookUrl = webhookUrl != null && !webhookUrl.isBlank();
|
boolean hasWebhookUrl = webhookUrl != null && !webhookUrl.isBlank();
|
||||||
@ -794,6 +799,21 @@ public class TelegramChannelAdapter extends AbstractChannelAdapter {
|
|||||||
return CHANNEL_TYPE;
|
return CHANNEL_TYPE;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Long-polling mode runs a {@code getUpdates(offset=…)} loop that
|
||||||
|
* acknowledges each delivered update; multiple nodes polling the same
|
||||||
|
* bot token would steal updates from each other (whichever node calls
|
||||||
|
* {@code getUpdates} next consumes the queue, and the others get
|
||||||
|
* nothing). The leader gate ensures only one node polls at a time.
|
||||||
|
*
|
||||||
|
* <p>Webhook mode is exempt: Telegram POSTs to a public URL fanned
|
||||||
|
* by the load balancer, so every node may safely receive callbacks.
|
||||||
|
*/
|
||||||
|
@Override
|
||||||
|
public boolean requiresSingleLeader() {
|
||||||
|
return !resolveWebhookMode();
|
||||||
|
}
|
||||||
|
|
||||||
/** RFC-025 Change 4 入站文本净化上限(防止 caption 含超长二进制撑爆 prompt)。 */
|
/** RFC-025 Change 4 入站文本净化上限(防止 caption 含超长二进制撑爆 prompt)。 */
|
||||||
private static final int INBOUND_TEXT_MAX = 4096;
|
private static final int INBOUND_TEXT_MAX = 4096;
|
||||||
|
|
||||||
|
|||||||
@ -3300,6 +3300,20 @@ public class WeComChannelAdapter extends AbstractChannelAdapter {
|
|||||||
return CHANNEL_TYPE;
|
return CHANNEL_TYPE;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* WeCom's AI-bot transport is WebSocket-only ({@code wss://openws.work.weixin.qq.com}
|
||||||
|
* with {@code aibot_subscribe} authenticated by {@code bot_id + secret}) — there is
|
||||||
|
* no HTTP webhook fallback. Multiple nodes subscribing with the same
|
||||||
|
* credentials would each receive every {@code aibot_msg_callback} and race
|
||||||
|
* to send {@code aibot_respond_msg}, producing duplicate replies and
|
||||||
|
* eventual gateway rejection. The leader gate ensures only one node
|
||||||
|
* holds the subscription at a time.
|
||||||
|
*/
|
||||||
|
@Override
|
||||||
|
public boolean requiresSingleLeader() {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
// ==================== 工具方法 ====================
|
// ==================== 工具方法 ====================
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|||||||
Loading…
Reference in New Issue
Block a user