fix(notify): 修复传输生命周期竞态,完善背压与协议边界
- 完善 stream/bulk DataID 分配、预留和双向命名空间,修复并发打开及 dedicated/shared 回退时的 ID 冲突 - 将收发、回复、恢复任务和 sidecar 绑定原始会话与物理连接,防止重连后的旧消息误操作新连接 - 加强 close/reset 身份校验及实例移除检查,修复 dedicated attach 失败、通道引用和资源回收竞态 - 收紧批量发送器停止准入,确保在途入队完成后统一清理请求、缓冲区和等待者 - 修复 record 满队列死锁、取消时序号消耗及关闭竞态,确保关闭有界并返回真实错误 - 增加协商式 record 逻辑半关闭,保留反向 ACK;通过 reset 传递 RecordFailure,避免背压掩盖原始失败原因 - 补齐帧长度、批次数量、序号溢出和未确认窗口校验,提前拒绝超限数据并按字节预算拆批 - 为入站分发增加全局及单连接的条数、字节预算和阻塞背压,关闭时唤醒等待者,消除正常断连日志噪音 - 完善 bulk 窗口释放失败处理与传输诊断,补充并发、重连、背压、协议边界及真实 TCP 回归覆盖
This commit is contained in:
+152
-25
@@ -8,53 +8,170 @@ import (
|
||||
|
||||
const defaultInboundDispatchSource = "_notify.default_inbound_source"
|
||||
|
||||
// Inbound dispatch runs one serial worker per source so messages from the same
|
||||
// connection keep their relative order. The pending queue is bounded both per
|
||||
// source and in total: a slow handler can never let one connection grow the
|
||||
// queue without limit, and a saturated connection can never starve the others.
|
||||
// Callers block until there is room, which pushes backpressure to the transport
|
||||
// reader instead of growing memory; CloseAndWait unblocks every caller.
|
||||
const (
|
||||
defaultInboundDispatchQueueLimit = 4096
|
||||
defaultInboundDispatchQueueBytes = 64 << 20
|
||||
defaultInboundSourceQueueLimit = 512
|
||||
defaultInboundSourceQueueBytes = 16 << 20
|
||||
)
|
||||
|
||||
type inboundDispatchItem struct {
|
||||
size int
|
||||
fn func()
|
||||
}
|
||||
|
||||
type inboundDispatcher struct {
|
||||
mu sync.Mutex
|
||||
closed bool
|
||||
workers map[string]*inboundDispatchWorker
|
||||
wg sync.WaitGroup
|
||||
mu sync.Mutex
|
||||
closed bool
|
||||
closeCh chan struct{}
|
||||
roomCh chan struct{}
|
||||
queued int
|
||||
queuedBytes int
|
||||
maxItems int
|
||||
maxBytes int
|
||||
sourceItems int
|
||||
sourceBytes int
|
||||
workers map[string]*inboundDispatchWorker
|
||||
wg sync.WaitGroup
|
||||
}
|
||||
|
||||
type inboundDispatchWorker struct {
|
||||
queue []func()
|
||||
running bool
|
||||
queue []inboundDispatchItem
|
||||
running bool
|
||||
queued int
|
||||
queuedBytes int
|
||||
}
|
||||
|
||||
func newInboundDispatcher() *inboundDispatcher {
|
||||
return newInboundDispatcherWithCaps(
|
||||
defaultInboundDispatchQueueLimit,
|
||||
defaultInboundDispatchQueueBytes,
|
||||
defaultInboundSourceQueueLimit,
|
||||
defaultInboundSourceQueueBytes,
|
||||
)
|
||||
}
|
||||
|
||||
// newInboundDispatcherWithLimits sizes a dispatcher for a single source: the
|
||||
// per-source caps equal the global caps.
|
||||
func newInboundDispatcherWithLimits(maxItems int, maxBytes int) *inboundDispatcher {
|
||||
return newInboundDispatcherWithCaps(maxItems, maxBytes, maxItems, maxBytes)
|
||||
}
|
||||
|
||||
func newInboundDispatcherWithCaps(maxItems int, maxBytes int, sourceItems int, sourceBytes int) *inboundDispatcher {
|
||||
if maxItems <= 0 {
|
||||
maxItems = defaultInboundDispatchQueueLimit
|
||||
}
|
||||
if maxBytes <= 0 {
|
||||
maxBytes = defaultInboundDispatchQueueBytes
|
||||
}
|
||||
if sourceItems <= 0 || sourceItems > maxItems {
|
||||
sourceItems = maxItems
|
||||
}
|
||||
if sourceBytes <= 0 || sourceBytes > maxBytes {
|
||||
sourceBytes = maxBytes
|
||||
}
|
||||
return &inboundDispatcher{
|
||||
workers: make(map[string]*inboundDispatchWorker),
|
||||
closeCh: make(chan struct{}),
|
||||
roomCh: make(chan struct{}, 1),
|
||||
maxItems: maxItems,
|
||||
maxBytes: maxBytes,
|
||||
sourceItems: sourceItems,
|
||||
sourceBytes: sourceBytes,
|
||||
workers: make(map[string]*inboundDispatchWorker),
|
||||
}
|
||||
}
|
||||
|
||||
// Dispatch queues fn for the given source without byte accounting. Prefer
|
||||
// DispatchSized when the queued payload size is known. Like DispatchSized it
|
||||
// blocks while the queue is at its limit.
|
||||
func (d *inboundDispatcher) Dispatch(source string, fn func()) bool {
|
||||
return d.DispatchSized(source, 0, fn)
|
||||
}
|
||||
|
||||
// DispatchSized queues fn for the given source. It blocks while the source or
|
||||
// the dispatcher is at its item or byte limit, and returns false once the
|
||||
// dispatcher is closed. A parked caller is released by CloseAndWait, so the
|
||||
// owner of the reader must keep CloseAndWait reachable (a concurrent closer, or
|
||||
// the reader's own stop path once the wait unblocks).
|
||||
func (d *inboundDispatcher) DispatchSized(source string, size int, fn func()) bool {
|
||||
if d == nil || fn == nil {
|
||||
return false
|
||||
}
|
||||
if source == "" {
|
||||
source = defaultInboundDispatchSource
|
||||
}
|
||||
d.mu.Lock()
|
||||
if d.closed {
|
||||
if size < 0 {
|
||||
size = 0
|
||||
}
|
||||
for {
|
||||
d.mu.Lock()
|
||||
if d.closed {
|
||||
d.mu.Unlock()
|
||||
return false
|
||||
}
|
||||
worker := d.workers[source]
|
||||
if worker == nil {
|
||||
worker = &inboundDispatchWorker{}
|
||||
d.workers[source] = worker
|
||||
}
|
||||
if d.roomLocked(worker, size) {
|
||||
worker.queue = append(worker.queue, inboundDispatchItem{size: size, fn: fn})
|
||||
worker.queued++
|
||||
worker.queuedBytes += size
|
||||
d.queued++
|
||||
d.queuedBytes += size
|
||||
if worker.running {
|
||||
d.mu.Unlock()
|
||||
return true
|
||||
}
|
||||
worker.running = true
|
||||
d.wg.Add(1)
|
||||
d.mu.Unlock()
|
||||
go d.run(source, worker)
|
||||
return true
|
||||
}
|
||||
d.mu.Unlock()
|
||||
// roomCh is signalled whenever a queued item is consumed; closeCh
|
||||
// releases the caller during shutdown.
|
||||
select {
|
||||
case <-d.roomCh:
|
||||
case <-d.closeCh:
|
||||
return false
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (d *inboundDispatcher) roomLocked(worker *inboundDispatchWorker, size int) bool {
|
||||
if d.maxItems > 0 && d.queued >= d.maxItems {
|
||||
return false
|
||||
}
|
||||
worker := d.workers[source]
|
||||
if worker == nil {
|
||||
worker = &inboundDispatchWorker{}
|
||||
d.workers[source] = worker
|
||||
if d.sourceItems > 0 && worker.queued >= d.sourceItems {
|
||||
return false
|
||||
}
|
||||
worker.queue = append(worker.queue, fn)
|
||||
if worker.running {
|
||||
d.mu.Unlock()
|
||||
return true
|
||||
// Always admit at least one item per scope so an oversized payload cannot
|
||||
// deadlock the reader behind an empty queue.
|
||||
if d.maxBytes > 0 && d.queued > 0 && d.queuedBytes+size > d.maxBytes {
|
||||
return false
|
||||
}
|
||||
if d.sourceBytes > 0 && worker.queued > 0 && worker.queuedBytes+size > d.sourceBytes {
|
||||
return false
|
||||
}
|
||||
worker.running = true
|
||||
d.wg.Add(1)
|
||||
d.mu.Unlock()
|
||||
go d.run(source, worker)
|
||||
return true
|
||||
}
|
||||
|
||||
func (d *inboundDispatcher) signalRoomLocked() {
|
||||
select {
|
||||
case d.roomCh <- struct{}{}:
|
||||
default:
|
||||
}
|
||||
}
|
||||
|
||||
func (d *inboundDispatcher) run(source string, worker *inboundDispatchWorker) {
|
||||
defer d.wg.Done()
|
||||
for {
|
||||
@@ -64,14 +181,20 @@ func (d *inboundDispatcher) run(source string, worker *inboundDispatchWorker) {
|
||||
if current := d.workers[source]; current == worker {
|
||||
delete(d.workers, source)
|
||||
}
|
||||
d.signalRoomLocked()
|
||||
d.mu.Unlock()
|
||||
return
|
||||
}
|
||||
fn := worker.queue[0]
|
||||
worker.queue[0] = nil
|
||||
item := worker.queue[0]
|
||||
worker.queue[0] = inboundDispatchItem{}
|
||||
worker.queue = worker.queue[1:]
|
||||
worker.queued--
|
||||
worker.queuedBytes -= item.size
|
||||
d.queued--
|
||||
d.queuedBytes -= item.size
|
||||
d.signalRoomLocked()
|
||||
d.mu.Unlock()
|
||||
fn()
|
||||
item.fn()
|
||||
}
|
||||
}
|
||||
|
||||
@@ -80,7 +203,11 @@ func (d *inboundDispatcher) CloseAndWait() {
|
||||
return
|
||||
}
|
||||
d.mu.Lock()
|
||||
d.closed = true
|
||||
if !d.closed {
|
||||
d.closed = true
|
||||
close(d.closeCh)
|
||||
}
|
||||
d.signalRoomLocked()
|
||||
d.mu.Unlock()
|
||||
d.wg.Wait()
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user