fix(notify): 修复传输生命周期竞态,完善背压与协议边界

- 完善 stream/bulk DataID 分配、预留和双向命名空间,修复并发打开及 dedicated/shared 回退时的 ID 冲突
- 将收发、回复、恢复任务和 sidecar 绑定原始会话与物理连接,防止重连后的旧消息误操作新连接
- 加强 close/reset 身份校验及实例移除检查,修复 dedicated attach 失败、通道引用和资源回收竞态
- 收紧批量发送器停止准入,确保在途入队完成后统一清理请求、缓冲区和等待者
- 修复 record 满队列死锁、取消时序号消耗及关闭竞态,确保关闭有界并返回真实错误
- 增加协商式 record 逻辑半关闭,保留反向 ACK;通过 reset 传递 RecordFailure,避免背压掩盖原始失败原因
- 补齐帧长度、批次数量、序号溢出和未确认窗口校验,提前拒绝超限数据并按字节预算拆批
- 为入站分发增加全局及单连接的条数、字节预算和阻塞背压,关闭时唤醒等待者,消除正常断连日志噪音
- 完善 bulk 窗口释放失败处理与传输诊断,补充并发、重连、背压、协议边界及真实 TCP 回归覆盖
This commit is contained in:
2026-09-23 15:33:17 +08:00
parent 0826e17063
commit 1f2e74acca
79 changed files with 9190 additions and 1013 deletions
+323 -36
View File
@@ -3,6 +3,7 @@ package notify
import (
"context"
"errors"
"fmt"
"io"
"net"
"strings"
@@ -32,6 +33,8 @@ const (
defaultBulkAcceptReadyTimeout = 10 * time.Second
defaultBulkResetNotifyTimeout = 30 * time.Second
defaultBulkDataWriteTimeout = 2 * time.Minute
bulkWindowReleaseRetryDelay = 25 * time.Millisecond
bulkWindowReleaseShutdownGrace = 100 * time.Millisecond
)
type BulkMetadata map[string]string
@@ -163,6 +166,7 @@ var (
errBulkRejected = errors.New("bulk open rejected")
errBulkReset = errors.New("bulk reset")
errBulkDataIDEmpty = errors.New("bulk data id is empty")
errBulkDataIDExhausted = errors.New("bulk data id exhausted")
errBulkDataPathNotReady = errors.New("bulk data path is not implemented yet")
errBulkRangeInvalid = errors.New("bulk range is invalid")
errBulkBackpressureExceeded = errors.New("bulk inbound backpressure exceeded")
@@ -291,7 +295,10 @@ type bulkHandle struct {
rangeSpec BulkRange
metadata BulkMetadata
sessionEpoch uint64
clientRoute clientSessionRoute
client *ClientCommon
debug atomic.Bool
debugSide string
logical *LogicalConn
transport *TransportConn
transportGeneration uint64
@@ -316,8 +323,11 @@ type bulkHandle struct {
writeCtxCancel context.CancelFunc
createdAt time.Time
writeMu sync.Mutex
mu sync.Mutex
writeMu sync.Mutex
mu sync.Mutex
negotiationMu sync.RWMutex
finalizeOnce sync.Once
acceptState atomic.Uint32 // 0=pending, 1=dispatched/handled, 2=reset before dispatch
writeQueue chan bulkAsyncWriteRequest
writeWorkerDone chan struct{}
@@ -355,6 +365,7 @@ type bulkHandle struct {
dedicatedReady chan struct{}
dedicatedWriteClosed bool
dedicatedActiveLease bool
dedicatedLaneLease bool
dedicatedState bulkDedicatedAttachState
dedicatedAttempts uint32
dedicatedLastCode string
@@ -411,6 +422,7 @@ func newBulkHandle(parent context.Context, runtime *bulkRuntime, runtimeScope st
ctx: ctx,
cancel: cancel,
createdAt: time.Now(),
debugSide: bulkDebugSide(logical),
readNotify: make(chan struct{}, 1),
flowNotify: make(chan struct{}, 1),
writeStateDone: make(chan struct{}),
@@ -454,11 +466,20 @@ func (b *bulkHandle) fastPathVersionSnapshot() uint8 {
if b == nil {
return bulkFastPathVersionV1
}
b.mu.Lock()
defer b.mu.Unlock()
b.negotiationMu.RLock()
defer b.negotiationMu.RUnlock()
return normalizeBulkFastPathVersion(b.fastPathVersion)
}
func (b *bulkHandle) setFastPathVersion(version uint8) {
if b == nil {
return
}
b.negotiationMu.Lock()
b.fastPathVersion = normalizeBulkFastPathVersion(version)
b.negotiationMu.Unlock()
}
func (b *bulkHandle) FastPathVersion() uint8 {
return b.fastPathVersionSnapshot()
}
@@ -502,9 +523,20 @@ func (b *bulkHandle) TransportGeneration() uint64 {
if b == nil {
return 0
}
b.negotiationMu.RLock()
defer b.negotiationMu.RUnlock()
return b.transportGeneration
}
func (b *bulkHandle) setTransportGeneration(generation uint64) {
if b == nil || generation == 0 {
return
}
b.negotiationMu.Lock()
b.transportGeneration = generation
b.negotiationMu.Unlock()
}
func (b *bulkHandle) Dedicated() bool {
if b == nil {
return false
@@ -618,6 +650,9 @@ func (b *bulkHandle) installDedicatedSender(sender *bulkDedicatedSender) *bulkDe
}
b.dedicatedMu.Lock()
defer b.dedicatedMu.Unlock()
if b.dedicatedState == bulkDedicatedAttachStateClosed {
return nil
}
if b.dedicatedSender != nil {
return b.dedicatedSender
}
@@ -689,6 +724,10 @@ func (b *bulkHandle) attachDedicatedConn(conn net.Conn) error {
return net.ErrClosed
}
b.dedicatedMu.Lock()
if b.dedicatedState == bulkDedicatedAttachStateClosed {
b.dedicatedMu.Unlock()
return b.dedicatedAttachClosedError()
}
if b.dedicatedConn != nil {
b.dedicatedMu.Unlock()
return errors.New("bulk dedicated conn already attached")
@@ -718,6 +757,10 @@ func (b *bulkHandle) attachDedicatedConnShared(conn net.Conn) error {
return net.ErrClosed
}
b.dedicatedMu.Lock()
if b.dedicatedState == bulkDedicatedAttachStateClosed {
b.dedicatedMu.Unlock()
return b.dedicatedAttachClosedError()
}
if b.dedicatedConn != nil {
if b.dedicatedConn == conn {
b.dedicatedConnOwned = false
@@ -754,6 +797,10 @@ func (b *bulkHandle) replaceDedicatedConn(conn net.Conn) (net.Conn, *bulkDedicat
return nil, nil, net.ErrClosed
}
b.dedicatedMu.Lock()
if b.dedicatedState == bulkDedicatedAttachStateClosed {
b.dedicatedMu.Unlock()
return nil, nil, b.dedicatedAttachClosedError()
}
oldConn := b.dedicatedConn
oldOwned := b.dedicatedConnOwned
oldSender := b.dedicatedSender
@@ -786,6 +833,10 @@ func (b *bulkHandle) replaceDedicatedConnShared(conn net.Conn) (net.Conn, *bulkD
return nil, nil, net.ErrClosed
}
b.dedicatedMu.Lock()
if b.dedicatedState == bulkDedicatedAttachStateClosed {
b.dedicatedMu.Unlock()
return nil, nil, b.dedicatedAttachClosedError()
}
oldConn := b.dedicatedConn
oldOwned := b.dedicatedConnOwned
oldSender := b.dedicatedSender
@@ -836,6 +887,13 @@ func (b *bulkHandle) bestEffortCloseDedicatedWriteHalf() {
}
}
func (b *bulkHandle) dedicatedAttachClosedError() error {
if err := b.resetErrSnapshot(); err != nil {
return err
}
return io.ErrClosedPipe
}
func (b *bulkHandle) dedicatedWriteHalfClosedSnapshot() bool {
if b == nil {
return false
@@ -850,6 +908,16 @@ func (b *bulkHandle) setClientSnapshotOwner(client *ClientCommon) {
return
}
b.client = client
if client != nil {
b.debug.Store(client.IsDebugMode())
}
}
func bulkDebugSide(logical *LogicalConn) string {
if logical != nil {
return "server"
}
return "client"
}
func (b *bulkHandle) clearDedicatedConn() (net.Conn, bool) {
@@ -888,10 +956,35 @@ func (b *bulkHandle) releaseDedicatedActiveReserved() bool {
return true
}
func (b *bulkHandle) markDedicatedLaneReserved() {
if b == nil {
return
}
b.dedicatedMu.Lock()
b.dedicatedLaneLease = true
b.dedicatedMu.Unlock()
}
func (b *bulkHandle) releaseDedicatedLaneReserved() bool {
if b == nil {
return false
}
b.dedicatedMu.Lock()
defer b.dedicatedMu.Unlock()
if !b.dedicatedLaneLease {
return false
}
b.dedicatedLaneLease = false
return true
}
func (b *bulkHandle) markAcceptDispatched() bool {
if b == nil {
return false
}
if !b.acceptState.CompareAndSwap(0, 1) {
return false
}
b.acceptMu.Lock()
defer b.acceptMu.Unlock()
if b.acceptDispatched {
@@ -905,6 +998,7 @@ func (b *bulkHandle) markAcceptHandled() {
if b == nil {
return
}
b.acceptState.CompareAndSwap(0, 1)
b.acceptMu.Lock()
b.acceptDispatched = true
b.acceptMu.Unlock()
@@ -1047,14 +1141,66 @@ func (b *bulkHandle) acceptsClientSessionEpoch(epoch uint64) bool {
return b.sessionEpoch == epoch
}
func (b *bulkHandle) setClientSessionRoute(route clientSessionRoute) {
if b == nil {
return
}
b.sessionEpoch = route.epoch
b.clientRoute = route
}
func (b *bulkHandle) clientSessionRouteSnapshot() clientSessionRoute {
if b == nil {
return clientSessionRoute{}
}
if !b.clientRoute.bound() && b.client != nil {
return b.client.clientSessionRouteSnapshot()
}
return b.clientRoute
}
func (b *bulkHandle) acceptsClientSessionRoute(route clientSessionRoute) bool {
if !b.acceptsClientSessionEpoch(route.epoch) {
return false
}
if b == nil || b.clientRoute.binding == nil || route.binding == nil {
return true
}
return b.clientRoute.binding == route.binding
}
func (b *bulkHandle) acceptsTransportGeneration(transport *TransportConn) bool {
if b == nil {
return false
}
if b.transportGeneration == 0 || transport == nil {
generation := b.TransportGeneration()
if generation == 0 || transport == nil {
return true
}
return b.transportGeneration == transport.TransportGeneration()
return generation == transport.TransportGeneration()
}
func (b *bulkHandle) acceptsCurrentTransport() bool {
if b == nil {
return false
}
if b.transport != nil {
return b.transport.IsCurrent()
}
if b.client != nil {
return b.client.clientSessionRouteCurrent(b.clientSessionRouteSnapshot())
}
return true
}
func (b *bulkHandle) acceptDispatchAllowed() bool {
if b == nil || b.acceptState.Load() == 2 {
return false
}
if err := b.resetErrSnapshot(); err != nil {
return false
}
return b.acceptsCurrentTransport()
}
func (b *bulkHandle) dataIDSnapshot() uint64 {
@@ -1429,6 +1575,7 @@ func (b *bulkHandle) markReset(err error) {
if b == nil {
return
}
b.acceptState.CompareAndSwap(0, 2)
b.applyResetState(bulkResetError(err))
b.finalize()
}
@@ -1631,6 +1778,73 @@ func (b *bulkHandle) takePendingWindowRelease() (int64, int, bulkReleaseSender)
return bytes, chunks, release
}
func (b *bulkHandle) restorePendingWindowRelease(bytes int64, chunks int) {
if b == nil || (bytes <= 0 && chunks <= 0) {
return
}
b.mu.Lock()
b.pendingReleaseBytes += bytes
b.pendingReleaseChunks += chunks
b.mu.Unlock()
}
func (b *bulkHandle) waitWindowReleaseRetry() bool {
if b == nil {
return false
}
timer := time.NewTimer(bulkWindowReleaseRetryDelay)
defer timer.Stop()
select {
case <-timer.C:
return true
case <-b.Context().Done():
return false
}
}
func (b *bulkHandle) shouldResetAfterWindowReleaseFailure() bool {
if b == nil {
return false
}
b.mu.Lock()
defer b.mu.Unlock()
return b.resetErr == nil && !b.remoteClosed && !b.peerReadClosed && !b.localReadClosed
}
func (b *bulkHandle) windowReleaseClosing() bool {
if b == nil {
return false
}
b.mu.Lock()
defer b.mu.Unlock()
return b.localClosed || b.remoteClosed || b.peerReadClosed || b.localReadClosed
}
func (b *bulkHandle) waitWindowReleaseShutdown() bool {
if b == nil || !b.windowReleaseClosing() {
return false
}
timer := time.NewTimer(bulkWindowReleaseShutdownGrace)
defer timer.Stop()
select {
case <-b.Context().Done():
return true
case <-timer.C:
return !b.shouldResetAfterWindowReleaseFailure()
}
}
func isBulkWindowReleaseClosedError(err error) bool {
if err == nil {
return false
}
if errors.Is(err, io.ErrClosedPipe) || errors.Is(err, net.ErrClosed) {
return true
}
message := strings.ToLower(err.Error())
return strings.Contains(message, "closed pipe") || strings.Contains(message, "closed network connection")
}
func (b *bulkHandle) runWindowReleaseLoop() {
if b == nil {
return
@@ -1647,12 +1861,61 @@ func (b *bulkHandle) runWindowReleaseLoop() {
if release == nil || (bytes <= 0 && chunks <= 0) {
break
}
_ = release(b, bytes, chunks)
debug := b.debugEnabled()
var releaseStarted time.Time
if debug {
releaseStarted = time.Now()
b.debugf("release begin bytes=%d chunks=%d", bytes, chunks)
}
err := release(b, bytes, chunks)
if debug {
b.mu.Lock()
pendingBytes, pendingChunks := b.pendingReleaseBytes, b.pendingReleaseChunks
b.mu.Unlock()
b.debugf("release end bytes=%d chunks=%d elapsed=%s pending-bytes=%d pending-chunks=%d error=%v", bytes, chunks, time.Since(releaseStarted), pendingBytes, pendingChunks, err)
}
if err != nil {
b.restorePendingWindowRelease(bytes, chunks)
if b.Context().Err() != nil {
return
}
if errors.Is(err, context.Canceled) || isTimeoutLikeError(err) {
if !b.waitWindowReleaseRetry() {
return
}
b.scheduleWindowRelease()
continue
}
if isBulkWindowReleaseClosedError(err) && b.waitWindowReleaseShutdown() {
return
}
if b.shouldResetAfterWindowReleaseFailure() {
b.markReset(err)
}
return
}
}
}
}
func (b *bulkHandle) acquireOutboundWindow(ctx context.Context, size int, chunks int) error {
func (b *bulkHandle) debugEnabled() bool {
if b == nil {
return false
}
if b.debug.Load() {
return true
}
if b.logical != nil && b.logical.server != nil {
return b.logical.server.IsDebugMode()
}
return false
}
func (b *bulkHandle) debugf(format string, args ...interface{}) {
fmt.Printf("[bulk-debug] at=%s side=%s id=%s data=%d age=%s %s\n", time.Now().Format(time.RFC3339Nano), b.debugSide, b.id, b.dataID, time.Since(b.createdAt), fmt.Sprintf(format, args...))
}
func (b *bulkHandle) acquireOutboundWindow(ctx context.Context, size int, chunks int) (retErr error) {
if b == nil || size <= 0 || !b.flowControlEnabled() {
return nil
}
@@ -1663,6 +1926,8 @@ func (b *bulkHandle) acquireOutboundWindow(ctx context.Context, size int, chunks
if chunks <= 0 {
chunks = 1
}
debug := b.debugEnabled()
var waitStarted time.Time
for {
b.mu.Lock()
if b.resetErr != nil {
@@ -1696,7 +1961,13 @@ func (b *bulkHandle) acquireOutboundWindow(ctx context.Context, size int, chunks
return nil
}
notify := b.flowNotify
avail, inFlight := b.outboundAvailBytes, b.outboundInFlight
b.mu.Unlock()
if debug && waitStarted.IsZero() {
waitStarted = time.Now()
b.debugf("window wait begin need=%d chunks=%d avail=%d inflight=%d", size, chunks, avail, inFlight)
defer func() { b.debugf("window wait end elapsed=%s error=%v", time.Since(waitStarted), retErr) }()
}
select {
case <-notify:
case <-ctx.Done():
@@ -1737,7 +2008,17 @@ func (b *bulkHandle) releaseOutboundWindow(bytes int64, chunks int) {
if b == nil || !b.flowControlEnabled() {
return
}
debug := b.debugEnabled()
var lockStarted time.Time
if debug {
lockStarted = time.Now()
}
b.mu.Lock()
var lockWait time.Duration
if debug {
lockWait = time.Since(lockStarted)
}
beforeBytes, beforeChunks := b.outboundAvailBytes, b.outboundInFlight
if b.windowBytes > 0 && bytes > 0 {
b.outboundAvailBytes += bytes
maxAvail := int64(b.windowBytes)
@@ -1752,7 +2033,11 @@ func (b *bulkHandle) releaseOutboundWindow(bytes int64, chunks int) {
}
}
b.notifyFlowLocked()
afterBytes, afterChunks := b.outboundAvailBytes, b.outboundInFlight
b.mu.Unlock()
if debug {
b.debugf("release received bytes=%d chunks=%d avail=%d->%d inflight=%d->%d lock-wait=%s", bytes, chunks, beforeBytes, afterBytes, beforeChunks, afterChunks, lockWait)
}
}
func (b *bulkHandle) bufferedChunkCountLocked() int {
@@ -1790,7 +2075,7 @@ func (b *bulkHandle) snapshot() BulkSnapshot {
snapshot := BulkSnapshot{
ID: b.id,
DataID: b.dataID,
FastPathVersion: normalizeBulkFastPathVersion(b.fastPathVersion),
FastPathVersion: b.fastPathVersionSnapshot(),
Scope: normalizeFileScope(b.runtimeScope),
Range: b.rangeSpec,
Metadata: cloneBulkMetadata(b.metadata),
@@ -1804,7 +2089,7 @@ func (b *bulkHandle) snapshot() BulkSnapshot {
DedicatedAttachLastCode: dedicatedLastCode,
DedicatedDataStarted: dedicatedDataStarted,
SessionEpoch: b.sessionEpoch,
TransportGeneration: b.transportGeneration,
TransportGeneration: b.TransportGeneration(),
LocalClosed: b.localClosed,
LocalReadClosed: b.localReadClosed,
RemoteClosed: b.remoteClosed,
@@ -1833,7 +2118,7 @@ func (b *bulkHandle) snapshot() BulkSnapshot {
var diag snapshotBindingDiagnostics
switch {
case b.logical != nil || b.transport != nil:
diag = snapshotBindingDiagnosticsFromLogical(b.logical, b.transport, b.transportGeneration)
diag = snapshotBindingDiagnosticsFromLogical(b.logical, b.transport, b.TransportGeneration())
case b.client != nil:
diag = snapshotBindingDiagnosticsFromClient(b.client, b.sessionEpoch)
}
@@ -1859,31 +2144,33 @@ func (b *bulkHandle) finalize() {
if b == nil {
return
}
b.markDedicatedAttachClosed()
b.maybeSendWindowRelease(0, true)
if b.cancel != nil {
b.cancel()
}
if b.writeCtxCancel != nil {
b.writeCtxCancel()
}
sender := b.clearDedicatedSender()
conn, owned := b.clearDedicatedConn()
if conn != nil && owned {
_ = conn.Close()
}
if sender != nil {
sender.stop()
}
if b.client != nil && b.releaseDedicatedActiveReserved() {
b.client.releaseBulkDedicatedActiveSlot()
}
if b.client != nil {
b.client.releaseBulkDedicatedLane(b.dedicatedLaneIDSnapshot())
}
if b.runtime != nil {
b.runtime.remove(b.runtimeScope, b.id)
}
b.finalizeOnce.Do(func() {
b.markDedicatedAttachClosed()
b.maybeSendWindowRelease(0, true)
if b.cancel != nil {
b.cancel()
}
if b.writeCtxCancel != nil {
b.writeCtxCancel()
}
sender := b.clearDedicatedSender()
conn, owned := b.clearDedicatedConn()
if conn != nil && owned {
_ = conn.Close()
}
if sender != nil {
sender.stop()
}
if b.client != nil && b.releaseDedicatedActiveReserved() {
b.client.releaseBulkDedicatedActiveSlot()
}
if b.client != nil && b.releaseDedicatedLaneReserved() {
b.client.releaseBulkDedicatedLaneAtRoute(b.dedicatedLaneIDSnapshot(), b.clientSessionRouteSnapshot())
}
if b.runtime != nil {
b.runtime.remove(b.runtimeScope, b)
}
})
}
func (b *bulkHandle) recordReadLocked(n int, now time.Time) {