fix(notify): 修复传输生命周期竞态,完善背压与协议边界
- 完善 stream/bulk DataID 分配、预留和双向命名空间,修复并发打开及 dedicated/shared 回退时的 ID 冲突 - 将收发、回复、恢复任务和 sidecar 绑定原始会话与物理连接,防止重连后的旧消息误操作新连接 - 加强 close/reset 身份校验及实例移除检查,修复 dedicated attach 失败、通道引用和资源回收竞态 - 收紧批量发送器停止准入,确保在途入队完成后统一清理请求、缓冲区和等待者 - 修复 record 满队列死锁、取消时序号消耗及关闭竞态,确保关闭有界并返回真实错误 - 增加协商式 record 逻辑半关闭,保留反向 ACK;通过 reset 传递 RecordFailure,避免背压掩盖原始失败原因 - 补齐帧长度、批次数量、序号溢出和未确认窗口校验,提前拒绝超限数据并按字节预算拆批 - 为入站分发增加全局及单连接的条数、字节预算和阻塞背压,关闭时唤醒等待者,消除正常断连日志噪音 - 完善 bulk 窗口释放失败处理与传输诊断,补充并发、重连、背压、协议边界及真实 TCP 回归覆盖
This commit is contained in:
+110
-27
@@ -61,11 +61,14 @@ type bulkBatchSender struct {
|
||||
stopCh chan struct{}
|
||||
doneCh chan struct{}
|
||||
|
||||
stopOnce sync.Once
|
||||
flushMu sync.Mutex
|
||||
queued atomic.Int64
|
||||
errMu sync.Mutex
|
||||
err error
|
||||
stopOnce sync.Once
|
||||
admissionMu sync.Mutex
|
||||
admitting sync.WaitGroup
|
||||
admissionClosed bool
|
||||
flushMu sync.Mutex
|
||||
queued atomic.Int64
|
||||
errMu sync.Mutex
|
||||
err error
|
||||
}
|
||||
|
||||
func newBulkBatchSender(binding *transportBinding, codec bulkBatchCodec, writeTimeoutProvider func() time.Duration) *bulkBatchSender {
|
||||
@@ -193,20 +196,15 @@ func (s *bulkBatchSender) submitFramesOwned(ctx context.Context, frames []bulkFa
|
||||
}
|
||||
req = cloneQueuedBulkBatchRequest(req)
|
||||
s.queued.Add(1)
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
if !s.enqueue(req) {
|
||||
s.queued.Add(-1)
|
||||
if req.release != nil {
|
||||
req.release()
|
||||
}
|
||||
return normalizeStreamDeadlineError(ctx.Err())
|
||||
case <-s.stopCh:
|
||||
s.queued.Add(-1)
|
||||
if req.release != nil {
|
||||
req.release()
|
||||
if err := ctx.Err(); err != nil {
|
||||
return normalizeStreamDeadlineError(err)
|
||||
}
|
||||
return s.stoppedErr()
|
||||
case s.reqCh <- req:
|
||||
}
|
||||
select {
|
||||
case err := <-req.done:
|
||||
@@ -261,7 +259,11 @@ func (s *bulkBatchSender) tryDirectSubmit(req bulkBatchRequest) (bool, error) {
|
||||
}
|
||||
err := s.flush([]bulkBatchRequest{req})
|
||||
if err != nil {
|
||||
s.setErr(err)
|
||||
if isBatchSenderQueueWaitError(err) {
|
||||
return true, err
|
||||
}
|
||||
s.markFailed(err)
|
||||
s.waitAdmissions()
|
||||
s.failPending(err)
|
||||
return true, err
|
||||
}
|
||||
@@ -288,7 +290,10 @@ func (s *bulkBatchSender) run() {
|
||||
if timerCh == nil {
|
||||
select {
|
||||
case <-s.stopCh:
|
||||
s.failPending(s.stoppedErr())
|
||||
err := s.stoppedErr()
|
||||
s.waitAdmissions()
|
||||
s.failBatch(batch, err)
|
||||
s.failPending(err)
|
||||
return
|
||||
case next := <-s.reqCh:
|
||||
batch = append(batch, next)
|
||||
@@ -303,7 +308,10 @@ func (s *bulkBatchSender) run() {
|
||||
if timer != nil {
|
||||
timer.Stop()
|
||||
}
|
||||
s.failPending(s.stoppedErr())
|
||||
err := s.stoppedErr()
|
||||
s.waitAdmissions()
|
||||
s.failBatch(batch, err)
|
||||
s.failPending(err)
|
||||
return
|
||||
case next := <-s.reqCh:
|
||||
batch = append(batch, next)
|
||||
@@ -344,10 +352,17 @@ func (s *bulkBatchSender) run() {
|
||||
}
|
||||
s.flushMu.Unlock()
|
||||
if err != nil {
|
||||
s.setErr(err)
|
||||
if isBatchSenderQueueWaitError(err) {
|
||||
for _, item := range active {
|
||||
s.finishRequest(item, err)
|
||||
}
|
||||
continue
|
||||
}
|
||||
s.markFailed(err)
|
||||
for _, item := range active {
|
||||
s.finishRequest(item, err)
|
||||
}
|
||||
s.waitAdmissions()
|
||||
s.failPending(err)
|
||||
return
|
||||
}
|
||||
@@ -360,6 +375,7 @@ func (s *bulkBatchSender) run() {
|
||||
func (s *bulkBatchSender) nextRequest() (bulkBatchRequest, bool) {
|
||||
select {
|
||||
case <-s.stopCh:
|
||||
s.waitAdmissions()
|
||||
s.failPending(s.stoppedErr())
|
||||
return bulkBatchRequest{}, false
|
||||
case req := <-s.reqCh:
|
||||
@@ -433,6 +449,9 @@ func (s *bulkBatchSender) flush(requests []bulkBatchRequest) error {
|
||||
lockAcquired, err := s.binding.withConnWriteLockContextStopDeadlineManaged(context.Background(), s.stopCh, writeDeadline, func(conn net.Conn) error {
|
||||
return writeFramedPayloadBatchUnlocked(conn, queue, frames)
|
||||
})
|
||||
if !lockAcquired && isBatchSenderQueueWaitCause(err) {
|
||||
return newBatchSenderQueueWaitError(err)
|
||||
}
|
||||
s.binding.observeBulkAdaptivePayloadWrite(payloadBytes, time.Since(started), writeTimeout, err)
|
||||
if lockAcquired && err != nil {
|
||||
// A failed framed write may have emitted only part of a frame.
|
||||
@@ -454,6 +473,15 @@ func (s *bulkBatchSender) encodeRequests(requests []bulkBatchRequest) ([]bulkBat
|
||||
return nil, nil
|
||||
}
|
||||
payloads := make([]bulkBatchEncodedPayload, 0, len(requests))
|
||||
released := false
|
||||
defer func() {
|
||||
if released {
|
||||
return
|
||||
}
|
||||
for index := range payloads {
|
||||
payloads[index].done()
|
||||
}
|
||||
}()
|
||||
batch := make([]bulkFastFrame, 0, minInt(len(requests), bulkFastBatchMaxItems))
|
||||
mixedBatchLimit := s.sharedMixedPayloadLimit()
|
||||
batchRequestIndex := -1
|
||||
@@ -465,6 +493,9 @@ func (s *bulkBatchSender) encodeRequests(requests []bulkBatchRequest) ([]bulkBat
|
||||
}
|
||||
payload, release, err := s.encodeBatch(batch)
|
||||
if err != nil {
|
||||
if release != nil {
|
||||
release()
|
||||
}
|
||||
return err
|
||||
}
|
||||
payloads = append(payloads, bulkBatchEncodedPayload{payload: payload, release: release})
|
||||
@@ -479,12 +510,15 @@ func (s *bulkBatchSender) encodeRequests(requests []bulkBatchRequest) ([]bulkBat
|
||||
for _, frame := range req.frames {
|
||||
if !bulkFastPathSupportsSharedBatch(req.fastPathVersion) {
|
||||
if err := flushBatch(); err != nil {
|
||||
return nil, err
|
||||
return payloads, err
|
||||
}
|
||||
batchBytes = bulkFastBatchHeaderLen
|
||||
payload, release, err := s.encodeSingle(frame)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
if release != nil {
|
||||
release()
|
||||
}
|
||||
return payloads, err
|
||||
}
|
||||
payloads = append(payloads, bulkBatchEncodedPayload{payload: payload, release: release})
|
||||
continue
|
||||
@@ -492,12 +526,15 @@ func (s *bulkBatchSender) encodeRequests(requests []bulkBatchRequest) ([]bulkBat
|
||||
frameLen := bulkFastBatchFrameLen(frame)
|
||||
if frameLen+bulkFastBatchHeaderLen > bulkFastBatchMaxPlainBytes {
|
||||
if err := flushBatch(); err != nil {
|
||||
return nil, err
|
||||
return payloads, err
|
||||
}
|
||||
batchBytes = bulkFastBatchHeaderLen
|
||||
payload, release, err := s.encodeSingle(frame)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
if release != nil {
|
||||
release()
|
||||
}
|
||||
return payloads, err
|
||||
}
|
||||
payloads = append(payloads, bulkBatchEncodedPayload{payload: payload, release: release})
|
||||
continue
|
||||
@@ -512,7 +549,7 @@ func (s *bulkBatchSender) encodeRequests(requests []bulkBatchRequest) ([]bulkBat
|
||||
}
|
||||
if len(batch) > 0 && (len(batch) >= bulkFastBatchMaxItems || batchBytes+frameLen > batchLimit) {
|
||||
if err := flushBatch(); err != nil {
|
||||
return nil, err
|
||||
return payloads, err
|
||||
}
|
||||
batchBytes = bulkFastBatchHeaderLen
|
||||
nextMixed = false
|
||||
@@ -529,8 +566,9 @@ func (s *bulkBatchSender) encodeRequests(requests []bulkBatchRequest) ([]bulkBat
|
||||
}
|
||||
}
|
||||
if err := flushBatch(); err != nil {
|
||||
return nil, err
|
||||
return payloads, err
|
||||
}
|
||||
released = true
|
||||
return payloads, nil
|
||||
}
|
||||
|
||||
@@ -621,10 +659,8 @@ func (s *bulkBatchSender) stop() {
|
||||
if s == nil {
|
||||
return
|
||||
}
|
||||
s.stopOnce.Do(func() {
|
||||
s.setErr(errTransportDetached)
|
||||
close(s.stopCh)
|
||||
})
|
||||
s.markFailed(errTransportDetached)
|
||||
s.waitAdmissions()
|
||||
<-s.doneCh
|
||||
// Direct submissions flush on the caller goroutine rather than run(). Wait
|
||||
// for that path too before declaring the binding safe to hand off.
|
||||
@@ -632,6 +668,28 @@ func (s *bulkBatchSender) stop() {
|
||||
s.flushMu.Unlock()
|
||||
}
|
||||
|
||||
func (s *bulkBatchSender) enqueue(req bulkBatchRequest) bool {
|
||||
if s == nil {
|
||||
return false
|
||||
}
|
||||
s.admissionMu.Lock()
|
||||
if s.admissionClosed {
|
||||
s.admissionMu.Unlock()
|
||||
return false
|
||||
}
|
||||
s.admitting.Add(1)
|
||||
s.admissionMu.Unlock()
|
||||
defer s.admitting.Done()
|
||||
select {
|
||||
case <-req.ctx.Done():
|
||||
return false
|
||||
case <-s.stopCh:
|
||||
return false
|
||||
case s.reqCh <- req:
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
func (s *bulkBatchSender) failPending(err error) {
|
||||
for {
|
||||
select {
|
||||
@@ -643,6 +701,12 @@ func (s *bulkBatchSender) failPending(err error) {
|
||||
}
|
||||
}
|
||||
|
||||
func (s *bulkBatchSender) failBatch(batch []bulkBatchRequest, err error) {
|
||||
for _, item := range batch {
|
||||
s.finishRequest(item, err)
|
||||
}
|
||||
}
|
||||
|
||||
func (s *bulkBatchSender) finishRequest(req bulkBatchRequest, err error) {
|
||||
if s != nil {
|
||||
s.queued.Add(-1)
|
||||
@@ -664,6 +728,25 @@ func (s *bulkBatchSender) setErr(err error) {
|
||||
s.errMu.Unlock()
|
||||
}
|
||||
|
||||
func (s *bulkBatchSender) markFailed(err error) {
|
||||
if s == nil {
|
||||
return
|
||||
}
|
||||
s.setErr(err)
|
||||
s.stopOnce.Do(func() {
|
||||
s.admissionMu.Lock()
|
||||
s.admissionClosed = true
|
||||
close(s.stopCh)
|
||||
s.admissionMu.Unlock()
|
||||
})
|
||||
}
|
||||
|
||||
func (s *bulkBatchSender) waitAdmissions() {
|
||||
if s != nil {
|
||||
s.admitting.Wait()
|
||||
}
|
||||
}
|
||||
|
||||
func (s *bulkBatchSender) errSnapshot() error {
|
||||
if s == nil {
|
||||
return errTransportDetached
|
||||
|
||||
Reference in New Issue
Block a user