6 changed files with 210 additions and 164 deletions
+8 -8
View File
@@ -1719,12 +1719,12 @@ issue #3 未关闭,`feat/fix-3-downlink-deadlock` 未合入 `main`。下面是
- 备选方案:为每个未测子项补验收用例(本波不做,避免为变绿放松断言)。
- 影响:汇总改为通过 19、部分通过 4(F03/F08/F21/F22)、失败 0。F19 仍引用仓库内 SDK 清单、本波不重跑。
### ¸´ÉóÐÞ¸´ R3-03
### 复审修复 R3-01
1. **д¶ÓÁÐÂúʱ Close ²»ÔÙÓë enqueue ËÀËø**
- ÈÕÆÚ£º2026-09-30
- Ô­Ìõ¿î£ºissue #67£»Í£»úÐè Drain/Close ÔÚ³¬Ê±ÄÚ·µ»Ø£»ÒÑÈë¶ÓÈÎÎñÈÔÓÉд goroutine ´¦ÀíÍê¡£
- ʵ¼Ê×ö·¨£ºQueue Ôö¼Ó closing Ðźš£enqueue ÔÚ³Ö sendMu.RLock ʱ select ͬʱµÈ´ý q.ch¡¢closing Óë ctx.Done()£¬Í¨µÀÂúʱ²»ÔÙÎÞÏÞ×èÈû¡£Close ÏÈ Swap(closed) ²¢ close(closing) »½ÐÑÔÚ;·¢ËÍ·½ÊͷŶÁËø£¬ÔÙÔÚÐ´ËøÄÚ close(q.ch)£¬×îºóµÈ loop Í˳ö¡£²»ÏòÒÑ¹Ø±Õ channel ·¢ËÍ¡£
- Ô­Òò£ºÍ¨µÀÂúʱ²¢·¢ Do Õ¼×ŶÁËø¶ÂÔÚ·¢ËÍÉÏ£¬Close µÄÐ´ËøÄò»µ½£¬Í£»ú¿¨ËÀ£»Î´Ìá½»µÄÔÚ;дҲ»á¶ª¡£
- ±¸Ñ¡·½°¸£ºÈë¶Ó¸ÄΪ·Ç×èÈû£¬ÂúÔòÁ¢¼´ ErrBusy£¨·ñ¾ö£º¸Ä±ä±³Ñ¹ÓïÒ壬Õý³£¸ß·å»áÎóÉËÌá½»£©¡£
- Ó°Ï죺½ö internal/store/queue.go Óë²âÊÔ£»²»¸ÄÇ¨ÒÆ±àºÅ¡£
1. **推送 worker 漏唤醒:定时分发回执、超限后续、撤回腾窗**
- 日期:2026-09-30
- 原条款:Gitea #65;推送 worker 只在 `WakePush` 时跑一轮 `PushPending`。
- 实际做法:`dispatchDueBatch` 本条已分发时除 pending 接收方外始终 `WakePush` 发送方;`PushPending` 遇 `too_large` 等跳过且本轮未占满窗口时再 `WakePush` 当前接收方(不在同一次递归扫表);`Recall` 成功改成 `recalled` 的接收方各 `WakePush` 一次,已推送撤回仍走 `flushRevokes`。不改 `PublishDown` 签名。
- 原因:终态回执、超限后的后续 pending、撤回腾出窗口后都依赖再唤醒,否则空闲在线端收不到。
- 备选方案:在同一次 `PushPending` 里循环扫完整 pending 表(否决,指令要求合并唤醒下一轮)。
- 影响:仅 `internal/app/message/`;相关单测见 `review_r3_65_test.go`。
+5
View File
@@ -90,6 +90,7 @@ func (a *App) Recall(ctx context.Context, senderID string, req *protocol.Recall)
nowMs := a.now().UnixMilli()
var data protocol.RecallData
var revokes []revokeJob
var wakeReceivers []string
err := a.db.Queue.Do(ctx, func(tx *sql.Tx) error {
var seq int64
var state string
@@ -156,6 +157,7 @@ WHERE seq = ? AND endpoint_id = ? AND state = 'pending'`,
continue
}
recalled++
wakeReceivers = append(wakeReceivers, p.ep)
if p.pushed.Valid && p.pushed.String != "" {
revokes = append(revokes, revokeJob{
endpointID: p.ep,
@@ -190,6 +192,9 @@ SELECT COUNT(*) FROM deliveries WHERE seq = ? AND state IN ('expired','dropped',
a.pendingRevoke = append(a.pendingRevoke, revokes...)
a.mu.Unlock()
a.flushRevokes(ctx)
for _, ep := range wakeReceivers {
a.WakePush(ep)
}
return data, nil
}
+8
View File
@@ -127,6 +127,7 @@ LIMIT ?`, nowMs, limit)
}
mu.Lock()
n++
wake[d.senderID] = struct{}{}
mu.Unlock()
rows2, qErr := a.db.Read.QueryContext(ctx, `
SELECT DISTINCT endpoint_id FROM deliveries WHERE seq = ? AND state = 'pending'`, d.seq)
@@ -203,8 +204,10 @@ LIMIT ?`, endpointID, room)
}
var toClaim []pushItem
skipped := false
for _, it := range items {
if it.body == nil {
skipped = true
continue
}
msg := protocol.Msg{
@@ -227,6 +230,7 @@ LIMIT ?`, endpointID, room)
return rejErr
}
a.WakePush(it.senderID)
skipped = true
continue
}
it.payload = payload
@@ -252,6 +256,10 @@ LIMIT ?`, endpointID, room)
a.observeDispatchToPush(it.sendAt, nowMs)
}
}
// 超限等跳过未占满窗口时再唤醒本端,让 worker 下一轮取后续 pending(不在本轮递归扫表)。
if skipped && len(claimed) < room {
a.WakePush(endpointID)
}
return a.pushReceipts(ctx, endpointID, connID, nowMs)
}
+180
View File
@@ -0,0 +1,180 @@
package message
import (
"context"
"database/sql"
"encoding/json"
"testing"
"time"
"git.asio.asia/nixevol/NixMsg/internal/app/port"
"git.asio.asia/nixevol/NixMsg/internal/protocol"
)
// 定时到点且接收方离线成终态时,在线发送方经 WakePush 拿到回执。
func TestR365DispatchDueWakesSenderForReceipt(t *testing.T) {
t.Parallel()
e := openDeliveryEnv(t, nil)
insertEndpoint(t, e.db, "alice", "", 1, 0)
insertEndpoint(t, e.db, "bob", "", 1, 0)
_ = e.db.Queue.Do(context.Background(), func(tx *sql.Tx) error {
_, err := tx.Exec(`UPDATE endpoints SET offline_since = ? WHERE id=?`, e.nowMs-120_000, "bob")
return err
})
ctx := context.Background()
aliceConn := LiveConn{ConnID: "c-alice", Ready: true}
e.conns.Set("alice", aliceConn)
if err := e.app.OnHandshakeComplete(ctx, "alice", aliceConn); err != nil {
t.Fatal(err)
}
t.Cleanup(func() { e.app.stopPushWorker("alice", "") })
delay := int64(5000)
req := baseSend("due-rcpt", "bob")
req.DelayMs = &delay
if _, err := e.app.Submit(ctx, "alice", port.ConnInfo{}, req); err != nil {
t.Fatal(err)
}
st, _ := e.msgState("alice", "due-rcpt")
if st != StateScheduled {
t.Fatalf("want scheduled got %s", st)
}
if e.down.FilterType(protocol.TypeReceipt) != 0 {
t.Fatal("receipt before due")
}
e.setNow(e.nowMs + 5000)
if _, err := e.app.DispatchDue(ctx, e.nowMs, 10); err != nil {
t.Fatal(err)
}
deadline := time.Now().Add(2 * time.Second)
for time.Now().Before(deadline) {
if e.down.FilterType(protocol.TypeReceipt) > 0 {
return
}
time.Sleep(20 * time.Millisecond)
}
t.Fatalf("sender got no receipt after scheduled dispatch, receipts=%d", e.down.FilterType(protocol.TypeReceipt))
}
// 窗口内首条超限拒收后,同连接后续小消息仍被 worker 推送。
func TestR365TooLargeThenSmallContinuesPush(t *testing.T) {
t.Parallel()
e := openDeliveryEnv(t, func(l *Limits) { l.DeliveryWindow = 1 })
insertEndpoint(t, e.db, "alice", "", 1, 0)
insertEndpoint(t, e.db, "bob", "", 1, 0)
ctx := context.Background()
// 整帧上限:小消息能过,大正文整帧超限被拒。
bobConn := LiveConn{ConnID: "c-bob", Ready: true, MaxReceiveBytes: 200}
e.conns.Set("bob", bobConn)
if err := e.app.OnHandshakeComplete(ctx, "bob", bobConn); err != nil {
t.Fatal(err)
}
t.Cleanup(func() { e.app.stopPushWorker("bob", "") })
big := baseSend("big-skip", "bob")
b := make([]byte, 400)
for i := range b {
b[i] = 'A'
}
big.Body.Data = string(b)
if _, err := e.app.Submit(ctx, "alice", port.ConnInfo{}, big); err != nil {
t.Fatal(err)
}
small := baseSend("small-ok", "bob")
small.Body.Data = "hi"
if _, err := e.app.Submit(ctx, "alice", port.ConnInfo{}, small); err != nil {
t.Fatal(err)
}
deadline := time.Now().Add(2 * time.Second)
for time.Now().Before(deadline) {
foundSmall := false
for _, p := range e.down.Snapshots() {
if payloadType(p.Payload) != protocol.TypeMsg {
continue
}
var m struct {
ID string `json:"id"`
}
_ = json.Unmarshal(p.Payload, &m)
if m.ID == "small-ok" {
foundSmall = true
}
}
if foundSmall {
seqBig := e.seqOf("alice", "big-skip")
st, reason := e.deliveryState(seqBig, "bob")
if st != DeliveryRejected || reason != ReasonTooLarge {
t.Fatalf("big: %s/%s", st, reason)
}
return
}
time.Sleep(20 * time.Millisecond)
}
t.Fatalf("small msg not pushed after too_large; msgs=%d snapshots=%d",
e.down.FilterType(protocol.TypeMsg), len(e.down.Snapshots()))
}
// 撤回已推在途消息后,同连接更晚的 pending 继续推。
func TestR365RecallFreesWindowForLaterPending(t *testing.T) {
t.Parallel()
e := openDeliveryEnv(t, func(l *Limits) { l.DeliveryWindow = 1 })
insertEndpoint(t, e.db, "alice", "", 1, 0)
insertEndpoint(t, e.db, "bob", "", 1, 0)
ctx := context.Background()
bobConn := LiveConn{ConnID: "c-bob", Ready: true}
e.conns.Set("bob", bobConn)
if err := e.app.OnHandshakeComplete(ctx, "bob", bobConn); err != nil {
t.Fatal(err)
}
t.Cleanup(func() { e.app.stopPushWorker("bob", "") })
if _, err := e.app.Submit(ctx, "alice", port.ConnInfo{}, baseSend("first-in-flight", "bob")); err != nil {
t.Fatal(err)
}
if _, err := e.app.Submit(ctx, "alice", port.ConnInfo{}, baseSend("second-wait", "bob")); err != nil {
t.Fatal(err)
}
deadline := time.Now().Add(2 * time.Second)
for time.Now().Before(deadline) {
if e.down.FilterType(protocol.TypeMsg) >= 1 {
break
}
time.Sleep(10 * time.Millisecond)
}
if e.down.FilterType(protocol.TypeMsg) < 1 {
t.Fatal("first msg not pushed")
}
seq1 := e.seqOf("alice", "first-in-flight")
var pushed sql.NullString
_ = e.db.Read.QueryRow(`SELECT pushed_conn FROM deliveries WHERE seq=? AND endpoint_id='bob'`, seq1).Scan(&pushed)
if !pushed.Valid {
t.Fatal("first not in-flight")
}
if _, err := e.app.Recall(ctx, "alice", &protocol.Recall{
V: protocol.Version, Type: protocol.TypeRecall, RID: "r1", ID: "first-in-flight",
}); err != nil {
t.Fatal(err)
}
deadline = time.Now().Add(2 * time.Second)
for time.Now().Before(deadline) {
for _, p := range e.down.Snapshots() {
if payloadType(p.Payload) != protocol.TypeMsg {
continue
}
var m struct {
ID string `json:"id"`
}
_ = json.Unmarshal(p.Payload, &m)
if m.ID == "second-wait" {
return
}
}
time.Sleep(20 * time.Millisecond)
}
t.Fatalf("second pending not pushed after recall; msgs=%d", e.down.FilterType(protocol.TypeMsg))
}
+2 -15
View File
@@ -38,7 +38,6 @@ type Queue struct {
ch chan writeJob
done chan struct{}
closing chan struct{} // Close 时关闭,唤醒持读锁阻塞在发送上的 enqueue
closed atomic.Bool
sendMu sync.RWMutex
@@ -53,15 +52,10 @@ type Queue struct {
// NewQueue 创建合并写入队列并启动写 goroutine。
func NewQueue(db *sql.DB) *Queue {
return newQueue(db, queueBuffSize)
}
func newQueue(db *sql.DB, buffSize int) *Queue {
q := &Queue{
db: db,
ch: make(chan writeJob, buffSize),
ch: make(chan writeJob, queueBuffSize),
done: make(chan struct{}),
closing: make(chan struct{}),
ready: true,
}
go q.loop()
@@ -121,15 +115,10 @@ func (q *Queue) enqueue(job writeJob) error {
return ErrQueueClosed
}
q.addPending(1)
// 通道满时不得只堵在发送上持有读锁:Close 需要写锁关闭 q.ch。
select {
case q.ch <- job:
q.sendMu.RUnlock()
return nil
case <-q.closing:
q.addPending(-1)
q.sendMu.RUnlock()
return ErrQueueClosed
case <-job.ctx.Done():
q.addPending(-1)
q.sendMu.RUnlock()
@@ -387,13 +376,11 @@ func (q *Queue) Drain(ctx context.Context) error {
}
// Close 关闭队列:不再接受新任务,并等待写 goroutine 处理完已入队任务后退出。
// 先关闭 closing 唤醒因通道满而阻塞的发送方并释放读锁,再在无发送者时关闭数据通道,
// 避免向已关闭 channel 发送而 panic,也避免与持读锁的 enqueue 死锁。
// 在写锁内关闭数据通道,避免并发 Do 向已关闭 channel 发送而 panic。
func (q *Queue) Close() error {
if q.closed.Swap(true) {
return nil
}
close(q.closing)
q.sendMu.Lock()
close(q.ch)
q.sendMu.Unlock()
-134
View File
@@ -289,137 +289,3 @@ ON CONFLICT(key) DO UPDATE SET value = excluded.value, updated_at = excluded.upd
}
}
}
// openSmallQueue 用小缓冲队列替换默认队列,便于测满通道时的 Close/Drain。
func openSmallQueue(t *testing.T, buf int) *DB {
t.Helper()
dir := t.TempDir()
db, err := Open(dir, "FULL")
if err != nil {
t.Fatal(err)
}
if err := db.Queue.Close(); err != nil {
t.Fatal(err)
}
db.Queue = newQueue(db.Write, buf)
return db
}
func TestQueueCloseUnblocksFullChannel(t *testing.T) {
t.Parallel()
const buf = 4
db := openSmallQueue(t, buf)
defer func() { _ = db.Close() }()
q := db.Queue
hold := make(chan struct{})
blockerStarted := make(chan struct{})
blockerErr := make(chan error, 1)
go func() {
blockerErr <- q.Do(context.Background(), func(tx *sql.Tx) error {
close(blockerStarted)
<-hold
return nil
})
}()
<-blockerStarted
ctx := context.Background()
var fillWG sync.WaitGroup
for i := 0; i < buf; i++ {
fillWG.Add(1)
go func() {
defer fillWG.Done()
_ = q.Do(ctx, func(tx *sql.Tx) error { return nil })
}()
}
deadline := time.Now().Add(2 * time.Second)
for q.Len() < buf+1 && time.Now().Before(deadline) {
time.Sleep(2 * time.Millisecond)
}
if q.Len() < buf+1 {
t.Fatalf("channel not full: len=%d", q.Len())
}
blockedErr := make(chan error, 1)
go func() {
blockedErr <- q.Do(ctx, func(tx *sql.Tx) error { return nil })
}()
// 等额外 Do 堵在 enqueue(pending 超过通道容量+正在执行的一条)。
deadline = time.Now().Add(2 * time.Second)
for q.Len() < buf+2 && time.Now().Before(deadline) {
time.Sleep(2 * time.Millisecond)
}
closeDone := make(chan error, 1)
go func() { closeDone <- q.Close() }()
select {
case err := <-blockedErr:
if !errors.Is(err, ErrQueueClosed) {
t.Fatalf("blocked Do: %v", err)
}
case <-time.After(2 * time.Second):
t.Fatal("Close did not unblock full-channel enqueue")
}
close(hold)
select {
case err := <-closeDone:
if err != nil {
t.Fatal(err)
}
case <-time.After(2 * time.Second):
t.Fatal("Close hung after writer released")
}
<-blockerErr
fillWG.Wait()
}
func TestQueueDrainTimeoutWhileWriterBlocked(t *testing.T) {
t.Parallel()
const buf = 4
db := openSmallQueue(t, buf)
defer func() { _ = db.Close() }()
q := db.Queue
hold := make(chan struct{})
started := make(chan struct{})
go func() {
_ = q.Do(context.Background(), func(tx *sql.Tx) error {
close(started)
<-hold
return nil
})
}()
<-started
ctx := context.Background()
var wg sync.WaitGroup
for i := 0; i < buf; i++ {
wg.Add(1)
go func() {
defer wg.Done()
_ = q.Do(ctx, func(tx *sql.Tx) error { return nil })
}()
}
deadline := time.Now().Add(2 * time.Second)
for q.Len() < buf+1 && time.Now().Before(deadline) {
time.Sleep(2 * time.Millisecond)
}
drainCtx, cancel := context.WithTimeout(context.Background(), 80*time.Millisecond)
defer cancel()
start := time.Now()
err := q.Drain(drainCtx)
elapsed := time.Since(start)
if !errors.Is(err, context.DeadlineExceeded) {
t.Fatalf("Drain err=%v want deadline", err)
}
if elapsed > 500*time.Millisecond {
t.Fatalf("Drain took %s, should return on timeout", elapsed)
}
close(hold)
wg.Wait()
}