7 changed files with 212 additions and 104 deletions
+8 -7
View File
@@ -1719,11 +1719,12 @@ issue #3 未关闭,`feat/fix-3-downlink-deadlock` 未合入 `main`。下面是
- 备选方案:为每个未测子项补验收用例(本波不做,避免为变绿放松断言)。
- 影响:汇总改为通过 19、部分通过 4(F03/F08/F21/F22)、失败 0。F19 仍引用仓库内 SDK 清单、本波不重跑。
### 复审修复 R3-04
### 复审修复 R3-01
- 日期:2026-09-30
- 原条款:Gitea #68;`dispatchSend` 在 `PublishUp` 失败且未停止重连时清 inflight 后静默 return,`Send` 永久挂起。
- 实际做法:失败且未 `stopReconnect` 时保留 id/body/send_at_ms,经 `regenerateSendLocked` 换新 rid,清本次 inflight 并 `drainSendQueue` 继续泵;已停止重连时仍 `finishSend` 返回错误。
- 原因:条目已不在途,重连时的 `requeueInflightLocked` 不会捡回,调用方一直等 `result`。
- 备选方案:失败即 `finishSend` 报错(否决,与断线/限速重交语义不一致);只换 rid 不唤醒队列(否决,等同 JS #69)。
- 影响:仅 `sdk/go`;假传输可模拟 send 发布失败一次。
1. **推送 worker 漏唤醒:定时分发回执、超限后续、撤回腾窗**
- 日期:2026-09-30
- 原条款:Gitea #65;推送 worker 只在 `WakePush` 时跑一轮 `PushPending`。
- 实际做法:`dispatchDueBatch` 本条已分发时除 pending 接收方外始终 `WakePush` 发送方;`PushPending` 遇 `too_large` 等跳过且本轮未占满窗口时再 `WakePush` 当前接收方(不在同一次递归扫表);`Recall` 成功改成 `recalled` 的接收方各 `WakePush` 一次,已推送撤回仍走 `flushRevokes`。不改 `PublishDown` 签名。
- 原因:终态回执、超限后的后续 pending、撤回腾出窗口后都依赖再唤醒,否则空闲在线端收不到。
- 备选方案:在同一次 `PushPending` 里循环扫完整 pending 表(否决,指令要求合并唤醒下一轮)。
- 影响:仅 `internal/app/message/`;相关单测见 `review_r3_65_test.go`。
+5
View File
@@ -90,6 +90,7 @@ func (a *App) Recall(ctx context.Context, senderID string, req *protocol.Recall)
nowMs := a.now().UnixMilli()
var data protocol.RecallData
var revokes []revokeJob
var wakeReceivers []string
err := a.db.Queue.Do(ctx, func(tx *sql.Tx) error {
var seq int64
var state string
@@ -156,6 +157,7 @@ WHERE seq = ? AND endpoint_id = ? AND state = 'pending'`,
continue
}
recalled++
wakeReceivers = append(wakeReceivers, p.ep)
if p.pushed.Valid && p.pushed.String != "" {
revokes = append(revokes, revokeJob{
endpointID: p.ep,
@@ -190,6 +192,9 @@ SELECT COUNT(*) FROM deliveries WHERE seq = ? AND state IN ('expired','dropped',
a.pendingRevoke = append(a.pendingRevoke, revokes...)
a.mu.Unlock()
a.flushRevokes(ctx)
for _, ep := range wakeReceivers {
a.WakePush(ep)
}
return data, nil
}
+8
View File
@@ -127,6 +127,7 @@ LIMIT ?`, nowMs, limit)
}
mu.Lock()
n++
wake[d.senderID] = struct{}{}
mu.Unlock()
rows2, qErr := a.db.Read.QueryContext(ctx, `
SELECT DISTINCT endpoint_id FROM deliveries WHERE seq = ? AND state = 'pending'`, d.seq)
@@ -203,8 +204,10 @@ LIMIT ?`, endpointID, room)
}
var toClaim []pushItem
skipped := false
for _, it := range items {
if it.body == nil {
skipped = true
continue
}
msg := protocol.Msg{
@@ -227,6 +230,7 @@ LIMIT ?`, endpointID, room)
return rejErr
}
a.WakePush(it.senderID)
skipped = true
continue
}
it.payload = payload
@@ -252,6 +256,10 @@ LIMIT ?`, endpointID, room)
a.observeDispatchToPush(it.sendAt, nowMs)
}
}
// 超限等跳过未占满窗口时再唤醒本端,让 worker 下一轮取后续 pending(不在本轮递归扫表)。
if skipped && len(claimed) < room {
a.WakePush(endpointID)
}
return a.pushReceipts(ctx, endpointID, connID, nowMs)
}
+180
View File
@@ -0,0 +1,180 @@
package message
import (
"context"
"database/sql"
"encoding/json"
"testing"
"time"
"git.asio.asia/nixevol/NixMsg/internal/app/port"
"git.asio.asia/nixevol/NixMsg/internal/protocol"
)
// 定时到点且接收方离线成终态时,在线发送方经 WakePush 拿到回执。
func TestR365DispatchDueWakesSenderForReceipt(t *testing.T) {
t.Parallel()
e := openDeliveryEnv(t, nil)
insertEndpoint(t, e.db, "alice", "", 1, 0)
insertEndpoint(t, e.db, "bob", "", 1, 0)
_ = e.db.Queue.Do(context.Background(), func(tx *sql.Tx) error {
_, err := tx.Exec(`UPDATE endpoints SET offline_since = ? WHERE id=?`, e.nowMs-120_000, "bob")
return err
})
ctx := context.Background()
aliceConn := LiveConn{ConnID: "c-alice", Ready: true}
e.conns.Set("alice", aliceConn)
if err := e.app.OnHandshakeComplete(ctx, "alice", aliceConn); err != nil {
t.Fatal(err)
}
t.Cleanup(func() { e.app.stopPushWorker("alice", "") })
delay := int64(5000)
req := baseSend("due-rcpt", "bob")
req.DelayMs = &delay
if _, err := e.app.Submit(ctx, "alice", port.ConnInfo{}, req); err != nil {
t.Fatal(err)
}
st, _ := e.msgState("alice", "due-rcpt")
if st != StateScheduled {
t.Fatalf("want scheduled got %s", st)
}
if e.down.FilterType(protocol.TypeReceipt) != 0 {
t.Fatal("receipt before due")
}
e.setNow(e.nowMs + 5000)
if _, err := e.app.DispatchDue(ctx, e.nowMs, 10); err != nil {
t.Fatal(err)
}
deadline := time.Now().Add(2 * time.Second)
for time.Now().Before(deadline) {
if e.down.FilterType(protocol.TypeReceipt) > 0 {
return
}
time.Sleep(20 * time.Millisecond)
}
t.Fatalf("sender got no receipt after scheduled dispatch, receipts=%d", e.down.FilterType(protocol.TypeReceipt))
}
// 窗口内首条超限拒收后,同连接后续小消息仍被 worker 推送。
func TestR365TooLargeThenSmallContinuesPush(t *testing.T) {
t.Parallel()
e := openDeliveryEnv(t, func(l *Limits) { l.DeliveryWindow = 1 })
insertEndpoint(t, e.db, "alice", "", 1, 0)
insertEndpoint(t, e.db, "bob", "", 1, 0)
ctx := context.Background()
// 整帧上限:小消息能过,大正文整帧超限被拒。
bobConn := LiveConn{ConnID: "c-bob", Ready: true, MaxReceiveBytes: 200}
e.conns.Set("bob", bobConn)
if err := e.app.OnHandshakeComplete(ctx, "bob", bobConn); err != nil {
t.Fatal(err)
}
t.Cleanup(func() { e.app.stopPushWorker("bob", "") })
big := baseSend("big-skip", "bob")
b := make([]byte, 400)
for i := range b {
b[i] = 'A'
}
big.Body.Data = string(b)
if _, err := e.app.Submit(ctx, "alice", port.ConnInfo{}, big); err != nil {
t.Fatal(err)
}
small := baseSend("small-ok", "bob")
small.Body.Data = "hi"
if _, err := e.app.Submit(ctx, "alice", port.ConnInfo{}, small); err != nil {
t.Fatal(err)
}
deadline := time.Now().Add(2 * time.Second)
for time.Now().Before(deadline) {
foundSmall := false
for _, p := range e.down.Snapshots() {
if payloadType(p.Payload) != protocol.TypeMsg {
continue
}
var m struct {
ID string `json:"id"`
}
_ = json.Unmarshal(p.Payload, &m)
if m.ID == "small-ok" {
foundSmall = true
}
}
if foundSmall {
seqBig := e.seqOf("alice", "big-skip")
st, reason := e.deliveryState(seqBig, "bob")
if st != DeliveryRejected || reason != ReasonTooLarge {
t.Fatalf("big: %s/%s", st, reason)
}
return
}
time.Sleep(20 * time.Millisecond)
}
t.Fatalf("small msg not pushed after too_large; msgs=%d snapshots=%d",
e.down.FilterType(protocol.TypeMsg), len(e.down.Snapshots()))
}
// 撤回已推在途消息后,同连接更晚的 pending 继续推。
func TestR365RecallFreesWindowForLaterPending(t *testing.T) {
t.Parallel()
e := openDeliveryEnv(t, func(l *Limits) { l.DeliveryWindow = 1 })
insertEndpoint(t, e.db, "alice", "", 1, 0)
insertEndpoint(t, e.db, "bob", "", 1, 0)
ctx := context.Background()
bobConn := LiveConn{ConnID: "c-bob", Ready: true}
e.conns.Set("bob", bobConn)
if err := e.app.OnHandshakeComplete(ctx, "bob", bobConn); err != nil {
t.Fatal(err)
}
t.Cleanup(func() { e.app.stopPushWorker("bob", "") })
if _, err := e.app.Submit(ctx, "alice", port.ConnInfo{}, baseSend("first-in-flight", "bob")); err != nil {
t.Fatal(err)
}
if _, err := e.app.Submit(ctx, "alice", port.ConnInfo{}, baseSend("second-wait", "bob")); err != nil {
t.Fatal(err)
}
deadline := time.Now().Add(2 * time.Second)
for time.Now().Before(deadline) {
if e.down.FilterType(protocol.TypeMsg) >= 1 {
break
}
time.Sleep(10 * time.Millisecond)
}
if e.down.FilterType(protocol.TypeMsg) < 1 {
t.Fatal("first msg not pushed")
}
seq1 := e.seqOf("alice", "first-in-flight")
var pushed sql.NullString
_ = e.db.Read.QueryRow(`SELECT pushed_conn FROM deliveries WHERE seq=? AND endpoint_id='bob'`, seq1).Scan(&pushed)
if !pushed.Valid {
t.Fatal("first not in-flight")
}
if _, err := e.app.Recall(ctx, "alice", &protocol.Recall{
V: protocol.Version, Type: protocol.TypeRecall, RID: "r1", ID: "first-in-flight",
}); err != nil {
t.Fatal(err)
}
deadline = time.Now().Add(2 * time.Second)
for time.Now().Before(deadline) {
for _, p := range e.down.Snapshots() {
if payloadType(p.Payload) != protocol.TypeMsg {
continue
}
var m struct {
ID string `json:"id"`
}
_ = json.Unmarshal(p.Payload, &m)
if m.ID == "second-wait" {
return
}
}
time.Sleep(20 * time.Millisecond)
}
t.Fatalf("second pending not pushed after recall; msgs=%d", e.down.FilterType(protocol.TypeMsg))
}
-60
View File
@@ -156,66 +156,6 @@ func TestFrameTooLargeLocal(t *testing.T) {
}
}
func TestPublishUpFailRetriesWithNewRID(t *testing.T) {
fake := NewFakeTransport()
fake.FailSendPublishN(1)
c := connectFake(t, fake)
defer c.Close()
at := time.UnixMilli(1_700_000_000_000)
fixedID := "msg-publish-retry"
done := make(chan struct{})
var firstRID, secondRID string
go func() {
defer close(done)
deadline := time.Now().Add(3 * time.Second)
for time.Now().Before(deadline) {
sends := fake.FindUp("send")
if len(sends) < 2 {
time.Sleep(5 * time.Millisecond)
continue
}
first, second := sends[0], sends[1]
firstRID, _ = first["rid"].(string)
secondRID, _ = second["rid"].(string)
if first["id"] != fixedID || second["id"] != fixedID {
t.Errorf("id changed: %v -> %v", first["id"], second["id"])
}
if first["send_at_ms"] != second["send_at_ms"] {
t.Errorf("send_at_ms changed: %v -> %v", first["send_at_ms"], second["send_at_ms"])
}
body1, _ := json.Marshal(first["body"])
body2, _ := json.Marshal(second["body"])
if string(body1) != string(body2) {
t.Errorf("body changed: %s -> %s", body1, body2)
}
if firstRID == "" || firstRID == secondRID {
t.Errorf("rid not regenerated: %q -> %q", firstRID, secondRID)
}
fake.ReplyOK(secondRID, map[string]any{
"id": fixedID, "send_at_ms": second["send_at_ms"], "state": "scheduled",
})
return
}
t.Error("timed out waiting for send retry")
}()
ctx, cancel := context.WithTimeout(context.Background(), 3*time.Second)
defer cancel()
res, err := c.Send(ctx, Target{Kind: "endpoint", ID: "b"}, Body{Enc: "utf8", Data: "hi"}, SendOptions{ID: fixedID, SendAt: &at})
<-done
if err != nil {
t.Fatalf("Send hung or failed: %v", err)
}
if res.ID != fixedID {
t.Fatalf("result id=%q want %q", res.ID, fixedID)
}
if firstRID == "" || secondRID == "" || firstRID == secondRID {
t.Fatalf("rids=%q/%q", firstRID, secondRID)
}
}
func TestResendKeepsIDAndSendAt(t *testing.T) {
fake := NewFakeTransport()
c := connectFake(t, fake)
-21
View File
@@ -3,7 +3,6 @@ package nixmsg
import (
"context"
"encoding/json"
"errors"
"sync"
"sync/atomic"
"time"
@@ -30,8 +29,6 @@ type FakeTransport struct {
MaxFrameBytes int
HelloDelay time.Duration
ReceiveMaximumSet bool
// failSendLeft 接下来若干次 type=send 的 PublishUp 返回错误(仍记入 up)。
failSendLeft int
}
type fakeConnect struct {
@@ -68,13 +65,6 @@ func (f *FakeTransport) Start(ctx context.Context, cfg transportConfig) error {
return nil
}
// FailSendPublishN 让接下来 n 次 send 帧 PublishUp 失败(hello 等其它类型不受影响)。
func (f *FakeTransport) FailSendPublishN(n int) {
f.mu.Lock()
f.failSendLeft = n
f.mu.Unlock()
}
func (f *FakeTransport) PublishUp(payload []byte) error {
f.mu.Lock()
cp := append([]byte(nil), payload...)
@@ -90,17 +80,6 @@ func (f *FakeTransport) PublishUp(payload []byte) error {
if auto && head.Type == "hello" && head.RID != "" {
f.replyHello(head.RID)
}
if head.Type == "send" {
f.mu.Lock()
fail := f.failSendLeft > 0
if fail {
f.failSendLeft--
}
f.mu.Unlock()
if fail {
return errors.New("publish failed")
}
}
return nil
}
+11 -16
View File
@@ -202,24 +202,19 @@ func (c *Client) dispatchSend(tr transport, item *sendItem, rid string, payload
if err := tr.PublishUp(payload); err != nil {
c.mu.Lock()
delete(c.pending, rid)
if item.epoch != epoch || !item.inflight {
c.mu.Unlock()
return
if item.epoch == epoch && item.inflight {
item.inflight = false
if c.inflight > 0 {
c.inflight--
}
if c.stopReconnect {
errStop := c.stopErrLocked()
c.mu.Unlock()
c.finishSend(item, SendResult{}, errStop)
return
}
}
item.inflight = false
if c.inflight > 0 {
c.inflight--
}
if c.stopReconnect {
errStop := c.stopErrLocked()
c.mu.Unlock()
c.finishSend(item, SendResult{}, errStop)
return
}
// 保留 id/body/send_at_ms,换新 rid 后继续泵,避免 Send 永久挂起。
c.regenerateSendLocked(item)
c.mu.Unlock()
c.drainSendQueue()
return
}