mirror of
https://github.com/luxfi/node.git
synced 2026-07-27 03:39:39 +00:00
fix(bootstrap): self-vote caught-up requires every connected beacon replied
RED CRITICAL: the frontier caught-up shortcut gated on CONNECTIVITY (fullyConnectedBeacons) while the tally keyed on REPLIES (CaughtUp over withSelfVote). The two diverge for a beacon that is CONNECTED but SILENT (TCP up, frontier reply delayed/withheld) — a capability strictly weaker than a full eclipse, and one that occurs naturally when an ahead beacon replays state on a mass co-restart. Such a beacon counts as fully connected yet is absent from the tally, so self's heavy weight backfills the naming floor and the node goes live at a forged STALE height. Add repliedCovers() set-containment guard: every connected beacon must have answered THIS frontier round before the self-vote shortcut can fire. A silent ahead-beacon now blocks caught-up -> node fails safe to FrontierConnecting (keeps waiting); a genuine fresh net (every connected beacon answers genesis) still completes. Keep RED's zz_red_probe_test.go as a permanent regression guard: connected-but-silent ahead beacon -> status=FrontierConnecting (was FrontierCaughtUp). Fresh-net + equal-stake self-vote tests still pass.
This commit is contained in:
@@ -202,6 +202,34 @@ func (b *blockHandler) withSelfVote(replies []BeaconReply, weights map[ids.NodeI
|
||||
return out
|
||||
}
|
||||
|
||||
// repliedCovers reports whether EVERY connected beacon answered THIS frontier
|
||||
// round — set containment {reply.NodeID} ⊇ connected, NOT a count (frontier
|
||||
// replies are not yet round/connected-filtered, so a stale/non-connected reply
|
||||
// could pad a length check). The self-vote caught-up shortcut requires this:
|
||||
// without it the gate keys on CONNECTIVITY (fullyConnectedBeacons) while CaughtUp
|
||||
// tallies REPLIES, and the two diverge for a beacon that is CONNECTED but SILENT
|
||||
// — TCP up, its frontier reply delayed or withheld. That is a capability strictly
|
||||
// WEAKER than a full eclipse (no TCP drop needed) and it occurs NATURALLY when an
|
||||
// ahead beacon replays state on a mass co-restart. Such a beacon counts as "fully
|
||||
// connected" yet is absent from the tally, so self's (heavy) weight backfills the
|
||||
// floor and the node goes live at a forged STALE height — the precise failure the
|
||||
// frontier-quorum gate exists to prevent (RED CRITICAL). Requiring every connected
|
||||
// beacon to have replied means a silent ahead-beacon blocks caught-up → the node
|
||||
// waits / fails safe, while a genuine fresh net (every connected beacon answers
|
||||
// genesis) still completes.
|
||||
func repliedCovers(replies []BeaconReply, connected []ids.NodeID) bool {
|
||||
replied := make(map[ids.NodeID]struct{}, len(replies))
|
||||
for _, r := range replies {
|
||||
replied[r.NodeID] = struct{}{}
|
||||
}
|
||||
for _, id := range connected {
|
||||
if _, ok := replied[id]; !ok {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// FrontierTip implements chainbootstrap.BlockSource. It returns a FrontierStatus that
|
||||
// decomplects the THREE reasons a tip may not be named — the fix for the mainnet canary
|
||||
// where a freshly-booted STALE node, asking the frontier BEFORE any beacon had connected,
|
||||
@@ -327,6 +355,7 @@ func (b *blockHandler) FrontierTip(ctx context.Context) (ids.ID, chainbootstrap.
|
||||
// naming) is untouched, and a genuinely behind node has an ahead peer in the full set → CaughtUp
|
||||
// is false → it keeps waiting/syncing.
|
||||
if haveLast && b.fullyConnectedBeacons(weights, connected) &&
|
||||
repliedCovers(replies, connected) &&
|
||||
policy.CaughtUp(b.withSelfVote(replies, weights, lastID), lastH, b.acceptedHeight) {
|
||||
b.logger.Debug("bootstrap frontier: CAUGHT UP at own tip (full beacon set + self all at/below it, peer-only weight below the naming floor)",
|
||||
log.Stringer("chainID", b.chainID),
|
||||
|
||||
@@ -0,0 +1,172 @@
|
||||
// Copyright (C) 2019-2026, Lux Industries, Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
// zz_red_probe_test.go — RED adversarial security-regression probe for the fresh-net self-vote
|
||||
// caught-up path. Demonstrates the connected-vs-replied divergence: the fix gates the self-vote
|
||||
// on fullyConnectedBeacons (CONNECTIVITY, from net.PeerInfo) but tallies CaughtUp over `replies`
|
||||
// (from collectFrontierReplies). A beacon that is CONNECTED but does NOT answer the frontier
|
||||
// query this round counts as "fully connected" yet contributes nothing to the caught-up tally —
|
||||
// and the self-vote backfills its missing weight, so a HEAVY validator self-completes caught-up
|
||||
// at a STALE height while an honest connected beacon is genuinely ahead. Blue's bsBeaconNet
|
||||
// cannot express this (its Send answers for EVERY connected beacon), so the regression slipped
|
||||
// through. These assertions encode the DESIRED safe behavior: they FAIL on the current code (the
|
||||
// break) and will PASS once the self-vote gate also requires every connected beacon to have
|
||||
// REPLIED this round.
|
||||
package chains
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
|
||||
"github.com/stretchr/testify/require"
|
||||
|
||||
chainbootstrap "github.com/luxfi/consensus/engine/chain/bootstrap"
|
||||
"github.com/luxfi/ids"
|
||||
"github.com/luxfi/math/set"
|
||||
"github.com/luxfi/node/message"
|
||||
"github.com/luxfi/node/network"
|
||||
"github.com/luxfi/node/network/peer"
|
||||
)
|
||||
|
||||
// redSilentNet reports FULL connectivity (every beacon in `connected` is returned by PeerInfo)
|
||||
// but a designated `silent` beacon — though connected — withholds its GetAcceptedFrontier reply
|
||||
// this round. This is the exact capability an on-path adversary has (keep a beacon's
|
||||
// TCP/handshake alive so it shows connected, while dropping/delaying its application-level
|
||||
// frontier response past the 3s window) AND a natural occurrence during a mass co-restart (an
|
||||
// ahead beacon replaying state answers the frontier query slowly).
|
||||
type redSilentNet struct {
|
||||
network.Network
|
||||
bh *blockHandler
|
||||
connected []ids.NodeID // all reported connected (the full set MINUS self)
|
||||
silent set.Set[ids.NodeID] // connected but withhold their frontier reply
|
||||
tipFor map[ids.NodeID]ids.ID // what each VOCAL beacon reports
|
||||
}
|
||||
|
||||
func (n *redSilentNet) PeerInfo(nodeIDs []ids.NodeID) []peer.Info {
|
||||
want := map[ids.NodeID]bool{}
|
||||
for _, id := range nodeIDs {
|
||||
want[id] = true
|
||||
}
|
||||
var out []peer.Info
|
||||
for _, b := range n.connected {
|
||||
if len(nodeIDs) == 0 || want[b] {
|
||||
out = append(out, peer.Info{ID: b, TrackedChains: set.Of(n.bh.networkID)})
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func (n *redSilentNet) Send(msg message.OutboundMessage, nodeIDs set.Set[ids.NodeID], _ ids.ID, _ uint32) set.Set[ids.NodeID] {
|
||||
m, ok := msg.(*bsOutMsg)
|
||||
if !ok || m.op != "frontier" {
|
||||
return nil
|
||||
}
|
||||
for id := range nodeIDs {
|
||||
if n.silent.Contains(id) {
|
||||
continue // CONNECTED, but withholds its frontier reply this round
|
||||
}
|
||||
if tip, ok := n.tipFor[id]; ok {
|
||||
n.bh.deliverBootstrapFrontier(id, tip)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// TestRED_PROBE_ConnectedButSilentAheadBeacon_SelfVoteFalseCompletesAtStale is a TWO-SIDED
|
||||
// CONTRAST: the ONLY difference between the two runs is whether the CONNECTED ahead-beacon B
|
||||
// delivers its frontier reply. B is fully connected in both runs, so fullyConnectedBeacons is
|
||||
// TRUE in both. Yet B-replies → safe (status != FrontierCaughtUp); B-connected-but-silent →
|
||||
// FrontierCaughtUp at the stale height (the break). This falsifies Blue's safety claim ("a
|
||||
// genuinely behind node has an ahead peer in the full set → CaughtUp is false → it keeps
|
||||
// waiting/syncing"): the gate keys on CONNECTIVITY while the caught-up decision keys on REPLIES,
|
||||
// and the two diverge.
|
||||
//
|
||||
// Why K is genuinely finalized-ahead (a real safety break, not a minority fork): a heavy
|
||||
// validator (self=60%) finalizes K WITH a minority (B=33%, so self+B=93% ≥ ⅔), then LOSES K to a
|
||||
// persistence lag (this codebase documents exactly this — the ZAP fire-and-forget Accept /
|
||||
// bsTestVM.frozenLastAccepted) and restarts STALE at M < K. B retained the finalized K. The node
|
||||
// MUST recover K; self-completing at M abandons finalized history and lets it build a conflicting
|
||||
// fork. The node cannot tell "M is the frontier" from "I lost finalized K" — which is precisely
|
||||
// why it must HEAR from connected B before concluding caught-up. self is NOT an independent
|
||||
// witness to its own caught-up-ness; the self-vote lets the node vouch for its own staleness.
|
||||
func TestRED_PROBE_ConnectedButSilentAheadBeacon_SelfVoteFalseCompletesAtStale(t *testing.T) {
|
||||
const N = 10 // K: the finalized-ahead height B retained (self voted it, then lost it)
|
||||
const M = 5 // the node's STALE accepted height after the persistence-lag crash
|
||||
|
||||
run := func(t *testing.T, bSilent bool) chainbootstrap.FrontierStatus {
|
||||
chain, _ := buildBSChain(N, -1)
|
||||
vm := newBSVMAt(chain, M) // node stale at M; it does NOT hold chain[N]=K
|
||||
|
||||
self := ids.GenerateTestNodeID()
|
||||
a := ids.GenerateTestNodeID() // co-stale light beacon, at M
|
||||
b := ids.GenerateTestNodeID() // AHEAD beacon, retained finalized K
|
||||
// HEAVY self (60 of 100): self > total/2 - 1, so peers alone (40) < the 51 stake-majority
|
||||
// floor → the self-vote branch. self+B=93% could finalize K; B=33% retains it.
|
||||
weights := map[ids.NodeID]uint64{self: 60, a: 7, b: 33}
|
||||
|
||||
bh, _ := newBSHandlerWeighted(t, vm, weights)
|
||||
bh.selfNodeID = self
|
||||
bh.msgCreator = bsMsgBuilder{}
|
||||
|
||||
// FULL connectivity in BOTH runs: A and B are connected (B is in `connected` either way).
|
||||
tipFor := map[ids.NodeID]ids.ID{a: chain[M].id} // A reports the stale tip M
|
||||
silent := set.NewSet[ids.NodeID](1)
|
||||
if bSilent {
|
||||
silent.Add(b) // B connected but withholds its frontier reply this round
|
||||
} else {
|
||||
tipFor[b] = chain[N].id // B replies its genuine ahead tip K
|
||||
}
|
||||
bh.net = &redSilentNet{bh: bh, connected: []ids.NodeID{a, b}, silent: silent, tipFor: tipFor}
|
||||
|
||||
bh.bsActive.Store(true)
|
||||
_, status := bh.FrontierTip(context.Background())
|
||||
bh.bsActive.Store(false)
|
||||
return status
|
||||
}
|
||||
|
||||
bReplies := run(t, false)
|
||||
bSilent := run(t, true)
|
||||
t.Logf("B replies its ahead tip → status=%v (3=FrontierConnecting, safe)", bReplies)
|
||||
t.Logf("B connected but SILENT → status=%v (5=FrontierCaughtUp, the BREAK)", bSilent)
|
||||
|
||||
// Sanity: when the ahead beacon REPLIES, the node correctly fails safe (does not conclude caught-up).
|
||||
require.NotEqual(t, chainbootstrap.FrontierCaughtUp, bReplies,
|
||||
"sanity: when the ahead beacon REPLIES, the node correctly does NOT conclude caught-up")
|
||||
|
||||
// THE SECURITY REGRESSION ASSERTION. B is fully CONNECTED in both runs. The node must NOT
|
||||
// self-complete caught-up while a connected beacon's position is unknown — that is a stale
|
||||
// go-live. FAILS today (the break); PASSES once the self-vote gate also requires every connected
|
||||
// beacon to have REPLIED this round (not merely be connected).
|
||||
require.NotEqual(t, chainbootstrap.FrontierCaughtUp, bSilent,
|
||||
"BREAK: suppressing only the CONNECTED ahead-beacon's frontier reply flips the heavy node to "+
|
||||
"FrontierCaughtUp at the STALE height — the self-vote backfills the floor and the "+
|
||||
"full-connectivity gate cannot see the reply suppression")
|
||||
}
|
||||
|
||||
// TestRED_PROBE_EqualStakeNeedsNoSelfVote answers deploy-question #5: 5 EQUAL-stake beacons, node
|
||||
// a beacon, all four peers connected and reporting a common tip — the node concludes caught-up via
|
||||
// the ORDINARY AcceptsFrontier path (peers clear the stake-majority floor: 4·w of 5·w = 80% >
|
||||
// 50%). The self-vote is NEVER needed for equal stake, so the equal-stake devnet hang is NOT this
|
||||
// self-exclusion floor (look at primaryNetworkReady / P-chain bootstrap / beacon connectivity).
|
||||
func TestRED_PROBE_EqualStakeNeedsNoSelfVote(t *testing.T) {
|
||||
chain, byID := buildBSChain(8, -1)
|
||||
vm := newBSVM(chain) // node at genesis (height 0)
|
||||
|
||||
self := ids.GenerateTestNodeID()
|
||||
p1, p2, p3, p4 := ids.GenerateTestNodeID(), ids.GenerateTestNodeID(), ids.GenerateTestNodeID(), ids.GenerateTestNodeID()
|
||||
weights := map[ids.NodeID]uint64{self: 100, p1: 100, p2: 100, p3: 100, p4: 100}
|
||||
|
||||
bh, chainID := newBSHandlerWeighted(t, vm, weights)
|
||||
bh.selfNodeID = self
|
||||
bh.msgCreator = bsMsgBuilder{}
|
||||
bh.net = &bsBeaconNet{bh: bh, chainID: chainID, connected: []ids.NodeID{p1, p2, p3, p4}, byID: byID, tip: chain[0]}
|
||||
|
||||
bh.bsActive.Store(true)
|
||||
tip, status := bh.FrontierTip(context.Background())
|
||||
bh.bsActive.Store(false)
|
||||
|
||||
t.Logf("equal-stake fresh net: status=%v tip=%v", status, tip)
|
||||
require.Contains(t, []chainbootstrap.FrontierStatus{chainbootstrap.FrontierNamed, chainbootstrap.FrontierCaughtUp}, status,
|
||||
"equal-stake peers clear the stake-majority floor unaided — no self-vote needed")
|
||||
require.Equal(t, chain[0].id, tip, "caught up at genesis")
|
||||
}
|
||||
Reference in New Issue
Block a user