mirror of
https://github.com/luxfi/node.git
synced 2026-07-29 08:36:26 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e708aaaf6b | ||
|
|
de53a2b23d | ||
|
|
dea4e94f51 | ||
|
|
be295b08b3 | ||
|
|
ebff9c751f | ||
|
|
1fba2feb35 | ||
|
|
579219e76e | ||
|
|
ac682665b3 | ||
|
|
ecc943f0ef | ||
|
|
ffd5f61993 | ||
|
|
031e400397 | ||
|
|
f7d5564a7a | ||
|
|
a84ea099fe | ||
|
|
8693ed4ba4 | ||
|
|
0d81038e82 | ||
|
|
f4e7236fac | ||
|
|
f97c880216 | ||
|
|
433099ce86 | ||
|
|
37b2a4a10c | ||
|
|
3c496a82b8 | ||
|
|
a268ab2b9e | ||
|
|
4f75480ba9 | ||
|
|
2ebaf260fa | ||
|
|
a37f0e1075 | ||
|
|
eda317ec0f | ||
|
|
ff034876d5 | ||
|
|
94f15375d0 | ||
|
|
8a8820b62f | ||
|
|
4c5314932b | ||
|
|
e4d0db372f | ||
|
|
a382f08d49 | ||
|
|
e54f2ec60c | ||
|
|
8fd7eccf7a | ||
|
|
d6f8f10f59 | ||
|
|
eaefec7a62 | ||
|
|
8437de9095 | ||
|
|
45681207fa | ||
|
|
8179f5a09e | ||
|
|
13714be91d | ||
|
|
a9eb2b0f43 | ||
|
|
91372ea3b9 | ||
|
|
f567af1fe3 | ||
|
|
277f162fb5 | ||
|
|
fbc6afd923 | ||
|
|
689f9fe5b8 | ||
|
|
02e7e12034 | ||
|
|
8f5578feb3 | ||
|
|
9666219074 | ||
|
|
47755174d9 | ||
|
|
f0576dcce7 | ||
|
|
3b94a71522 | ||
|
|
aa632eff8e | ||
|
|
ed531b8e1a | ||
|
|
2e568e3741 | ||
|
|
fdf86622d6 | ||
|
|
a86d80dd5e | ||
|
|
16b297a0fa | ||
|
|
34e738e65c | ||
|
|
a2f28eb92e | ||
|
|
c6314ed10e | ||
|
|
d8677aa48c | ||
|
|
5ca04a3f3d | ||
|
|
d1c9bcd868 | ||
|
|
c596c79212 | ||
|
|
e962558156 | ||
|
|
9ee97461cf | ||
|
|
17756e27b5 | ||
|
|
8284625fc6 | ||
|
|
7a4a99db49 | ||
|
|
1ea66944af | ||
|
|
51a460caff | ||
|
|
27ea0bb9ff | ||
|
|
76a3f2c3f9 | ||
|
|
845d990d5c | ||
|
|
ea34ffae2f | ||
|
|
9fe0ad265a | ||
|
|
c172000ae2 | ||
|
|
c0928d3c08 | ||
|
|
7675a09341 | ||
|
|
24f4cdf506 | ||
|
|
c0ffea88c0 | ||
|
|
4eed7a691a | ||
|
|
9699b6c706 | ||
|
|
ed3a7ddcc8 | ||
|
|
ccf0f65fc9 | ||
|
|
00726eb169 | ||
|
|
e037ef48fb | ||
|
|
caefd0de02 | ||
|
|
73def6a3ac | ||
|
|
89d899bd56 | ||
|
|
ba864e1d5e | ||
|
|
b16527eb8c | ||
|
|
908c358611 | ||
|
|
ef446d31ad | ||
|
|
ce2f9f2ae4 | ||
|
|
fde0770401 | ||
|
|
f553a9cd33 | ||
|
|
131d904caa | ||
|
|
47675cc9de | ||
|
|
078601bf8e | ||
|
|
15796b5054 | ||
|
|
f3fe8ef94b | ||
|
|
553fbf1717 | ||
|
|
e25a93b67c | ||
|
|
3ad00d482a | ||
|
|
2cef1e8db7 | ||
|
|
f2d5477e5d | ||
|
|
3802d6578c | ||
|
|
319d9a8994 | ||
|
|
9a147a2f92 | ||
|
|
75fd93551e | ||
|
|
1c55ab0521 | ||
|
|
0f6f500be0 | ||
|
|
9e36abdf5a | ||
|
|
fba9c6df9c | ||
|
|
d9845e35f6 | ||
|
|
e5881fd98d | ||
|
|
26de532757 | ||
|
|
43255f740f | ||
|
|
8b6fbdab37 | ||
|
|
a91f508f75 | ||
|
|
fe7da9616e | ||
|
|
97df98327e | ||
|
|
e5639d439a | ||
|
|
b1d9a38967 | ||
|
|
bd3ea46c2d | ||
|
|
76591875ba | ||
|
|
6e52c7e8fd | ||
|
|
4d7b421777 | ||
|
|
760252b25d | ||
|
|
b59fe61fc4 | ||
|
|
0d3e9ab4d6 | ||
|
|
26e67fd9f4 | ||
|
|
2ec16817f7 | ||
|
|
f6884dd3ec | ||
|
|
29cff375a8 | ||
|
|
5f1425cd22 | ||
|
|
99eafdf9ff | ||
|
|
3830479ca5 | ||
|
|
48b461caf6 | ||
|
|
5e42f34369 | ||
|
|
8b9ba00c78 | ||
|
|
8d9075cead | ||
|
|
00c8f31fe5 | ||
|
|
24ffb5e406 | ||
|
|
2c2c9188f0 | ||
|
|
981c2b234c | ||
|
|
6df9578947 | ||
|
|
1f1a152d00 | ||
|
|
e28eae435e | ||
|
|
656b1d42e6 | ||
|
|
871ce80a8a | ||
|
|
a5a73152b3 | ||
|
|
aa616374a9 | ||
|
|
73c23f9b06 | ||
|
|
465bd28952 | ||
|
|
36d7b6b5f4 | ||
|
|
75f8042102 | ||
|
|
0437a6c599 | ||
|
|
8716735960 | ||
|
|
a8c58da22a | ||
|
|
5d9d334957 | ||
|
|
78b4233b1a | ||
|
|
60c50a0e32 | ||
|
|
e40ac9d41f | ||
|
|
b6958135af | ||
|
|
94f7a06503 | ||
|
|
5e487305d8 | ||
|
|
0c6a7e6914 | ||
|
|
35c03c9a32 | ||
|
|
f27f48e248 | ||
|
|
86167a8cbd | ||
|
|
4e3c858b0e | ||
|
|
9bb87c9c26 | ||
|
|
9745896c57 | ||
|
|
25d624c385 | ||
|
|
23094ae9b0 | ||
|
|
fb38e10cea | ||
|
|
a48261c298 | ||
|
|
205b4013d2 | ||
|
|
017471e548 | ||
|
|
ed5bf5ccb3 | ||
|
|
913690cedd | ||
|
|
e16350cfcd | ||
|
|
f2ec4f2e25 | ||
|
|
653e9b2b7f | ||
|
|
04a461ca0a | ||
|
|
5d9e6637f5 | ||
|
|
71c98255bf | ||
|
|
5a755e4bec | ||
|
|
f6f909d22c | ||
|
|
6f6e3d01c0 | ||
|
|
d36cc83206 | ||
|
|
58426e1e77 | ||
|
|
f15b71db45 | ||
|
|
9af3dae914 | ||
|
|
b5d91f4f3e | ||
|
|
75d2ad7c26 | ||
|
|
96a77d9698 | ||
|
|
07b5a4028b | ||
|
|
59822392b6 | ||
|
|
bcd3141d9d | ||
|
|
3dd7a87cb1 | ||
|
|
81a752c36f | ||
|
|
04abac267a | ||
|
|
69756068fb | ||
|
|
74ddea71f1 | ||
|
|
c78d5a3ade | ||
|
|
3c780e5080 | ||
|
|
c686146c7f | ||
|
|
98476e7fbe | ||
|
|
b0313cb56e | ||
|
|
15ebcecf04 | ||
|
|
7ab2a404bd | ||
|
|
8cbf34331c | ||
|
|
c246f3508f | ||
|
|
8da3da09bc | ||
|
|
d22e87f68b | ||
|
|
38528099c8 | ||
|
|
82fc618afd | ||
|
|
69f9f23d2d | ||
|
|
51aa38bb33 | ||
|
|
fcbb8a4556 | ||
|
|
6470097f17 | ||
|
|
c2468897ef | ||
|
|
8314e36288 | ||
|
|
02f5ef1525 | ||
|
|
3278d481e6 | ||
|
|
794ccff8e4 | ||
|
|
fde31ee5a9 | ||
|
|
66f81ab795 | ||
|
|
27ca79a74f | ||
|
|
f1283a2ed9 | ||
|
|
5e0090337e | ||
|
|
81801ca236 | ||
|
|
6a8be2ee92 | ||
|
|
d062d03714 | ||
|
|
b2b786de63 | ||
|
|
858a03f26a | ||
|
|
6338041056 | ||
|
|
0462393602 | ||
|
|
be5c766848 | ||
|
|
b1d8e630dd | ||
|
|
9fc82dc0e5 | ||
|
|
a805551f57 | ||
|
|
3bebb93092 | ||
|
|
020209dc5f | ||
|
|
06729f6926 | ||
|
|
1032578ebf | ||
|
|
eb0e48c698 | ||
|
|
a1d40cca71 | ||
|
|
c1ecd70506 | ||
|
|
40f2e6cdc3 | ||
|
|
313fe35e44 | ||
|
|
24644701e0 | ||
|
|
3e7eeedb39 | ||
|
|
4d47176ce9 | ||
|
|
3b5885c018 | ||
|
|
a0ba39d6aa | ||
|
|
a705bcad6e | ||
|
|
e69ab3f50d | ||
|
|
a8d45db350 | ||
|
|
57b1c7fe42 | ||
|
|
9ebd686b11 | ||
|
|
e81c38896e | ||
|
|
70334e13bd | ||
|
|
c2e0c7dba1 | ||
|
|
adccfc73e6 | ||
|
|
6fb7b274bb | ||
|
|
797950eac2 | ||
|
|
fa932ebfad | ||
|
|
c970b553d2 | ||
|
|
a3358e9d7c | ||
|
|
dfd6525be5 | ||
|
|
7af65f4c69 | ||
|
|
c1dec826a5 | ||
|
|
3f02b2a1c8 | ||
|
|
11ae00f7dc | ||
|
|
4cf006a88e | ||
|
|
c668e7b9bb | ||
|
|
64f5eb03aa | ||
|
|
721af2509a | ||
|
|
920459ecb7 | ||
|
|
58b211deab | ||
|
|
d83ea482e2 | ||
|
|
2517148102 | ||
|
|
72e3df205e | ||
|
|
fd1b6b6871 | ||
|
|
a0636c8220 | ||
|
|
4a0049a40c | ||
|
|
b0b79a7fe4 | ||
|
|
217d3fb3d0 | ||
|
|
9d1409d66c | ||
|
|
3ceee1f38c | ||
|
|
85fde32f3c | ||
|
|
6896a69e20 | ||
|
|
3aa69e413f | ||
|
|
9d63838063 | ||
|
|
c139b4bc4e | ||
|
|
807dfaad99 | ||
|
|
a23d339687 | ||
|
|
aada7ee2f3 | ||
|
|
66a1d65316 | ||
|
|
04ae2f991e | ||
|
|
41de365f32 | ||
|
|
5862f71495 | ||
|
|
2726881e4e | ||
|
|
b9fa7b094f | ||
|
|
866f8b2c6a | ||
|
|
e772a7fc5f | ||
|
|
a45e8d7c5e | ||
|
|
2a645ccc92 | ||
|
|
90309f1221 | ||
|
|
7e1630761a | ||
|
|
285901e0c1 | ||
|
|
afc8c82e51 | ||
|
|
2b3614dd37 | ||
|
|
ca3ca97294 | ||
|
|
08a1137221 | ||
|
|
3d2ef49965 | ||
|
|
7d15753a40 | ||
|
|
63a7e33514 |
@@ -0,0 +1 @@
|
||||
# CI Status Check - 2025-09-23 23:30:58
|
||||
@@ -0,0 +1 @@
|
||||
# Triggering CI - Version 1.13.5 Ready
|
||||
@@ -1,9 +0,0 @@
|
||||
<svg xmlns="http://www.w3.org/2000/svg" width="1280" height="640" viewBox="0 0 1280 640" role="img" aria-label="node">
|
||||
<rect width="1280" height="640" fill="#0A0A0A"/>
|
||||
<svg x="96" y="215" width="210" height="210" viewBox="17 26 66 66"><path d="M50 88 L17.09 31 L82.91 31 Z" fill="#fff"/></svg>
|
||||
<text x="378" y="276" font-family="Inter,system-ui,-apple-system,sans-serif" font-size="78" font-weight="800" letter-spacing="-2" fill="#ffffff">node</text>
|
||||
<text x="378" y="322" font-family="Inter,system-ui,sans-serif" font-size="30" fill="#ffffff" opacity=".66">Lux blockchain node — multi-consensus, post-quantum ready</text>
|
||||
<rect x="378" y="338" width="806" height="3" rx="1.5" fill="#ffffff" opacity=".9"/>
|
||||
<text x="378" y="390" font-family="Inter,system-ui,sans-serif" font-size="24" font-weight="600" fill="#ffffff" opacity=".5">github.com/luxfi</text>
|
||||
<text x="1184" y="390" text-anchor="end" font-family="Inter,system-ui,sans-serif" font-size="24" font-weight="600" fill="#ffffff" opacity=".5">lux.network</text>
|
||||
</svg>
|
||||
|
Before Width: | Height: | Size: 1.0 KiB |
@@ -0,0 +1,34 @@
|
||||
name: Lint proto files
|
||||
|
||||
on:
|
||||
push:
|
||||
paths:
|
||||
- 'proto/**/*.proto'
|
||||
- '.github/workflows/buf-lint.yml'
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
buf-lint:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- name: Check for .proto files
|
||||
id: proto-check
|
||||
run: |
|
||||
if find proto -name '*.proto' -type f 2>/dev/null | grep -q .; then
|
||||
echo "has_proto=true" >> $GITHUB_OUTPUT
|
||||
else
|
||||
echo "has_proto=false" >> $GITHUB_OUTPUT
|
||||
echo "No .proto files found — skipping buf-lint (node is ZAP-native by default; protobuf is opt-in)."
|
||||
fi
|
||||
- uses: bufbuild/buf-setup-action@v1
|
||||
if: steps.proto-check.outputs.has_proto == 'true'
|
||||
with:
|
||||
github_token: ${{ github.token }}
|
||||
version: "1.47.2"
|
||||
- uses: bufbuild/buf-lint-action@v1
|
||||
if: steps.proto-check.outputs.has_proto == 'true'
|
||||
with:
|
||||
input: "proto"
|
||||
@@ -9,7 +9,7 @@ on:
|
||||
|
||||
jobs:
|
||||
push:
|
||||
runs-on: lux-build-amd64
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: bufbuild/buf-setup-action@v1.31.0
|
||||
|
||||
@@ -50,7 +50,7 @@ jobs:
|
||||
TIMEOUT: ${{ env.TIMEOUT }}
|
||||
CGO_ENABLED: '0'
|
||||
Fuzz:
|
||||
runs-on: lux-build-amd64
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
GOPRIVATE: github.com/luxfi/*
|
||||
GONOSUMDB: github.com/luxfi/*
|
||||
@@ -71,7 +71,7 @@ jobs:
|
||||
# e2e_pre_etna, e2e_post_etna, e2e_existing_network, Upgrade
|
||||
# These will be re-enabled once tmpnet is properly configured
|
||||
Lint:
|
||||
runs-on: lux-build-amd64
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
GOPRIVATE: github.com/luxfi/*
|
||||
GONOSUMDB: github.com/luxfi/*
|
||||
@@ -94,9 +94,56 @@ jobs:
|
||||
- name: Run actionlint
|
||||
shell: bash
|
||||
run: scripts/actionlint.sh
|
||||
buf-lint:
|
||||
name: Protobuf Lint
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- name: Install buf
|
||||
shell: bash
|
||||
run: |
|
||||
BUF_VERSION="1.47.2"
|
||||
curl -sSL "https://github.com/bufbuild/buf/releases/download/v${BUF_VERSION}/buf-Linux-x86_64" -o /usr/local/bin/buf
|
||||
chmod +x /usr/local/bin/buf
|
||||
buf --version
|
||||
- name: Lint protobuf
|
||||
shell: bash
|
||||
run: buf lint proto
|
||||
check_generated_protobuf:
|
||||
name: Up-to-date protobuf
|
||||
runs-on: ubuntu-latest
|
||||
continue-on-error: true
|
||||
env:
|
||||
GOPRIVATE: github.com/luxfi/*
|
||||
GONOSUMDB: github.com/luxfi/*
|
||||
GOWORK: off
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: ./.github/actions/setup-go-for-project
|
||||
- name: Configure Git for private modules
|
||||
shell: bash
|
||||
run: git config --global url."https://${{ github.token }}@github.com/".insteadOf "https://github.com/"
|
||||
- name: Install buf
|
||||
shell: bash
|
||||
run: |
|
||||
BUF_VERSION="1.47.2"
|
||||
curl -sSL "https://github.com/bufbuild/buf/releases/download/v${BUF_VERSION}/buf-Linux-x86_64" -o /usr/local/bin/buf
|
||||
chmod +x /usr/local/bin/buf
|
||||
- name: Install protoc-gen-go tools
|
||||
shell: bash
|
||||
run: |
|
||||
go install google.golang.org/protobuf/cmd/protoc-gen-go@v1.35.1
|
||||
go install google.golang.org/grpc/cmd/protoc-gen-go-grpc@v1.3.0
|
||||
go install connectrpc.com/connect/cmd/protoc-gen-connect-go@latest
|
||||
- shell: bash
|
||||
run: scripts/protobuf_codegen.sh
|
||||
env:
|
||||
CGO_ENABLED: '0'
|
||||
- shell: bash
|
||||
run: .github/workflows/check-clean-branch.sh
|
||||
check_mockgen:
|
||||
name: Up-to-date mocks
|
||||
runs-on: lux-build-amd64
|
||||
runs-on: ubuntu-latest
|
||||
continue-on-error: true
|
||||
env:
|
||||
GOPRIVATE: github.com/luxfi/*
|
||||
@@ -116,7 +163,7 @@ jobs:
|
||||
run: .github/workflows/check-clean-branch.sh
|
||||
go_mod_tidy:
|
||||
name: Up-to-date go.mod and go.sum
|
||||
runs-on: lux-build-amd64
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
GOPRIVATE: github.com/luxfi/*
|
||||
GONOSUMDB: github.com/luxfi/*
|
||||
@@ -133,7 +180,7 @@ jobs:
|
||||
run: .github/workflows/check-clean-branch.sh
|
||||
test_build_image:
|
||||
name: Image build
|
||||
runs-on: lux-build-amd64
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- name: Set up Docker Buildx
|
||||
|
||||
@@ -24,7 +24,7 @@ on:
|
||||
jobs:
|
||||
analyze:
|
||||
name: Analyze
|
||||
runs-on: lux-build-amd64
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
actions: read
|
||||
contents: read
|
||||
|
||||
@@ -2,11 +2,6 @@ name: Docker
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
tag:
|
||||
description: 'existing version tag to (re)build as a multi-arch manifest, e.g. v1.32.1'
|
||||
required: false
|
||||
default: ''
|
||||
push:
|
||||
tags: ['v*']
|
||||
|
||||
@@ -16,27 +11,16 @@ permissions:
|
||||
id-token: write
|
||||
|
||||
jobs:
|
||||
build:
|
||||
build-amd64:
|
||||
# In-cluster ARC pool (lux-build autoscalingrunnerset in lux-k8s, amd64
|
||||
# DOKS nodes, DinD sidecar). Replaces the offline evo classic runner.
|
||||
# ARC matches on the scale-set name, NOT classic [self-hosted,linux,amd64]
|
||||
# labels — the org runner group + arcd repo allowlist enforce isolation.
|
||||
# arm64 is produced by Go cross-compile (CGO_ENABLED=0, Dockerfile
|
||||
# TARGETARCH path) on the amd64 runner — no QEMU emulation of the build.
|
||||
runs-on: lux-build
|
||||
outputs:
|
||||
digest: ${{ steps.build.outputs.digest }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
# On workflow_dispatch with an explicit `tag`, (re)build that tag
|
||||
# as a multi-arch manifest; otherwise build the pushed ref.
|
||||
ref: ${{ github.event.inputs.tag || github.ref }}
|
||||
|
||||
- name: Set up QEMU
|
||||
uses: docker/setup-qemu-action@v3
|
||||
with:
|
||||
platforms: arm64
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
@@ -66,7 +50,6 @@ jobs:
|
||||
latest=false
|
||||
tags: |
|
||||
type=ref,event=tag
|
||||
type=raw,value=${{ github.event.inputs.tag }},enable=${{ github.event.inputs.tag != '' }}
|
||||
type=sha,format=short,prefix=sha-
|
||||
|
||||
- name: Resolve cross-repo PAT for private luxfi/* deps
|
||||
@@ -102,13 +85,13 @@ jobs:
|
||||
echo "$tok" >> "$GITHUB_OUTPUT"
|
||||
echo "EOF" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Build & push (multi-arch)
|
||||
- name: Build & push (amd64)
|
||||
id: build
|
||||
uses: docker/build-push-action@v6
|
||||
with:
|
||||
context: .
|
||||
file: Dockerfile
|
||||
platforms: linux/amd64,linux/arm64
|
||||
platforms: linux/amd64
|
||||
push: true
|
||||
build-args: |
|
||||
CGO_ENABLED=0
|
||||
@@ -120,11 +103,11 @@ jobs:
|
||||
tags: ${{ steps.meta.outputs.tags }}
|
||||
labels: ${{ steps.meta.outputs.labels }}
|
||||
provenance: false
|
||||
cache-from: type=registry,ref=ghcr.io/luxfi/node:buildcache
|
||||
cache-to: type=registry,ref=ghcr.io/luxfi/node:buildcache,mode=max
|
||||
cache-from: type=registry,ref=ghcr.io/luxfi/node:buildcache-amd64
|
||||
cache-to: type=registry,ref=ghcr.io/luxfi/node:buildcache-amd64,mode=max
|
||||
|
||||
notify-universe:
|
||||
needs: build
|
||||
needs: build-amd64
|
||||
if: startsWith(github.ref, 'refs/tags/v')
|
||||
runs-on: lux-build
|
||||
steps:
|
||||
|
||||
@@ -9,7 +9,7 @@ permissions:
|
||||
|
||||
jobs:
|
||||
fuzz:
|
||||
runs-on: lux-build-amd64
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
GOPRIVATE: github.com/luxfi/*
|
||||
GONOSUMDB: github.com/luxfi/*
|
||||
|
||||
@@ -11,7 +11,7 @@ permissions:
|
||||
|
||||
jobs:
|
||||
MerkleDB:
|
||||
runs-on: lux-build-amd64
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
GOPRIVATE: github.com/luxfi/*
|
||||
GONOSUMDB: github.com/luxfi/*
|
||||
|
||||
@@ -16,7 +16,7 @@ jobs:
|
||||
permissions:
|
||||
contents: read
|
||||
issues: write
|
||||
runs-on: lux-build-amd64
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: crazy-max/ghaction-github-labeler@548a7c3603594ec17c819e1239f281a3b801ab4d #v6.0.0
|
||||
|
||||
@@ -4,7 +4,7 @@ on:
|
||||
- cron: '0 0 * * 0' # Run every day at midnight UTC on Sunday
|
||||
jobs:
|
||||
stale:
|
||||
runs-on: lux-build-amd64
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/stale@v10
|
||||
with:
|
||||
|
||||
@@ -15,7 +15,7 @@ env:
|
||||
|
||||
jobs:
|
||||
test-zapdb-replay:
|
||||
runs-on: lux-build-amd64
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v4
|
||||
@@ -87,7 +87,7 @@ jobs:
|
||||
--api-admin-enabled=true 2>&1 | grep -E "(genesis-db|Genesis)" || true
|
||||
|
||||
test-database-factory:
|
||||
runs-on: lux-build-amd64
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v4
|
||||
|
||||
@@ -5,22 +5,6 @@ All notable changes to this project will be documented in this file.
|
||||
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/),
|
||||
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
||||
|
||||
## [1.36.31]
|
||||
|
||||
### Fixed
|
||||
- **`/v1/health` reported a chain the node denies exists.** The `bootstrapped` check publishes `chains.Nets.Bootstrapping()` verbatim as its message, but `Nets.chains` is keyed by **net** ID and the aggregate appended the map **key**, not the chain. Every primary-network chain that failed to converge therefore surfaced as the single ID `11111111111111111111111111111111LpoYY` — `constants.PrimaryNetworkID` (`ids.Empty`) — so an operator resolving it got "there is no chain with alias/ID", and N stuck chains collapsed into one indistinguishable entry (measured on devnet/testnet: `"message":["11111111111111111111111111111111LpoYY"],"contiguousFailures":3213`). The net owns the bootstrapping set, so the net now names its own chains: new `nets.Net.Bootstrapping() []ids.ID` (`nets/net.go`), and `chains/chains.go` aggregates those instead of the keys. `TestNetsBootstrappingReportsChainsNotNets` asserts `ids.Empty` never appears and that two stuck chains are individually named; `TestNetsBootstrapping` no longer asserts the bug (it demanded the net ID).
|
||||
- Ships the GET `/v1/health` encoder fix from `dada5a31` (`{"healthy":…,"error":"health reply encode failed"}` on every node — jsonv2 has no representation for `apihealth.Result.Duration`), which was on `main` but had never been tagged.
|
||||
|
||||
## [1.36.13]
|
||||
|
||||
### Fixed
|
||||
- **The durable rejoin fix (#66/#74) shipped INERT for the C-Chain — the exact chain whose 40h mainnet freeze motivated it.** `chains/manager.go` gated `expectsStakedBeacons` on `ids.IsNativeChain(chainParams.ID)` — the *blockchain* ID. But `ids.IsNativeChain` only matches the symbolic `111…C` alias form, and every deployed C/X/Q carries a **hash** blockchain ID (devnet C `21HieZng…`, mainnet C `2wRdZG…`); only the P-chain has a symbolic blockchain ID (`111…P`), and it is excluded as the platform chain. So `isNativeChain` was **always false** for the real C-Chain → `expectsStakedBeacons=false` → under the production `--skip-bootstrap=true` the beacon set was emptied → a behind C-Chain named its STALE local tip the frontier and never caught up (verified on devnet v1.36.12: C-Chain wedged at height 0 with "using empty beacons for single-node mode"). Fix: discriminate on the **validating Net** (`chainParams.ChainID == constants.PrimaryNetworkID`) via new `chainValidatesOnPrimaryNetwork`, which is `PrimaryNetworkID` for C/X/Q and each sovereign L1's own net ID for an L2 — so C/X/Q now keep their staked beacons under `--skip-bootstrap` (peer-sync a behind validator) while L2s keep the empty-beacon single-node path. Regression test `TestChainValidatesOnPrimaryNetwork_RealHashChainID` exercises the real discriminator with hash chain IDs (the prior hardcoded-bool tests could not catch this); `TestRED_EmptyStakedSetFailsSafe` still passes (forged-frontier gate intact).
|
||||
|
||||
## [1.36.12]
|
||||
|
||||
### Fixed
|
||||
- **EVM chains (C/D/L2) failed to initialize on v1.36.11 — bundled VM plugins were built against a stale `luxfi/api`.** The node pins `luxfi/api v1.0.16`, whose `InitializeResponse` carries the appended `Capabilities uint64` field (Quasar-export handshake, api commit `1f2dc5a`). The image baked the C-Chain EVM plugin from `luxfi/evm@v1.104.8` and the D-Chain dexvm plugin from `luxfi/dex@v1.5.15`, both of which resolve `luxfi/api v1.0.15` (no `Capabilities`). Their `InitializeResponse.Encode` writes a shorter payload than the node's strict `InitializeResponse.Decode` expects, so `vms/rpcchainvm/zap/client.go` fails every EVM VM handshake with `zap decode initialize response: unexpected EOF`. Native VMs (P/X/Q) were unaffected. Fix: bump `EVM_VERSION` `v1.104.8 → v1.104.9` (api v1.0.16) and force `luxfi/api@v1.0.16` in the dexvm build stage. The api bump is code-free for plugins (`luxfi/chains v1.7.4 → v1.7.5` adopted it with a go.mod-only diff; `CHAINS_REF=v1.7.6` already carries v1.0.16). No node source change beyond the version bump — this is v1.36.11's node binary (durable rejoin fix, commit `63f61429d1`) rebuilt with plugin↔node ZAP-wire alignment.
|
||||
|
||||
## [1.28.0]
|
||||
|
||||
### Added
|
||||
|
||||
+32
-90
@@ -124,27 +124,22 @@ RUN --mount=type=secret,id=ghtok,required=false \
|
||||
# linked into the C-Chain plugin via github.com/luxfi/chains/evm/cevm
|
||||
# and become the default execution backend (parallel + GPU EVM).
|
||||
#
|
||||
# The release assets live in the PRIVATE lux-private/cevm repo, so the fetch is
|
||||
# authenticated with the same `ghtok` secret as the private go modules and
|
||||
# lux-accel above. Best-effort, like lux-accel: only a build that asks for the
|
||||
# cevm backend (EVM_CGO=1 below) actually needs these.
|
||||
# CI/RELEASE GAP: the luxcpp/cevm release artifacts MUST publish per-arch
|
||||
# tarballs at the URL below for both linux-x86_64 and linux-arm64. Until
|
||||
# those tarballs exist, builds with CGO_ENABLED=1 will fail at this step and
|
||||
# operators must build with CGO_ENABLED=0 (pure-Go fallback).
|
||||
ARG CGO_ENABLED=1
|
||||
ARG CEVM_VERSION=v0.51.10
|
||||
ARG CEVM_REPO=lux-private/cevm
|
||||
RUN --mount=type=secret,id=ghtok,required=false \
|
||||
if [ "${CGO_ENABLED}" = "1" ]; then \
|
||||
ARG CEVM_VERSION=v0.19.0
|
||||
RUN if [ "${CGO_ENABLED}" = "1" ]; then \
|
||||
ARCH=$(echo ${TARGETPLATFORM} | cut -d / -f2) && \
|
||||
if [ "$ARCH" = "amd64" ]; then CEVM_ARCH="linux-x86_64"; else CEVM_ARCH="linux-arm64"; fi && \
|
||||
AUTH=""; [ -s /run/secrets/ghtok ] && AUTH="--header=Authorization: Bearer $(cat /run/secrets/ghtok)"; \
|
||||
( wget -q ${AUTH:+"$AUTH"} \
|
||||
"https://github.com/${CEVM_REPO}/releases/download/${CEVM_VERSION}/luxcpp-cevm-${CEVM_ARCH}.tar.gz" \
|
||||
-O /tmp/cevm.tar.gz \
|
||||
&& tar -xzf /tmp/cevm.tar.gz -C /usr/local \
|
||||
&& rm /tmp/cevm.tar.gz \
|
||||
&& ldconfig 2>/dev/null \
|
||||
) || echo "WARN: cevm ${CEVM_VERSION} fetch skipped (unreachable; needed only when EVM_CGO=1)" ; \
|
||||
wget -q "https://github.com/luxcpp/cevm/releases/download/${CEVM_VERSION}/luxcpp-cevm-${CEVM_ARCH}.tar.gz" \
|
||||
-O /tmp/cevm.tar.gz && \
|
||||
tar -xzf /tmp/cevm.tar.gz -C /usr/local && \
|
||||
rm /tmp/cevm.tar.gz && \
|
||||
ldconfig 2>/dev/null || true ; \
|
||||
else \
|
||||
echo "CGO_ENABLED=0: skipping cevm fetch (pure-Go fallback build)" ; \
|
||||
echo "CGO_ENABLED=0: skipping luxcpp/cevm fetch (pure-Go fallback build)" ; \
|
||||
fi
|
||||
|
||||
# Build node. CGO_ENABLED=1 (default) links luxcpp/cevm for parallel + GPU EVM.
|
||||
@@ -262,35 +257,7 @@ RUN . ./build_env.sh && \
|
||||
# failed ValidateState, and BRICKED the node. Proven on-node: real swap → kill -9 →
|
||||
# clean reboot, state intact. v1.99.40 = v1.99.39 + deps to latest. consensus v1.25.21 =
|
||||
# stake-weighted alpha-of-K quorum finality + per-height single-finalize + epoch-bound certs.
|
||||
# v1.104.9 bumps luxfi/api v1.0.15 -> v1.0.16 (indirect via luxfi/vm). v1.0.16
|
||||
# APPENDED InitializeResponse.Capabilities (uint64) for the Quasar-export
|
||||
# handshake (api commit 1f2dc5a). The node pins api v1.0.16 and DECODES that
|
||||
# field; an EVM plugin built at v1.104.8 (api v1.0.15) omits it, so the node's
|
||||
# strict struct decode hits "zap decode initialize response: unexpected EOF"
|
||||
# and EVERY EVM chain (C + L2s hanzo/zoo/pars/spc, all mgj786) fails to
|
||||
# initialize. The api bump is code-free for plugins (chains v1.7.4->v1.7.5
|
||||
# adopted it with a go.mod-only diff), so v1.104.9 = v1.104.8 + api alignment.
|
||||
# C-Chain execution backend. The default is the pure-Go EVM, which is what
|
||||
# every image has shipped so far. EVM_CGO=1 EVM_TAGS=cevm links luxcpp/cevm
|
||||
# instead — that pair is what makes AutoEVM resolve to CppEVM. The libraries
|
||||
# come from the cevm fetch in this same stage above.
|
||||
ARG EVM_CGO=0
|
||||
ARG EVM_TAGS=""
|
||||
|
||||
# v1.104.22 realigns the plugin with THIS node's own go.mod: api v1.0.16 ->
|
||||
# v1.1.1, vm v1.2.6 -> v1.3.1, geth v1.17.12 -> v1.20.1. Node main pins api
|
||||
# v1.1.1, so pinning an evm at api v1.0.16 reintroduces the exact
|
||||
# InitializeResponse decode mismatch described above, just in the opposite
|
||||
# direction — keep this ARG and node's go.mod on the same api/vm/geth line.
|
||||
#
|
||||
# v1.104.14 is also the FIRST evm tag carrying the C-Chain fee-split seam
|
||||
# (core/fee_split.go creditTxFee + extras.FeeSplitTimestamp/FeeRewardVault).
|
||||
# Below it, encoding/json silently DISCARDS the genesis "feeSplitTimestamp"
|
||||
# key: a chain configured for the 50/50 split instead routes 100% of every fee
|
||||
# to the block coinbase and the reward vault 0x0100..0002 stays 0 forever, with
|
||||
# nothing burned. The split stays dormant wherever feeSplitTimestamp is absent
|
||||
# (mainnet), so this bump is behaviour-preserving there.
|
||||
ARG EVM_VERSION=v1.104.22
|
||||
ARG EVM_VERSION=v1.99.40
|
||||
ARG EVM_VM_ID=mgj786NP7uDwBCcq6YwThhaN8FLyybkCa4zBWTQbNgmK6k9A6
|
||||
# the pinned evm go.mod may pin a dead luxfi/upgrade pseudo-version
|
||||
# (v1.0.1-0.20260603055252-f51810805436 — commit pruned from origin). Heal it to
|
||||
@@ -303,22 +270,15 @@ RUN --mount=type=cache,target=/root/.cache/go-build \
|
||||
cd /tmp/evm && \
|
||||
. /build/build_env.sh && \
|
||||
go mod edit -require=github.com/luxfi/upgrade@v1.0.1 && \
|
||||
# evm v1.99.52 = v1.99.51 + the idle-builder bounded-wake backstop
|
||||
# (startPendingTxPoll: a 500ms mempool re-poll so a tx after idle wakes the
|
||||
# builder within a bounded window — closes the lost-wakeup/subscribe-gap;
|
||||
# deterministic, wake-timing only). v1.99.51 = the block-production stall fix
|
||||
# (WaitForEvent builder-ready race + target-rate build pacing — un-freezes the
|
||||
# C-Chain that stalled at the imported frontier) on top of precompile v0.16.0
|
||||
# enable-everything
|
||||
# (wallet curves + standard precompiles enabled; fflonk fail-closed; accel
|
||||
# byte-identity; DEX big.Rat + determinism). Pin chains v1.4.8 (warp
|
||||
# consolidated to one luxfi/warp helper; graphvm genesis-last-accepted fix)
|
||||
# to match the chain-VM plugin stage (CHAINS_REF) below.
|
||||
go mod edit -require=github.com/luxfi/chains@v1.7.0 && \
|
||||
# evm v1.99.40 pins luxfi/chains v1.3.19, whose dexvm/registry was an incomplete
|
||||
# refactor (forbidden.go deleted -> AssertNoForbiddenAssetRefs/looksLikeASCIITickerID/
|
||||
# toHex/fromHex undefined => won't compile). Force v1.3.21 (forbidden.go restored),
|
||||
# matching node's go.mod and the chain-VM plugin stage (CHAINS_REF) below.
|
||||
go mod edit -require=github.com/luxfi/chains@v1.3.21 && \
|
||||
find /tmp/evm -name go.sum -exec sed -i -E '/^github.com\/(luxfi|hanzoai)\//d' {} + && \
|
||||
GOARCH=$(echo ${TARGETPLATFORM} | cut -d / -f2) \
|
||||
CGO_ENABLED=${EVM_CGO} GOFLAGS=-mod=mod \
|
||||
go build -ldflags="-s -w" -tags "${EVM_TAGS}" -o /luxd/build/plugins/${EVM_VM_ID} ./plugin && \
|
||||
CGO_ENABLED=0 GOFLAGS=-mod=mod \
|
||||
go build -ldflags="-s -w" -o /luxd/build/plugins/${EVM_VM_ID} ./plugin && \
|
||||
chmod +x /luxd/build/plugins/${EVM_VM_ID} && \
|
||||
rm -rf /tmp/evm
|
||||
|
||||
@@ -335,16 +295,15 @@ RUN --mount=type=cache,target=/root/.cache/go-build \
|
||||
# oraclevm -> r5m1ujrmXxVcQetG3CQfuDLHp2RHKh6vCDaFgBRQfUcTZh7eS
|
||||
# quantumvm -> ry9Sg8rZdT26iEKvJDmC2wkESs4SDKgZEhk5BgLSwg1EpcNug
|
||||
# relayvm -> sP6dLqrrBR9w3soP18fbJ3YzZecZdD7DDdfH2cFhhLq7Hy9bz
|
||||
# mpcvm -> qCURact1n41FcoNBch8iMVBwc9AWie48D118ZNJ5tBdWrvryS
|
||||
# thresholdvm -> tGVBwRxpmD2aFdg3iYjgRvrCe8Jcmq9UNKxyHMus2NZ8WcD8t
|
||||
# zkvm -> vv3qPfyTVXZ5ArRZA9Jh4hbYDTBe43f7sgQg4CHfNg1rnnvX9
|
||||
|
||||
# MUST track node's go.mod luxfi/chains (the D-Chain dexvm + 10 VM plugins).
|
||||
# v1.3.14 = the native-atomic seam (rail-bound D->C atomic, LP committed-liquidity).
|
||||
# Bump with every chains release or the bundled VM plugins go stale vs node's deps.
|
||||
# v1.4.7 == node go.mod's luxfi/chains pin: warp consolidated to ONE luxfi/warp
|
||||
# helper (bridgevm/zkvm/mpcvm), graphvm genesis-last-accepted fix, built on
|
||||
# evm v1.99.48 + precompile v0.16.0 (enable-everything builder surface). Keeps the
|
||||
# baked VM plugins in lockstep with the host node.
|
||||
ARG CHAINS_REF=v1.7.6
|
||||
# v1.3.21 == node go.mod's luxfi/chains pin (the finality-complete go-live); keeps the
|
||||
# 10 baked VM plugins (incl. the FATAL-gated bridgevm) in lockstep with the host node.
|
||||
ARG CHAINS_REF=v1.3.21
|
||||
RUN --mount=type=cache,target=/root/.cache/go-build \
|
||||
git clone --depth 1 --branch ${CHAINS_REF} https://github.com/luxfi/chains.git /tmp/chains && \
|
||||
find /tmp/chains -name go.sum -exec sed -i -E '/^github.com\/(luxfi|hanzoai)\//d' {} +
|
||||
@@ -378,25 +337,13 @@ RUN --mount=type=cache,target=/root/.cache/go-build \
|
||||
-o /luxd/build/plugins/ry9Sg8rZdT26iEKvJDmC2wkESs4SDKgZEhk5BgLSwg1EpcNug ./cmd/plugin ) || echo "WARN: quantumvm plugin build skipped" ; \
|
||||
( cd /tmp/chains/relayvm && CGO_ENABLED=0 GOFLAGS=-mod=mod go build -ldflags="-s -w" \
|
||||
-o /luxd/build/plugins/sP6dLqrrBR9w3soP18fbJ3YzZecZdD7DDdfH2cFhhLq7Hy9bz ./cmd/plugin ) || echo "WARN: relayvm plugin build skipped" ; \
|
||||
( cd /tmp/chains/mpcvm && CGO_ENABLED=0 GOFLAGS=-mod=mod go build -ldflags="-s -w" \
|
||||
-o /luxd/build/plugins/qCURact1n41FcoNBch8iMVBwc9AWie48D118ZNJ5tBdWrvryS ./cmd/plugin ) || echo "WARN: mpcvm plugin build skipped" ; \
|
||||
( cd /tmp/chains/thresholdvm && CGO_ENABLED=0 GOFLAGS=-mod=mod go build -ldflags="-s -w" \
|
||||
-o /luxd/build/plugins/tGVBwRxpmD2aFdg3iYjgRvrCe8Jcmq9UNKxyHMus2NZ8WcD8t ./cmd/plugin ) || echo "WARN: thresholdvm plugin build skipped" ; \
|
||||
( cd /tmp/chains/zkvm && CGO_ENABLED=0 GOFLAGS=-mod=mod go build -ldflags="-s -w" \
|
||||
-o /luxd/build/plugins/vv3qPfyTVXZ5ArRZA9Jh4hbYDTBe43f7sgQg4CHfNg1rnnvX9 ./cmd/plugin ) || echo "WARN: zkvm plugin build skipped" ; \
|
||||
( chmod +x /luxd/build/plugins/* 2>/dev/null || true ) && \
|
||||
for p in \
|
||||
juFxSrbCM4wszxddKepj1GWwmrn9YgN1g4n3VUWPpRo9JjERA \
|
||||
kMhHABHM8j4bH94MCc4rsTNdo5E9En37MMyiujk4WdNxgXFsY \
|
||||
nZQm4Dmg1rjX18rb8maL9gamYyXPf1xCvF7ymWzxp6a1nSQTt \
|
||||
oR6tnZHezwogyf9fRnomNXC9ojwCEBAU6jdUzpgy2PB1tD7fM \
|
||||
pJJCSV7hHYVY6TUZwR8qUPAfuhX8JLb2C1AzNSezrYNbgau8M \
|
||||
r5m1ujrmXxVcQetG3CQfuDLHp2RHKh6vCDaFgBRQfUcTZh7eS \
|
||||
ry9Sg8rZdT26iEKvJDmC2wkESs4SDKgZEhk5BgLSwg1EpcNug \
|
||||
sP6dLqrrBR9w3soP18fbJ3YzZecZdD7DDdfH2cFhhLq7Hy9bz \
|
||||
qCURact1n41FcoNBch8iMVBwc9AWie48D118ZNJ5tBdWrvryS \
|
||||
vv3qPfyTVXZ5ArRZA9Jh4hbYDTBe43f7sgQg4CHfNg1rnnvX9 ; do \
|
||||
test -s /luxd/build/plugins/$p \
|
||||
|| { echo "FATAL: required chain-VM plugin $p missing/empty — its build failed above (see the matching WARN line); the runtime-stage hard COPY would otherwise fail cryptically. Surface & fix the real Go build error, or remove the plugin from BOTH the build list and the runtime COPY."; exit 1; } ; \
|
||||
done && \
|
||||
test -s /luxd/build/plugins/kMhHABHM8j4bH94MCc4rsTNdo5E9En37MMyiujk4WdNxgXFsY \
|
||||
|| { echo "FATAL: bridgevm (B-Chain) plugin missing — the v1.30.16 regression would recur"; exit 1; } && \
|
||||
rm -rf /tmp/chains
|
||||
|
||||
# ============= Native D-Chain DEX VM Plugin Stage ================
|
||||
@@ -412,7 +359,7 @@ RUN --mount=type=cache,target=/root/.cache/go-build \
|
||||
# accelerator in pkg/lx is a separate concern gated by its own cuda/metal tags and
|
||||
# is NOT linked here. v1.5.10 is the first tag whose cmd/dchain builds CGO=0
|
||||
# (drops the phantom dchain+cgo gate); v1.5.11 wires CLOB order ingestion over the
|
||||
# node HTTP router (VM.CreateHandlers -> /v1/bc/D/dex/<method>, pkg/dchain/ingest.go)
|
||||
# node HTTP router (VM.CreateHandlers -> /ext/bc/D/dex/<method>, pkg/dchain/ingest.go)
|
||||
# so an order POSTed to the node flows submitTx -> mempool -> consensus -> Verify
|
||||
# (match) -> Accept; v1.5.12 persists the head block so the VM survives a restart
|
||||
# once advanced past genesis (GetBlock(lastAccepted) no longer ErrNotFound);
|
||||
@@ -424,7 +371,7 @@ RUN --mount=type=cache,target=/root/.cache/go-build \
|
||||
# DB (every plugin VM via vm/rpc) — that stranded the native D-Chain order book
|
||||
# (rebuildBookFromDB folded empty -> 0 fills despite committed asks). v1.5.15 adds
|
||||
# the committed-state READ surface (clob_get_trades/orders/markets/book over
|
||||
# /v1/bc/D/dex/<method>, pkg/dchain/read.go): read-only JSON of the durable trade
|
||||
# /ext/bc/D/dex/<method>, pkg/dchain/read.go): read-only JSON of the durable trade
|
||||
# log / resting book / markets, served beside the writes with ZERO consensus
|
||||
# impact. Needed to VERIFY a fill replicated identically across validators (query
|
||||
# every node, diff the trade rows + head root) and to feed markets-display (native
|
||||
@@ -435,11 +382,6 @@ RUN --mount=type=cache,target=/root/.cache/go-build \
|
||||
git clone --depth 1 --branch ${DEX_REF} https://github.com/luxfi/dex.git /tmp/dex && \
|
||||
find /tmp/dex -name go.sum -exec sed -i -E '/^github.com\/(luxfi|hanzoai)\//d' {} + && \
|
||||
cd /tmp/dex && \
|
||||
# Align the dexvm plugin's ZAP wire (luxfi/api) with the node. No dex release
|
||||
# pins api v1.0.16 yet (all <=v1.5.20 carry v1.0.15), so force it here; the
|
||||
# bump is code-free (adds InitializeResponse.Capabilities, capability-gated),
|
||||
# so dexvm emits the field the node decodes -> no "unexpected EOF" on D-Chain.
|
||||
go mod edit -require=github.com/luxfi/api@v1.0.16 && \
|
||||
. /build/build_env.sh && \
|
||||
GOARCH=$(echo ${TARGETPLATFORM} | cut -d / -f2) \
|
||||
CGO_ENABLED=0 GOFLAGS=-mod=mod \
|
||||
@@ -502,7 +444,7 @@ COPY --from=builder \
|
||||
/luxd/build/plugins/r5m1ujrmXxVcQetG3CQfuDLHp2RHKh6vCDaFgBRQfUcTZh7eS \
|
||||
/luxd/build/plugins/ry9Sg8rZdT26iEKvJDmC2wkESs4SDKgZEhk5BgLSwg1EpcNug \
|
||||
/luxd/build/plugins/sP6dLqrrBR9w3soP18fbJ3YzZecZdD7DDdfH2cFhhLq7Hy9bz \
|
||||
/luxd/build/plugins/qCURact1n41FcoNBch8iMVBwc9AWie48D118ZNJ5tBdWrvryS \
|
||||
/luxd/build/plugins/tGVBwRxpmD2aFdg3iYjgRvrCe8Jcmq9UNKxyHMus2NZ8WcD8t \
|
||||
/luxd/build/plugins/vv3qPfyTVXZ5ArRZA9Jh4hbYDTBe43f7sgQg4CHfNg1rnnvX9 \
|
||||
/luxd/build/plugins/
|
||||
WORKDIR /luxd/build
|
||||
|
||||
@@ -0,0 +1,324 @@
|
||||
# Lux Mainnet Launch Checklist
|
||||
|
||||
Node: luxfi/node v1.24.11
|
||||
Consensus: Quasar (BLS+Corona, slashing, stake-weighted sampling)
|
||||
EVM: GPU ecrecover, 18 precompiles
|
||||
Genesis: networkID=1, startTime=2025-12-12T21:06:51Z (mainnet), networkID=2, startTime=2026-02-10T16:00:00Z (testnet)
|
||||
Precompile constraint: all activations MUST be after 2025-12-25
|
||||
|
||||
---
|
||||
|
||||
## Infrastructure
|
||||
|
||||
| Environment | Cluster | DOKS ID | K8s Version | Namespace | Validators |
|
||||
|-------------|---------|---------|-------------|-----------|------------|
|
||||
| Testnet | do-sfo3-lux-test-k8s | `005ec3c4` | 1.35.1-do.0 | lux-testnet | 11 (target) |
|
||||
| Rehearsal | do-sfo3-lux-dev-k8s | `0ff340e1` | 1.35.1-do.0 | lux-devnet | 21 (target) |
|
||||
| Mainnet | do-sfo3-lux-k8s | `04c46df5` | 1.34.1-do.4 | lux-mainnet | 21 (target) |
|
||||
|
||||
Current state: lux-k8s runs 5 validators (v1.23.31) via LuxNetwork CRD. Testnet cluster (lux-test-k8s) has a 3-replica StatefulSet (v1.23.40).
|
||||
|
||||
Image: `ghcr.io/luxfi/node:v1.24.11` (built via CI/CD, linux/amd64+arm64)
|
||||
Staking keys: KMS at `kms.lux.network`, project `lux-infra`, synced via KMSSecret CRD
|
||||
Secrets: never in manifests, never in env files, never committed
|
||||
Genesis configs: `/Users/z/work/lux/genesis/configs/{testnet,mainnet,devnet}/`
|
||||
K8s manifests: `/Users/z/work/lux/universe/k8s/`
|
||||
Profiles: `standard.json` (~100MB/node), `max.json` (~512MB/node)
|
||||
|
||||
Bootstrappers:
|
||||
- Mainnet: 5 seeds (ports 9631) -- `209.38.118.46`, `209.38.174.69`, `24.144.69.101`, `134.199.187.56`, `143.198.246.173`
|
||||
- Testnet: 2 seeds (ports 9641) -- `134.199.187.16`, `209.38.174.84`
|
||||
|
||||
Consensus parameters (mainnet): K=20, AlphaPreference=15, AlphaConfidence=15, Beta=20, ConcurrentPolls=4
|
||||
|
||||
Tokenomics: 10B total supply (9 decimals), min validator stake 1M LUX, min delegator stake 25K LUX, combined staking allowed (NFT+delegation), 80% uptime threshold
|
||||
|
||||
C-Chain: chainId=96369 (mainnet), 96368 (testnet), gasLimit=12M, targetBlockRate=2s, minBaseFee=25gwei
|
||||
|
||||
---
|
||||
|
||||
## Phase 1: Testnet (lux-test-k8s, networkID=2)
|
||||
|
||||
Target: validate all consensus, EVM, and staking behavior with K=11 validators.
|
||||
|
||||
### 1.1 Deployment
|
||||
|
||||
- [ ] Update LuxNetwork CRD in `universe/k8s/lux-k8s/validators/statefulset.yaml` (testnet section): `validators: 11`, `image.tag: v1.24.11`
|
||||
- [ ] Update lux-test-k8s StatefulSet in `universe/k8s/lux-test-k8s/testnet/statefulset.yaml`: replicas=11, image=v1.24.11
|
||||
- [ ] Generate 11 staking key pairs via `lux cli` and store in KMS (`lux-infra/testnet/staking/`)
|
||||
- [ ] Add 9 new bootstrapper entries to `genesis/configs/testnet/bootstrappers.json` (currently 2)
|
||||
- [ ] Apply `max.json` profile for testnet validators (512MB/node for stress testing headroom)
|
||||
- [ ] Regenerate testnet genesis with 11 initial validators via `genesis` tool
|
||||
- [ ] Verify all precompile activation timestamps are after 2025-12-25
|
||||
- [ ] Deploy via PaaS (platform.hanzo.ai), not manual kubectl
|
||||
- [ ] Verify all 11 pods reach Running state
|
||||
- [ ] Verify all 11 nodes report healthy via `/ext/health/liveness`
|
||||
|
||||
### 1.2 Bootstrap and Connectivity
|
||||
|
||||
- [ ] Verify all 11 validators discover each other via P2P (check `info.peers` RPC, expect 10 peers per node)
|
||||
- [ ] Verify staking port 9641 reachable between all pods (`luxd-{0..10}.luxd-headless.testnet.svc.cluster.local:9641`)
|
||||
- [ ] Verify P-chain bootstraps and all validators appear in `platform.getCurrentValidators`
|
||||
- [ ] Verify C-chain bootstraps and produces blocks
|
||||
- [ ] Verify X-chain bootstraps and processes UTXO transactions
|
||||
|
||||
### 1.3 Quasar Consensus Verification
|
||||
|
||||
- [ ] Submit transactions, verify Quasar finalization with K=11
|
||||
- [ ] Verify BLS aggregate signatures in block headers
|
||||
- [ ] Verify Corona optimistic fast path activates when all 11 validators are online
|
||||
- [ ] Measure finality latency (target: sub-second with Corona)
|
||||
- [ ] Verify stake-weighted sampling: validators with more stake get polled proportionally
|
||||
|
||||
### 1.4 EVM Execution
|
||||
|
||||
- [ ] Deploy a test contract, call all standard opcodes
|
||||
- [ ] Submit 100 sequential transactions, verify correct nonce ordering
|
||||
- [ ] Verify `eth_call` and `eth_estimateGas` return correct results
|
||||
- [ ] Verify block gas limit is 12M (from cchain.json config)
|
||||
- [ ] Verify minBaseFee=25gwei is enforced
|
||||
|
||||
### 1.5 GPU ecrecover
|
||||
|
||||
- [ ] Verify GPU backend auto-detection: CUDA on Linux DOKS nodes, Metal on macOS
|
||||
- [ ] Run ecrecover-heavy workload (1000 signature verifications per block)
|
||||
- [ ] Compare ecrecover throughput: GPU vs CPU fallback
|
||||
- [ ] Verify graceful fallback to CPU when GPU unavailable (set `--gpu-backend=cpu`)
|
||||
|
||||
### 1.6 Precompiles (all 18)
|
||||
|
||||
- [ ] Test each precompile individually via contract calls
|
||||
- [ ] Verify DEX precompile (LP-9010 PoolManager): pool creation, swaps, flash loans
|
||||
- [ ] Verify DEX router precompile (LP-9012): multi-hop routing
|
||||
- [ ] Verify all precompile addresses are deterministic and match spec
|
||||
- [ ] Verify precompile gas metering is correct (no underpriced or overpriced ops)
|
||||
- [ ] Verify precompiles revert correctly on invalid input
|
||||
|
||||
### 1.7 Slashing
|
||||
|
||||
- [ ] Craft equivocation evidence: have a validator sign two different blocks at same height
|
||||
- [ ] Submit equivocation proof to P-chain slashing precompile
|
||||
- [ ] Verify slashed validator's stake is burned
|
||||
- [ ] Verify slashed validator is removed from active set
|
||||
- [ ] Verify honest validators are unaffected
|
||||
|
||||
### 1.8 Uptime and Rewards
|
||||
|
||||
- [ ] Stop 1 validator (scale pod to 0)
|
||||
- [ ] Wait for reward period to elapse
|
||||
- [ ] Verify stopped validator's uptime drops below 80%
|
||||
- [ ] Verify rewards are withheld for the stopped validator
|
||||
- [ ] Restart the validator, verify it re-bootstraps and resumes
|
||||
- [ ] Verify validators with >80% uptime receive expected rewards
|
||||
|
||||
### 1.9 Stress Test
|
||||
|
||||
- [ ] Run stress test: maximum TPS with 1B gas blocks (increase gas limit temporarily)
|
||||
- [ ] Measure sustained TPS over 1 hour (target: verify consensus is the bottleneck, not EVM)
|
||||
- [ ] Monitor memory usage per node (should stay within `max.json` profile ~512MB)
|
||||
- [ ] Monitor disk I/O and database growth rate
|
||||
- [ ] Verify no consensus stalls under load
|
||||
- [ ] Verify block production rate stays at targetBlockRate=2s
|
||||
|
||||
### 1.10 Validator Join/Leave
|
||||
|
||||
- [ ] Add a 12th validator via `platform.addPermissionlessValidator` (permissionless staking)
|
||||
- [ ] Verify new validator bootstraps from existing state
|
||||
- [ ] Verify new validator begins participating in consensus
|
||||
- [ ] Remove a validator via unstaking (wait for stake period to end or use testnet short periods)
|
||||
- [ ] Verify removed validator exits gracefully
|
||||
- [ ] Verify remaining validators continue producing blocks
|
||||
|
||||
### 1.11 Formal Verification
|
||||
|
||||
- [ ] Run Lean proofs for Quasar consensus safety and liveness
|
||||
- [ ] Run TLA+ model checker for consensus state machine
|
||||
- [ ] Run Tamarin prover for BLS+Corona security properties
|
||||
- [ ] Run Halmos for EVM precompile correctness (symbolic execution)
|
||||
- [ ] All proofs pass with zero counterexamples
|
||||
|
||||
---
|
||||
|
||||
## Phase 2: Mainnet Rehearsal (lux-dev-k8s, networkID=3)
|
||||
|
||||
Target: full mainnet simulation with real parameters for 72 hours.
|
||||
|
||||
### 2.1 Deployment
|
||||
|
||||
- [ ] Update LuxNetwork CRD (devnet section): `validators: 21`, `image.tag: v1.24.11`
|
||||
- [ ] Generate 21 staking key pairs, store in KMS (`lux-infra/devnet/staking/`)
|
||||
- [ ] Use mainnet genesis parameters (networkID=3, but same tokenomics, same stake amounts)
|
||||
- [ ] Apply `max.json` profile
|
||||
- [ ] Deploy via PaaS
|
||||
- [ ] Verify all 21 pods healthy
|
||||
|
||||
### 2.2 Real Staking Parameters
|
||||
|
||||
- [ ] Configure minimum validator stake: 1M LUX
|
||||
- [ ] Configure minimum delegator stake: 25K LUX
|
||||
- [ ] Configure max delegation ratio: 10x
|
||||
- [ ] Configure NFT staking tiers (Genesis 500K/2x, Pioneer 750K/1.5x, Standard 1M/1x)
|
||||
- [ ] Verify combined staking logic: NFT value + delegation + staked >= 1M
|
||||
- [ ] Verify B-chain validators require 100M LUX + KYC
|
||||
|
||||
### 2.3 72-Hour Soak Test
|
||||
|
||||
- [ ] Start clock. Record block height and timestamp.
|
||||
- [ ] Continuous transaction load: 50 TPS sustained
|
||||
- [ ] Monitor: CPU, memory, disk, network per node (Prometheus + Grafana via PaaS)
|
||||
- [ ] Monitor: consensus latency p50/p95/p99
|
||||
- [ ] Monitor: block production rate (target: 1 block per 2s)
|
||||
- [ ] Monitor: peer count stability (all 21 connected)
|
||||
- [ ] Monitor: no OOMKills, no pod restarts, no crashloops
|
||||
- [ ] At hour 24: rolling restart of 5 validators (verify zero downtime)
|
||||
- [ ] At hour 48: simulate network partition (isolate 7 nodes), verify chain halts (< 2/3 online)
|
||||
- [ ] Restore partition, verify chain resumes within 30s
|
||||
- [ ] At hour 72: record final block height, calculate actual vs expected blocks
|
||||
- [ ] Pass criteria: zero consensus faults, zero data loss, <1% block time variance
|
||||
|
||||
### 2.4 Security Audit
|
||||
|
||||
- [ ] External security audit firm engaged (Red team)
|
||||
- [ ] Audit scope: consensus, EVM, precompiles, staking, slashing, P2P networking
|
||||
- [ ] Audit result: 0 critical findings, 0 high findings
|
||||
- [ ] All medium findings remediated or accepted with documented risk
|
||||
- [ ] Audit report signed and archived
|
||||
|
||||
### 2.5 Bridge / Teleport (B-Chain + T-Chain)
|
||||
|
||||
- [ ] Deploy MPC threshold signing (5 nodes, threshold 3) in `lux-mpc` namespace
|
||||
- [ ] Deploy bridge UI and API in `lux-bridge` namespace
|
||||
- [ ] Verify CGGMP21 keygen: 5 parties generate shared key
|
||||
- [ ] Verify threshold signing: 3-of-5 produces valid signature
|
||||
- [ ] Test cross-chain transfer: lock on source chain, mint on Lux
|
||||
- [ ] Test reverse: burn on Lux, unlock on source chain
|
||||
- [ ] Verify MPC API at `mpc-api.lux.network` responds
|
||||
- [ ] Verify bridge handles partial MPC node failure (2 down, 3 still sign)
|
||||
|
||||
### 2.6 DEX (D-Chain + Precompiles)
|
||||
|
||||
- [ ] Deploy DEX precompile PoolManager (LP-9010) -- already active from genesis
|
||||
- [ ] Deploy DEX Router precompile (LP-9012) -- already active from genesis
|
||||
- [ ] Deploy off-chain CLOB matching engine
|
||||
- [ ] Create liquidity pool via precompile
|
||||
- [ ] Execute swap via router precompile
|
||||
- [ ] Verify AMM pricing matches expected curve
|
||||
- [ ] Verify flash loan execution and repayment
|
||||
- [ ] Test CLOB: place limit order, verify fill
|
||||
- [ ] Verify DEX on lux.exchange frontend connects to devnet
|
||||
|
||||
---
|
||||
|
||||
## Phase 3: Mainnet Launch (lux-k8s, networkID=1)
|
||||
|
||||
Target: production network with real value.
|
||||
|
||||
### 3.1 Pre-launch
|
||||
|
||||
- [ ] All Phase 1 items passed
|
||||
- [ ] All Phase 2 items passed
|
||||
- [ ] Security audit sign-off received
|
||||
- [ ] Formal verification suite green
|
||||
- [ ] Legal review complete (terms of service, validator agreements)
|
||||
- [ ] Incident response runbook written and tested
|
||||
|
||||
### 3.2 Genesis Ceremony
|
||||
|
||||
- [ ] Final genesis config reviewed: `genesis/configs/mainnet/genesis.json` (networkID=1)
|
||||
- [ ] Genesis startTime confirmed: 2025-12-12T21:06:51Z
|
||||
- [ ] Initial allocations verified (500M initial + unlock schedule)
|
||||
- [ ] All 5 bootstrapper IPs confirmed reachable on port 9631
|
||||
- [ ] Genesis hash computed and published to lux.network
|
||||
- [ ] Genesis block signed by founding validators
|
||||
|
||||
### 3.3 Validator Onboarding
|
||||
|
||||
- [ ] Update LuxNetwork CRD (mainnet section): `validators: 21`, `image.tag: v1.24.11`
|
||||
- [ ] Scale from 5 current validators to 21
|
||||
- [ ] Generate 16 new staking key pairs in KMS (`lux-infra/mainnet/staking/`)
|
||||
- [ ] Update bootstrappers.json with all 21 validator endpoints
|
||||
- [ ] Deploy via PaaS with rolling update strategy
|
||||
- [ ] Verify all 21 validators healthy and in consensus
|
||||
- [ ] Publish validator onboarding guide for external operators
|
||||
- [ ] Open permissionless staking after initial stabilization period
|
||||
|
||||
### 3.4 Public RPC Endpoints
|
||||
|
||||
- [ ] Deploy KrakenD API gateway in `lux-gateway` namespace
|
||||
- [ ] Configure rate limiting per IP and per API key
|
||||
- [ ] Configure Cloudflare DNS (proxied, full SSL):
|
||||
- `api.lux.network` -> gateway (C-chain + P-chain + X-chain RPC)
|
||||
- `ws.lux.network` -> gateway (WebSocket subscriptions)
|
||||
- [ ] Verify `eth_chainId` returns `0x17871` (96369)
|
||||
- [ ] Verify `net_version` returns `96369`
|
||||
- [ ] Verify RPC endpoints handle 10K req/s without degradation
|
||||
- [ ] Verify WebSocket subscriptions for `newHeads`, `logs`, `pendingTransactions`
|
||||
|
||||
### 3.5 Explorer Deployment
|
||||
|
||||
- [ ] Deploy explorer (luxfi/explorer) in `lux-explorer` namespace (already has manifests for 5 chains)
|
||||
- [ ] Configure for C-chain (chainId 96369)
|
||||
- [ ] Configure indexers for all active chains
|
||||
- [ ] Configure Cloudflare DNS: `explore.lux.network`
|
||||
- [ ] Verify block display, transaction search, contract verification
|
||||
- [ ] Deploy exchange frontend: `lux.exchange`
|
||||
|
||||
### 3.6 Bridge Activation
|
||||
|
||||
- [ ] Deploy MPC production cluster (5 nodes, threshold 3)
|
||||
- [ ] Generate production MPC keys (CGGMP21 keygen ceremony)
|
||||
- [ ] Store MPC key shares in KMS (`lux-infra/mainnet/mpc/`)
|
||||
- [ ] Deploy bridge contracts on supported chains (ETH, BNB, Polygon, Arbitrum, Base, Optimism)
|
||||
- [ ] Deploy bridge UI at bridge domain
|
||||
- [ ] Configure Cloudflare DNS
|
||||
- [ ] Enable deposits (one chain at a time, small limits first)
|
||||
- [ ] Monitor for 24h, then raise limits
|
||||
|
||||
### 3.7 Post-Launch Monitoring
|
||||
|
||||
- [ ] Prometheus + Grafana dashboards live (via PaaS o11y stack)
|
||||
- [ ] Alerts configured:
|
||||
- Validator down (any pod not Ready for >5min)
|
||||
- Consensus stall (no new block for >30s)
|
||||
- Peer count drop (any node <15 peers)
|
||||
- Memory usage >80% of limit
|
||||
- Disk usage >70%
|
||||
- Error rate >1% on RPC endpoints
|
||||
- [ ] On-call rotation established
|
||||
- [ ] Runbook covers: validator restart, chain halt recovery, emergency upgrade, key rotation
|
||||
|
||||
---
|
||||
|
||||
## Port Reference
|
||||
|
||||
| Network | HTTP | Staking | Metrics |
|
||||
|---------|------|---------|---------|
|
||||
| Mainnet | 9630 | 9631 | 9090 |
|
||||
| Testnet | 9640 | 9641 | 9090 |
|
||||
| Devnet | 9650 | 9651 | 9090 |
|
||||
|
||||
## Chain IDs
|
||||
|
||||
| Chain | Mainnet | Testnet | Devnet |
|
||||
|-------|---------|---------|--------|
|
||||
| C-Chain | 96369 | 96368 | 96370 |
|
||||
| Zoo EVM | 200200 | 200201 | 200202 |
|
||||
| Hanzo EVM | 36963 | 36964 | 36964 |
|
||||
| SPC EVM | 36911 | 36910 | 36912 |
|
||||
| Pars EVM | 494949 | 7071 | 494951 |
|
||||
|
||||
## File References
|
||||
|
||||
| What | Path |
|
||||
|------|------|
|
||||
| Node source | `~/work/lux/node/` |
|
||||
| Genesis configs | `~/work/lux/genesis/configs/{mainnet,testnet,devnet}/` |
|
||||
| Chain configs | `~/work/lux/genesis/configs/chain-configs/` |
|
||||
| K8s manifests | `~/work/lux/universe/k8s/` |
|
||||
| Validator CRD | `~/work/lux/universe/k8s/lux-k8s/validators/statefulset.yaml` |
|
||||
| Testnet StatefulSet | `~/work/lux/universe/k8s/lux-test-k8s/testnet/statefulset.yaml` |
|
||||
| Node profiles | `~/work/lux/node/config/profiles/{standard,max}.json` |
|
||||
| Tokenomics config | `~/work/lux/node/config/tokenomics.go` |
|
||||
| GPU config | `~/work/lux/node/config/gpu.go` |
|
||||
| Health/consensus params | `~/work/lux/node/config/health.go` |
|
||||
| Network registry | `~/work/lux/universe/NETWORKS.yaml` |
|
||||
@@ -10,24 +10,3 @@ For the canonical Lux IP and licensing strategy, see:
|
||||
|
||||
For commercial inquiries that go beyond BSD-3 (e.g. private moat
|
||||
acceleration kernels), contact `licensing@lux.network`.
|
||||
|
||||
## Upstream attribution
|
||||
|
||||
See [NOTICE](NOTICE) for the full attribution. In summary:
|
||||
|
||||
- **avalanchego** (Ava Labs, Inc.) — this repository is derived from
|
||||
[ava-labs/avalanchego](https://github.com/ava-labs/avalanchego), licensed
|
||||
under the **BSD 3-Clause License** (© 2019 Ava Labs, Inc.). BSD-3 is
|
||||
permissive; the Lux additions here are likewise BSD-3-Clause.
|
||||
- **go-ethereum** (The go-ethereum Authors) — EVM support derives from
|
||||
[go-ethereum](https://github.com/ethereum/go-ethereum). It is **not**
|
||||
vendored in-tree; it is consumed as the external Go module
|
||||
`github.com/luxfi/geth`, which retains go-ethereum's original licenses:
|
||||
the library is **LGPL-3.0-or-later** and the command-line tools are
|
||||
**GPL-3.0**.
|
||||
|
||||
**Copyleft flag:** the LGPL-3.0/GPL-3.0 terms of the go-ethereum-derived code
|
||||
(via `github.com/luxfi/geth`) are **not** superseded by this repository's
|
||||
BSD-3-Clause license. Distributing compiled node binaries must honor LGPL-3.0
|
||||
for the linked geth library (published source of that code and its
|
||||
modifications at <https://github.com/luxfi/geth>, and user ability to relink).
|
||||
|
||||
@@ -65,10 +65,10 @@ selection, and EVM contract auth.
|
||||
### Where to look for X
|
||||
- Profile resolve at boot: `node/node.go:initSecurityProfile`
|
||||
- Profile RPC + REST + metrics: `service/security/`
|
||||
- JSON-RPC namespace: `security` at `POST /v1/security`
|
||||
- JSON-RPC namespace: `security` at `POST /ext/security`
|
||||
(methods `securityProfile`, `blockSecurity`)
|
||||
- REST sidecars: `GET /v1/security/profile`, `GET /v1/security/block/{n}`
|
||||
- Prometheus gauges: `/v1/metrics` under the `security_*` family
|
||||
- REST sidecars: `GET /ext/security/profile`, `GET /ext/security/block/{n}`
|
||||
- Prometheus gauges: `/ext/metrics` under the `security_*` family
|
||||
- Peer scheme gate: `network/peer/scheme_gate.go`
|
||||
- Classical-compat registry: `vms/txs/auth/policy.go`
|
||||
- Mempool gate (P-Chain): `vms/platformvm/mempool/*.go`
|
||||
@@ -85,29 +85,6 @@ selection, and EVM contract auth.
|
||||
|
||||
## FeePolicy — canonical user-tx fee gate
|
||||
|
||||
> **Topology + UTXO ownership + cross-chain fee model** are normatively
|
||||
> specified by [**LP-0130** (Chain Topology, UTXO Ownership, and Fee
|
||||
> Model)](https://github.com/luxfi/lps/blob/main/LPs/lp-0130-chain-topology-utxo-ownership-and-fee-model.md).
|
||||
> Read that LP before touching any VM's fee/settlement path or any
|
||||
> cross-chain import/export flow. In particular:
|
||||
>
|
||||
> - Only **P** and **X** are canonical UTXO state machines (LP-0130 §2).
|
||||
> - **X** is the money rail; **P** is the staking/reward rail; **LUX**
|
||||
> is the fee currency everywhere (LP-0130 §3, §5).
|
||||
> - **Q-Chain has no user-payable blockspace** — finality is a
|
||||
> validator obligation paid via P (LP-0130 §6). `quantumvm` MUST
|
||||
> use `NoUserTxPolicy{}` — enforced in chains/quantumvm/feegate.go as of 2026-07-03.
|
||||
> - **M-Chain fees are service fees** deducted from the originating
|
||||
> chain's fee pool, not a user M-balance (LP-0130 §7). `mpcvm`
|
||||
> already runs `NoUserTxPolicy{}` — correct.
|
||||
> - **B-Chain fees** are deducted from the bridged amount (LP-0130 §8).
|
||||
> - Every non-P/X chain settles worker rewards to X (asset payouts) or
|
||||
> P (staker rewards) via epoch fee roots reconciled at Q finality
|
||||
> (LP-0130 §4, §11).
|
||||
> - **Σ-escrow invariant** (LP-0130 I-8): `Σ non-P/X fee balances ==
|
||||
> Σ X-side fee escrow` at every Q checkpoint. Drift is a
|
||||
> finality-blocking fault.
|
||||
|
||||
Every Lux VM that accepts user-submitted txs declares a `fee.Policy`
|
||||
(package `vms/types/fee`). There is one interface and one validator —
|
||||
no per-VM bespoke fee structs.
|
||||
@@ -120,14 +97,14 @@ no per-VM bespoke fee structs.
|
||||
| zkvm | Z-Chain | user-tx | `FlatPolicy{Fee: MinTxFeeFloor, ...}` |
|
||||
| aivm | A-Chain | user-tx | `FlatPolicy{Fee: MinTxFeeFloor, ...}` |
|
||||
| keyvm | K-Chain | user-tx | `FlatPolicy{Fee: MinTxFeeFloor, ...}` |
|
||||
| bridgevm | B-Chain | user-tx | `FlatPolicy{Fee: MinTxFeeFloor, ...}` (deducted from bridged amount, LP-0130 §8) |
|
||||
| quantumvm | Q-Chain | **service-only** (LP-0130 §6) | `NoUserTxPolicy{}` — validator obligation, no user-payable blockspace |
|
||||
| bridgevm | B-Chain | user-tx | `FlatPolicy{Fee: MinTxFeeFloor, ...}` |
|
||||
| quantumvm | Q-Chain | user-tx | `FlatPolicy{Fee: MinTxFeeFloor, ...}` |
|
||||
| identityvm | I-Chain | user-tx | `FlatPolicy{Fee: MinTxFeeFloor, ...}` |
|
||||
| mpcvm | M-Chain | service-only (LP-0130 §7) | `NoUserTxPolicy{}` — fees pulled from originating chain's fee pool |
|
||||
| thresholdvm | M-Chain | service-only | `NoUserTxPolicy{}` |
|
||||
| oraclevm | O-Chain | service-only | `NoUserTxPolicy{}` |
|
||||
| relayvm | R-Chain | service-only | `NoUserTxPolicy{}` |
|
||||
| graphvm | G-Chain | read-only | `NoUserTxPolicy{}` (GraphQL refuses `mutation`) |
|
||||
| evm | C-Chain | user-tx | native EVM gas (gas * gasPrice >= 0 enforced upstream); balance is X-imported LUX (LP-0130 §10) |
|
||||
| evm | C-Chain | user-tx | native EVM gas (gas * gasPrice >= 0 enforced upstream) |
|
||||
| platformvm | P-Chain | user-tx | native `TxFee` field on Config |
|
||||
| avm | X-Chain | user-tx | native `TxFee` field on Config |
|
||||
|
||||
@@ -158,101 +135,6 @@ charge less.
|
||||
- Relay (R-Chain): `~/work/lux/relay/vm/feegate.go` (re-exported by `~/work/lux/chains/relayvm/`)
|
||||
- Graph (G-Chain): `~/work/lux/chains/graphvm/feegate.go` (read-only; NoUserTxPolicy)
|
||||
|
||||
## C-Chain tx-fee routing — RewardManager to DAO Safe (P-Chain: NO CHANGE)
|
||||
|
||||
Owner tokenomics pivoted: **100% of C-Chain tx fees accrue to the chain's DAO Gov
|
||||
Safe** via the existing `rewardmanager` precompile (C-Chain only). This needs **no
|
||||
P-Chain change** — routing is `GetCoinbaseAt` → reward address on the C-Chain. See
|
||||
`~/work/lux/evm/LLM.md` → "C-Chain Tx-Fee Routing — RewardManager". The P-Chain
|
||||
`feeRewardPool` fold-in below is **NOT built** (design-only, superseded); no
|
||||
`vms/platformvm/state` change was made.
|
||||
|
||||
<details><summary>Superseded design — 50/50 burn + P-Chain staking-reward fold (dormant option)</summary>
|
||||
|
||||
If the DAO ever chooses an in-protocol 50/50 split, the C-Chain half exists (dormant,
|
||||
`FeeSplitTimestamp` gated off) and the P-Chain fold-in would be: system-triggered epoch
|
||||
export of the C-Chain vault C→P → persisted `feeRewardPool` in `vms/platformvm/state`
|
||||
(mirror the `accruedFees` singleton, upgrade-safe) → pro-rata payout at
|
||||
`vms/platformvm/txs/executor/proposal_tx_executor.go` `rewardValidatorTx` (~line 285),
|
||||
unified into `PotentialReward`, NO second mint → decrement `currentSupply` by the epoch
|
||||
burn. Model R1 (move-not-mint), conservation-exact; R2 (burn+re-mint) rejected.
|
||||
</details>
|
||||
|
||||
## v1.36.12 fleet rollout — durable rejoin fix + RewardManager→DAO Safe (IN PROGRESS 2026-07-15)
|
||||
|
||||
Rolling the durable rejoin fix (node `63f61429d1`) across all Lux nets, gated
|
||||
devnet→testnet→mainnet, + activating RewardManager fees. Two hard facts were found
|
||||
on the devnet canary that change the naive "swap image" plan:
|
||||
|
||||
**BLOCKER (fixed in v1.36.12): published `node:v1.36.11` (digest `c3cf92a6`) cannot
|
||||
run ANY EVM chain.** Its baked VM plugins were built against a stale `luxfi/api`:
|
||||
C-Chain EVM from `luxfi/evm@v1.104.8` and D-Chain dexvm from `luxfi/dex@v1.5.15`
|
||||
both resolve `api v1.0.15`; the node pins `api v1.0.16`, which APPENDED
|
||||
`InitializeResponse.Capabilities` (Quasar-export handshake, api `1f2dc5a`). Node
|
||||
decodes the field, stale plugins never encode it → `vms/rpcchainvm/zap/client.go`
|
||||
fails every EVM `Initialize` with `zap decode initialize response: unexpected EOF`.
|
||||
Native VMs (P/X/Q) unaffected. **Fix = v1.36.12** (this repo, tag pushed, ARC docker
|
||||
build run 29381442539): `EVM_VERSION v1.104.8→v1.104.9`, force `api@v1.0.16` in the
|
||||
dexvm build stage; `CHAINS_REF=v1.7.6` was already v1.0.16. Node binary unchanged
|
||||
(still the durable fix). api bump is code-free for plugins (`chains v1.7.4→v1.7.5`
|
||||
adopted it go.mod-only). **Roll v1.36.12, NOT v1.36.11.** (Also: `api 1f2dc5a` added
|
||||
`Capabilities` WITHOUT bumping `version.RPCChainVMProtocol` (42) → skew is invisible
|
||||
at handshake, only fails at Initialize decode. Consider bumping the protocol next
|
||||
api-wire change so skews fail fast.)
|
||||
|
||||
**MIGRATION: v1.36.2→v1.36.x is a P-Chain codec change (linearcodec→ZAP-native),
|
||||
one-time DB wipe + re-bootstrap.** v1.36.11/12 cannot read a v1.36.2 P-Chain zapdb
|
||||
(`loadMetadata: feeState: zap: invalid magic bytes`; `state_commit.go:116` "database
|
||||
must be wiped"). Recovery = wipe `/data/db`+`/data/chainData`, re-bootstrap from
|
||||
peers. **Cross-version bootstrap (v1.36.11 node ← v1.36.2 peers) is PROVEN working**
|
||||
(devnet luxd-1: P/X re-bootstrapped from the 4 v1.36.2 peers). Devnet startup got a
|
||||
marker-gated one-time wipe (`/data/.zap-native-migrated`): absent→wipe+set marker,
|
||||
present→NO wipe (the durable-fix no-wipe restart path). mainnet/testnet/zoo use
|
||||
`startup.sh` which already has `.wipe-cchain` (C-Chain only) + `.allow-bootstrap`
|
||||
(flips skip-bootstrap=false + EVM state-sync); for the codec migration the P-Chain
|
||||
zapdb (`/data/db`) must also be cleared. **Mainnet C-Chain is 1.08M blocks → MUST
|
||||
enable EVM state-sync for the re-sync (full replay is too slow); native VMs are tiny.**
|
||||
|
||||
**Durable fix (`63f61429d1`):** discriminator for keeping the staked beacon set is
|
||||
SYBIL-PROTECTION, not `--skip-bootstrap`. So a behind validator on a sybil-protected
|
||||
net catches up from peers even with `--skip-bootstrap=true` (which prod hardcodes),
|
||||
no wipe. Devnet added `--skip-bootstrap=true` to the inline cmd to exercise this.
|
||||
|
||||
**RewardManager (C-Chain precompile, per-net `cchain-upgrade.json` → append one
|
||||
`precompileUpgrades` entry, dated `blockTimestamp`):** proven testnet shape is
|
||||
`{"rewardManagerConfig":{"blockTimestamp":<ts>,"adminAddresses":["<admin>"],
|
||||
"initialRewardConfig":{"rewardAddress":"<reward>"}}}`. Reward addr = coinbase; 100%
|
||||
fees land there, blackhole `0x0100…00` goes flat. Addresses: **testnet ALREADY LIVE**
|
||||
(reward `0xEAbCC110fAcBfebabC66Ad6f9E7B67288e720B59`, admin `0x9011…94714`); **mainnet**
|
||||
reward+admin = DAO Gov Safe `0x8E29b816c6C35b13cE1ff68D33E245C2bda8ac3D`; **zoo**
|
||||
reward `0x229599f227231d8C90fcF1a78589F5DC4b7A6962`; **devnet** reward
|
||||
`0x8d5081153aE1cfb41f5c932fe0b6Beb7E159cF84` (idx2), admin `0x9011…94714` (idx0).
|
||||
Source ConfigMaps: devnet `luxd-chain-upgrades/cchain-upgrade.json`; mainnet+testnet
|
||||
`luxd-startup/cchain-upgrade.json`; zoo `zood-mv-genesis/upgrade.json` (`--upgrade-file`).
|
||||
|
||||
**Rollout levers (lux-operator is scaled 0/0 — sts/cm are the live source of truth;
|
||||
CRs are STALE, do not scale operator up mid-roll):** ports devnet 9650 / testnet 9640
|
||||
/ mainnet 9630 / zoo 9630; RPC path `/v1/bc/C/rpc`; container `luxd` (`zood` on zoo).
|
||||
Devnet uses an inline generated cmd; testnet/mainnet/zoo use `/scripts/startup.sh`.
|
||||
**Zoo `zood-mv` trap: RollingUpdate + hardcoded `--skip-bootstrap=true` with NO
|
||||
`.allow-bootstrap` gate → switch to OnDelete BEFORE rolling.** Lux sts are OnDelete.
|
||||
Master funded key = LUX_MNEMONIC (secret `lux-deployer`) idx0 `0x9011…94714`;
|
||||
`genesis/cmd/derivekey -mnemonic "$M"` (path m/44'/9000'/0'/0/i, `CGO_ENABLED=0`).
|
||||
|
||||
**Per-node roll protocol (ALL nets, one at a time, NEVER 2 mainnet down — quorum
|
||||
4/5):** set sts image v1.36.12 (+ rewardManager cm edit) → delete ONE pod → WAIT
|
||||
until it is back at **TIP HEIGHT matching the others** (NOT pod-Ready; a wedged node
|
||||
false-reports Ready) AND C-Chain serves RPC → only then the next. If any node fails
|
||||
to return to tip, STOP that net and report.
|
||||
|
||||
**State at pause:** v1.36.12 tag pushed + ARC build dispatched (run 29381442539).
|
||||
Devnet sts = v1.36.11 + skip-bootstrap + wipe-marker; luxd-1 migrated (P/X up on
|
||||
v1.36.11, C-Chain down = the plugin bug → will heal on v1.36.12); luxd-0/2/3/4 still
|
||||
v1.36.2 healthy (devnet C-Chain 4/5). Nothing rolled on testnet/mainnet/zoo. NEXT:
|
||||
when v1.36.12 image is ready → set devnet sts image v1.36.12, delete luxd-1, confirm
|
||||
C-Chain inits + reaches tip; then finish devnet (durable-fix proof + RewardManager),
|
||||
then gated testnet→mainnet→zoo.
|
||||
|
||||
## Essential Commands
|
||||
|
||||
### Release & build (canonical) — via platform.hanzo.ai, NOT GitHub Actions
|
||||
@@ -348,7 +230,7 @@ Located in `/vms/`:
|
||||
- **platformvm**: Staking, validation, network management
|
||||
- **xvm**: Asset transfers, UTXO model
|
||||
- **dexvm**: DEX with order book, perpetuals, AMM
|
||||
- **mpcvm**: Threshold MPC and FHE for confidential computing
|
||||
- **thresholdvm**: Threshold MPC and FHE for confidential computing
|
||||
- **quantumvm**: PQ consensus coordination (ML-DSA, Corona)
|
||||
- **identityvm**: Decentralized identity (DID, verifiable credentials)
|
||||
- **keyvm**: Post-quantum key management (ML-KEM, ML-DSA)
|
||||
@@ -438,11 +320,11 @@ github.com/luxfi/genesis (JSON config) → github.com/luxfi/node/genesis/build
|
||||
### CGO Dependencies
|
||||
These require CGO for full functionality (graceful fallback when disabled):
|
||||
- `consensus/quasar` - GPU NTT acceleration
|
||||
- `vms/mpcvm/fhe` - GPU FHE operations
|
||||
- `vms/thresholdvm/fhe` - GPU FHE operations
|
||||
- `x/blockdb` - zstd compression
|
||||
|
||||
### FHE (Fully Homomorphic Encryption)
|
||||
Located in `vms/mpcvm/fhe/`:
|
||||
Located in `vms/thresholdvm/fhe/`:
|
||||
- Uses `github.com/luxfi/lattice/multiparty` for DKG
|
||||
- Lattice-based cryptography only (no fallbacks)
|
||||
- Threshold decryption via Warp messaging
|
||||
@@ -757,7 +639,7 @@ strings.Contains(errStr, "not found") // parent block not in local state
|
||||
### Known CGO Stubs
|
||||
When CGO disabled, these use CPU fallbacks:
|
||||
- `consensus/quasar/gpu_ntt_nocgo.go`
|
||||
- `vms/mpcvm/fhe/gpu_fhe_nocgo.go`
|
||||
- `vms/thresholdvm/fhe/gpu_fhe_nocgo.go`
|
||||
- `vms/zkvm/accel/accel_mlx.go`
|
||||
|
||||
### 8. ZAP CreateHandlers for VM HTTP Endpoints
|
||||
@@ -777,7 +659,7 @@ When CGO disabled, these use CPU fallbacks:
|
||||
```bash
|
||||
curl -s -X POST -H "Content-Type: application/json" \
|
||||
-d '{"jsonrpc":"2.0","method":"eth_chainId","params":[],"id":1}' \
|
||||
http://localhost:9640/v1/bc/C/rpc
|
||||
http://localhost:9640/ext/bc/C/rpc
|
||||
# Returns: {"jsonrpc":"2.0","id":1,"result":"0x17870"}
|
||||
```
|
||||
|
||||
@@ -786,7 +668,7 @@ curl -s -X POST -H "Content-Type: application/json" \
|
||||
|
||||
**Behavior**:
|
||||
- **GET /**: Returns JSON node information (nodeId, networkId, version, chains, endpoints)
|
||||
- **POST /**: Proxies JSON-RPC requests directly to C-chain `/v1/bc/C/rpc`
|
||||
- **POST /**: Proxies JSON-RPC requests directly to C-chain `/ext/bc/C/rpc`
|
||||
- **OPTIONS /**: Returns CORS preflight headers
|
||||
|
||||
**Files Modified**: `server/http/router.go`, `server/http/server.go`
|
||||
@@ -853,7 +735,7 @@ if s.validators.NumNets() != 0 {
|
||||
|
||||
**Verification**:
|
||||
```bash
|
||||
curl -s http://localhost:9650/v1/health | jq '.checks.bls'
|
||||
curl -s http://localhost:9650/ext/health | jq '.checks.bls'
|
||||
# Should show: "message": "node has the correct BLS key"
|
||||
```
|
||||
|
||||
@@ -887,7 +769,7 @@ Testing conducted on a single Lux validator node (testnet mode, macOS):
|
||||
**Benchmark Command:**
|
||||
```bash
|
||||
cd ~/work/lux/benchmarks
|
||||
NODE_ENDPOINT="http://localhost:9640/v1/bc/C/rpc" \
|
||||
NODE_ENDPOINT="http://localhost:9640/ext/bc/C/rpc" \
|
||||
PRIVATE_KEY="<funded_key>" \
|
||||
./bin/bench tps --chains=lux --duration=60s --concurrency=5
|
||||
```
|
||||
@@ -935,14 +817,4 @@ v2 semantic differences worth knowing (these change wire shape):
|
||||
|
||||
---
|
||||
|
||||
## Housekeeping
|
||||
|
||||
Removed 6 generated write-ups / stale root artifacts (`LAUNCH_CHECKLIST.md`,
|
||||
`rename_app.sh`, `replace_imports.sh`, `gen_zoo_addr` binary, `.ci-status-check.md`,
|
||||
`.ci-trigger`) plus the 73MB `.claude/worktrees/` agent scratch tree. Release and
|
||||
launch state live in this file, `CHANGELOG.md`, `RELEASE.md`; chain IDs/ports in
|
||||
`~/work/lux/universe/NETWORKS.yaml` and `~/work/lux/genesis/configs/`.
|
||||
|
||||
---
|
||||
|
||||
*Last Updated*: 2026-06-06
|
||||
|
||||
@@ -7,15 +7,8 @@ CGO_ENABLED ?= 1
|
||||
FIPS_STRICT ?= 0
|
||||
|
||||
# Go 1.26 experimental features:
|
||||
# runtimesecret - zeroes stack/register state after secret.Do() for forward secrecy.
|
||||
# It SIGSEGVs at startup under the WSL2 kernel (confirmed on go1.26.3 and go1.26.4), so enable
|
||||
# it only off-WSL; forward secrecy stays on for real Linux/macOS/production builds.
|
||||
WSL := $(shell grep -qiE 'microsoft|WSL' /proc/sys/kernel/osrelease 2>/dev/null && echo 1)
|
||||
ifeq ($(WSL),1)
|
||||
GOEXPERIMENT ?= none
|
||||
else
|
||||
# runtimesecret - zeroes stack/register state after secret.Do() for forward secrecy
|
||||
GOEXPERIMENT ?= runtimesecret
|
||||
endif
|
||||
export GOEXPERIMENT
|
||||
|
||||
# FIPS 140-3 always enabled (required for blockchain/financial systems)
|
||||
@@ -219,7 +212,7 @@ run-testnet: build-fips init-chains
|
||||
node-status:
|
||||
@echo "$(GREEN)Checking node status...$(NC)"
|
||||
@curl -s -X POST --data '{"jsonrpc":"2.0","id":1,"method":"info.isBootstrapped","params":{}}' \
|
||||
-H 'content-type:application/json;' http://localhost:9630/v1/info | jq
|
||||
-H 'content-type:application/json;' http://localhost:9630/ext/info | jq
|
||||
|
||||
stop-node:
|
||||
@echo "$(YELLOW)Stopping Lux node...$(NC)"
|
||||
|
||||
@@ -1,38 +0,0 @@
|
||||
Lux Node
|
||||
Copyright (c) 2019-2025 Lux Industries Inc.
|
||||
|
||||
This product includes software from avalanchego by Ava Labs, Inc.
|
||||
(https://github.com/ava-labs/avalanchego), licensed under the BSD 3-Clause
|
||||
License:
|
||||
|
||||
Copyright (C) 2019, Ava Labs, Inc.
|
||||
|
||||
Lux Node is derived from avalanchego. The Lux additions and modifications in
|
||||
this repository are licensed under the BSD 3-Clause License (see the LICENSE
|
||||
file). The BSD 3-Clause terms of the upstream avalanchego code are retained;
|
||||
this NOTICE preserves the required Ava Labs copyright attribution at the
|
||||
repository level.
|
||||
|
||||
--------------------------------------------------------------------------
|
||||
|
||||
Ethereum Virtual Machine support is derived from go-ethereum by The
|
||||
go-ethereum Authors (https://github.com/ethereum/go-ethereum). In this
|
||||
repository that code is NOT vendored in-tree; it is consumed as an external
|
||||
Go module through the Lux fork github.com/luxfi/geth, which retains
|
||||
go-ethereum's original licenses:
|
||||
|
||||
* The go-ethereum library packages are licensed under the GNU Lesser
|
||||
General Public License, version 3 (LGPL-3.0-or-later).
|
||||
* The go-ethereum command-line tools are licensed under the GNU General
|
||||
Public License, version 3 (GPL-3.0).
|
||||
|
||||
Copyright (C) The go-ethereum Authors
|
||||
|
||||
COPYLEFT NOTICE: The LGPL-3.0/GPL-3.0 terms continue to apply to the
|
||||
go-ethereum-derived portions provided via github.com/luxfi/geth, and are NOT
|
||||
superseded by the BSD 3-Clause license of this repository. Distribution of
|
||||
compiled Lux Node binaries must honor LGPL-3.0 for the linked go-ethereum
|
||||
library code (source availability for that code and its modifications, and
|
||||
the ability for users to relink against a modified library). The luxfi/geth
|
||||
source, including Lux modifications, is published at
|
||||
https://github.com/luxfi/geth.
|
||||
@@ -1,5 +1,3 @@
|
||||
<p align="center"><img src=".github/hero.svg" alt="node" width="880"></p>
|
||||
|
||||
<div align="center">
|
||||
<img src="resources/LuxLogoRed.png?raw=true">
|
||||
</div>
|
||||
|
||||
+1
-1
@@ -3066,7 +3066,7 @@ This version is backwards compatible to [v1.9.0](https://github.com/luxfi/node/r
|
||||
- Fixed `x/merkledb.ChangeProof#getLargestKey` to correctly handle no changes
|
||||
- Added test for `xvm/txs/executor.SemanticVerifier#verifyFxUsage` with multiple valid fxs
|
||||
- Fixed CPU + bandwidth performance regression during vertex processing
|
||||
- Added example usage of the `/v1/index/X/block` API
|
||||
- Added example usage of the `/ext/index/X/block` API
|
||||
- Reduced the default value of `--consensus-optimal-processing` from `50` to `10`
|
||||
- Updated the year in the license header
|
||||
|
||||
|
||||
+1
-1
@@ -145,7 +145,7 @@ Production implementations live in `lux/crypto/` and `lux/lattice/`. Formal veri
|
||||
| `audits/2025-12-30-other-vms-audit.md` | Secondary VMs |
|
||||
| `audits/2025-12-30-platformvm-audit.md` | PlatformVM (P-Chain) |
|
||||
| `audits/2025-12-30-proposervm-evm-audit.md` | ProposerVM and EVM integration |
|
||||
| `audits/2025-12-30-mpcvm-audit.md` | ThresholdVM (T-Chain) |
|
||||
| `audits/2025-12-30-thresholdvm-audit.md` | ThresholdVM (T-Chain) |
|
||||
| `audits/2025-12-30-warp-audit.md` | Warp cross-chain messaging |
|
||||
| `audits/2025-12-30-zkvm-audit.md` | ZKVM (Z-Chain) |
|
||||
|
||||
|
||||
@@ -73,6 +73,12 @@ tasks:
|
||||
- task: generate-mocks
|
||||
- task: check-clean-branch
|
||||
|
||||
check-generate-protobuf:
|
||||
desc: Checks that generated protobuf is up-to-date (requires a clean git working tree)
|
||||
cmds:
|
||||
- task: generate-protobuf
|
||||
- task: check-clean-branch
|
||||
|
||||
check-go-mod-tidy:
|
||||
desc: Checks that go.mod and go.sum are up-to-date (requires a clean git working tree)
|
||||
cmds:
|
||||
@@ -95,6 +101,10 @@ tasks:
|
||||
- cmd: grep -lr -E '^// Code generated - DO NOT EDIT\.$' tests/load/c | xargs -r rm
|
||||
- cmd: go generate ./tests/load/c/...
|
||||
|
||||
generate-protobuf:
|
||||
desc: Generates protobuf
|
||||
cmd: ./scripts/protobuf_codegen.sh
|
||||
|
||||
ginkgo-build:
|
||||
desc: Runs ginkgo against the current working directory
|
||||
cmd: ./bin/ginkgo build {{.USER_WORKING_DIR}}
|
||||
|
||||
@@ -1,107 +0,0 @@
|
||||
// Copyright (C) 2019-2025, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
package chains
|
||||
|
||||
import (
|
||||
"context"
|
||||
"sync"
|
||||
"testing"
|
||||
|
||||
"github.com/stretchr/testify/require"
|
||||
|
||||
"github.com/luxfi/ids"
|
||||
"github.com/luxfi/log"
|
||||
version "github.com/luxfi/version"
|
||||
"github.com/luxfi/vm/chain"
|
||||
)
|
||||
|
||||
// recordingConnector is a chain.ChainVM that records the version it is handed on
|
||||
// Connected. Every other method comes from the embedded (nil) interface and is
|
||||
// never invoked by the connect path.
|
||||
type recordingConnector struct {
|
||||
chain.ChainVM
|
||||
|
||||
mu sync.Mutex
|
||||
called bool
|
||||
gotVersion *version.Application
|
||||
}
|
||||
|
||||
func (c *recordingConnector) Connected(_ context.Context, _ ids.NodeID, nodeVersion *chain.VersionInfo) error {
|
||||
c.mu.Lock()
|
||||
defer c.mu.Unlock()
|
||||
c.called = true
|
||||
c.gotVersion = nodeVersion
|
||||
return nil
|
||||
}
|
||||
|
||||
func (c *recordingConnector) Disconnected(context.Context, ids.NodeID) error { return nil }
|
||||
|
||||
// TestBlockHandlerConnectedWithVersionDeliversRealVersion is the regression
|
||||
// guard for RED CRITICAL #1 (C-Chain nil-version panic).
|
||||
//
|
||||
// Before the fix blockHandler.Connected forwarded connector.Connected(ctx,
|
||||
// nodeID, nil) — a hardcoded nil version. proposervm promotes Connected to the
|
||||
// inner C-Chain VM (coreth), whose state-sync peer tracker compares peer
|
||||
// versions; a nil version there dereferences nil (version.Application.Compare)
|
||||
// and PANICS a state-syncing node (a fresh join, or a validator rejoining after
|
||||
// falling behind — the launch's core invariant).
|
||||
//
|
||||
// The fix plumbs the REAL peer version through ConnectedWithVersion to the
|
||||
// connector. This test drives the real blockHandler with a real version and
|
||||
// asserts the connector receives that non-nil version — never nil.
|
||||
func TestBlockHandlerConnectedWithVersionDeliversRealVersion(t *testing.T) {
|
||||
require := require.New(t)
|
||||
|
||||
rc := &recordingConnector{}
|
||||
bh := newBlockHandler(
|
||||
nil, // BlockBuilder — unused by the connect path
|
||||
rc, // connector under test
|
||||
log.Noop(), // logger
|
||||
nil, // engine
|
||||
nil, // net
|
||||
nil, // msgCreator
|
||||
ids.Empty, // chainID
|
||||
ids.Empty, // networkID
|
||||
nil, // beacons
|
||||
ids.NodeID{}, // selfNodeID
|
||||
false, // expectsStakedBeacons
|
||||
)
|
||||
|
||||
nodeID := ids.GenerateTestNodeID()
|
||||
peerVersion := &version.Application{Name: "lux", Major: 1, Minor: 36, Patch: 27}
|
||||
|
||||
require.NoError(bh.ConnectedWithVersion(context.Background(), nodeID, peerVersion))
|
||||
|
||||
rc.mu.Lock()
|
||||
defer rc.mu.Unlock()
|
||||
require.True(rc.called, "connector.Connected must be invoked")
|
||||
require.NotNil(rc.gotVersion, "connector must receive a NON-nil version (coreth dereferences it in state-sync)")
|
||||
require.Equal("lux", rc.gotVersion.Name)
|
||||
require.Equal(1, rc.gotVersion.Major)
|
||||
require.Equal(36, rc.gotVersion.Minor)
|
||||
require.Equal(27, rc.gotVersion.Patch)
|
||||
}
|
||||
|
||||
// TestBlockHandlerConnectedDedupsButStillCarriesVersion confirms the version
|
||||
// survives the once-only dedup: the first (versioned) dispatch reaches the
|
||||
// connector; a duplicate is a no-op (not a nil-version overwrite).
|
||||
func TestBlockHandlerConnectedDedupsButStillCarriesVersion(t *testing.T) {
|
||||
require := require.New(t)
|
||||
|
||||
rc := &recordingConnector{}
|
||||
bh := newBlockHandler(nil, rc, log.Noop(), nil, nil, nil, ids.Empty, ids.Empty, nil, ids.NodeID{}, false)
|
||||
|
||||
nodeID := ids.GenerateTestNodeID()
|
||||
peerVersion := &version.Application{Name: "lux", Major: 1, Minor: 36, Patch: 27}
|
||||
|
||||
require.NoError(bh.ConnectedWithVersion(context.Background(), nodeID, peerVersion))
|
||||
// A duplicate connect (e.g. dispatched again per tracked network) must be a
|
||||
// no-op — never a second call that could clobber the stored version with nil.
|
||||
require.NoError(bh.ConnectedWithVersion(context.Background(), nodeID, nil))
|
||||
|
||||
rc.mu.Lock()
|
||||
defer rc.mu.Unlock()
|
||||
require.NotNil(rc.gotVersion, "the deduped duplicate must not overwrite the real version with nil")
|
||||
require.Equal(27, rc.gotVersion.Patch)
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -1,584 +0,0 @@
|
||||
// Copyright (C) 2019-2026, Lux Industries, Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
// bootstrap_trust.go — the SEPARATE trust object for INITIAL SYNC, decomplected from
|
||||
// consensus finality.
|
||||
//
|
||||
// The mass-recovery DEADLOCK this fixes: the prior FrontierTip required a ⅔-by-stake quorum
|
||||
// of the CURRENT total validator set to be CONNECTED before it would name a sync frontier.
|
||||
// When the recovery TARGETS are themselves validators (a node that crashed IS one of the 5),
|
||||
// taking them down drops connected stake below ⅔ of the whole set — so on a network of 5
|
||||
// equal-weight validators, losing 2 leaves 3 (60% < ⅔) and NO node can ever name a frontier
|
||||
// to recover from. Bootstrap trust was braided into consensus finality, and finality's ⅔ rule
|
||||
// is mathematically unsatisfiable during a mass outage.
|
||||
//
|
||||
// The fix is a type split, NOT a renamed threshold. FinalityQuorum decides FINALITY
|
||||
// (> ⅔ of CURRENT stake — UNCHANGED). BootstrapTrust decides whether a fetched frontier is
|
||||
// SAFE TO BEGIN SYNC FROM: a quorum of AUTHENTICATED CONFIGURED beacons that RESPOND, gated by
|
||||
// a response FLOOR (MinResponses) and an agreement threshold over the RESPONDERS (not over the
|
||||
// whole set). 3 of 5 reachable beacons all agreeing is a valid sync anchor even though 3 of 5
|
||||
// stake is not a finalizing supermajority. The two decisions have different threat models and
|
||||
// are different objects.
|
||||
package chains
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/binary"
|
||||
"errors"
|
||||
"fmt"
|
||||
"math"
|
||||
"math/bits"
|
||||
"sort"
|
||||
"time"
|
||||
|
||||
consensusconfig "github.com/luxfi/consensus/config"
|
||||
"github.com/luxfi/ids"
|
||||
)
|
||||
|
||||
// BootstrapTrust is not a consensus-finality oracle.
|
||||
// It selects a weak-subjective sync frontier from authenticated configured beacons.
|
||||
// Live block acceptance remains governed exclusively by FinalityQuorum.
|
||||
type BootstrapTrust interface {
|
||||
// AcceptsFrontier returns the block an empty/behind node may BEGIN SYNCING FROM, selected
|
||||
// from the authenticated configured beacons' frontier replies — or an error
|
||||
// (ErrInsufficientBootstrapResponses / ErrNoBootstrapQuorum) when no trusted frontier can be
|
||||
// named this round. The returned Frontier is a sync ANCHOR, never a consensus certificate
|
||||
// (see the type comment): the node must still re-execute every block it descends to before
|
||||
// re-entering live consensus, where FinalityQuorum alone governs acceptance.
|
||||
AcceptsFrontier(ctx context.Context, replies []BeaconReply) (*Frontier, error)
|
||||
}
|
||||
|
||||
// FinalityQuorum decides FINALITY: whether a weight is a finalizing supermajority (> ⅔) of the
|
||||
// CURRENT validator set. This is the live-consensus rule; bootstrap does NOT change it. It is a
|
||||
// SEPARATE named type from BootstrapTrust precisely so the distinction is explicit and testable:
|
||||
// a frontier that AcceptsFrontier admits is "safe to sync from", and in general it does NOT
|
||||
// satisfy HasFinality (3 of 5 responders is a valid sync anchor; 3 of 5 stake is not finality).
|
||||
type FinalityQuorum interface {
|
||||
HasFinality(weight, total StakeWeight) bool
|
||||
}
|
||||
|
||||
// StakeWeight is validator stake in the units the validator manager reports (Weight/Light).
|
||||
type StakeWeight = uint64
|
||||
|
||||
// twoThirdsFinality is the production FinalityQuorum: > ⅔ of the CURRENT total stake, exactly
|
||||
// the rule the live cert-gate uses (consensusconfig.TwoThirdsStakeFloor). Defined here only to
|
||||
// give the live rule a name to CONTRAST bootstrap trust against — it is not wired into the live
|
||||
// path (that already enforces ⅔ inside consensus), and bootstrap never calls it to ACCEPT.
|
||||
type twoThirdsFinality struct{}
|
||||
|
||||
func (twoThirdsFinality) HasFinality(weight, total StakeWeight) bool {
|
||||
return weight > consensusconfig.TwoThirdsStakeFloor(total)
|
||||
}
|
||||
|
||||
// DefaultFinalityQuorum returns the live ⅔-of-current-stake finality rule — the thing bootstrap
|
||||
// trust is explicitly NOT. Used by the test suite to prove a bootstrap-accepted frontier does
|
||||
// not constitute finality.
|
||||
func DefaultFinalityQuorum() FinalityQuorum { return twoThirdsFinality{} }
|
||||
|
||||
var (
|
||||
// ErrInsufficientBootstrapResponses: fewer than MinResponses configured beacons answered.
|
||||
// Not a partition-capture-safe quorum — the node must keep waiting for more beacons (or use
|
||||
// an operator checkpoint), never sync from the captured few. INVARIANT 2's response floor.
|
||||
ErrInsufficientBootstrapResponses = errors.New("bootstrap: insufficient configured-beacon responses")
|
||||
// ErrNoBootstrapQuorum: enough beacons responded, but no block clears the agreement threshold
|
||||
// over the responders (a genuine partition, or a transient bleeding-edge split the loop retries).
|
||||
ErrNoBootstrapQuorum = errors.New("bootstrap: no responder-agreed frontier")
|
||||
)
|
||||
|
||||
// BeaconReply is one authenticated configured beacon's report of its accepted frontier tip
|
||||
// during initial sync. NodeID is authenticated at the transport handshake (a peer cannot forge
|
||||
// another's identity); Weight is the beacon's CONFIGURED stake from the trust anchor — NOT a
|
||||
// self-reported value. A reply whose NodeID is not in the policy's TrustedBeacons is ignored.
|
||||
type BeaconReply struct {
|
||||
NodeID ids.NodeID
|
||||
Tip ids.ID
|
||||
Weight StakeWeight
|
||||
}
|
||||
|
||||
// Frontier is the weak-subjective sync anchor BootstrapTrust selects: the block a node descends
|
||||
// to and re-executes. It is NOT a consensus certificate (see BootstrapTrust). Height is the
|
||||
// tallied height when named via the ancestor-tolerant path, and 0 (unknown) when named via the
|
||||
// exact fast path before any ancestry fetch — the sync loop uses ID; Height is diagnostic.
|
||||
type Frontier struct {
|
||||
ID ids.ID
|
||||
Height uint64
|
||||
Weight StakeWeight // responder stake whose accepted chain contains this block
|
||||
Responders int // distinct configured beacons that backed it
|
||||
FromCheckpoint bool // selected from an operator checkpoint (too few beacons responded)
|
||||
}
|
||||
|
||||
// BlockRef is a parsed block's CONTENT-ADDRESSED identity (id, height, parent) — the only thing
|
||||
// the ancestor-tolerant tally needs. Decouples the policy from the VM/block types.
|
||||
type BlockRef struct {
|
||||
ID ids.ID
|
||||
Height uint64
|
||||
Parent ids.ID
|
||||
}
|
||||
|
||||
// AncestrySource resolves a tip's CONTENT-ADDRESSED ancestry for the ancestor-tolerant tally —
|
||||
// the SAME parent-linked descent the sync loop trusts. Injected so the trust DECISION (which
|
||||
// beacons count, the response floor, the agreement threshold) stays separate from the transport.
|
||||
type AncestrySource interface {
|
||||
// Ancestry returns up to max blocks ending at tip, parsed to (id, height, parent). An empty
|
||||
// result (no error) means the tip's ancestry was not served — that anchor contributes nothing.
|
||||
Ancestry(ctx context.Context, tip ids.ID, max int) ([]BlockRef, error)
|
||||
}
|
||||
|
||||
// Checkpoint is an operator-pinned (id, height) the recovering node may anchor to when too few
|
||||
// beacons respond to form a quorum — the EXPLICIT override for INVARIANT 2's "1 of N reachable"
|
||||
// case. Absent (nil) ⇒ the default policy REJECTS rather than trusting a captured minority.
|
||||
//
|
||||
// INVARIANT 4 (a checkpoint is a SIGNED weak-subjectivity anchor, not a bare config value): the
|
||||
// checkpoint carries a cryptographic Signature by the configured checkpoint AUTHORITY over its
|
||||
// (id, height), and AcceptsFrontier trusts it ONLY when CheckpointVerifier authenticates that
|
||||
// signature. A (id,height) present in a flag/config but UNSIGNED — or signed by a non-authority key
|
||||
// — is REJECTED (fail closed). This is the crucial hardening: the checkpoint is the one path that
|
||||
// bypasses the beacon quorum, so a compromised flag must NOT be able to inject a false sync anchor
|
||||
// without ALSO forging the authority's signature. It is a cryptographic vouch, NEVER a ⅔-live-stake
|
||||
// tally (that conflation is the very deadlock BootstrapTrust exists to avoid).
|
||||
type Checkpoint struct {
|
||||
ID ids.ID
|
||||
Height uint64
|
||||
// Signature is the checkpoint authority's signature over this checkpoint's canonical (id,height)
|
||||
// bytes. Verified by CheckpointVerifier before the anchor is trusted; an empty signature is
|
||||
// never accepted.
|
||||
Signature []byte
|
||||
}
|
||||
|
||||
// CheckpointVerifier authenticates a Checkpoint's Signature against the configured checkpoint
|
||||
// AUTHORITY key(s). The node injects a real implementation backed by a PROVEN primitive (Ed25519 /
|
||||
// BLS — never custom crypto); the policy stays free of any crypto dependency, exactly like
|
||||
// AncestrySource and heightOf. A nil verifier means no signed anchor is configured, so any
|
||||
// Checkpoint is untrusted and the below-floor case fails closed.
|
||||
type CheckpointVerifier interface {
|
||||
// VerifyCheckpoint reports whether sig is a valid signature over (id, height) by the configured
|
||||
// checkpoint authority. It MUST reject an empty signature and be signature-safe (constant-time
|
||||
// compare on the primitive). It is the sole authority on whether a pinned anchor may be trusted.
|
||||
VerifyCheckpoint(id ids.ID, height uint64, sig []byte) bool
|
||||
}
|
||||
|
||||
// CanonicalCheckpointMessage is the exact byte string a checkpoint authority signs and
|
||||
// CheckpointVerifier authenticates: a domain-separated, fixed-layout encoding of (id, height) so a
|
||||
// signature can never be transplanted from another context. 8-byte big-endian height after the
|
||||
// 32-byte id, under a distinct domain tag.
|
||||
func CanonicalCheckpointMessage(id ids.ID, height uint64) []byte {
|
||||
const domain = "lux-bootstrap-checkpoint-v1\x00"
|
||||
msg := make([]byte, 0, len(domain)+len(id)+8)
|
||||
msg = append(msg, domain...)
|
||||
msg = append(msg, id[:]...)
|
||||
var h [8]byte
|
||||
binary.BigEndian.PutUint64(h[:], height)
|
||||
return append(msg, h[:]...)
|
||||
}
|
||||
|
||||
// Ratio is an exact rational threshold (e.g. 2/3, 3/4). A value clears it iff
|
||||
// value > floorOf(whole) — strictly greater, matching the consensus ⅔ floor's semantics.
|
||||
type Ratio struct{ Num, Den uint64 }
|
||||
|
||||
// floorOf returns ⌊whole · Num / Den⌋ without floating point, overflow-safe for the sub-unity
|
||||
// thresholds used here. Ratio{2,3}.floorOf(w) == consensusconfig.TwoThirdsStakeFloor(w) exactly,
|
||||
// so the responder-⅔ agreement reuses the same strict-greater floor the live rule uses.
|
||||
func (r Ratio) floorOf(whole uint64) uint64 {
|
||||
if r.Den == 0 {
|
||||
return whole // degenerate guard; constructors always set a real ratio
|
||||
}
|
||||
hi, lo := bits.Mul64(whole, r.Num)
|
||||
if hi >= r.Den {
|
||||
return math.MaxUint64 // Num ≥ Den: not a sub-unity threshold — nothing can exceed it
|
||||
}
|
||||
q, _ := bits.Div64(hi, lo, r.Den)
|
||||
return q
|
||||
}
|
||||
|
||||
// BootstrapPolicy is the default BootstrapTrust: a CONFIGURED-BEACON quorum with a response
|
||||
// FLOOR and an agreement threshold over the RESPONDERS — a SEPARATE object from FinalityQuorum
|
||||
// with a SEPARATE threat model. It does NOT pass "reachable stake" into the ⅔-of-current-stake
|
||||
// finality rule (that conflation IS the mass-recovery deadlock). It reuses the ancestor-tolerant
|
||||
// common-ancestor tally only for HOW to find the agreed frontier; the ACCEPTANCE gate is the
|
||||
// response floor + responder agreement here.
|
||||
//
|
||||
// The three invariants:
|
||||
// - INVARIANT 1 (non-circular beacon eligibility): only NodeIDs in TrustedBeacons count, and
|
||||
// TrustedBeacons comes from the configured/checkpointed/genesis anchor — NEVER peer
|
||||
// self-report. A recovering node never lets arbitrary peers define who is a beacon.
|
||||
// - INVARIANT 2 (a floor prevents partition-capture): MinResponses authenticated beacons must
|
||||
// respond before any frontier is named; an attacker who partitions the node down to a few
|
||||
// beacons cannot capture the frontier. Below the floor, REJECT (or use Checkpoint).
|
||||
// - INVARIANT 3 (acceptance ≠ finality): the named Frontier is "safe to begin sync from", not
|
||||
// finalized. The node independently re-executes the descent before re-entering consensus.
|
||||
type BootstrapPolicy struct {
|
||||
// TrustedBeacons is the trust anchor: configured-beacon NodeID → configured stake (INVARIANT
|
||||
// 1). Resolved from --bootstrap-nodes / a finalized P-chain checkpoint / the genesis set —
|
||||
// never from peer self-report.
|
||||
TrustedBeacons map[ids.NodeID]StakeWeight
|
||||
// AgreementThreshold is the fraction of the RESPONDER weight a named block must exceed
|
||||
// (default 2/3). Over RESPONDERS, not the whole set — that is what permits mass recovery.
|
||||
AgreementThreshold Ratio
|
||||
// MinResponses is the FLOOR on distinct configured-beacon responders (INVARIANT 2). Default:
|
||||
// a MAJORITY of the configured set (the largest floor that still lets a node recover when a
|
||||
// minority of validators is down). Capped at the set size.
|
||||
MinResponses int
|
||||
// MinResponseWeight is an OPTIONAL floor on the total responder weight (0 ⇒ disabled).
|
||||
MinResponseWeight StakeWeight
|
||||
// MinResponders is the minimum DISTINCT beacons that must back a NAMED block (default 2), so a
|
||||
// single beacon cannot alone name the frontier. Capped at the responder count.
|
||||
MinResponders int
|
||||
// MinFrontierHeight is the node's current last-accepted height. The ANCESTOR-TOLERANT path
|
||||
// names only a block STRICTLY ABOVE it — a frontier genuinely AHEAD. A common ancestor BELOW it
|
||||
// is history the node has (a partition above, not a frontier ahead). A block AT exactly this
|
||||
// height is ALSO not named here (the M1 eclipse-stale fix): an eclipse can throttle the honest
|
||||
// ahead-tips below the ⅔ naming threshold while the node's OWN height accrues ⅔ as their shared
|
||||
// ANCESTOR — naming it would go Ready stale. Excluding own height routes that case to CaughtUp,
|
||||
// which distinguishes a legit all-at-N fleet from an eclipse with ahead-tips the node lacks. So
|
||||
// nothing at or below own height is named (→ ErrNoBootstrapQuorum, fail safe), never a
|
||||
// false-complete at the stale height. The exact fast path is exempt: a tip a responder
|
||||
// supermajority ACTIVELY reports is a real frontier even at own height (a genuinely fresh
|
||||
// network, or a fleet unanimously AT the tip).
|
||||
MinFrontierHeight uint64
|
||||
// Checkpoint is the OPTIONAL operator override for the below-floor case (INVARIANT 2). nil ⇒
|
||||
// reject below the floor. When set, it is trusted ONLY if CheckpointVerifier authenticates its
|
||||
// signature (INVARIANT 4).
|
||||
Checkpoint *Checkpoint
|
||||
// CheckpointVerifier authenticates the Checkpoint's authority signature (INVARIANT 4). nil ⇒ a
|
||||
// configured Checkpoint is NOT trusted (fail closed) — a bare (id,height) is never enough.
|
||||
CheckpointVerifier CheckpointVerifier
|
||||
// NamingWindow bounds the ancestry fetched per anchor; MaxAnchors bounds how many distinct
|
||||
// reported tips are resolved. Both default to the package constants when zero.
|
||||
NamingWindow int
|
||||
MaxAnchors int
|
||||
// NamingTimeout TOTAL-bounds the ancestor-tolerant resolution (all anchor fetches combined) so
|
||||
// a partition that ANSWERS the frontier query but WITHHOLDS ancestry cannot make the decision
|
||||
// hang — it returns what it found (or nothing → ErrNoBootstrapQuorum) and the caller's bounded
|
||||
// retry tries a fresh sample next round. Zero ⇒ the package default.
|
||||
NamingTimeout time.Duration
|
||||
// Source resolves content-addressed ancestry for the ancestor-tolerant tally. When nil, the
|
||||
// policy decides on the exact fast path alone (no split resolution).
|
||||
Source AncestrySource
|
||||
}
|
||||
|
||||
// compile-time: the default policy IS a BootstrapTrust.
|
||||
var _ BootstrapTrust = (*BootstrapPolicy)(nil)
|
||||
|
||||
func (p *BootstrapPolicy) effectiveMinResponses() int {
|
||||
n := len(p.TrustedBeacons)
|
||||
if p.MinResponses > 0 {
|
||||
if p.MinResponses > n {
|
||||
return n
|
||||
}
|
||||
return p.MinResponses
|
||||
}
|
||||
return n/2 + 1 // default: a MAJORITY of the configured beacon set
|
||||
}
|
||||
|
||||
func (p *BootstrapPolicy) effectiveAgreement() Ratio {
|
||||
if p.AgreementThreshold.Den == 0 {
|
||||
return Ratio{Num: 2, Den: 3} // default: ⅔ of the RESPONDERS
|
||||
}
|
||||
return p.AgreementThreshold
|
||||
}
|
||||
|
||||
func (p *BootstrapPolicy) effectiveMinResponders(responders int) int {
|
||||
r := p.MinResponders
|
||||
if r <= 0 {
|
||||
r = bootstrapMinAgreeingBeacons // default 2
|
||||
}
|
||||
if r > responders {
|
||||
r = responders
|
||||
}
|
||||
if r < 1 {
|
||||
r = 1
|
||||
}
|
||||
return r
|
||||
}
|
||||
|
||||
func (p *BootstrapPolicy) namingWindow() int {
|
||||
if p.NamingWindow > 0 {
|
||||
return p.NamingWindow
|
||||
}
|
||||
return bootstrapNamingWindow
|
||||
}
|
||||
|
||||
func (p *BootstrapPolicy) maxAnchors() int {
|
||||
if p.MaxAnchors > 0 {
|
||||
return p.MaxAnchors
|
||||
}
|
||||
return maxNamingAnchors
|
||||
}
|
||||
|
||||
func (p *BootstrapPolicy) namingTimeout() time.Duration {
|
||||
if p.NamingTimeout > 0 {
|
||||
return p.NamingTimeout
|
||||
}
|
||||
return bootstrapNamingTimeout
|
||||
}
|
||||
|
||||
// tallyResponders applies INVARIANT 1 (only CONFIGURED beacons count, deduplicated by NodeID — a
|
||||
// reply from a peer not in TrustedBeacons, a repeat, or an empty tip is dropped) and returns the
|
||||
// distinct responder count + total responder stake plus the per-tip stake / voter maps the naming
|
||||
// tally walks. The authenticated NodeID (transport handshake) is what makes "configured"
|
||||
// unforgeable. Shared by AcceptsFrontier (which names a frontier AHEAD) and CaughtUp (which
|
||||
// concludes NONE is ahead) so both judge the IDENTICAL responder set under the SAME eligibility
|
||||
// rule — the eligibility decision lives in exactly one place.
|
||||
func (p *BootstrapPolicy) tallyResponders(replies []BeaconReply) (responders int, responderWeight StakeWeight, stakeOnTip map[ids.ID]StakeWeight, votersOf map[ids.ID]map[ids.NodeID]struct{}) {
|
||||
seen := make(map[ids.NodeID]struct{}, len(replies))
|
||||
stakeOnTip = make(map[ids.ID]StakeWeight)
|
||||
votersOf = make(map[ids.ID]map[ids.NodeID]struct{})
|
||||
for _, r := range replies {
|
||||
w, ok := p.TrustedBeacons[r.NodeID]
|
||||
if !ok || r.Tip == ids.Empty {
|
||||
continue
|
||||
}
|
||||
if _, dup := seen[r.NodeID]; dup {
|
||||
continue
|
||||
}
|
||||
seen[r.NodeID] = struct{}{}
|
||||
responders++
|
||||
responderWeight += w
|
||||
stakeOnTip[r.Tip] += w
|
||||
if votersOf[r.Tip] == nil {
|
||||
votersOf[r.Tip] = make(map[ids.NodeID]struct{})
|
||||
}
|
||||
votersOf[r.Tip][r.NodeID] = struct{}{}
|
||||
}
|
||||
return responders, responderWeight, stakeOnTip, votersOf
|
||||
}
|
||||
|
||||
// floorMet reports whether the responder set clears INVARIANT 2's partition-capture FLOOR: at
|
||||
// least MinResponses distinct configured beacons AND (when MinResponseWeight is configured) at
|
||||
// least that much total responder stake. AcceptsFrontier gates NAMING a frontier on it and
|
||||
// CaughtUp gates concluding NONE-AHEAD on the SAME floor — so an eclipse that suppresses the
|
||||
// honest ahead-nodes to fake EITHER outcome must drop the responder set below it and fail safe.
|
||||
func (p *BootstrapPolicy) floorMet(responders int, responderWeight StakeWeight) bool {
|
||||
if responders < p.effectiveMinResponses() {
|
||||
return false
|
||||
}
|
||||
if p.MinResponseWeight > 0 && responderWeight < p.MinResponseWeight {
|
||||
return false
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// AcceptsFrontier implements BootstrapTrust. It (1) keeps ONLY configured-beacon replies
|
||||
// (INVARIANT 1, tallyResponders), (2) enforces the MinResponses / MinResponseWeight floor or falls
|
||||
// back to the operator checkpoint (INVARIANT 2, floorMet), then (3) names the highest block a
|
||||
// responder supermajority shares via the ancestor-tolerant tally. It never consults the
|
||||
// ⅔-of-current-stake finality rule (INVARIANT 3): the decision is the response floor + responder
|
||||
// agreement, a separate threat model.
|
||||
func (p *BootstrapPolicy) AcceptsFrontier(ctx context.Context, replies []BeaconReply) (*Frontier, error) {
|
||||
responders, responderWeight, stakeOnTip, votersOf := p.tallyResponders(replies)
|
||||
|
||||
// INVARIANT 2: the response FLOOR prevents partition-capture. Below MinResponses (or below
|
||||
// MinResponseWeight) the node has not heard from enough authenticated beacons to trust ANY
|
||||
// frontier — an attacker may have partitioned it down to a captured few. REJECT, unless the
|
||||
// operator explicitly pinned a checkpoint to anchor from.
|
||||
if !p.floorMet(responders, responderWeight) {
|
||||
if p.Checkpoint != nil {
|
||||
// INVARIANT 4: the checkpoint bypasses the beacon quorum, so trust it ONLY when the
|
||||
// configured authority SIGNED this exact (id,height). A checkpoint present in config but
|
||||
// unsigned, or signed by a non-authority key, is REJECTED (fail closed) — a compromised
|
||||
// flag cannot inject a false sync anchor without also forging the authority's signature.
|
||||
if p.CheckpointVerifier == nil || len(p.Checkpoint.Signature) == 0 ||
|
||||
!p.CheckpointVerifier.VerifyCheckpoint(p.Checkpoint.ID, p.Checkpoint.Height, p.Checkpoint.Signature) {
|
||||
return nil, fmt.Errorf("%w: a checkpoint is pinned but its authority signature did not verify",
|
||||
ErrInsufficientBootstrapResponses)
|
||||
}
|
||||
return &Frontier{
|
||||
ID: p.Checkpoint.ID,
|
||||
Height: p.Checkpoint.Height,
|
||||
Responders: responders,
|
||||
FromCheckpoint: true,
|
||||
}, nil
|
||||
}
|
||||
return nil, fmt.Errorf("%w: %d configured beacons responded (weight %d), need %d",
|
||||
ErrInsufficientBootstrapResponses, responders, responderWeight, p.effectiveMinResponses())
|
||||
}
|
||||
|
||||
// The agreement threshold is over the RESPONDERS, not the whole configured set — this is what
|
||||
// lets a node recover when validators are down (3 of 5 reachable, all 3 agreeing, is a valid
|
||||
// sync anchor). The ⅔-of-current-stake finality rule is never used here (INVARIANT 3).
|
||||
floor := p.effectiveAgreement().floorOf(responderWeight)
|
||||
required := p.effectiveMinResponders(responders)
|
||||
|
||||
id, height, weight, ok := p.nameFrontier(ctx, stakeOnTip, votersOf, floor, required)
|
||||
if !ok {
|
||||
return nil, ErrNoBootstrapQuorum
|
||||
}
|
||||
return &Frontier{ID: id, Height: height, Weight: weight, Responders: responders}, nil
|
||||
}
|
||||
|
||||
// CaughtUp reports whether the responder set PROVES the node is already AT OR ABOVE the network
|
||||
// frontier — the dual of AcceptsFrontier ("nobody is ahead" vs "here is the block ahead to sync
|
||||
// to"). It is the go-live path for a TIP-HOLDER on a mixed-height co-restart: when producers
|
||||
// restart together the responder set SPLITS (the tip-holders are exactly half — below the ⅔
|
||||
// naming threshold), so AcceptsFrontier names NOTHING (ErrNoBootstrapQuorum), yet the node is
|
||||
// plainly not behind. Without this determination such a producer fails safe DOWN at its own tip —
|
||||
// the exact OPPOSITE of the stale-go-live bug, and just as wrong. THREE conditions, ALL required:
|
||||
//
|
||||
// - (a) the SAME response FLOOR AcceptsFrontier uses is met (floorMet: MinResponses distinct
|
||||
// beacons AND MinResponseWeight stake-majority). An eclipse that hides the higher (real) tips
|
||||
// to fake caught-up must SUPPRESS the ahead-nodes' replies, dropping the responder set below
|
||||
// the floor → NOT caught up, fail safe. No partition-capture: faking caught-up costs the same
|
||||
// stake-majority of honest beacons that faking a NAMED frontier does.
|
||||
// - (b) every responder's reported ACCEPTED tip is at height ≤ lastAccepted. A genuinely STALE
|
||||
// node has at least one honest responder AHEAD (height > lastAccepted) → NOT caught up: it
|
||||
// still syncs, so the stale-go-live bug stays fixed. (GetAcceptedFrontier reports a beacon's
|
||||
// last-ACCEPTED block, so an un-finalized N+1 a producer is merely processing is never reported
|
||||
// — the ±1 pending-tip skew cannot fake "ahead", and a producer one ACCEPTED block ahead
|
||||
// correctly defeats caught-up so the node syncs that block.)
|
||||
// - (c) the node has ACCEPTED every reported tip — heightOf returns ok ONLY for a block on the
|
||||
// node's FINALIZED chain, so a tip the node lacks OR merely holds-in-store-but-has-not-accepted
|
||||
// (someone genuinely ahead, a gossiped-ahead block, or a same-height sibling/fork it never
|
||||
// finalized) makes the conclusion fail. The node declares caught-up only to blocks it ACCEPTED.
|
||||
//
|
||||
// heightOf resolves a tip's height from the node's ACCEPTED chain (ok=false when the tip is not
|
||||
// accepted — including a block merely PRESENT in the store but unaccepted, the luxd-2 freeze case),
|
||||
// injected so the trust DECISION stays free of any VM/block dependency — the same separation as
|
||||
// AncestrySource. It is NEVER a network fetch: an unaccepted/absent tip simply makes the node
|
||||
// not-caught-up (the safe direction — it syncs). Because (c) requires the node to have ACCEPTED
|
||||
// every reported tip, the heights (b) compares are the blocks' canonical (content-addressed)
|
||||
// heights read from the finalized chain — store presence can never fake "caught up".
|
||||
func (p *BootstrapPolicy) CaughtUp(replies []BeaconReply, lastAccepted uint64, heightOf func(ids.ID) (uint64, bool)) bool {
|
||||
responders, responderWeight, stakeOnTip, _ := p.tallyResponders(replies)
|
||||
if !p.floorMet(responders, responderWeight) {
|
||||
return false // (a) below the floor — an eclipse/partition can never fake caught-up
|
||||
}
|
||||
sawTip := false
|
||||
for tip := range stakeOnTip {
|
||||
sawTip = true
|
||||
h, held := heightOf(tip)
|
||||
if !held || h > lastAccepted {
|
||||
return false // (c) a tip we do not hold, or (b) a responder ahead → NOT caught up
|
||||
}
|
||||
}
|
||||
return sawTip // ≥1 responder tip evaluated (floor already implies this; guards an empty set)
|
||||
}
|
||||
|
||||
// nameFrontier finds the block a responder supermajority shares — by CONTENT, reusing the
|
||||
// parent-link descent the sync loop trusts (HOW to find the agreed frontier; the ACCEPTANCE gate
|
||||
// already passed in AcceptsFrontier). A beacon reporting tip T vouches for every ANCESTOR of T,
|
||||
// so the named frontier is the HIGHEST block whose backing stake exceeds floor (the responder
|
||||
// agreement threshold) with ≥ required distinct voters.
|
||||
//
|
||||
// - EXACT FAST PATH: if a single reported tip clears the floor outright, name it with NO
|
||||
// ancestry fetch (the whole responding quorum already agrees on the same tip). Exempt from
|
||||
// MinFrontierHeight: an actively-reported tip is a real frontier even when low.
|
||||
// - ANCESTOR-TOLERANT PATH: otherwise, fetch the distinct tips' ancestries into ONE union index
|
||||
// and globally credit each tip's stake to every block on its content-addressed chain. The
|
||||
// highest block clearing the floor AND at a height STRICTLY ABOVE MinFrontierHeight (a frontier
|
||||
// genuinely ahead — never the node's own height, which an eclipse could over-credit as a shared
|
||||
// ancestor; that routes to CaughtUp) is named. A sibling split converges to the common committed
|
||||
// ancestor; a partition that shares nothing ⅔-backed names nothing (→ fail safe).
|
||||
//
|
||||
// C1 (a forged chain finalizes ZERO) is preserved: a block is credited a beacon's stake only when
|
||||
// that beacon's tip lies on the block's CONTENT-ADDRESSED descendant chain (parent ids are bound
|
||||
// to block content), so a peer cannot fake linkage to over-credit; a block is named only with
|
||||
// backing > ⅔ of the responder weight; a minority (< ⅓) forged tip can only RATIFY real ancestors
|
||||
// it builds on, never name itself or raise the named height above the honest common block.
|
||||
func (p *BootstrapPolicy) nameFrontier(ctx context.Context, stakeOnTip map[ids.ID]StakeWeight, votersOf map[ids.ID]map[ids.NodeID]struct{}, floor StakeWeight, required int) (ids.ID, uint64, StakeWeight, bool) {
|
||||
// EXACT fast path: a single reported tip already clears the floor — name it, no fetch.
|
||||
for tip, st := range stakeOnTip {
|
||||
if st > floor && len(votersOf[tip]) >= required {
|
||||
return tip, 0, st, true
|
||||
}
|
||||
}
|
||||
if p.Source == nil {
|
||||
return ids.Empty, 0, 0, false
|
||||
}
|
||||
|
||||
// TOTAL-bound all anchor fetches so a partition that answers the frontier query but withholds
|
||||
// ancestry cannot hang the decision — the caller's bounded retry handles it next round.
|
||||
ctx, cancel := context.WithTimeout(ctx, p.namingTimeout())
|
||||
defer cancel()
|
||||
|
||||
// Build ONE union index from the distinct reported tips' ancestries (most stake first; skip a
|
||||
// tip already present from an earlier fetch — a nested tip covers its ancestors). Bounded by
|
||||
// MaxAnchors × NamingWindow blocks, so a Byzantine swarm reporting many forged tips cannot
|
||||
// induce unbounded work.
|
||||
index := make(map[ids.ID]BlockRef)
|
||||
fetches := 0
|
||||
for _, tip := range sortedByStakeDesc(stakeOnTip) {
|
||||
if _, have := index[tip]; have {
|
||||
continue
|
||||
}
|
||||
if fetches >= p.maxAnchors() {
|
||||
break
|
||||
}
|
||||
fetches++
|
||||
refs, err := p.Source.Ancestry(ctx, tip, p.namingWindow())
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
for _, ref := range refs {
|
||||
if _, ok := index[ref.ID]; !ok {
|
||||
index[ref.ID] = ref
|
||||
}
|
||||
}
|
||||
}
|
||||
if len(index) == 0 {
|
||||
return ids.Empty, 0, 0, false
|
||||
}
|
||||
|
||||
// Global credit: each reported tip vouches for every block on its content-addressed ancestry.
|
||||
// The running backing at block B = the responder stake whose accepted chain contains B.
|
||||
backing := make(map[ids.ID]StakeWeight)
|
||||
voters := make(map[ids.ID]map[ids.NodeID]struct{})
|
||||
for tip, st := range stakeOnTip {
|
||||
cur := tip
|
||||
for {
|
||||
ref, ok := index[cur]
|
||||
if !ok {
|
||||
break // the served ancestry does not extend further down (or this tip was unserved)
|
||||
}
|
||||
backing[cur] += st
|
||||
if voters[cur] == nil {
|
||||
voters[cur] = make(map[ids.NodeID]struct{})
|
||||
}
|
||||
for v := range votersOf[tip] {
|
||||
voters[cur][v] = struct{}{}
|
||||
}
|
||||
if ref.Parent == ids.Empty {
|
||||
break
|
||||
}
|
||||
cur = ref.Parent
|
||||
}
|
||||
}
|
||||
|
||||
// Name the HIGHEST block clearing the floor with ≥ required distinct voters, at a height
|
||||
// STRICTLY ABOVE MinFrontierHeight — a genuine frontier AHEAD. A block AT the node's own
|
||||
// last-accepted height is NOT named here (it is history the node already holds, reachable as a
|
||||
// ⅔-backed ANCESTOR of higher tips an eclipse can suppress below the naming threshold — the M1
|
||||
// stale-go-live path): that case routes to CaughtUp, which alone can distinguish a legit
|
||||
// all-at-N fleet (→ Ready at N) from an eclipse with ahead-tips the node lacks (→ sync). A block
|
||||
// BELOW own height is a partition diverged beneath the node. Both fail safe, never false-complete.
|
||||
var bestID ids.ID
|
||||
var bestHeight, bestStake uint64
|
||||
found := false
|
||||
for id, st := range backing {
|
||||
ref := index[id]
|
||||
if st <= floor || len(voters[id]) < required || ref.Height <= p.MinFrontierHeight {
|
||||
continue
|
||||
}
|
||||
if !found || ref.Height > bestHeight || (ref.Height == bestHeight && st > bestStake) {
|
||||
bestID, bestHeight, bestStake, found = id, ref.Height, st, true
|
||||
}
|
||||
}
|
||||
return bestID, bestHeight, bestStake, found
|
||||
}
|
||||
|
||||
// sortedByStakeDesc returns the reported tips most-stake-first (stable id tiebreak) — the order
|
||||
// the ancestor-tolerant tally fetches anchors in, so the well-supported honest tips are covered
|
||||
// first and a forged low-stake outlier swarm falls outside the anchor cap.
|
||||
func sortedByStakeDesc(stakeOnTip map[ids.ID]StakeWeight) []ids.ID {
|
||||
tips := make([]ids.ID, 0, len(stakeOnTip))
|
||||
for t := range stakeOnTip {
|
||||
tips = append(tips, t)
|
||||
}
|
||||
sort.Slice(tips, func(i, j int) bool {
|
||||
if stakeOnTip[tips[i]] != stakeOnTip[tips[j]] {
|
||||
return stakeOnTip[tips[i]] > stakeOnTip[tips[j]]
|
||||
}
|
||||
return bytes.Compare(tips[i][:], tips[j][:]) < 0
|
||||
})
|
||||
return tips
|
||||
}
|
||||
@@ -1,918 +0,0 @@
|
||||
// Copyright (C) 2019-2026, Lux Industries, Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
// bootstrap_trust_test.go — the A–G proof matrix for the BootstrapTrust policy: the SEPARATE
|
||||
// trust object (distinct from consensus finality) that lets a node recover when validators are
|
||||
// down (mass recovery) while refusing partition-capture and never weakening finality.
|
||||
//
|
||||
// Most cases test the POLICY decision (AcceptsFrontier) directly — deterministic, no network
|
||||
// timing — since that IS the acceptance gate the owner specified. The mass-recovery success (A)
|
||||
// and the global-tally height-floor guard also run the FULL fetch+execute loop over the real
|
||||
// transport to prove the node converges (or fails safe) end to end. Each is load-bearing: revert
|
||||
// the response-floor policy to the prior ⅔-of-current-total-stake gate and A deadlocks; drop the
|
||||
// configured-beacon filter and D/E capture; drop the MinFrontierHeight floor and the shared-
|
||||
// genesis fork false-completes.
|
||||
package chains
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/ed25519"
|
||||
"testing"
|
||||
|
||||
"github.com/stretchr/testify/require"
|
||||
|
||||
consensusconfig "github.com/luxfi/consensus/config"
|
||||
chainbootstrap "github.com/luxfi/consensus/engine/chain/bootstrap"
|
||||
"github.com/luxfi/ids"
|
||||
)
|
||||
|
||||
// ----- policy test helpers --------------------------------------------------
|
||||
|
||||
// stubAncestry is an in-memory AncestrySource: it walks a parent-linked BlockRef map down from a
|
||||
// tip, exactly as the real wire transport would serve content-addressed ancestry. Modeling the
|
||||
// transport this way keeps the policy unit tests deterministic while exercising the real ancestor-
|
||||
// tolerant tally. `withhold` models a beacon that names a tip but does NOT serve its ancestry.
|
||||
type stubAncestry struct {
|
||||
byID map[ids.ID]BlockRef
|
||||
withhold map[ids.ID]bool
|
||||
}
|
||||
|
||||
func (s *stubAncestry) Ancestry(_ context.Context, tip ids.ID, max int) ([]BlockRef, error) {
|
||||
if s.withhold[tip] {
|
||||
return nil, nil
|
||||
}
|
||||
var out []BlockRef
|
||||
cur := tip
|
||||
for i := 0; i < max; i++ {
|
||||
ref, ok := s.byID[cur]
|
||||
if !ok {
|
||||
break
|
||||
}
|
||||
out = append(out, ref)
|
||||
if ref.Parent == ids.Empty {
|
||||
break
|
||||
}
|
||||
cur = ref.Parent
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// refChain builds genesis..n as content-addressed BlockRefs (parent-linked), returning the slice
|
||||
// and an id→ref index for the stub AncestrySource.
|
||||
func refChain(n int) ([]BlockRef, map[ids.ID]BlockRef) {
|
||||
refs := make([]BlockRef, 0, n+1)
|
||||
byID := map[ids.ID]BlockRef{}
|
||||
var parent ids.ID
|
||||
for h := 0; h <= n; h++ {
|
||||
r := BlockRef{ID: ids.GenerateTestID(), Height: uint64(h), Parent: parent}
|
||||
refs = append(refs, r)
|
||||
byID[r.ID] = r
|
||||
parent = r.ID
|
||||
}
|
||||
return refs, byID
|
||||
}
|
||||
|
||||
// childRef makes a block extending `parent` at height parentHeight+1 — used to forge a "higher"
|
||||
// sibling tip built on a real block.
|
||||
func childRef(parent BlockRef) BlockRef {
|
||||
return BlockRef{ID: ids.GenerateTestID(), Height: parent.Height + 1, Parent: parent.ID}
|
||||
}
|
||||
|
||||
func nodeIDs(n int) []ids.NodeID {
|
||||
out := make([]ids.NodeID, n)
|
||||
for i := range out {
|
||||
out[i] = ids.GenerateTestNodeID()
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// equalBeacons builds a TrustedBeacons map of equal-weight validators.
|
||||
func equalBeacons(beacons []ids.NodeID, w uint64) map[ids.NodeID]StakeWeight {
|
||||
m := make(map[ids.NodeID]StakeWeight, len(beacons))
|
||||
for _, id := range beacons {
|
||||
m[id] = w
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
func reply(id ids.NodeID, tip ids.ID, w uint64) BeaconReply {
|
||||
return BeaconReply{NodeID: id, Tip: tip, Weight: w}
|
||||
}
|
||||
|
||||
// equalStake is the owner's mainnet shape: 5 validators each 0.5e18, total 2.5e18.
|
||||
const equalStake uint64 = 500_000_000_000_000_000
|
||||
|
||||
// ----- A: MASS RECOVERY SUCCESS ---------------------------------------------
|
||||
|
||||
// TestBootstrapTrust_A_MassRecoverySucceeds is THE deadlock fix. 5 EQUAL-weight validators; the 2
|
||||
// stranded recovery targets are down, so only 3 are reachable; the 3 reachable agree on the
|
||||
// frontier. With MinResponses=3 the policy ACCEPTS — even though 3 of 5 stake (1.5e18) is BELOW
|
||||
// the ⅔-of-current-total floor (1.667e18) that the prior code required to be CONNECTED. That old
|
||||
// floor was mathematically unsatisfiable here (the down nodes ARE validators), which is exactly
|
||||
// why no node could recover. This test pins both: the policy accepts, AND the old gate would have
|
||||
// rejected (the deadlock), AND the full loop converges over the real transport.
|
||||
func TestBootstrapTrust_A_MassRecoverySucceeds(t *testing.T) {
|
||||
// The deadlock the fix escapes: 3-of-5 connected stake does NOT clear ⅔ of the total set.
|
||||
require.LessOrEqual(t, 3*equalStake, consensusconfig.TwoThirdsStakeFloor(5*equalStake),
|
||||
"precondition: 3 of 5 equal validators is BELOW ⅔ of total — the prior connect gate's deadlock")
|
||||
|
||||
// Policy decision: 5 configured, 3 reachable agree on the frontier (mainnet analog 1082796).
|
||||
beacons := nodeIDs(5)
|
||||
frontier := ids.GenerateTestID()
|
||||
policy := &BootstrapPolicy{
|
||||
TrustedBeacons: equalBeacons(beacons, equalStake),
|
||||
MinResponses: 3,
|
||||
}
|
||||
replies := []BeaconReply{
|
||||
reply(beacons[0], frontier, equalStake),
|
||||
reply(beacons[1], frontier, equalStake),
|
||||
reply(beacons[2], frontier, equalStake),
|
||||
// beacons[3], beacons[4] are down/stranded — no reply.
|
||||
}
|
||||
f, err := policy.AcceptsFrontier(context.Background(), replies)
|
||||
require.NoError(t, err, "3 of 5 reachable beacons agreeing MUST be accepted — the mass-recovery case")
|
||||
require.Equal(t, frontier, f.ID)
|
||||
require.Equal(t, 3, f.Responders)
|
||||
require.False(t, f.FromCheckpoint)
|
||||
|
||||
// End to end over the real GetAcceptedFrontier/GetAncestors transport: a STALE node with only
|
||||
// 3 of its 5 equal-weight validators reachable converges to the frontier N (not stuck at M).
|
||||
const N = 40
|
||||
const M = 23
|
||||
chain, byID := buildBSChain(N, -1)
|
||||
vm := newBSVMAt(chain, M)
|
||||
v := nodeIDs(5)
|
||||
weights := equalBeacons(v, equalStake)
|
||||
bh, chainID := newBSHandlerWeighted(t, vm, weights)
|
||||
bh.bootstrapMinResponses = 3 // the owner's MinBootstrapResponses=3
|
||||
bh.net = &bsBeaconNet{
|
||||
bh: bh, chainID: chainID, connected: []ids.NodeID{v[0], v[1], v[2]}, // 2 stranded down
|
||||
byID: byID, tip: chain[N], serveAncestors: true,
|
||||
}
|
||||
bh.msgCreator = bsMsgBuilder{}
|
||||
ctx := context.Background()
|
||||
|
||||
bh.bsActive.Store(true)
|
||||
tip, status := bh.FrontierTip(ctx)
|
||||
bh.bsActive.Store(false)
|
||||
require.Equal(t, chainbootstrap.FrontierNamed, status,
|
||||
"MASS RECOVERY: 3 of 5 equal validators reachable + agreeing must NAME the frontier (no deadlock)")
|
||||
require.Equal(t, chain[N].id, tip)
|
||||
|
||||
require.NoError(t, runBS(t, bh), "mass-recovery node must converge")
|
||||
last, _ := vm.LastAccepted(ctx)
|
||||
require.Equal(t, chain[N].id, last, "RECOVERED: converged to the frontier N=%d despite 2 of 5 validators down", N)
|
||||
require.True(t, bh.Accepted(ctx, chain[N].id))
|
||||
}
|
||||
|
||||
// ----- B: ONE-BEACON CAPTURE REJECTED ---------------------------------------
|
||||
|
||||
// TestBootstrapTrust_B_OneBeaconCaptureRejected: 5 configured, only 1 reachable. A single beacon —
|
||||
// even an authentic configured one — cannot name the frontier (it could be the attacker's lone
|
||||
// peer in an eclipse). The response FLOOR rejects it.
|
||||
func TestBootstrapTrust_B_OneBeaconCaptureRejected(t *testing.T) {
|
||||
beacons := nodeIDs(5)
|
||||
policy := &BootstrapPolicy{TrustedBeacons: equalBeacons(beacons, equalStake), MinResponses: 3}
|
||||
replies := []BeaconReply{reply(beacons[0], ids.GenerateTestID(), equalStake)}
|
||||
|
||||
f, err := policy.AcceptsFrontier(context.Background(), replies)
|
||||
require.Nil(t, f)
|
||||
require.ErrorIs(t, err, ErrInsufficientBootstrapResponses,
|
||||
"1 of 5 reachable must be REJECTED (capture) — below the MinResponses floor")
|
||||
}
|
||||
|
||||
// ----- C: TWO-BEACON PARTITION REJECTED -------------------------------------
|
||||
|
||||
// TestBootstrapTrust_C_TwoBeaconPartitionRejected: 5 configured, 2 reachable AGREEING. Two beacons
|
||||
// is still below MinResponses=3, so the policy rejects by default — an attacker who partitions the
|
||||
// node down to 2 beacons cannot capture the frontier even if both agree.
|
||||
func TestBootstrapTrust_C_TwoBeaconPartitionRejected(t *testing.T) {
|
||||
beacons := nodeIDs(5)
|
||||
frontier := ids.GenerateTestID()
|
||||
policy := &BootstrapPolicy{TrustedBeacons: equalBeacons(beacons, equalStake), MinResponses: 3}
|
||||
replies := []BeaconReply{
|
||||
reply(beacons[0], frontier, equalStake),
|
||||
reply(beacons[1], frontier, equalStake),
|
||||
}
|
||||
f, err := policy.AcceptsFrontier(context.Background(), replies)
|
||||
require.Nil(t, f)
|
||||
require.ErrorIs(t, err, ErrInsufficientBootstrapResponses,
|
||||
"2 of 5 reachable + agreeing must be REJECTED by default — the partition-capture floor is MinResponses=3")
|
||||
}
|
||||
|
||||
// ----- D: NON-CONFIGURED PEER IGNORED ---------------------------------------
|
||||
|
||||
// TestBootstrapTrust_D_NonConfiguredPeerIgnored: an attacker peer that is NOT in the configured
|
||||
// beacon set reports a higher forged tip. INVARIANT 1 (non-circular eligibility): peers never
|
||||
// define who is a beacon, so the forged reply is dropped entirely and the configured beacons name
|
||||
// the real frontier.
|
||||
func TestBootstrapTrust_D_NonConfiguredPeerIgnored(t *testing.T) {
|
||||
beacons := nodeIDs(5)
|
||||
real := ids.GenerateTestID()
|
||||
forgedHigher := ids.GenerateTestID()
|
||||
attacker := ids.GenerateTestNodeID() // NOT in TrustedBeacons
|
||||
|
||||
policy := &BootstrapPolicy{TrustedBeacons: equalBeacons(beacons, equalStake), MinResponses: 3}
|
||||
replies := []BeaconReply{
|
||||
reply(beacons[0], real, equalStake),
|
||||
reply(beacons[1], real, equalStake),
|
||||
reply(beacons[2], real, equalStake),
|
||||
reply(attacker, forgedHigher, 9_000_000_000_000_000_000), // huge self-reported weight, ignored
|
||||
}
|
||||
f, err := policy.AcceptsFrontier(context.Background(), replies)
|
||||
require.NoError(t, err)
|
||||
require.Equal(t, real, f.ID, "the non-configured attacker's forged tip must be IGNORED")
|
||||
require.NotEqual(t, forgedHigher, f.ID)
|
||||
require.Equal(t, 3, f.Responders, "only the 3 configured beacons count toward the quorum")
|
||||
}
|
||||
|
||||
// ----- E: MINORITY CONFIGURED FORGERY REJECTED ------------------------------
|
||||
|
||||
// TestBootstrapTrust_E_MinorityConfiguredForgeryRejected: 3 honest configured beacons report
|
||||
// frontier A; 2 configured beacons report a FORGED tip B built directly on A (a forged higher
|
||||
// sibling). C1: the forgers can only RATIFY A (the real block they built on); B itself holds only
|
||||
// the Byzantine minority's stake and is NEVER named. The policy selects A.
|
||||
func TestBootstrapTrust_E_MinorityConfiguredForgeryRejected(t *testing.T) {
|
||||
const w uint64 = 100
|
||||
refs, byID := refChain(30) // genesis..30; A := refs[30]
|
||||
A := refs[30]
|
||||
forgedB := childRef(A) // forged sibling at height 31, parent = real A
|
||||
byID[forgedB.ID] = forgedB
|
||||
|
||||
beacons := nodeIDs(5)
|
||||
policy := &BootstrapPolicy{
|
||||
TrustedBeacons: equalBeacons(beacons, w),
|
||||
MinResponses: 3,
|
||||
Source: &stubAncestry{byID: byID},
|
||||
}
|
||||
replies := []BeaconReply{
|
||||
reply(beacons[0], A.ID, w),
|
||||
reply(beacons[1], A.ID, w),
|
||||
reply(beacons[2], A.ID, w), // 3 honest on A (300)
|
||||
reply(beacons[3], forgedB.ID, w), // 2 Byzantine on the forged child (200)
|
||||
reply(beacons[4], forgedB.ID, w),
|
||||
}
|
||||
// floor = ⅔ of 500 = 333. Neither A (300) nor forgedB (200) clears it directly, so the
|
||||
// ancestor-tolerant tally runs: the forgers' stake flows DOWN through A (its real parent),
|
||||
// crediting A with 500 while forgedB keeps only 200 → A named, forgedB never.
|
||||
f, err := policy.AcceptsFrontier(context.Background(), replies)
|
||||
require.NoError(t, err)
|
||||
require.Equal(t, A.ID, f.ID, "C1: the forged child only RATIFIES A — A is named")
|
||||
require.NotEqual(t, forgedB.ID, f.ID, "C1: the Byzantine-minority forged tip is NEVER named")
|
||||
require.Equal(t, A.Height, f.Height)
|
||||
}
|
||||
|
||||
// ----- F: SPLIT REACHABLE ANCESTRY ------------------------------------------
|
||||
|
||||
// TestBootstrapTrust_F_SplitReachableAncestrySelectsCommonAncestor: 3 reachable configured beacons
|
||||
// each report a DIFFERENT sibling tip (three pending blocks built on the same committed block H —
|
||||
// the healthy bleeding edge). No single tip holds a supermajority, but H is in all three accepted
|
||||
// chains, so the policy names H (the highest ⅔-of-responders common committed block), NOT any
|
||||
// isolated tip.
|
||||
func TestBootstrapTrust_F_SplitReachableAncestrySelectsCommonAncestor(t *testing.T) {
|
||||
const w uint64 = 100
|
||||
refs, byID := refChain(39) // genesis..39; H := refs[39] (the common committed block)
|
||||
H := refs[39]
|
||||
a1, a2, a3 := childRef(H), childRef(H), childRef(H) // three sibling pending blocks at height 40
|
||||
for _, c := range []BlockRef{a1, a2, a3} {
|
||||
byID[c.ID] = c
|
||||
}
|
||||
|
||||
beacons := nodeIDs(3)
|
||||
policy := &BootstrapPolicy{
|
||||
TrustedBeacons: equalBeacons(beacons, w),
|
||||
MinResponses: 3,
|
||||
Source: &stubAncestry{byID: byID},
|
||||
}
|
||||
replies := []BeaconReply{
|
||||
reply(beacons[0], a1.ID, w),
|
||||
reply(beacons[1], a2.ID, w),
|
||||
reply(beacons[2], a3.ID, w),
|
||||
}
|
||||
// floor = ⅔ of 300 = 200. Each sibling holds only 100, but H is shared by all three → 300 > 200.
|
||||
f, err := policy.AcceptsFrontier(context.Background(), replies)
|
||||
require.NoError(t, err)
|
||||
require.Equal(t, H.ID, f.ID, "must select the common committed ancestor H")
|
||||
require.Equal(t, H.Height, f.Height)
|
||||
require.NotEqual(t, a1.ID, f.ID)
|
||||
require.NotEqual(t, a2.ID, f.ID)
|
||||
require.NotEqual(t, a3.ID, f.ID)
|
||||
}
|
||||
|
||||
// ----- G: FINALITY UNCHANGED ------------------------------------------------
|
||||
|
||||
// TestBootstrapTrust_G_FinalityUnchanged proves INVARIANT 3: a bootstrap-accepted frontier is NOT
|
||||
// finality. The SAME 3-of-5 support that AcceptsFrontier admits as a sync anchor does NOT satisfy
|
||||
// FinalityQuorum.HasFinality — live block acceptance still requires > ⅔ of CURRENT validator
|
||||
// stake (4 of 5 here). The bootstrap quorum cannot finalize a block.
|
||||
func TestBootstrapTrust_G_FinalityUnchanged(t *testing.T) {
|
||||
const w uint64 = 100
|
||||
const total = 5 * w
|
||||
beacons := nodeIDs(5)
|
||||
frontier := ids.GenerateTestID()
|
||||
policy := &BootstrapPolicy{TrustedBeacons: equalBeacons(beacons, w), MinResponses: 3}
|
||||
|
||||
// BootstrapTrust ACCEPTS 3 of 5 (a sync anchor).
|
||||
f, err := policy.AcceptsFrontier(context.Background(), []BeaconReply{
|
||||
reply(beacons[0], frontier, w),
|
||||
reply(beacons[1], frontier, w),
|
||||
reply(beacons[2], frontier, w),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
require.Equal(t, frontier, f.ID)
|
||||
require.Equal(t, StakeWeight(3*w), f.Weight, "the frontier is backed by exactly the 3 responders")
|
||||
|
||||
// FinalityQuorum says that SAME 3-of-5 weight is NOT finality — the decisions are different
|
||||
// objects with different thresholds. Finality is unchanged: it still needs > ⅔ (4 of 5).
|
||||
cq := DefaultFinalityQuorum()
|
||||
require.False(t, cq.HasFinality(3*w, total),
|
||||
"INVARIANT 3: a bootstrap-accepted frontier (3 of 5) is NOT a finalizing supermajority")
|
||||
require.True(t, cq.HasFinality(4*w, total),
|
||||
"finality UNCHANGED: > ⅔ of current stake (4 of 5) still finalizes")
|
||||
require.False(t, cq.HasFinality(f.Weight, total),
|
||||
"the bootstrap quorum's own backing weight cannot finalize a block")
|
||||
|
||||
// AFTER-SYNC (the owner's "bootstrap is not a finality bypass"): the node has now SYNCED to the
|
||||
// frontier via BootstrapTrust and re-entered live consensus. A NEW block that collects the SAME
|
||||
// 3-of-5 stake STILL does not finalize — bootstrap admitted a sync ANCHOR, it did not lower the
|
||||
// finality bar. Live acceptance returns to strict > ⅔ of CURRENT stake, exactly as before any
|
||||
// bootstrap. HasFinality is stateless in the bootstrap outcome, which is the whole point: there
|
||||
// is no code path by which "we bootstrapped from 3/5" leaks into the finality decision.
|
||||
require.False(t, cq.HasFinality(3*w, total),
|
||||
"AFTER syncing from a 3-of-5 bootstrap frontier, live finality STILL needs > ⅔ (4 of 5) — no bypass")
|
||||
}
|
||||
|
||||
// ----- checkpoint override (complements B) ----------------------------------
|
||||
|
||||
// edCheckpointAuthority is a test checkpoint authority backed by Ed25519 — a PROVEN primitive, no
|
||||
// custom crypto. It signs a checkpoint's canonical (id,height) message and verifies against its own
|
||||
// public key, rejecting an empty signature and any key that is not the configured authority.
|
||||
type edCheckpointAuthority struct {
|
||||
priv ed25519.PrivateKey
|
||||
pub ed25519.PublicKey
|
||||
}
|
||||
|
||||
func newEdCheckpointAuthority(t *testing.T) *edCheckpointAuthority {
|
||||
t.Helper()
|
||||
pub, priv, err := ed25519.GenerateKey(nil)
|
||||
require.NoError(t, err)
|
||||
return &edCheckpointAuthority{priv: priv, pub: pub}
|
||||
}
|
||||
|
||||
func (a *edCheckpointAuthority) sign(id ids.ID, height uint64) []byte {
|
||||
return ed25519.Sign(a.priv, CanonicalCheckpointMessage(id, height))
|
||||
}
|
||||
|
||||
// VerifyCheckpoint implements CheckpointVerifier: authenticate against the authority's public key.
|
||||
func (a *edCheckpointAuthority) VerifyCheckpoint(id ids.ID, height uint64, sig []byte) bool {
|
||||
return len(sig) != 0 && ed25519.Verify(a.pub, CanonicalCheckpointMessage(id, height), sig)
|
||||
}
|
||||
|
||||
// TestBootstrapTrust_CheckpointOverride: below the response floor (1 of 5), the DEFAULT is reject
|
||||
// (test B), but an operator who pins a SIGNED checkpoint gets the explicit override — the node
|
||||
// anchors to the authenticated (id,height) instead of trusting the lone beacon. This is the
|
||||
// sanctioned escape hatch for a deeply-partitioned node, NEVER an open-ended ≥1-beacon acceptance,
|
||||
// and (INVARIANT 4) NEVER a bare unsigned config value.
|
||||
func TestBootstrapTrust_CheckpointOverride(t *testing.T) {
|
||||
beacons := nodeIDs(5)
|
||||
authority := newEdCheckpointAuthority(t)
|
||||
ckptID := ids.GenerateTestID()
|
||||
const ckptHeight = uint64(1_082_796)
|
||||
policy := &BootstrapPolicy{
|
||||
TrustedBeacons: equalBeacons(beacons, equalStake),
|
||||
MinResponses: 3,
|
||||
Checkpoint: &Checkpoint{ID: ckptID, Height: ckptHeight, Signature: authority.sign(ckptID, ckptHeight)},
|
||||
CheckpointVerifier: authority,
|
||||
}
|
||||
// 1 reachable beacon — below the floor — but a SIGNED checkpoint is pinned.
|
||||
f, err := policy.AcceptsFrontier(context.Background(), []BeaconReply{
|
||||
reply(beacons[0], ids.GenerateTestID(), equalStake),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
require.True(t, f.FromCheckpoint, "below the floor with a SIGNED checkpoint → anchor to the checkpoint")
|
||||
require.Equal(t, ckptID, f.ID)
|
||||
require.Equal(t, ckptHeight, f.Height)
|
||||
|
||||
// Without the checkpoint the same 1-of-5 is rejected (the default — never trust the lone beacon).
|
||||
policy.Checkpoint = nil
|
||||
_, err = policy.AcceptsFrontier(context.Background(), []BeaconReply{
|
||||
reply(beacons[0], ids.GenerateTestID(), equalStake),
|
||||
})
|
||||
require.ErrorIs(t, err, ErrInsufficientBootstrapResponses)
|
||||
}
|
||||
|
||||
// TestBootstrapTrust_CheckpointMustBeSigned is INVARIANT 4: a checkpoint that is present but not
|
||||
// AUTHENTICATED is REJECTED (fail closed). A compromised flag/config that pins a false (id,height)
|
||||
// cannot inject a sync anchor without the authority's signature. Four rejection modes, one accept.
|
||||
func TestBootstrapTrust_CheckpointMustBeSigned(t *testing.T) {
|
||||
beacons := nodeIDs(5)
|
||||
authority := newEdCheckpointAuthority(t)
|
||||
attacker := newEdCheckpointAuthority(t) // a DIFFERENT key — not the configured authority
|
||||
ckptID := ids.GenerateTestID()
|
||||
const h = uint64(500_000)
|
||||
lone := []BeaconReply{reply(beacons[0], ids.GenerateTestID(), equalStake)} // 1-of-5, below floor
|
||||
|
||||
base := func() *BootstrapPolicy {
|
||||
return &BootstrapPolicy{TrustedBeacons: equalBeacons(beacons, equalStake), MinResponses: 3, CheckpointVerifier: authority}
|
||||
}
|
||||
|
||||
// (1) UNSIGNED checkpoint (empty signature) → rejected even with a verifier wired.
|
||||
p := base()
|
||||
p.Checkpoint = &Checkpoint{ID: ckptID, Height: h}
|
||||
_, err := p.AcceptsFrontier(context.Background(), lone)
|
||||
require.ErrorIs(t, err, ErrInsufficientBootstrapResponses, "an UNSIGNED checkpoint must be rejected")
|
||||
|
||||
// (2) signed by a NON-AUTHORITY (attacker) key → rejected.
|
||||
p = base()
|
||||
p.Checkpoint = &Checkpoint{ID: ckptID, Height: h, Signature: attacker.sign(ckptID, h)}
|
||||
_, err = p.AcceptsFrontier(context.Background(), lone)
|
||||
require.ErrorIs(t, err, ErrInsufficientBootstrapResponses, "a checkpoint signed by a non-authority key must be rejected")
|
||||
|
||||
// (3) authority signature over a DIFFERENT (id,height) — replay onto a forged anchor → rejected.
|
||||
p = base()
|
||||
forgedID := ids.GenerateTestID()
|
||||
p.Checkpoint = &Checkpoint{ID: forgedID, Height: h, Signature: authority.sign(ckptID, h)} // sig binds ckptID, not forgedID
|
||||
_, err = p.AcceptsFrontier(context.Background(), lone)
|
||||
require.ErrorIs(t, err, ErrInsufficientBootstrapResponses, "a signature transplanted to a different (id,height) must be rejected")
|
||||
|
||||
// (4) NO verifier configured → any checkpoint is untrusted (fail closed).
|
||||
p = base()
|
||||
p.CheckpointVerifier = nil
|
||||
p.Checkpoint = &Checkpoint{ID: ckptID, Height: h, Signature: authority.sign(ckptID, h)}
|
||||
_, err = p.AcceptsFrontier(context.Background(), lone)
|
||||
require.ErrorIs(t, err, ErrInsufficientBootstrapResponses, "no verifier ⇒ even a validly-signed checkpoint is untrusted")
|
||||
|
||||
// (accept) authority signs the exact pinned (id,height) → trusted.
|
||||
p = base()
|
||||
p.Checkpoint = &Checkpoint{ID: ckptID, Height: h, Signature: authority.sign(ckptID, h)}
|
||||
f, err := p.AcceptsFrontier(context.Background(), lone)
|
||||
require.NoError(t, err)
|
||||
require.True(t, f.FromCheckpoint)
|
||||
require.Equal(t, ckptID, f.ID)
|
||||
}
|
||||
|
||||
// ----- safety guard for the global ancestor-tolerant tally ------------------
|
||||
|
||||
// TestBootstrapTrust_ForkAtSharedGenesisFailsSafe is the load-bearing guard for the
|
||||
// MinFrontierHeight floor — the safety property the global cross-anchor tally (which makes case F
|
||||
// work) would otherwise break. Two branches fork at a DEEP shared ancestor H (height 5), and the
|
||||
// node is stale ABOVE the fork (height 23). The tally credits H with the union of BOTH halves'
|
||||
// stake (all responders share H), so without the floor it would name H — and since the node
|
||||
// already HOLDS H, the loop would FALSE-COMPLETE at the stale height instead of recognizing it has
|
||||
// no ⅔-agreed frontier ahead. The MinFrontierHeight floor refuses to name any block beneath the
|
||||
// node's last-accepted height, turning the partition into a safe ErrNoBootstrapQuorum.
|
||||
//
|
||||
// Asserted deterministically at the POLICY level (a stub AncestrySource serves BOTH branches'
|
||||
// shared ancestry — the real wire transport's rotated sampling may only serve one, masking the
|
||||
// vulnerability, so the integration path is NOT a faithful test of this guard). Revert the floor
|
||||
// (set MinFrontierHeight: 0) and this names H instead of failing safe.
|
||||
func TestBootstrapTrust_ForkAtSharedGenesisFailsSafe(t *testing.T) {
|
||||
const w uint64 = 100
|
||||
const nodeHeight = 23
|
||||
|
||||
// Shared prefix genesis..H (H at height 5), then two divergent branches to height 40.
|
||||
shared, byID := refChain(5)
|
||||
H := shared[5]
|
||||
branchA := []BlockRef{H}
|
||||
branchB := []BlockRef{H}
|
||||
for h := 6; h <= 40; h++ {
|
||||
a := childRef(branchA[len(branchA)-1])
|
||||
b := childRef(branchB[len(branchB)-1])
|
||||
byID[a.ID], byID[b.ID] = a, b
|
||||
branchA = append(branchA, a)
|
||||
branchB = append(branchB, b)
|
||||
}
|
||||
tipA, tipB := branchA[len(branchA)-1], branchB[len(branchB)-1]
|
||||
|
||||
beacons := nodeIDs(6)
|
||||
policy := &BootstrapPolicy{
|
||||
TrustedBeacons: equalBeacons(beacons, w),
|
||||
MinResponses: 4,
|
||||
MinFrontierHeight: nodeHeight, // the node is stale at height 23, ABOVE the fork at 5
|
||||
Source: &stubAncestry{byID: byID},
|
||||
}
|
||||
replies := []BeaconReply{
|
||||
reply(beacons[0], tipA.ID, w), reply(beacons[1], tipA.ID, w), reply(beacons[2], tipA.ID, w),
|
||||
reply(beacons[3], tipB.ID, w), reply(beacons[4], tipB.ID, w), reply(beacons[5], tipB.ID, w),
|
||||
}
|
||||
// H (height 5) is shared by all 6 → 600 > floor(400). But it is BELOW the node's height, so the
|
||||
// floor refuses it; no block at/above height 23 has ⅔ → fail safe.
|
||||
f, err := policy.AcceptsFrontier(context.Background(), replies)
|
||||
require.Nil(t, f, "must not name the deep shared ancestor — that would false-complete at the stale height")
|
||||
require.ErrorIs(t, err, ErrNoBootstrapQuorum,
|
||||
"a fork sharing only blocks BELOW the node's height must fail safe, never name the deep common ancestor")
|
||||
|
||||
// The same split with the node BELOW the fork (a fresh node) legitimately names H — the floor
|
||||
// only blocks naming history the node already has, never a real frontier ahead.
|
||||
policy.MinFrontierHeight = 0
|
||||
f, err = policy.AcceptsFrontier(context.Background(), replies)
|
||||
require.NoError(t, err)
|
||||
require.Equal(t, H.ID, f.ID, "with the node below the fork, H IS the ⅔-common frontier to sync to")
|
||||
}
|
||||
|
||||
// ----- H: SKEWED-WEIGHT PARTITION-CAPTURE (the re-red HIGH; MinResponseWeight floor) ---------
|
||||
|
||||
// weightedBeacons builds a TrustedBeacons map from an explicit per-node weight list — for
|
||||
// modeling a SKEWED (non-uniform) validator stake distribution.
|
||||
func weightedBeacons(beacons []ids.NodeID, w []uint64) map[ids.NodeID]StakeWeight {
|
||||
m := make(map[ids.NodeID]StakeWeight, len(beacons))
|
||||
for i, id := range beacons {
|
||||
m[id] = StakeWeight(w[i])
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
// TestBootstrapTrust_H_SkewedWeightPartitionRejected is the load-bearing regression for the re-red
|
||||
// HIGH finding. Under SKEWED validator weights the MinResponses COUNT floor and the ⅔-of-responders
|
||||
// WEIGHT agreement diverge: an attacker who eclipses the HEAVY honest beacon but lets enough LIGHT
|
||||
// honest beacons through to satisfy the count can shrink the responder-WEIGHT denominator until his
|
||||
// < ⅓-of-total Byzantine stake clears ⅔-of-responders and NAMES A FORGED FRONTIER. The MinResponseWeight
|
||||
// stake-majority floor (> ½ of TOTAL configured beacon stake) closes this — a < ⅓-stake adversary can
|
||||
// never make the responders carry a ⅔ weight majority once they must also carry > ½ of the total.
|
||||
//
|
||||
// Red's PoC: 6 beacons w={3,3,13,1,1,1}, total 22, Byzantine {B0,B1}=6 (27% < ⅓). The attacker
|
||||
// partitions to {B0,B1 on forgedF} + {H2,H3 on realR} = 4 responders (= the majority count floor),
|
||||
// responderWeight=8, ⅔-floor=5, backing[forgedF]=6 > 5 → forgedF would be named. The heavy honest H1
|
||||
// (weight 13, on the real tip) is eclipsed. With MinResponseWeight=⌈22/2⌉=12, responderWeight=8 < 12
|
||||
// → the partition is rejected (the node waits for / re-samples a stake-majority of beacons).
|
||||
func TestBootstrapTrust_H_SkewedWeightPartitionRejected(t *testing.T) {
|
||||
refs, byID := refChain(30)
|
||||
realR := refs[30]
|
||||
forgedF := childRef(realR) // forged sibling at height 31 (its only honest ancestor is realR)
|
||||
byID[forgedF.ID] = forgedF
|
||||
|
||||
b := nodeIDs(6)
|
||||
weights := []uint64{3, 3, 13, 1, 1, 1} // total 22; Byzantine b[0],b[1]=6 (<⅓)
|
||||
var total uint64
|
||||
for _, w := range weights {
|
||||
total += w
|
||||
}
|
||||
tb := weightedBeacons(b, weights)
|
||||
|
||||
// The eclipse: only the 2 Byzantine + 2 LIGHT honest answer; the HEAVY honest b[2] (the real
|
||||
// tip's weight-13 voter) and b[5] are partitioned away.
|
||||
replies := []BeaconReply{
|
||||
reply(b[0], forgedF.ID, weights[0]), // Byzantine, light
|
||||
reply(b[1], forgedF.ID, weights[1]), // Byzantine, light
|
||||
reply(b[3], realR.ID, weights[3]), // honest, light
|
||||
reply(b[4], realR.ID, weights[4]), // honest, light
|
||||
}
|
||||
|
||||
// WITHOUT the stake-majority floor (the bug): the forged tip is named.
|
||||
vuln := &BootstrapPolicy{TrustedBeacons: tb, MinResponses: 4, Source: &stubAncestry{byID: byID}}
|
||||
if f, err := vuln.AcceptsFrontier(context.Background(), replies); err == nil && f != nil {
|
||||
require.Equal(t, forgedF.ID, f.ID,
|
||||
"VULN PRECONDITION: without MinResponseWeight the eclipsed skewed partition names the forged tip (proves the floor is load-bearing)")
|
||||
}
|
||||
|
||||
// WITH the stake-majority floor (the fix, exactly as bootstrapPolicy() now wires it): rejected.
|
||||
fixed := &BootstrapPolicy{
|
||||
TrustedBeacons: tb,
|
||||
MinResponses: 4,
|
||||
MinResponseWeight: StakeWeight(total/2 + 1), // ⌈total/2⌉ = 12
|
||||
Source: &stubAncestry{byID: byID},
|
||||
}
|
||||
_, err := fixed.AcceptsFrontier(context.Background(), replies)
|
||||
require.ErrorIs(t, err, ErrInsufficientBootstrapResponses,
|
||||
"FIX: responderWeight 8 < ½-stake floor 12 → the skewed partition cannot name a frontier (forged or otherwise)")
|
||||
|
||||
// And the fix still admits an HONEST stake-majority: add the heavy honest H1 (weight 13) on realR.
|
||||
full := append(replies, reply(b[2], realR.ID, weights[2])) // responderWeight 8+13 = 21 ≥ 12
|
||||
f, err := fixed.AcceptsFrontier(context.Background(), full)
|
||||
require.NoError(t, err, "an honest stake-majority of responders still names the real frontier")
|
||||
require.Equal(t, realR.ID, f.ID, "the real tip is named once a stake-majority is reachable; forged never")
|
||||
}
|
||||
|
||||
// TestBootstrapTrust_D2_NonConfiguredSwarmNamesNothing is the cleaner load-bearing isolation of the
|
||||
// configured-beacon filter (INVARIANT 1) that the re-red asked for: a SWARM of non-configured peers
|
||||
// (enough to clear any count floor on their own) all shouting a forged frontier names NOTHING,
|
||||
// because none is in TrustedBeacons. This proves the filter, not merely the MinResponders floor.
|
||||
func TestBootstrapTrust_D2_NonConfiguredSwarmNamesNothing(t *testing.T) {
|
||||
const w uint64 = 100
|
||||
refs, byID := refChain(30)
|
||||
real := refs[30]
|
||||
forged, fbyID := refChain(40) // a wholly forged chain from a fresh genesis
|
||||
for id, r := range fbyID {
|
||||
byID[id] = r
|
||||
}
|
||||
|
||||
configured := nodeIDs(3) // the real beacon set
|
||||
policy := &BootstrapPolicy{
|
||||
TrustedBeacons: equalBeacons(configured, w),
|
||||
MinResponses: 2,
|
||||
MinResponseWeight: StakeWeight(w*3/2 + 1),
|
||||
Source: &stubAncestry{byID: byID},
|
||||
}
|
||||
|
||||
// 50 non-configured peers, each heavy, all on the forged tip — NOT in TrustedBeacons.
|
||||
swarm := nodeIDs(50)
|
||||
var replies []BeaconReply
|
||||
for _, p := range swarm {
|
||||
replies = append(replies, reply(p, forged[40].ID, 9_000_000))
|
||||
}
|
||||
_, err := policy.AcceptsFrontier(context.Background(), replies)
|
||||
require.ErrorIs(t, err, ErrInsufficientBootstrapResponses,
|
||||
"INVARIANT 1: non-configured peers carry ZERO weight — a forged swarm names nothing")
|
||||
|
||||
// Add the 3 real configured beacons on the real tip → the real tip is named, swarm invisible.
|
||||
for _, c := range configured {
|
||||
replies = append(replies, reply(c, real.ID, w))
|
||||
}
|
||||
f, err := policy.AcceptsFrontier(context.Background(), replies)
|
||||
require.NoError(t, err)
|
||||
require.Equal(t, real.ID, f.ID, "only configured beacons name the frontier; the 50-peer forged swarm is ignored")
|
||||
}
|
||||
|
||||
// TestBootstrapPolicy_WiresStakeMajorityFloor is the regression guard the re-red flagged (LOW):
|
||||
// the production constructor bootstrapPolicy() MUST emit MinResponseWeight = ⌈total/2⌉. The H/D2
|
||||
// tests construct policies directly, so a mutation that stops the constructor from setting the
|
||||
// floor would not be caught — this asserts the WIRING on the real production path. Mutation-proof:
|
||||
// neuter `if total > 0` in bootstrapPolicy() and this test fails (MinResponseWeight==0).
|
||||
func TestBootstrapPolicy_WiresStakeMajorityFloor(t *testing.T) {
|
||||
refs, _ := refChain(5)
|
||||
vm := newBSVMAt(refs5BSBlocks(refs), 0)
|
||||
bh, _ := newBSHandlerWeighted(t, vm, map[ids.NodeID]uint64{}) // handler shell; we call bootstrapPolicy directly
|
||||
|
||||
// SKEWED set: total = 22, ⌈total/2⌉ = 12.
|
||||
b := nodeIDs(6)
|
||||
weights := map[ids.NodeID]uint64{b[0]: 3, b[1]: 3, b[2]: 13, b[3]: 1, b[4]: 1, b[5]: 1}
|
||||
var total uint64
|
||||
for _, w := range weights {
|
||||
total += w
|
||||
}
|
||||
|
||||
pol := bh.bootstrapPolicy(weights)
|
||||
require.Equal(t, StakeWeight(total/2+1), pol.MinResponseWeight,
|
||||
"REGRESSION: bootstrapPolicy() must wire MinResponseWeight = ⌈total/2⌉ (skewed-weight floor)")
|
||||
require.Equal(t, len(weights)/2+1, pol.MinResponses,
|
||||
"bootstrapPolicy() must wire the count-majority floor too")
|
||||
require.NotNil(t, pol.Source, "the policy must carry an AncestrySource")
|
||||
|
||||
// EQUAL-weight: 5 × 0.5e18 — the floor must not re-deadlock 3-of-5 (= 0.6 ≥ 0.5).
|
||||
eq := equalBeacons(nodeIDs(5), 500_000_000_000_000_000)
|
||||
var eqTotal uint64
|
||||
for _, w := range eq {
|
||||
eqTotal += uint64(w)
|
||||
}
|
||||
eqPol := bh.bootstrapPolicy(eq)
|
||||
require.Equal(t, StakeWeight(eqTotal/2+1), eqPol.MinResponseWeight)
|
||||
require.Less(t, eqPol.MinResponseWeight, StakeWeight(3*500_000_000_000_000_000),
|
||||
"3-of-5 equal stake (0.6·total) must clear the ½ floor — no re-deadlock")
|
||||
|
||||
// DEGENERATE: empty weights → floor disabled (0), no panic.
|
||||
require.Equal(t, StakeWeight(0), bh.bootstrapPolicy(map[ids.NodeID]uint64{}).MinResponseWeight,
|
||||
"empty weights → MinResponseWeight disabled (pre-P-chain / single-node fallback)")
|
||||
}
|
||||
|
||||
// refs5BSBlocks adapts a BlockRef chain to the []*bsTestBlock the bsTestVM needs (genesis only
|
||||
// accepted), so newBSHandlerWeighted has a VM. The handler is used only to call bootstrapPolicy().
|
||||
func refs5BSBlocks(refs []BlockRef) []*bsTestBlock {
|
||||
out := make([]*bsTestBlock, len(refs))
|
||||
var parent ids.ID
|
||||
for i, r := range refs {
|
||||
out[i] = &bsTestBlock{id: r.ID, parent: parent, height: r.Height, bytes: []byte(r.ID.String()), valid: true}
|
||||
parent = r.ID
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// ----- CaughtUp: the tip-holder go-live determination (RED CRITICAL fix) ----
|
||||
//
|
||||
// CaughtUp is the DUAL of AcceptsFrontier — "nobody is ahead" vs "here is the block ahead to sync
|
||||
// to". It is the go-live path for a TIP-HOLDER on a mixed-height co-restart, where the responders
|
||||
// SPLIT below the ⅔ naming threshold so AcceptsFrontier names NOTHING yet the node is plainly not
|
||||
// behind. Getting its SAFETY exactly right is the hinge between "fixes the freeze" and "reopens the
|
||||
// stale-go-live bug": these pin all three conditions (floor met, none-ahead, holds-every-tip) and
|
||||
// prove the two adversarial fake-caught-up attempts FAIL.
|
||||
|
||||
// heldOracle builds the height ORACLE CaughtUp injects: a block's height, ok=false when not held.
|
||||
func heldOracle(held map[ids.ID]uint64) func(ids.ID) (uint64, bool) {
|
||||
return func(id ids.ID) (uint64, bool) { h, ok := held[id]; return h, ok }
|
||||
}
|
||||
|
||||
// TestBootstrapTrust_CaughtUp_TipHolderSplitGoesReady is the CRITICAL regression at the policy layer:
|
||||
// the EXACT mainnet co-restart shape. A producer at N sees 4 responders split {N, N, N-16, genesis};
|
||||
// the tip-holders are only ½ (< ⅔), so AcceptsFrontier names NOTHING (ErrNoBootstrapQuorum) — yet the
|
||||
// node holds every reported tip and none is above N, so CaughtUp is TRUE. It pins BOTH halves: the
|
||||
// SAME replies yield no NAMED frontier (the case the tip-holder fails safe DOWN without this fix) but
|
||||
// ARE caught-up.
|
||||
func TestBootstrapTrust_CaughtUp_TipHolderSplitGoesReady(t *testing.T) {
|
||||
const N = 40
|
||||
refs, byID := refChain(N) // genesis..N
|
||||
b := nodeIDs(5) // 5 equal-weight beacons (the node is the 5th, not a responder)
|
||||
const w = uint64(100) // total 500 → MinResponseWeight ⌈500/2⌉=251, MinResponses majority=3
|
||||
policy := &BootstrapPolicy{
|
||||
TrustedBeacons: equalBeacons(b, w),
|
||||
MinResponses: 3,
|
||||
MinResponseWeight: 251,
|
||||
MinFrontierHeight: N,
|
||||
Source: &stubAncestry{byID: byID},
|
||||
}
|
||||
|
||||
// 4 connected responders: 2 at the tip N, one stale at N-16, one at genesis — the production shape.
|
||||
replies := []BeaconReply{
|
||||
reply(b[0], refs[N].ID, w),
|
||||
reply(b[1], refs[N].ID, w),
|
||||
reply(b[2], refs[N-16].ID, w),
|
||||
reply(b[3], refs[0].ID, w),
|
||||
}
|
||||
|
||||
// HALF 1: AcceptsFrontier names NOTHING — the tip-holders (200) do not clear ⅔ (266), and the
|
||||
// ⅔-backed common ancestor N-16 is below MinFrontierHeight=N (history the node already has).
|
||||
_, err := policy.AcceptsFrontier(context.Background(), replies)
|
||||
require.ErrorIs(t, err, ErrNoBootstrapQuorum,
|
||||
"the mixed-height split names no frontier — exactly the case the tip-holder froze on without CaughtUp")
|
||||
|
||||
// HALF 2: the node HOLDS its accepted chain 0..N, so it holds every reported tip and none is above
|
||||
// N → CaughtUp is TRUE. This is the go-live path the regression was missing.
|
||||
held := map[ids.ID]uint64{refs[N].ID: N, refs[N-16].ID: N - 16, refs[0].ID: 0}
|
||||
require.True(t, policy.CaughtUp(replies, N, heldOracle(held)),
|
||||
"a tip-holder that holds every reported tip and is at the top of all of them IS caught up")
|
||||
}
|
||||
|
||||
// TestBootstrapTrust_CaughtUp_StaleNodeNotCaughtUp is the FORWARD safety guard (the stale-go-live bug
|
||||
// staying FIXED): a STALE node at N-16 with honest producers at N PRESENT must NOT be caught-up — an
|
||||
// honest responder is ahead, so it still SYNCS. CaughtUp must not fire merely because SOME responders
|
||||
// are at/below the node.
|
||||
func TestBootstrapTrust_CaughtUp_StaleNodeNotCaughtUp(t *testing.T) {
|
||||
const N = 40
|
||||
refs, _ := refChain(N)
|
||||
b := nodeIDs(5)
|
||||
const w = uint64(100)
|
||||
policy := &BootstrapPolicy{TrustedBeacons: equalBeacons(b, w), MinResponses: 3, MinResponseWeight: 251, MinFrontierHeight: N - 16}
|
||||
|
||||
replies := []BeaconReply{
|
||||
reply(b[0], refs[N].ID, w), // honest, AHEAD
|
||||
reply(b[1], refs[N].ID, w), // honest, AHEAD
|
||||
reply(b[2], refs[N-16].ID, w), // at the node's height
|
||||
reply(b[3], refs[N-20].ID, w), // below
|
||||
}
|
||||
// The node holds only 0..N-16 — it does NOT hold the producers' tip N.
|
||||
held := map[ids.ID]uint64{refs[N-16].ID: N - 16, refs[N-20].ID: N - 20}
|
||||
require.False(t, policy.CaughtUp(replies, N-16, heldOracle(held)),
|
||||
"a stale node with an honest responder ahead must NOT be caught up — it syncs (stale-go-live stays fixed)")
|
||||
}
|
||||
|
||||
// TestBootstrapTrust_CaughtUp_StaleNodeMinorityFakeRejected is adversarial fake-caught-up #1 (honest
|
||||
// present): a node at N-16 where a <⅓-stake set of beacons reports ≤ N-16 to fake caught-up WHILE the
|
||||
// honest producers at N are also present. The honest max is ahead (and the node lacks tip N) → NOT
|
||||
// caught up. The minority cannot fake it past the honest responders.
|
||||
func TestBootstrapTrust_CaughtUp_StaleNodeMinorityFakeRejected(t *testing.T) {
|
||||
const N = 40
|
||||
refs, _ := refChain(N)
|
||||
b := nodeIDs(5)
|
||||
const w = uint64(100)
|
||||
policy := &BootstrapPolicy{TrustedBeacons: equalBeacons(b, w), MinResponses: 3, MinResponseWeight: 251, MinFrontierHeight: N - 16}
|
||||
|
||||
// 3 honest producers at N (ahead) + 1 Byzantine at N-16 trying to fake "everyone is at my height".
|
||||
replies := []BeaconReply{
|
||||
reply(b[0], refs[N].ID, w),
|
||||
reply(b[1], refs[N].ID, w),
|
||||
reply(b[2], refs[N].ID, w),
|
||||
reply(b[3], refs[N-16].ID, w), // the < ⅓ liar
|
||||
}
|
||||
held := map[ids.ID]uint64{refs[N-16].ID: N - 16} // node holds only up to N-16
|
||||
require.False(t, policy.CaughtUp(replies, N-16, heldOracle(held)),
|
||||
"a <⅓ minority reporting ≤N-16 cannot fake caught-up while honest producers at N are present")
|
||||
}
|
||||
|
||||
// TestBootstrapTrust_CaughtUp_EclipsedMinorityFailsSafe is adversarial fake-caught-up #2 (honest
|
||||
// eclipsed): the honest producers (at N) are SUPPRESSED and only a <½-stake set of beacons reports
|
||||
// ≤ N-16. The response FLOOR (the SAME one AcceptsFrontier uses) is not met → CaughtUp is FALSE →
|
||||
// fail safe. Faking caught-up costs the same stake-majority of honest beacons that faking a NAMED
|
||||
// frontier does — no partition-capture.
|
||||
func TestBootstrapTrust_CaughtUp_EclipsedMinorityFailsSafe(t *testing.T) {
|
||||
const N = 40
|
||||
refs, _ := refChain(N)
|
||||
b := nodeIDs(5)
|
||||
const w = uint64(100) // total 500 → MinResponseWeight 251
|
||||
policy := &BootstrapPolicy{TrustedBeacons: equalBeacons(b, w), MinResponses: 3, MinResponseWeight: 251, MinFrontierHeight: N - 16}
|
||||
|
||||
// Only 2 of 5 beacons answer (the honest producers at N are eclipsed). Their weight 200 < 251.
|
||||
replies := []BeaconReply{
|
||||
reply(b[2], refs[N-16].ID, w),
|
||||
reply(b[3], refs[N-20].ID, w),
|
||||
}
|
||||
held := map[ids.ID]uint64{refs[N-16].ID: N - 16, refs[N-20].ID: N - 20}
|
||||
require.False(t, policy.CaughtUp(replies, N-16, heldOracle(held)),
|
||||
"an eclipsed <½-stake responder set cannot fake caught-up — the floor is not met (fail safe)")
|
||||
|
||||
// Sanity: AcceptsFrontier ALSO rejects this set below the floor (the SAME floor gates both paths).
|
||||
_, err := policy.AcceptsFrontier(context.Background(), replies)
|
||||
require.ErrorIs(t, err, ErrInsufficientBootstrapResponses, "the same floor gates naming and caught-up")
|
||||
}
|
||||
|
||||
// TestBootstrapTrust_CaughtUp_OneAcceptedBlockBehindSyncs proves condition (b) uses the ACCEPTED
|
||||
// height: a node at accepted N that has merely PROCESSED N+1 (holds it) is NOT caught up when a
|
||||
// producer has ACCEPTED N+1 — it must sync that block. heightOf reads the block's canonical height,
|
||||
// so a held-but-above-lastAccepted tip correctly defeats caught-up (the ±1 pending skew cannot fake it).
|
||||
func TestBootstrapTrust_CaughtUp_OneAcceptedBlockBehindSyncs(t *testing.T) {
|
||||
const N = 40
|
||||
refs, _ := refChain(N + 1) // includes N+1
|
||||
b := nodeIDs(5)
|
||||
const w = uint64(100)
|
||||
policy := &BootstrapPolicy{TrustedBeacons: equalBeacons(b, w), MinResponses: 3, MinResponseWeight: 251, MinFrontierHeight: N}
|
||||
replies := []BeaconReply{
|
||||
reply(b[0], refs[N+1].ID, w), // a producer ACCEPTED N+1
|
||||
reply(b[1], refs[N].ID, w),
|
||||
reply(b[2], refs[N].ID, w),
|
||||
}
|
||||
// The node holds 0..N AND has processed N+1 (held), but its ACCEPTED height is N.
|
||||
held := map[ids.ID]uint64{refs[N+1].ID: N + 1, refs[N].ID: N}
|
||||
require.False(t, policy.CaughtUp(replies, N, heldOracle(held)),
|
||||
"a node one ACCEPTED block behind (even if it processed N+1) must NOT be caught up — it syncs")
|
||||
}
|
||||
|
||||
// TestBootstrapTrust_CaughtUp_SameHeightForkNotHeld proves condition (c): a responder reporting a
|
||||
// DIFFERENT block at the node's height (a fork the node never finalized) defeats caught-up — the node
|
||||
// must HOLD every reported tip, not merely match heights numerically.
|
||||
func TestBootstrapTrust_CaughtUp_SameHeightForkNotHeld(t *testing.T) {
|
||||
const N = 40
|
||||
refs, _ := refChain(N)
|
||||
fork := BlockRef{ID: ids.GenerateTestID(), Height: N} // a sibling at height N the node does NOT hold
|
||||
b := nodeIDs(5)
|
||||
const w = uint64(100)
|
||||
policy := &BootstrapPolicy{TrustedBeacons: equalBeacons(b, w), MinResponses: 3, MinResponseWeight: 251, MinFrontierHeight: N}
|
||||
replies := []BeaconReply{
|
||||
reply(b[0], refs[N].ID, w),
|
||||
reply(b[1], refs[N].ID, w),
|
||||
reply(b[2], fork.ID, w), // a fork at the same height N
|
||||
}
|
||||
held := map[ids.ID]uint64{refs[N].ID: N} // the node holds its tip N but NOT the fork
|
||||
require.False(t, policy.CaughtUp(replies, N, heldOracle(held)),
|
||||
"a same-height fork the node does not hold defeats caught-up (condition c: holds every reported tip)")
|
||||
}
|
||||
|
||||
// ----- M1: the pre-existing eclipse-stale own-height path (red fast-follow) ------------------
|
||||
//
|
||||
// M1 is the pre-existing path the own-height filter tightening closes. BEFORE: nameFrontier filtered
|
||||
// the ancestor-tolerant tally with `ref.Height < MinFrontierHeight` (== the node's own last-accepted),
|
||||
// so a block AT the node's own height PASSED the filter and could be NAMED. An eclipse that throttles
|
||||
// the genuinely-ahead responders below the ⅔ naming threshold — while letting the at-height responders
|
||||
// through — makes the node's OWN height accrue ⅔ purely as the shared ANCESTOR of those ahead tips, so
|
||||
// nameFrontier names it → FrontierNamed at own height → the node goes Ready STALE (here, 5 blocks
|
||||
// behind a finalized N+5). AFTER: the filter is `ref.Height <= MinFrontierHeight`, so own height is
|
||||
// EXCLUDED from naming; the at-own-height decision routes to CaughtUp, which SEES the N+5 ahead tips
|
||||
// (un-held, above) and REFUSES → the node syncs/fails safe instead of going Ready stale.
|
||||
//
|
||||
// Deterministic, no network timing. Revert the filter to `<` and the first assertion (own height
|
||||
// NOT named → ErrNoBootstrapQuorum) FAILS — that revert IS the M1 bug, so this is the RED-before /
|
||||
// GREEN-after pin. The boundary sub-assertion (one notch lower DOES name N) proves it is precisely
|
||||
// the OWN-HEIGHT exclusion doing the work, not some unrelated filter.
|
||||
func TestBootstrapTrust_EclipseOwnHeightNotNamedRoutesToCaughtUp(t *testing.T) {
|
||||
const N = 40 // the node's own last-accepted height
|
||||
const ahead = N + 5 // a GENUINELY FINALIZED block 5 ahead — the eclipse throttles its visibility
|
||||
refs, byID := refChain(ahead) // genesis..N+5, parent-linked; the ahead set's tip descends through N
|
||||
const w = uint64(100)
|
||||
|
||||
b := nodeIDs(6) // 6 configured beacons @100 → total 600; MinResponseWeight ⌈600/2⌉=301, MinResponses majority=4
|
||||
policy := &BootstrapPolicy{
|
||||
TrustedBeacons: equalBeacons(b, w),
|
||||
MinResponses: 4,
|
||||
MinResponseWeight: 301,
|
||||
MinFrontierHeight: N, // the node's own last-accepted height — exactly the M1 boundary
|
||||
Source: &stubAncestry{byID: byID},
|
||||
}
|
||||
|
||||
// THE ECLIPSE CONSTRUCTION (red's, verbatim numbers): the ahead responders are throttled to
|
||||
// R_a = 300 (3 beacons at N+5, BELOW the ⅔-of-responders naming threshold), the behind/at-height
|
||||
// responders R_b = 200 (2 beacons at N) all get through; the 6th beacon is eclipsed (no reply).
|
||||
// R = R_a + R_b = 500 > ½·600 (floor met). R_a = 300 < ⅔R = 333 (so N+5 is NOT named). YET block N
|
||||
// accrues R_a + R_b = 500 > ⅔R because the ahead nodes credit N as an ANCESTOR of N+5.
|
||||
replies := []BeaconReply{
|
||||
reply(b[0], refs[N].ID, w), // at the node's own height N
|
||||
reply(b[1], refs[N].ID, w), // at the node's own height N (R_b = 200)
|
||||
reply(b[2], refs[ahead].ID, w), // genuinely ahead at N+5
|
||||
reply(b[3], refs[ahead].ID, w), // genuinely ahead at N+5
|
||||
reply(b[4], refs[ahead].ID, w), // genuinely ahead at N+5 (R_a = 300, < ⅔·500 = 333)
|
||||
// b[5] eclipsed — no reply.
|
||||
}
|
||||
|
||||
// Sanity pins on the construction (so a future edit that breaks the eclipse shape is caught).
|
||||
require.Equal(t, uint64(333), Ratio{2, 3}.floorOf(500), "⅔-of-responders floor over R=500 is 333")
|
||||
require.Less(t, uint64(300), uint64(333), "R_a=300 is BELOW the ⅔ naming threshold — N+5 is not nameable")
|
||||
|
||||
// AFTER (the fix): own height N is EXCLUDED from naming → no ⅔-backed block ABOVE N exists
|
||||
// (N+5 is sub-⅔) → ErrNoBootstrapQuorum. (Revert `<=`→`<` and this names refs[N] — the M1 bug.)
|
||||
_, err := policy.AcceptsFrontier(context.Background(), replies)
|
||||
require.ErrorIs(t, err, ErrNoBootstrapQuorum,
|
||||
"M1 FIX: the node's OWN height must NOT be named even when ahead tips credit it as a ⅔-backed ancestor")
|
||||
|
||||
// …and the decision routes to CaughtUp, which SEES the genuinely-ahead N+5 tips (un-held, above
|
||||
// the node's height) and REFUSES — so the node syncs toward N+5, never goes Ready at stale N.
|
||||
held := map[ids.ID]uint64{} // the node holds 0..N, NOT N+1..N+5
|
||||
for h := 0; h <= N; h++ {
|
||||
held[refs[h].ID] = uint64(h)
|
||||
}
|
||||
require.False(t, policy.CaughtUp(replies, N, heldOracle(held)),
|
||||
"M1 FIX: routed to CaughtUp, the eclipse's ahead tips (un-held, above N) correctly defeat caught-up → sync")
|
||||
|
||||
// BOUNDARY: the SAME replies with MinFrontierHeight one notch lower (N-1) DO name N (height N is
|
||||
// now STRICTLY ABOVE the floor). This proves the refusal above is precisely the OWN-HEIGHT
|
||||
// exclusion — not the ⅔ tally, the responder floor, or the voter count — doing the work.
|
||||
policy.MinFrontierHeight = N - 1
|
||||
f, err := policy.AcceptsFrontier(context.Background(), replies)
|
||||
require.NoError(t, err, "one notch below own height, N is strictly above the floor and IS the ⅔-common frontier")
|
||||
require.Equal(t, refs[N].ID, f.ID, "boundary: N is named iff its height is STRICTLY ABOVE MinFrontierHeight")
|
||||
require.Equal(t, uint64(N), f.Height)
|
||||
}
|
||||
@@ -1,110 +0,0 @@
|
||||
// Copyright (C) 2019-2026, Lux Industries, Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
// catchup_frame_test.go — the CERT-CARRYING catch-up wire format. These prove the
|
||||
// load-bearing property: a v2 (block,cert) entry round-trips, an entry with no cert
|
||||
// routes to the vote path, and a cross-version exchange fails CLEANLY (a legacy
|
||||
// decoder cannot misparse a v2 frame, and the v2 decoder treats a legacy raw block
|
||||
// as legacy — never a partial/garbage parse). The cert-accept SEMANTICS are proven
|
||||
// in the consensus engine tests; here we pin only the framing.
|
||||
package chains
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/binary"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestCatchupEntry_RoundTrip(t *testing.T) {
|
||||
block := []byte("\xf9\x02\x00 a realistic-ish block body")
|
||||
cert := []byte("a-marshaled-quorum-cert")
|
||||
|
||||
blk, crt, ok := decodeCatchupEntry(encodeCatchupEntry(block, cert))
|
||||
if !ok {
|
||||
t.Fatal("a v2 entry must decode as a v2 entry")
|
||||
}
|
||||
if !bytes.Equal(blk, block) {
|
||||
t.Fatalf("block bytes corrupted: got %q want %q", blk, block)
|
||||
}
|
||||
if !bytes.Equal(crt, cert) {
|
||||
t.Fatalf("cert bytes corrupted: got %q want %q", crt, cert)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCatchupEntry_EmptyCertRoutesToVotePath(t *testing.T) {
|
||||
// A still-pending block served on the live missing-parent path carries no cert.
|
||||
// It must decode as a v2 entry with an EMPTY cert — the requester then votes on
|
||||
// it (handleContext routes certLen==0 to Put), the legacy behaviour.
|
||||
block := []byte("pending-block-no-cert")
|
||||
blk, crt, ok := decodeCatchupEntry(encodeCatchupEntry(block, nil))
|
||||
if !ok {
|
||||
t.Fatal("a v2 entry with empty cert must still decode as v2")
|
||||
}
|
||||
if !bytes.Equal(blk, block) {
|
||||
t.Fatalf("block bytes corrupted: got %q", blk)
|
||||
}
|
||||
if len(crt) != 0 {
|
||||
t.Fatalf("cert must be empty, got %d bytes", len(crt))
|
||||
}
|
||||
}
|
||||
|
||||
func TestCatchupEntry_LegacyRawBlockIsNotV2(t *testing.T) {
|
||||
// A legacy responder sends the raw block as the container (no magic). The v2
|
||||
// decoder must report ok=false so handleContext treats it as a raw block (Put),
|
||||
// never as a malformed v2 entry. Cover several real block-prefix shapes.
|
||||
for _, raw := range [][]byte{
|
||||
{0xf9, 0x02, 0x00, 0x11, 0x22}, // EVM/RLP list header
|
||||
{0x00, 0x00, 0x00, 0x2a}, // P/X-chain codec version prefix
|
||||
{}, // empty
|
||||
[]byte("LCU"), // 3 bytes — too short to even hold the magic
|
||||
} {
|
||||
if _, _, ok := decodeCatchupEntry(raw); ok {
|
||||
t.Fatalf("legacy raw block %x must NOT decode as a v2 entry", raw)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestCatchupEntry_MagicIsNotAPlausibleLength(t *testing.T) {
|
||||
// CROSS-VERSION SAFETY (new responder → legacy requester): a legacy decoder reads
|
||||
// the first 4 bytes of a v2 entry as a uint32 length. The magic must be so large
|
||||
// that it always exceeds the remaining buffer, so the legacy loop rejects the
|
||||
// frame (0 blocks processed) rather than consuming garbage. 0x4C435532 ≈ 1.28 GB.
|
||||
asLen := binary.BigEndian.Uint32(catchupEntryMagic[:])
|
||||
if asLen < (1 << 30) {
|
||||
t.Fatalf("magic read as a length (%d) is too small — a legacy decoder could misparse a v2 frame", asLen)
|
||||
}
|
||||
// And a full v2 frame's leading length-word (the magic) dwarfs the frame itself,
|
||||
// so the legacy `blockLen > remaining` guard always fires.
|
||||
frame := encodeCatchupEntry([]byte("blk"), []byte("crt"))
|
||||
if uint64(binary.BigEndian.Uint32(frame[:4])) <= uint64(len(frame)) {
|
||||
t.Fatal("magic-as-length must exceed the frame length so a legacy decoder self-rejects")
|
||||
}
|
||||
}
|
||||
|
||||
func TestCatchupEntry_CorruptV2FramesRejected(t *testing.T) {
|
||||
good := encodeCatchupEntry([]byte("block-body"), []byte("cert-body"))
|
||||
|
||||
// Truncations inside the v2 structure must fail to ok=false (never a partial
|
||||
// parse): drop the trailing cert byte, drop the certLen word, etc.
|
||||
for _, bad := range [][]byte{
|
||||
good[:len(good)-1], // last cert byte missing → certLen != remaining
|
||||
good[:len(good)-5], // certLen word + cert missing
|
||||
good[:8], // magic + blockLen only, block missing
|
||||
good[:6], // magic + 2 bytes of blockLen
|
||||
append(append([]byte(nil), good...), 0x00), // trailing byte → does not consume exactly
|
||||
} {
|
||||
if _, _, ok := decodeCatchupEntry(bad); ok {
|
||||
t.Fatalf("a corrupt v2 frame (len %d) must be rejected, not partial-parsed", len(bad))
|
||||
}
|
||||
}
|
||||
|
||||
// An overflowing blockLen (claims more block than the buffer holds) is rejected.
|
||||
overflow := append([]byte(nil), catchupEntryMagic[:]...)
|
||||
var u32 [4]byte
|
||||
binary.BigEndian.PutUint32(u32[:], 0xFFFFFFFF)
|
||||
overflow = append(overflow, u32[:]...)
|
||||
overflow = append(overflow, []byte("tiny")...)
|
||||
if _, _, ok := decodeCatchupEntry(overflow); ok {
|
||||
t.Fatal("a v2 frame with an overflowing blockLen must be rejected")
|
||||
}
|
||||
}
|
||||
@@ -1,56 +0,0 @@
|
||||
// Copyright (C) 2019-2025, Lux Industries, Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
package chains
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
|
||||
"github.com/luxfi/ids"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// *blockHandler must satisfy ancestorRequester, or the catch-up wire silently
|
||||
// loses its transport. Catch it at compile time, not in production.
|
||||
var _ ancestorRequester = (*blockHandler)(nil)
|
||||
|
||||
// catchupSpy records the requestContext calls networkCatchup bridges to.
|
||||
type catchupSpy struct {
|
||||
calls int
|
||||
lastFrom ids.NodeID
|
||||
lastBlock ids.ID
|
||||
}
|
||||
|
||||
func (s *catchupSpy) requestContext(_ context.Context, from ids.NodeID, blockID ids.ID) {
|
||||
s.calls++
|
||||
s.lastFrom = from
|
||||
s.lastBlock = blockID
|
||||
}
|
||||
|
||||
// TestNetworkCatchup_BridgesEngineSignalToWire is the regression for the
|
||||
// stranded-follower bug: node never set netCfg.Catchup, so the engine's
|
||||
// requestCatchup hit a nil interface and a follower that fell behind during
|
||||
// consensus never fetched the missing ancestors — it looped on an unfinalizable
|
||||
// orphan forever (observed live: mainnet luxd-0/2 stuck at 1082780 while peers
|
||||
// reached 1082793). The wire must (a) be nil-safe before its handler is
|
||||
// late-bound and (b) route the engine's RequestAncestors to the handler's
|
||||
// GetAncestors transport (requestContext), once, for the right block and peer.
|
||||
func TestNetworkCatchup_BridgesEngineSignalToWire(t *testing.T) {
|
||||
missing := ids.GenerateTestID()
|
||||
peer := ids.GenerateTestNodeID()
|
||||
|
||||
// (a) Before late-binding (handler nil): a harmless no-op, never a panic.
|
||||
c := &networkCatchup{}
|
||||
require.NoError(t, c.RequestAncestors(ids.Empty, ids.Empty, missing, peer))
|
||||
|
||||
// (b) Once wired: the engine's catch-up signal reaches the GetAncestors wire
|
||||
// exactly once — for the missing block, addressed to the peer that advertised
|
||||
// its child. RED before the fix: handler is never set, calls stays 0.
|
||||
spy := &catchupSpy{}
|
||||
c.handler = spy
|
||||
require.NoError(t, c.RequestAncestors(ids.Empty, ids.Empty, missing, peer))
|
||||
require.Equal(t, 1, spy.calls, "RequestAncestors must route to requestContext — a nil wire IS the stranded-follower bug")
|
||||
require.Equal(t, missing, spy.lastBlock)
|
||||
require.Equal(t, peer, spy.lastFrom)
|
||||
}
|
||||
+4
-35
@@ -46,48 +46,17 @@ func (s *Nets) GetOrCreate(chainID ids.ID) (nets.Net, bool) {
|
||||
return chain, true
|
||||
}
|
||||
|
||||
// IsChainBootstrapped reports whether the given chain has finished initial sync
|
||||
// (reached the network frontier and transitioned its VM to normal operation) on
|
||||
// this node — Bootstrapped(chainID) was called for it in its validation net. A
|
||||
// chain that is merely tracked (its sync goroutine launched) but has NOT converged
|
||||
// reads false. This is the per-chain truth manager.IsBootstrapped / info.isBootstrapped
|
||||
// key on, replacing the mere-existence test that returned true the instant a chain
|
||||
// was added to the manager (the premature-true masking bug: a C-Chain stalled at
|
||||
// genesis reported bootstrapped=true). A chainID is added to exactly one net's
|
||||
// tracking, so the first net that reports it bootstrapped is authoritative.
|
||||
func (s *Nets) IsChainBootstrapped(chainID ids.ID) bool {
|
||||
if s == nil {
|
||||
return false // no net tracking wired ⇒ nothing has been marked bootstrapped
|
||||
}
|
||||
s.lock.RLock()
|
||||
defer s.lock.RUnlock()
|
||||
|
||||
for _, chain := range s.chains {
|
||||
if chain.IsChainBootstrapped(chainID) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// Bootstrapping returns the chainIDs of any chains that are still
|
||||
// bootstrapping.
|
||||
//
|
||||
// s.chains is keyed by NET id, not chain id. Reporting the key named the net
|
||||
// instead of the chain: every primary-network chain that failed to converge
|
||||
// surfaced in /v1/health as the single ID
|
||||
// "11111111111111111111111111111111LpoYY" — constants.PrimaryNetworkID, i.e.
|
||||
// ids.Empty — a "chain" the chain manager has never heard of, so the operator
|
||||
// chasing it got "there is no chain with alias/ID". Worse, N stuck chains
|
||||
// collapsed into one indistinguishable entry. Ask each net which of ITS chains
|
||||
// are still bootstrapping; the net owns that set.
|
||||
func (s *Nets) Bootstrapping() []ids.ID {
|
||||
s.lock.RLock()
|
||||
defer s.lock.RUnlock()
|
||||
|
||||
chainsBootstrapping := make([]ids.ID, 0, len(s.chains))
|
||||
for _, chain := range s.chains {
|
||||
chainsBootstrapping = append(chainsBootstrapping, chain.Bootstrapping()...)
|
||||
for chainID, chain := range s.chains {
|
||||
if !chain.IsBootstrapped() {
|
||||
chainsBootstrapping = append(chainsBootstrapping, chainID)
|
||||
}
|
||||
}
|
||||
|
||||
return chainsBootstrapping
|
||||
|
||||
+2
-44
@@ -158,54 +158,12 @@ func TestNetsBootstrapping(t *testing.T) {
|
||||
chain, ok := chains.GetOrCreate(netID)
|
||||
require.True(ok)
|
||||
|
||||
// Start bootstrapping. What comes back is the CHAIN that is syncing, never
|
||||
// the net that holds it — this assertion used to demand netID, which is
|
||||
// how the phantom "11111111111111111111111111111111LpoYY" survived review.
|
||||
// Start bootstrapping
|
||||
chain.AddChain(chainID)
|
||||
bootstrapping := chains.Bootstrapping()
|
||||
require.Equal([]ids.ID{chainID}, bootstrapping)
|
||||
require.NotContains(bootstrapping, netID)
|
||||
require.Contains(bootstrapping, netID)
|
||||
|
||||
// Finish bootstrapping
|
||||
chain.Bootstrapped(chainID)
|
||||
require.Empty(chains.Bootstrapping())
|
||||
}
|
||||
|
||||
// The "bootstrapped" health check publishes Nets.Bootstrapping() verbatim as
|
||||
// its message. s.chains is keyed by NET id, so returning the key reported
|
||||
// constants.PrimaryNetworkID (ids.Empty, cb58
|
||||
// "11111111111111111111111111111111LpoYY") as though it were an unbootstrapped
|
||||
// CHAIN. Operators saw a chain ID the chain manager denies exists, and every
|
||||
// stuck chain on the net collapsed into that one phantom entry.
|
||||
func TestNetsBootstrappingReportsChainsNotNets(t *testing.T) {
|
||||
require := require.New(t)
|
||||
|
||||
chains, err := NewNets(ids.EmptyNodeID, map[ids.ID]nets.Config{
|
||||
constants.PrimaryNetworkID: {},
|
||||
})
|
||||
require.NoError(err)
|
||||
|
||||
primary, _ := chains.GetOrCreate(constants.PrimaryNetworkID)
|
||||
cChainID := ids.GenerateTestID()
|
||||
dChainID := ids.GenerateTestID()
|
||||
primary.AddChain(cChainID)
|
||||
primary.AddChain(dChainID)
|
||||
|
||||
bootstrapping := chains.Bootstrapping()
|
||||
|
||||
// The phantom: never the net's own ID.
|
||||
require.NotContains(bootstrapping, constants.PrimaryNetworkID)
|
||||
require.NotContains(
|
||||
bootstrapping,
|
||||
ids.Empty,
|
||||
"health check reported the primary NET id as an unbootstrapped chain",
|
||||
)
|
||||
// Both stuck chains must be individually nameable.
|
||||
require.ElementsMatch([]ids.ID{cChainID, dChainID}, bootstrapping)
|
||||
|
||||
primary.Bootstrapped(cChainID)
|
||||
require.Equal([]ids.ID{dChainID}, chains.Bootstrapping())
|
||||
|
||||
primary.Bootstrapped(dChainID)
|
||||
require.Empty(chains.Bootstrapping())
|
||||
}
|
||||
|
||||
@@ -1,139 +0,0 @@
|
||||
// Copyright (C) 2019-2026, Lux Industries, Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
// context_chunk_test.go — the heavy-block self-heal fix: GetContext must bound its Ancestors
|
||||
// response by SERIALIZED SIZE (not just block COUNT) so it stays under the peer message cap.
|
||||
//
|
||||
// Benchmark-proven live bug: under 250-trader DEX load, heavy blocks made a 256-block context
|
||||
// response sum to 3.4-5.7 MB > the 2 MB compressor cap; msgCreator.Ancestors FAILED to build,
|
||||
// so a behind validator received NOTHING and could never resync (permanently stuck while the tip
|
||||
// advanced). This pins the fix: the response is chunked to fit the budget, always serving at
|
||||
// least one block so the behind node makes progress every round.
|
||||
|
||||
package chains
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
consensuschain "github.com/luxfi/consensus/engine/chain"
|
||||
consensusblock "github.com/luxfi/consensus/engine/chain/block"
|
||||
"github.com/luxfi/constants"
|
||||
"github.com/luxfi/ids"
|
||||
"github.com/luxfi/log"
|
||||
"github.com/luxfi/node/message"
|
||||
)
|
||||
|
||||
// heavyBlock is a consensuschain.Block of a controlled byte size; only the methods GetContext
|
||||
// touches (ID/Parent/Bytes) are overridden — the rest of the interface is embedded and unused.
|
||||
type heavyBlock struct {
|
||||
consensusblock.Block
|
||||
id, parent ids.ID
|
||||
bytes []byte
|
||||
}
|
||||
|
||||
func (b *heavyBlock) ID() ids.ID { return b.id }
|
||||
func (b *heavyBlock) Parent() ids.ID { return b.parent }
|
||||
func (b *heavyBlock) Bytes() []byte { return b.bytes }
|
||||
|
||||
// sizeStubVM serves heavyBlocks by id (only GetBlock is called by GetContext).
|
||||
type sizeStubVM struct {
|
||||
consensuschain.BlockBuilder
|
||||
blocks map[ids.ID]consensusblock.Block
|
||||
}
|
||||
|
||||
func (v *sizeStubVM) GetBlock(_ context.Context, id ids.ID) (consensusblock.Block, error) {
|
||||
b, ok := v.blocks[id]
|
||||
if !ok {
|
||||
return nil, errors.New("not found")
|
||||
}
|
||||
return b, nil
|
||||
}
|
||||
|
||||
// sizeRecMsg records the containers GetContext hands to Ancestors, so the test can measure the
|
||||
// assembled response size (the input the real zstd compressor would reject above the cap).
|
||||
type sizeRecMsg struct {
|
||||
message.OutboundMsgBuilder
|
||||
containers [][]byte
|
||||
}
|
||||
|
||||
func (m *sizeRecMsg) Ancestors(_ ids.ID, _ uint32, containers [][]byte) (message.OutboundMessage, error) {
|
||||
m.containers = containers
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
func TestGetContext_ChunksBySize_FitsUnderCap(t *testing.T) {
|
||||
// A 100-block chain of ~150 KiB blocks. Packed by COUNT alone (up to maxContextBlocks=256,
|
||||
// i.e. all 100), the response is ~15 MB — 7x over the 2 MB cap, so the old handler's message
|
||||
// failed to build. Bounded by SIZE, it serves only as many as fit.
|
||||
const nBlocks = 100
|
||||
const blkSize = 150 * 1024
|
||||
|
||||
vm := &sizeStubVM{blocks: map[ids.ID]consensusblock.Block{}}
|
||||
parent := ids.Empty
|
||||
var tip ids.ID
|
||||
for i := 0; i < nBlocks; i++ {
|
||||
id := ids.GenerateTestID()
|
||||
vm.blocks[id] = &heavyBlock{id: id, parent: parent, bytes: make([]byte, blkSize)}
|
||||
parent = id
|
||||
tip = id
|
||||
}
|
||||
|
||||
msg := &sizeRecMsg{}
|
||||
bh := &blockHandler{
|
||||
logger: log.NewNoOpLogger(),
|
||||
vm: vm,
|
||||
msgCreator: msg,
|
||||
net: &redStubNet{},
|
||||
chainID: ids.GenerateTestID(),
|
||||
networkID: ids.GenerateTestID(),
|
||||
maxContextBlocks: 256,
|
||||
}
|
||||
|
||||
if err := bh.GetContext(context.Background(), ids.GenerateTestNodeID(), 1, time.Now().Add(time.Second), tip); err != nil {
|
||||
t.Fatalf("GetContext: %v", err)
|
||||
}
|
||||
|
||||
total := 0
|
||||
for _, c := range msg.containers {
|
||||
total += len(c)
|
||||
}
|
||||
budget := constants.DefaultMaxMessageSize - 128*1024 // the handler's byteBudget
|
||||
|
||||
if len(msg.containers) == 0 {
|
||||
t.Fatal("must serve at least one block (a behind node must always make progress)")
|
||||
}
|
||||
if total > budget {
|
||||
t.Fatalf("context response %d bytes exceeds budget %d — the real Ancestors compressor would REJECT it "+
|
||||
"(cap %d), stranding a behind validator (the live bug)", total, budget, constants.DefaultMaxMessageSize)
|
||||
}
|
||||
if len(msg.containers) >= nBlocks {
|
||||
t.Fatalf("expected SIZE truncation (fewer than %d blocks), got %d — response not chunked", nBlocks, len(msg.containers))
|
||||
}
|
||||
t.Logf("size-chunked: %d blocks, %d payload bytes (budget %d, cap %d) — fits, so the compressor accepts it",
|
||||
len(msg.containers), total, budget, constants.DefaultMaxMessageSize)
|
||||
}
|
||||
|
||||
// A single heavy block is ALWAYS served even if it alone exceeds the budget — the walk must never
|
||||
// deadlock (the trust-tiered validator cap gives such a block the send headroom; a stranger's
|
||||
// tight cap correctly rejects it downstream).
|
||||
func TestGetContext_SingleOversizeBlock_StillServed(t *testing.T) {
|
||||
oversize := constants.DefaultMaxMessageSize + 1<<20 // > the cap on its own
|
||||
id := ids.GenerateTestID()
|
||||
vm := &sizeStubVM{blocks: map[ids.ID]consensusblock.Block{
|
||||
id: &heavyBlock{id: id, parent: ids.Empty, bytes: make([]byte, oversize)},
|
||||
}}
|
||||
msg := &sizeRecMsg{}
|
||||
bh := &blockHandler{
|
||||
logger: log.NewNoOpLogger(), vm: vm, msgCreator: msg, net: &redStubNet{},
|
||||
chainID: ids.GenerateTestID(), networkID: ids.GenerateTestID(), maxContextBlocks: 256,
|
||||
}
|
||||
if err := bh.GetContext(context.Background(), ids.GenerateTestNodeID(), 1, time.Now().Add(time.Second), id); err != nil {
|
||||
t.Fatalf("GetContext: %v", err)
|
||||
}
|
||||
if len(msg.containers) != 1 {
|
||||
t.Fatalf("a single (even oversize) block must be served so the walk never deadlocks, got %d blocks", len(msg.containers))
|
||||
}
|
||||
}
|
||||
+195
-1355
File diff suppressed because it is too large
Load Diff
@@ -123,14 +123,8 @@ func (m *manager) authorizeChainActivation(chainID ids.ID) (authorized bool, rea
|
||||
// authorization from an NFT held at an address it does not control.
|
||||
|
||||
// The gate consults the X-Chain UTXO set, so the X-Chain must already be
|
||||
// bootstrapped on this node. IsBootstrapped now keys on REAL convergence
|
||||
// (sb.Bootstrapped), not mere presence — safe here because X-Chain is a DAG
|
||||
// chain marked bootstrapped SYNCHRONOUSLY inside createChain (Engine == nil
|
||||
// path) before the sequential chain-creator dequeues any re-pushed gated chain,
|
||||
// so IsBootstrapped(X) is already true when a gated chain re-runs. INVARIANT: if
|
||||
// X-Chain is ever linearized into an engine chain (async Bootstrapped via
|
||||
// monitorBootstrap), retryPendingGatedChains must be made to re-drain after X
|
||||
// converges, else gated chains park forever. If not bootstrapped yet, defer.
|
||||
// bootstrapped (created and tracked) on this node. If not, we cannot decide
|
||||
// yet — signal the caller to defer.
|
||||
if !m.IsBootstrapped(m.XChainID) {
|
||||
return false, false
|
||||
}
|
||||
|
||||
@@ -9,12 +9,10 @@ import (
|
||||
|
||||
"github.com/stretchr/testify/require"
|
||||
|
||||
"github.com/luxfi/constants"
|
||||
"github.com/luxfi/container/buffer"
|
||||
"github.com/luxfi/ids"
|
||||
"github.com/luxfi/log"
|
||||
"github.com/luxfi/math/set"
|
||||
"github.com/luxfi/node/nets"
|
||||
lux "github.com/luxfi/utxo"
|
||||
"github.com/luxfi/utxo/nftfx"
|
||||
"github.com/luxfi/utxo/secp256k1fx"
|
||||
@@ -165,10 +163,8 @@ func TestHoldsAuthorizationNFT(t *testing.T) {
|
||||
|
||||
// newGateManager builds the minimal manager needed to exercise the gate:
|
||||
// only the fields authorizeChainActivation reads. xChainVM, when non-nil, is
|
||||
// installed as the X-Chain's tracked VM AND marked bootstrapped in its net so
|
||||
// IsBootstrapped(XChainID) is true (the gate consults the X-Chain UTXO set, which
|
||||
// is only valid once the X-Chain has finished initial sync — mere presence in
|
||||
// m.chains is no longer sufficient) and xChainUTXOReader() resolves to it.
|
||||
// installed as the X-Chain's tracked VM so IsBootstrapped(XChainID) is true and
|
||||
// xChainUTXOReader() resolves to it.
|
||||
func newGateManager(
|
||||
xChainID ids.ID,
|
||||
stakingAddr ids.ShortID,
|
||||
@@ -176,29 +172,17 @@ func newGateManager(
|
||||
critical set.Set[ids.ID],
|
||||
xChainVM xChainUTXOReader,
|
||||
) *manager {
|
||||
netsTracker, err := NewNets(ids.GenerateTestNodeID(), map[ids.ID]nets.Config{
|
||||
constants.PrimaryNetworkID: {},
|
||||
})
|
||||
if err != nil {
|
||||
panic(err)
|
||||
}
|
||||
m := &manager{
|
||||
chains: make(map[ids.ID]*chainInfo),
|
||||
gatedAttempts: make(map[ids.ID]int),
|
||||
}
|
||||
m.Log = log.NewNoOpLogger()
|
||||
m.Nets = netsTracker
|
||||
m.XChainID = xChainID
|
||||
m.StakingXAddress = stakingAddr
|
||||
m.ChainAuthorizations = authz
|
||||
m.CriticalChains = critical
|
||||
if xChainVM != nil {
|
||||
m.chains[xChainID] = &chainInfo{Name: "X-Chain", VM: xChainVM}
|
||||
// The X-Chain has finished initial sync (its UTXO set is valid) — the real
|
||||
// precondition the gate depends on, now reflected in the net tracking.
|
||||
sb, _ := netsTracker.GetOrCreate(constants.PrimaryNetworkID)
|
||||
sb.AddChain(xChainID)
|
||||
sb.Bootstrapped(xChainID)
|
||||
}
|
||||
return m
|
||||
}
|
||||
@@ -294,14 +278,11 @@ func TestAuthorizeChainActivation(t *testing.T) {
|
||||
})
|
||||
|
||||
t.Run("gated chain, X-Chain tracked but VM lacks reader, opts out", func(t *testing.T) {
|
||||
// X-Chain is bootstrapped (in m.chains AND marked bootstrapped in its net, so
|
||||
// IsBootstrapped true) but its VM does not satisfy xChainUTXOReader. The gate must
|
||||
// fail closed (ready, !authz), not panic.
|
||||
// X-Chain is in m.chains (IsBootstrapped true) but its VM does not
|
||||
// satisfy xChainUTXOReader. The gate must fail closed (ready, !authz),
|
||||
// not panic.
|
||||
m := newGateManager(xChainID, addr, collection, nil, nil)
|
||||
m.chains[xChainID] = &chainInfo{Name: "X-Chain", VM: struct{}{}}
|
||||
sb, _ := m.Nets.GetOrCreate(constants.PrimaryNetworkID)
|
||||
sb.AddChain(xChainID)
|
||||
sb.Bootstrapped(xChainID)
|
||||
authorized, ready := m.authorizeChainActivation(gatedChain)
|
||||
require.True(t, ready)
|
||||
require.False(t, authorized)
|
||||
|
||||
@@ -177,69 +177,6 @@ func TestIsBootstrapped(t *testing.T) {
|
||||
require.False(m.IsBootstrapped(chainID))
|
||||
}
|
||||
|
||||
// TestIsBootstrappedTracksRealConvergence is the regression guard for the
|
||||
// premature-true masking bug: manager.IsBootstrapped must report true ONLY once the
|
||||
// chain has ACTUALLY finished initial sync (its net marked it Bootstrapped), NOT the
|
||||
// instant it is merely tracked (added to m.chains with its sync goroutine launched).
|
||||
// Before the fix, a C-Chain stalled at genesis (head 0x0) reported
|
||||
// info.isBootstrapped(C)=true, masking the stall from any readiness gate.
|
||||
func TestIsBootstrappedTracksRealConvergence(t *testing.T) {
|
||||
require := require.New(t)
|
||||
|
||||
chainConfigs := map[ids.ID]nets.Config{
|
||||
constants.PrimaryNetworkID: {},
|
||||
}
|
||||
netsTracker, err := NewNets(ids.GenerateTestNodeID(), chainConfigs)
|
||||
require.NoError(err)
|
||||
|
||||
config := &ManagerConfig{
|
||||
Log: log.NewNoOpLogger(),
|
||||
Metrics: metric.NewMultiGatherer(),
|
||||
VMManager: vms.NewManager(),
|
||||
ChainDataDir: t.TempDir(),
|
||||
Nets: netsTracker,
|
||||
}
|
||||
m, err := New(config)
|
||||
require.NoError(err)
|
||||
mImpl := m.(*manager)
|
||||
|
||||
// A native chain validated by the primary network — the C-Chain shape.
|
||||
chainID := ids.GenerateTestID()
|
||||
|
||||
// Simulate createChain's tracking: the chain EXISTS in m.chains and is registered
|
||||
// as bootstrapping in its validation net — but has NOT converged (initial sync is
|
||||
// still driving, e.g. stalled at genesis fetching ancestry).
|
||||
mImpl.chainsLock.Lock()
|
||||
mImpl.chains[chainID] = &chainInfo{Name: "C-Chain"}
|
||||
mImpl.chainsLock.Unlock()
|
||||
sb, _ := netsTracker.GetOrCreate(constants.PrimaryNetworkID)
|
||||
require.True(sb.AddChain(chainID))
|
||||
|
||||
// THE FIX: exists-but-not-converged must be FALSE (was true — the masking bug).
|
||||
require.False(m.IsBootstrapped(chainID),
|
||||
"a tracked-but-still-syncing chain must not report bootstrapped")
|
||||
for _, ci := range m.(*manager).GetChains() {
|
||||
if ci.ID == chainID {
|
||||
require.False(ci.Bootstrapped, "GetChains must not report a syncing chain bootstrapped")
|
||||
}
|
||||
}
|
||||
|
||||
// Initial sync reaches the frontier → monitorBootstrap calls sb.Bootstrapped.
|
||||
sb.Bootstrapped(chainID)
|
||||
|
||||
// Now — and only now — it reports bootstrapped (head advanced to frontier, VM live).
|
||||
require.True(m.IsBootstrapped(chainID),
|
||||
"a converged chain must report bootstrapped")
|
||||
found := false
|
||||
for _, ci := range m.(*manager).GetChains() {
|
||||
if ci.ID == chainID {
|
||||
found = true
|
||||
require.True(ci.Bootstrapped, "GetChains must report a converged chain bootstrapped")
|
||||
}
|
||||
}
|
||||
require.True(found)
|
||||
}
|
||||
|
||||
// TestToEngineChannelFlow verifies the toEngine channel notification flow
|
||||
// This tests the goroutine that reads from toEngine and triggers block building
|
||||
func TestToEngineChannelFlow(t *testing.T) {
|
||||
|
||||
@@ -365,12 +365,6 @@ func (vm *pChainHeightVM) LastAccepted(ctx context.Context) (ids.ID, error) {
|
||||
return vm.inner.LastAccepted(ctx)
|
||||
}
|
||||
|
||||
// NOTE: this wrapper deliberately does NOT forward GetBlockIDAtHeight. The bootstrap acceptance
|
||||
// oracle's fork-sibling check reads the IN-PROCESS consensus finalized ledger
|
||||
// (blockHandler.finalizedBlockAtHeight → engine.FinalizedBlockAtHeight), NOT a VM height index —
|
||||
// because the VM index is dead over ZAP (the zap server has no MsgGetBlockIDAtHeight handler, so
|
||||
// the real C-Chain returns nothing). A forwarder here would only re-expose that dead path.
|
||||
|
||||
// SetPreference delegates to the inner VM.
|
||||
func (vm *pChainHeightVM) SetPreference(ctx context.Context, id ids.ID) error {
|
||||
return vm.inner.SetPreference(ctx, id)
|
||||
|
||||
@@ -1,130 +0,0 @@
|
||||
// Copyright (C) 2019-2026, Lux Industries, Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
// proposervm_wrap_test.go — pins the single policy gate that decides whether a
|
||||
// linear chain.ChainVM is re-wrapped in proposervm for single-proposer-per-height
|
||||
// block production (the consensus-safety fix for the equivocation crash). The
|
||||
// gate is a pure function so the policy is verifiable without standing up a chain.
|
||||
package chains
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"github.com/luxfi/constants"
|
||||
"github.com/luxfi/ids"
|
||||
)
|
||||
|
||||
func TestShouldWrapInProposerVM(t *testing.T) {
|
||||
// A native chain ID that is NOT the P-Chain (e.g. the C-Chain): first 31
|
||||
// bytes zero, last byte the chain letter. Any non-platform ID exercises the
|
||||
// chainID condition; this mirrors how native chain IDs are shaped.
|
||||
cChainID := ids.ID{}
|
||||
cChainID[ids.IDLen-1] = 'C'
|
||||
xChainID := ids.ID{}
|
||||
xChainID[ids.IDLen-1] = 'X'
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
k int
|
||||
chainID ids.ID
|
||||
innerIsDAGNative bool
|
||||
want bool
|
||||
why string
|
||||
}{
|
||||
{
|
||||
name: "C-Chain devnet K=4",
|
||||
k: 4, // LocalBFTParams
|
||||
chainID: cChainID,
|
||||
innerIsDAGNative: false,
|
||||
want: true,
|
||||
why: "multi-validator EVM, not the P-Chain, not DAG — the chain that crashed; must be wrapped",
|
||||
},
|
||||
{
|
||||
name: "C-Chain mainnet K=21",
|
||||
k: 21, // MainnetParams
|
||||
chainID: cChainID,
|
||||
innerIsDAGNative: false,
|
||||
want: true,
|
||||
why: "large multi-validator EVM is wrapped exactly as avalanchego wraps it",
|
||||
},
|
||||
{
|
||||
name: "P-Chain is excluded even at K>1",
|
||||
k: 4,
|
||||
chainID: constants.PlatformChainID,
|
||||
innerIsDAGNative: false,
|
||||
want: false,
|
||||
why: "P-Chain validator state is published mid-create; its windower would be empty — keep newPChainHeightVM",
|
||||
},
|
||||
{
|
||||
name: "X-Chain (DAG-native) is excluded",
|
||||
k: 4,
|
||||
chainID: xChainID,
|
||||
innerIsDAGNative: true,
|
||||
want: false,
|
||||
why: "linearized DAG VM uses a push-notification bridge that does not compose with proposervm's window",
|
||||
},
|
||||
{
|
||||
name: "single-node K=1 is not wrapped",
|
||||
k: 1, // SingleValidatorParams
|
||||
chainID: cChainID,
|
||||
innerIsDAGNative: false,
|
||||
want: false,
|
||||
why: "one validator is trivially the sole proposer; no equivocation to prevent",
|
||||
},
|
||||
{
|
||||
name: "K=1 P-Chain is not wrapped",
|
||||
k: 1,
|
||||
chainID: constants.PlatformChainID,
|
||||
innerIsDAGNative: false,
|
||||
want: false,
|
||||
why: "K==1 short-circuits regardless of chain",
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
got := shouldWrapInProposerVM(tt.k, tt.chainID, tt.innerIsDAGNative)
|
||||
if got != tt.want {
|
||||
t.Fatalf("shouldWrapInProposerVM(k=%d, chainID=%s, dag=%v) = %v, want %v — %s",
|
||||
tt.k, tt.chainID, tt.innerIsDAGNative, got, tt.want, tt.why)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestShouldWrapInProposerVM_FlowsFromSelectedParams proves the K the gate reads
|
||||
// is the one selectConsensusParams produces per network, so the wrap decision is
|
||||
// consistent with the BFT committee actually chosen: every sybil-protected
|
||||
// network yields K>1 (C-Chain wrapped), and single-node yields K==1 (not wrapped).
|
||||
func TestShouldWrapInProposerVM_FlowsFromSelectedParams(t *testing.T) {
|
||||
cChainID := ids.ID{}
|
||||
cChainID[ids.IDLen-1] = 'C'
|
||||
|
||||
cases := []struct {
|
||||
name string
|
||||
sybilProtection bool
|
||||
networkID uint32
|
||||
wantWrapCChain bool
|
||||
}{
|
||||
{"single-node dev", false, constants.LocalID, false},
|
||||
{"devnet sybil", true, constants.DevnetID, true},
|
||||
{"localnet sybil", true, constants.LocalID, true},
|
||||
{"mainnet", true, constants.MainnetID, true},
|
||||
{"testnet", true, constants.TestnetID, true},
|
||||
}
|
||||
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
params := selectConsensusParams(c.sybilProtection, c.networkID)
|
||||
got := shouldWrapInProposerVM(params.K, cChainID, false)
|
||||
if got != c.wantWrapCChain {
|
||||
t.Fatalf("network %s (sybil=%v): K=%d → wrap=%v, want %v",
|
||||
c.name, c.sybilProtection, params.K, got, c.wantWrapCChain)
|
||||
}
|
||||
// The P-Chain is never wrapped, whatever the committee.
|
||||
if shouldWrapInProposerVM(params.K, constants.PlatformChainID, false) {
|
||||
t.Fatalf("network %s: P-Chain must never be wrapped (K=%d)", c.name, params.K)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -87,32 +87,6 @@ func selectConsensusParams(sybilProtection bool, networkID uint32) consensusconf
|
||||
}
|
||||
}
|
||||
|
||||
// shouldWrapInProposerVM decides whether a linear chain.ChainVM is wrapped in
|
||||
// proposervm to enforce single-proposer-per-height block production (the
|
||||
// proposer window). It is the SINGLE policy gate (the manager calls it once);
|
||||
// keeping it a pure function makes the policy unit-testable without standing up
|
||||
// a whole chain. All three conditions must hold:
|
||||
//
|
||||
// - k > 1: a multi-validator quorum. K==1 (single-node --dev) has exactly one
|
||||
// proposer already, so there is no equivocation to prevent and no schedule
|
||||
// to compute (proposervm would only add a wrapper with no safety value).
|
||||
// - chainID is NOT the P-Chain: the P-Chain publishes its OWN height-indexed
|
||||
// validators.State DURING its createChain, AFTER the chainRuntime snapshot
|
||||
// proposervm's windower reads — so its windower would see an empty set and
|
||||
// fall back to anyone-can-propose with a P-chain-height-0 stamp. The P-Chain
|
||||
// keeps the existing newPChainHeightVM path (which DOES get the live state).
|
||||
// - inner is NOT DAG-native (no Linearize): a linearized DAG VM (X-Chain) is
|
||||
// driven by a push-notification bridge that does not compose with
|
||||
// proposervm's pull/window model without avalanchego's initializeOnLinearizeVM
|
||||
// machinery. It keeps the existing path.
|
||||
//
|
||||
// The C-Chain and sovereign-L1 EVM chains satisfy all three (multi-validator,
|
||||
// not the P-Chain, not DAG) — they are exactly the chains that exhibited the
|
||||
// equivocation crash, and exactly the chains avalanchego wraps in proposervm.
|
||||
func shouldWrapInProposerVM(k int, chainID ids.ID, innerIsDAGNative bool) bool {
|
||||
return k > 1 && chainID != constants.PlatformChainID && !innerIsDAGNative
|
||||
}
|
||||
|
||||
// --- BLS vote verifier -------------------------------------------------------
|
||||
|
||||
// blsVoteVerifier verifies a validator's BLS signature over the canonical vote
|
||||
@@ -267,16 +241,6 @@ func (s *validatorStakeSource) TotalStake(height uint64) uint64 {
|
||||
return total
|
||||
}
|
||||
|
||||
// ValidatorCount implements consensuschain.StakeSource. The number of DISTINCT
|
||||
// validators in the set IN FORCE AT height — the round-scoped view-change's BFT
|
||||
// committee size (it sizes its POL/precommit quorum to bftAlpha over this count,
|
||||
// NOT the oversized sample K). Read from the SAME height-indexed set as
|
||||
// Weight/TotalStake so every node computes the identical committee and the
|
||||
// count-quorum matches the ⅔-by-stake set exactly.
|
||||
func (s *validatorStakeSource) ValidatorCount(height uint64) int {
|
||||
return len(validatorSetAtHeight(s.state, s.networkID, height))
|
||||
}
|
||||
|
||||
var _ consensuschain.StakeSource = (*validatorStakeSource)(nil)
|
||||
|
||||
// --- validator-set-root source (MEDIUM: epoch binding) -----------------------
|
||||
@@ -374,15 +338,11 @@ func hashValidatorSet(set map[ids.NodeID]*validators.GetValidatorOutput) ids.ID
|
||||
//
|
||||
// kind 1 = signed vote (payload = engine encodeSignedVote: nodeID+sig)
|
||||
// kind 2 = finality cert (payload = engine cert MarshalBinary)
|
||||
// (round-scoped view-change prevotes are engine-INTERNAL since consensus v1.36 —
|
||||
// the node no longer frames or routes a prevote kind)
|
||||
var quorumGossipMagic = [4]byte{'L', 'X', 'Q', 0x01}
|
||||
|
||||
const (
|
||||
quorumKindVote byte = 1
|
||||
quorumKindCert byte = 2
|
||||
// kind 3 (prevote) was DELETED with the v1.36 view-change rip-out (174af3c31); Nova sampling
|
||||
// decides and the ⅔ Quasar attestation rides quorumKindVote. Do not reuse 3 — keep the braid dead.
|
||||
)
|
||||
|
||||
// ErrNotQuorumGossip signals a payload is not a quorum envelope (so the caller
|
||||
|
||||
@@ -1,102 +0,0 @@
|
||||
// Copyright (C) 2019-2026, Lux Industries, Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
// red_pendingcontext_dos_test.go — regression guard for the catch-up DoS RED found
|
||||
// on the cert-carrying catch-up branch.
|
||||
//
|
||||
// The frontier-sync wiring connects the AcceptedFrontier handler to
|
||||
// requestContext(), which records each requested blockID in b.pendingContext.
|
||||
// Originally NOTHING evicted from that map, so a Byzantine peer streaming
|
||||
// AcceptedFrontier frames each naming a distinct random tip grew it without bound
|
||||
// → OOM (and a peer that took a request then withheld Context re-stranded the
|
||||
// victim forever). requestContext now reaps entries past pendingContextTTL and
|
||||
// hard-caps the map at maxPendingContext. These tests pin both properties; before
|
||||
// the fix the first asserted N=50_000 entries with ZERO eviction.
|
||||
package chains
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/luxfi/ids"
|
||||
"github.com/luxfi/log"
|
||||
"github.com/luxfi/math/set"
|
||||
"github.com/luxfi/node/message"
|
||||
"github.com/luxfi/node/network"
|
||||
"github.com/luxfi/node/proto/p2p"
|
||||
)
|
||||
|
||||
// redStubNet implements the one Network method requestContext uses (Send); every
|
||||
// other method is inherited from the embedded nil interface and is never called.
|
||||
type redStubNet struct {
|
||||
network.Network
|
||||
sends int
|
||||
}
|
||||
|
||||
func (s *redStubNet) Send(_ message.OutboundMessage, nodeIDs set.Set[ids.NodeID], _ ids.ID, _ uint32) set.Set[ids.NodeID] {
|
||||
s.sends++
|
||||
return nodeIDs
|
||||
}
|
||||
|
||||
// redStubMsg implements the one OutboundMsgBuilder method requestContext uses.
|
||||
type redStubMsg struct {
|
||||
message.OutboundMsgBuilder
|
||||
}
|
||||
|
||||
func (redStubMsg) GetAncestors(_ ids.ID, _ uint32, _ time.Duration, _ ids.ID, _ p2p.EngineType) (message.OutboundMessage, error) {
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
func newRedTestHandler(net network.Network) *blockHandler {
|
||||
return &blockHandler{
|
||||
logger: log.NewNoOpLogger(),
|
||||
net: net,
|
||||
msgCreator: redStubMsg{},
|
||||
chainID: ids.GenerateTestID(),
|
||||
pendingContext: make(map[ids.ID]contextRequest),
|
||||
}
|
||||
}
|
||||
|
||||
// TestPendingContext_BoundedUnderFlood: a peer streaming distinct fake tips into
|
||||
// requestContext can NEVER grow pendingContext past maxPendingContext. Inverts the
|
||||
// original RED PoC, which asserted unbounded growth to 50_000 → OOM.
|
||||
func TestPendingContext_BoundedUnderFlood(t *testing.T) {
|
||||
stubNet := &redStubNet{}
|
||||
bh := newRedTestHandler(stubNet)
|
||||
|
||||
const N = 10_000 // ≫ the maxPendingContext cap — enough to prove "stays bounded"
|
||||
from := ids.GenerateTestNodeID()
|
||||
for i := 0; i < N; i++ {
|
||||
bh.requestContext(context.Background(), from, ids.GenerateTestID())
|
||||
}
|
||||
|
||||
if got := len(bh.pendingContext); got > maxPendingContext {
|
||||
t.Fatalf("pendingContext unbounded: %d entries exceeds cap %d (the RED HIGH DoS)", got, maxPendingContext)
|
||||
}
|
||||
t.Logf("bounded: %d entries after a %d-distinct-tip flood (cap %d, sends=%d)",
|
||||
len(bh.pendingContext), N, maxPendingContext, stubNet.sends)
|
||||
}
|
||||
|
||||
// TestPendingContext_StaleEntriesReaped: a request whose Context is withheld past
|
||||
// its TTL is reaped on the next requestContext, so the block is re-requestable from
|
||||
// an honest peer (fixes the RED MEDIUM re-strand).
|
||||
func TestPendingContext_StaleEntriesReaped(t *testing.T) {
|
||||
bh := newRedTestHandler(&redStubNet{})
|
||||
from := ids.GenerateTestNodeID()
|
||||
|
||||
stale := ids.GenerateTestID()
|
||||
bh.pendingContext[stale] = contextRequest{
|
||||
nodeID: from,
|
||||
requestID: 1,
|
||||
blockID: stale,
|
||||
timestamp: time.Now().Add(-2 * pendingContextTTL),
|
||||
}
|
||||
|
||||
// Any later request runs the reaper before recording its own entry.
|
||||
bh.requestContext(context.Background(), from, ids.GenerateTestID())
|
||||
|
||||
if _, stillThere := bh.pendingContext[stale]; stillThere {
|
||||
t.Fatalf("stale pendingContext entry (%v old) not reaped → re-strand persists", 2*pendingContextTTL)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,244 @@
|
||||
# Robust RPC Handler Registration System
|
||||
|
||||
## Overview
|
||||
|
||||
This package provides a bulletproof RPC handler registration system for the Lux node, designed to handle the complexities of local development where nodes are frequently restarted. It replaces the fragile inline registration logic with a robust, maintainable solution.
|
||||
|
||||
## Key Features
|
||||
|
||||
### 🔄 Automatic Retry Logic
|
||||
- Exponential backoff for transient failures
|
||||
- Configurable retry count and wait times
|
||||
- Context-aware cancellation support
|
||||
|
||||
### ✅ Built-in Health Checks
|
||||
- Automatic validation after registration
|
||||
- Batch health checking for all chains
|
||||
- Detailed diagnostics for failures
|
||||
|
||||
### 🎯 Single Source of Truth
|
||||
- Centralized route construction logic
|
||||
- Consistent path formatting
|
||||
- No duplicate code or magic strings
|
||||
|
||||
### 🛡️ Defensive Programming
|
||||
- Nil checks on all inputs
|
||||
- Handler validation before registration
|
||||
- Graceful degradation on failures
|
||||
|
||||
### 📊 Developer-Friendly Debugging
|
||||
- Clear, actionable error messages
|
||||
- Comprehensive logging at appropriate levels
|
||||
- Built-in diagnostic tools
|
||||
|
||||
## Architecture
|
||||
|
||||
```
|
||||
┌─────────────────────┐
|
||||
│ Chain Manager │
|
||||
└──────────┬──────────┘
|
||||
│ Creates Chain
|
||||
▼
|
||||
┌─────────────────────┐
|
||||
│ ChainHandlerRegistrar│
|
||||
└──────────┬──────────┘
|
||||
│ Extracts Handlers
|
||||
▼
|
||||
┌─────────────────────┐
|
||||
│ Handler Manager │
|
||||
└──────────┬──────────┘
|
||||
│ Registers with Retries
|
||||
▼
|
||||
┌─────────────────────┐
|
||||
│ API Server │
|
||||
└─────────────────────┘
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
### Basic Integration
|
||||
|
||||
Replace the handler registration code in `chains/manager.go` (lines 941-990) with:
|
||||
|
||||
```go
|
||||
// Create robust registrar
|
||||
registrar := rpc.NewChainHandlerRegistrar(
|
||||
m.Server,
|
||||
m.Log,
|
||||
m.CChainID,
|
||||
m.PChainID,
|
||||
)
|
||||
|
||||
// Register handlers
|
||||
if err := registrar.RegisterChainHandlers(ctx, chainParams.ID, chain.VM); err != nil {
|
||||
m.Log.Error("Failed to register handlers", log.Err(err))
|
||||
// Decide if this should be fatal or not
|
||||
}
|
||||
```
|
||||
|
||||
### Configuration
|
||||
|
||||
```go
|
||||
// Development environment - fail fast
|
||||
registrar.SetRetryConfig(2, 50*time.Millisecond)
|
||||
|
||||
// Production environment - more robust
|
||||
registrar.SetRetryConfig(5, 200*time.Millisecond)
|
||||
```
|
||||
|
||||
### Debugging
|
||||
|
||||
```go
|
||||
// Get route information
|
||||
info, exists := registrar.GetRouteInfo(chainID)
|
||||
if exists {
|
||||
fmt.Printf("Chain %s routes: %v\n", chainID, info.Endpoints)
|
||||
}
|
||||
|
||||
// Run health checks
|
||||
results := registrar.HealthCheckAll()
|
||||
for chainID, healthy := range results {
|
||||
fmt.Printf("Chain %s: %v\n", chainID, healthy)
|
||||
}
|
||||
|
||||
// Validate specific endpoint
|
||||
err := registrar.ValidateEndpoint(chainID, "/rpc")
|
||||
```
|
||||
|
||||
### Using the Debug Tool
|
||||
|
||||
```go
|
||||
// Quick diagnosis from CLI
|
||||
rpc.QuickDiagnose("localhost:9650", chainID, "C")
|
||||
|
||||
// Programmatic diagnosis
|
||||
tool := rpc.NewDebugTool("localhost:9650", logger)
|
||||
report := tool.DiagnoseEndpoint(chainID, "C")
|
||||
fmt.Println(report.String())
|
||||
```
|
||||
|
||||
## Components
|
||||
|
||||
### HandlerManager (`handler_manager.go`)
|
||||
Core registration logic with retry mechanism and health checks.
|
||||
|
||||
**Key Methods:**
|
||||
- `RegisterChainHandlers()` - Main registration entry point
|
||||
- `HealthCheckRoute()` - Validates handler responsiveness
|
||||
- `GetRouteInfo()` - Retrieves registration details
|
||||
|
||||
### ChainHandlerRegistrar (`chain_integration.go`)
|
||||
Bridge between chain manager and handler manager.
|
||||
|
||||
**Key Methods:**
|
||||
- `RegisterChainHandlers()` - Extracts and registers handlers
|
||||
- `ValidateEndpoint()` - Tests specific endpoints
|
||||
- `GetAllRoutes()` - Returns all registered routes
|
||||
|
||||
### DebugTool (`debug_tool.go`)
|
||||
Comprehensive endpoint diagnostics for developers.
|
||||
|
||||
**Key Methods:**
|
||||
- `DiagnoseEndpoint()` - Full endpoint analysis
|
||||
- `QuickDiagnose()` - CLI-friendly diagnosis
|
||||
|
||||
## Error Handling
|
||||
|
||||
The system uses clear, actionable errors:
|
||||
|
||||
```go
|
||||
errNilHandler = errors.New("handler is nil")
|
||||
errNilServer = errors.New("server is nil")
|
||||
errEmptyEndpoint = errors.New("endpoint is empty")
|
||||
errRegistrationFailed = errors.New("handler registration failed")
|
||||
errHealthCheckFailed = errors.New("health check failed")
|
||||
```
|
||||
|
||||
Each error includes context about what failed and why.
|
||||
|
||||
## Testing
|
||||
|
||||
Comprehensive test coverage including:
|
||||
- Successful registration scenarios
|
||||
- Validation failure cases
|
||||
- Retry logic verification
|
||||
- Health check validation
|
||||
- Context cancellation
|
||||
- Performance benchmarks
|
||||
|
||||
Run tests:
|
||||
```bash
|
||||
go test ./chains/rpc/... -v
|
||||
```
|
||||
|
||||
## Common Issues and Solutions
|
||||
|
||||
### Issue: Handlers not accessible after registration
|
||||
**Solution:** Check health status with `HealthCheckAll()` and review debug output.
|
||||
|
||||
### Issue: Registration fails with "already exists"
|
||||
**Solution:** The retry logic handles this. If persistent, check for duplicate registration attempts.
|
||||
|
||||
### Issue: Slow registration during development
|
||||
**Solution:** Reduce retry count and wait time using `SetRetryConfig()`.
|
||||
|
||||
### Issue: Can't find the correct endpoint URL
|
||||
**Solution:** Use `DebugTool.DiagnoseEndpoint()` to test all URL patterns.
|
||||
|
||||
## Migration Guide
|
||||
|
||||
1. **Update imports:**
|
||||
```go
|
||||
import "github.com/luxfi/node/chains/rpc"
|
||||
```
|
||||
|
||||
2. **Replace inline registration (lines 941-990 in manager.go):**
|
||||
```go
|
||||
// Old code: complex type checking and manual registration
|
||||
// New code: single function call
|
||||
registrar := rpc.NewChainHandlerRegistrar(...)
|
||||
registrar.RegisterChainHandlers(...)
|
||||
```
|
||||
|
||||
3. **Add health monitoring (optional):**
|
||||
```go
|
||||
go func() {
|
||||
time.Sleep(5 * time.Second)
|
||||
registrar.HealthCheckAll()
|
||||
}()
|
||||
```
|
||||
|
||||
4. **Add debugging endpoints (optional):**
|
||||
```go
|
||||
http.HandleFunc("/debug/handlers", func(w http.ResponseWriter, r *http.Request) {
|
||||
routes := registrar.GetAllRoutes()
|
||||
json.NewEncoder(w).Encode(routes)
|
||||
})
|
||||
```
|
||||
|
||||
## Performance
|
||||
|
||||
- Registration: ~1ms per handler (without retries)
|
||||
- Health check: ~10ms per chain
|
||||
- Memory overhead: ~1KB per registered chain
|
||||
- No goroutine leaks or resource issues
|
||||
|
||||
## Future Improvements
|
||||
|
||||
Potential enhancements:
|
||||
- Metrics integration for registration success/failure rates
|
||||
- Automatic re-registration on failure
|
||||
- WebSocket-specific health checks
|
||||
- gRPC handler support
|
||||
- Handler versioning for upgrades
|
||||
|
||||
## Philosophy
|
||||
|
||||
This implementation follows core Go principles:
|
||||
- **Explicit over implicit** - Clear registration flow
|
||||
- **Errors are values** - Proper error handling throughout
|
||||
- **Simple over clever** - Straightforward retry logic
|
||||
- **Composition over inheritance** - Small, focused components
|
||||
- **Documentation is code** - Self-documenting with clear names
|
||||
|
||||
The system is designed to be bulletproof for development while remaining simple to understand and maintain.
|
||||
@@ -0,0 +1,168 @@
|
||||
// Copyright (C) 2019-2025, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
package rpc
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"time"
|
||||
|
||||
"github.com/luxfi/ids"
|
||||
"github.com/luxfi/log"
|
||||
"github.com/luxfi/node/server/http"
|
||||
"github.com/luxfi/node/vms"
|
||||
)
|
||||
|
||||
// ChainHandlerRegistrar provides a clean interface for chain manager to register handlers.
|
||||
// This replaces the inline registration logic with a more robust, testable solution.
|
||||
type ChainHandlerRegistrar struct {
|
||||
manager *HandlerManager
|
||||
server server.Server
|
||||
log log.Logger
|
||||
cChainID ids.ID // Special handling for C-Chain
|
||||
pChainID ids.ID // Platform chain ID for validation
|
||||
}
|
||||
|
||||
// NewChainHandlerRegistrar creates a registrar for chain handler registration.
|
||||
// Encapsulates all the registration logic in one place.
|
||||
func NewChainHandlerRegistrar(
|
||||
server server.Server,
|
||||
logger log.Logger,
|
||||
cChainID ids.ID,
|
||||
pChainID ids.ID,
|
||||
) *ChainHandlerRegistrar {
|
||||
return &ChainHandlerRegistrar{
|
||||
manager: NewHandlerManager(server, logger),
|
||||
server: server,
|
||||
log: logger,
|
||||
cChainID: cChainID,
|
||||
pChainID: pChainID,
|
||||
}
|
||||
}
|
||||
|
||||
// RegisterChainHandlers is the main entry point from chain manager.
|
||||
// Handles all the complexity of VM type checking and handler extraction.
|
||||
func (r *ChainHandlerRegistrar) RegisterChainHandlers(
|
||||
ctx context.Context,
|
||||
chainID ids.ID,
|
||||
vm interface{},
|
||||
) error {
|
||||
r.log.Info("Attempting to register chain handlers",
|
||||
log.Stringer("chainID", chainID),
|
||||
log.String("vmType", fmt.Sprintf("%T", vm)))
|
||||
|
||||
// Don't register handlers for Platform VM
|
||||
if chainID == r.pChainID {
|
||||
r.log.Debug("Skipping handler registration for Platform VM")
|
||||
return nil
|
||||
}
|
||||
|
||||
// Extract handlers from VM
|
||||
handlers, err := r.extractHandlers(ctx, vm)
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to extract handlers: %w", err)
|
||||
}
|
||||
|
||||
if len(handlers) == 0 {
|
||||
r.log.Info("VM does not provide any handlers",
|
||||
log.Stringer("chainID", chainID))
|
||||
return nil
|
||||
}
|
||||
|
||||
// Determine chain alias (special case for C-Chain)
|
||||
alias := r.getChainAlias(chainID)
|
||||
|
||||
// Register with robust handler manager
|
||||
return r.manager.RegisterChainHandlers(ctx, chainID, alias, handlers)
|
||||
}
|
||||
|
||||
// extractHandlers attempts to get handlers from the VM using multiple strategies.
|
||||
// Handles different VM wrapper types gracefully.
|
||||
func (r *ChainHandlerRegistrar) extractHandlers(
|
||||
ctx context.Context,
|
||||
vm interface{},
|
||||
) (map[string]http.Handler, error) {
|
||||
// First try direct interface check
|
||||
if provider, ok := vm.(vms.HandlerProvider); ok {
|
||||
r.log.Debug("VM directly implements HandlerProvider")
|
||||
return provider.CreateHandlers(ctx)
|
||||
}
|
||||
|
||||
// Try using the delegate helper (handles wrapped VMs)
|
||||
handlers, err := vms.DelegateHandlers(ctx, vm)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("handler delegation failed: %w", err)
|
||||
}
|
||||
|
||||
if len(handlers) > 0 {
|
||||
r.log.Debug("Successfully extracted handlers via delegation",
|
||||
log.Int("count", len(handlers)))
|
||||
}
|
||||
|
||||
return handlers, nil
|
||||
}
|
||||
|
||||
// getChainAlias returns the appropriate alias for a chain.
|
||||
// C-Chain gets special treatment, others use their ID.
|
||||
func (r *ChainHandlerRegistrar) getChainAlias(chainID ids.ID) string {
|
||||
if chainID == r.cChainID {
|
||||
return "C"
|
||||
}
|
||||
// Could extend this for X-Chain and P-Chain if needed
|
||||
return ""
|
||||
}
|
||||
|
||||
// GetRouteInfo returns information about a specific chain's registered routes.
|
||||
// Useful for debugging and operational visibility.
|
||||
func (r *ChainHandlerRegistrar) GetRouteInfo(chainID ids.ID) (*RouteInfo, bool) {
|
||||
return r.manager.GetRouteInfo(chainID)
|
||||
}
|
||||
|
||||
// GetAllRoutes returns all registered routes across all chains.
|
||||
// Complete visibility for monitoring and debugging.
|
||||
func (r *ChainHandlerRegistrar) GetAllRoutes() map[string]*RouteInfo {
|
||||
return r.manager.GetAllRoutes()
|
||||
}
|
||||
|
||||
// HealthCheckAll performs health checks on all registered routes.
|
||||
// Returns a map of chainID -> healthy status.
|
||||
func (r *ChainHandlerRegistrar) HealthCheckAll() map[string]bool {
|
||||
return r.manager.HealthCheckAll()
|
||||
}
|
||||
|
||||
// SetRetryConfig allows tuning of retry behavior for different environments.
|
||||
// Production might want more retries, dev might want faster failures.
|
||||
func (r *ChainHandlerRegistrar) SetRetryConfig(maxRetries int, initialWait time.Duration) {
|
||||
r.manager.SetRetryConfig(maxRetries, initialWait)
|
||||
}
|
||||
|
||||
// ValidateEndpoint performs a test request against a specific endpoint.
|
||||
// Useful for debugging specific handler issues.
|
||||
func (r *ChainHandlerRegistrar) ValidateEndpoint(
|
||||
chainID ids.ID,
|
||||
endpoint string,
|
||||
) error {
|
||||
info, exists := r.manager.GetRouteInfo(chainID)
|
||||
if !exists {
|
||||
return fmt.Errorf("no routes registered for chain %s", chainID)
|
||||
}
|
||||
|
||||
// Build the full URL
|
||||
fullURL := fmt.Sprintf("/ext/%s%s", info.Base, endpoint)
|
||||
|
||||
r.log.Info("Validating endpoint",
|
||||
log.Stringer("chainID", chainID),
|
||||
log.String("url", fullURL))
|
||||
|
||||
// Validates endpoint registration
|
||||
for _, registered := range info.Endpoints {
|
||||
if registered == endpoint {
|
||||
return nil
|
||||
}
|
||||
}
|
||||
|
||||
return fmt.Errorf("endpoint %s not found in registered endpoints: %v",
|
||||
endpoint, info.Endpoints)
|
||||
}
|
||||
@@ -0,0 +1,302 @@
|
||||
// Copyright (C) 2019-2025, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
package rpc
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"github.com/go-json-experiment/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/luxfi/ids"
|
||||
"github.com/luxfi/log"
|
||||
)
|
||||
|
||||
// DebugTool provides utilities for debugging RPC handler issues.
|
||||
// Developer-friendly diagnostics with clear, actionable output.
|
||||
type DebugTool struct {
|
||||
baseURL string
|
||||
client *http.Client
|
||||
log log.Logger
|
||||
}
|
||||
|
||||
// NewDebugTool creates a debug tool for RPC endpoint testing.
|
||||
func NewDebugTool(baseURL string, logger log.Logger) *DebugTool {
|
||||
if !strings.HasPrefix(baseURL, "http") {
|
||||
baseURL = "http://" + baseURL
|
||||
}
|
||||
|
||||
return &DebugTool{
|
||||
baseURL: strings.TrimSuffix(baseURL, "/"),
|
||||
client: &http.Client{
|
||||
Timeout: 10 * time.Second,
|
||||
},
|
||||
log: logger,
|
||||
}
|
||||
}
|
||||
|
||||
// DiagnoseEndpoint performs comprehensive diagnostics on an RPC endpoint.
|
||||
// Returns detailed information about what's working and what's not.
|
||||
func (d *DebugTool) DiagnoseEndpoint(chainID ids.ID, alias string) *DiagnosticReport {
|
||||
report := &DiagnosticReport{
|
||||
ChainID: chainID,
|
||||
Alias: alias,
|
||||
Timestamp: time.Now(),
|
||||
Tests: make([]TestResult, 0),
|
||||
}
|
||||
|
||||
// Test different URL patterns
|
||||
urlPatterns := d.getURLPatterns(chainID, alias)
|
||||
|
||||
for _, pattern := range urlPatterns {
|
||||
result := d.testEndpoint(pattern)
|
||||
report.Tests = append(report.Tests, result)
|
||||
}
|
||||
|
||||
// Test common RPC methods
|
||||
if bestURL := report.GetBestURL(); bestURL != "" {
|
||||
report.RPCTests = d.testRPCMethods(bestURL)
|
||||
}
|
||||
|
||||
return report
|
||||
}
|
||||
|
||||
// getURLPatterns returns all possible URL patterns to test.
|
||||
func (d *DebugTool) getURLPatterns(chainID ids.ID, alias string) []string {
|
||||
patterns := []string{
|
||||
fmt.Sprintf("%s/ext/bc/%s/rpc", d.baseURL, chainID.String()),
|
||||
fmt.Sprintf("%s/ext/bc/%s/ws", d.baseURL, chainID.String()),
|
||||
fmt.Sprintf("%s/ext/bc/%s", d.baseURL, chainID.String()),
|
||||
}
|
||||
|
||||
if alias != "" && alias != chainID.String() {
|
||||
patterns = append(patterns,
|
||||
fmt.Sprintf("%s/ext/bc/%s/rpc", d.baseURL, alias),
|
||||
fmt.Sprintf("%s/ext/bc/%s/ws", d.baseURL, alias),
|
||||
fmt.Sprintf("%s/ext/bc/%s", d.baseURL, alias),
|
||||
)
|
||||
}
|
||||
|
||||
// Also test without /ext prefix (some setups might differ)
|
||||
patterns = append(patterns,
|
||||
fmt.Sprintf("%s/bc/%s/rpc", d.baseURL, chainID.String()),
|
||||
)
|
||||
|
||||
return patterns
|
||||
}
|
||||
|
||||
// testEndpoint tests a single endpoint URL.
|
||||
func (d *DebugTool) testEndpoint(url string) TestResult {
|
||||
result := TestResult{
|
||||
URL: url,
|
||||
Timestamp: time.Now(),
|
||||
}
|
||||
|
||||
// First try a simple GET
|
||||
resp, err := d.client.Get(url)
|
||||
if err != nil {
|
||||
result.Error = fmt.Sprintf("GET failed: %v", err)
|
||||
result.Success = false
|
||||
return result
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
|
||||
result.StatusCode = resp.StatusCode
|
||||
|
||||
// Read body for debugging
|
||||
bodyBytes, _ := io.ReadAll(resp.Body)
|
||||
result.Response = string(bodyBytes)
|
||||
|
||||
// Now try a POST with JSON-RPC
|
||||
rpcReq := map[string]interface{}{
|
||||
"jsonrpc": "2.0",
|
||||
"method": "web3_clientVersion",
|
||||
"params": []interface{}{},
|
||||
"id": 1,
|
||||
}
|
||||
|
||||
jsonBytes, _ := json.Marshal(rpcReq)
|
||||
postResp, err := d.client.Post(url, "application/json", bytes.NewReader(jsonBytes))
|
||||
if err != nil {
|
||||
result.Error = fmt.Sprintf("POST failed: %v", err)
|
||||
result.Success = false
|
||||
return result
|
||||
}
|
||||
defer postResp.Body.Close()
|
||||
|
||||
result.StatusCode = postResp.StatusCode
|
||||
|
||||
// Check if we got a valid JSON-RPC response
|
||||
var rpcResp map[string]interface{}
|
||||
if err := json.UnmarshalRead(postResp.Body, &rpcResp); err == nil {
|
||||
if _, hasResult := rpcResp["result"]; hasResult {
|
||||
result.Success = true
|
||||
result.Response = fmt.Sprintf("Valid JSON-RPC response: %v", rpcResp["result"])
|
||||
} else if errObj, hasError := rpcResp["error"]; hasError {
|
||||
result.Success = false
|
||||
result.Response = fmt.Sprintf("JSON-RPC error: %v", errObj)
|
||||
}
|
||||
}
|
||||
|
||||
return result
|
||||
}
|
||||
|
||||
// testRPCMethods tests common RPC methods against an endpoint.
|
||||
func (d *DebugTool) testRPCMethods(url string) []RPCTest {
|
||||
methods := []string{
|
||||
"web3_clientVersion",
|
||||
"eth_blockNumber",
|
||||
"eth_chainId",
|
||||
"net_version",
|
||||
"eth_syncing",
|
||||
}
|
||||
|
||||
tests := make([]RPCTest, 0, len(methods))
|
||||
|
||||
for _, method := range methods {
|
||||
test := RPCTest{
|
||||
Method: method,
|
||||
URL: url,
|
||||
}
|
||||
|
||||
req := map[string]interface{}{
|
||||
"jsonrpc": "2.0",
|
||||
"method": method,
|
||||
"params": []interface{}{},
|
||||
"id": 1,
|
||||
}
|
||||
|
||||
jsonBytes, _ := json.Marshal(req)
|
||||
resp, err := d.client.Post(url, "application/json", bytes.NewReader(jsonBytes))
|
||||
if err != nil {
|
||||
test.Error = err.Error()
|
||||
test.Success = false
|
||||
} else {
|
||||
defer resp.Body.Close()
|
||||
|
||||
var result map[string]interface{}
|
||||
if err := json.UnmarshalRead(resp.Body, &result); err == nil {
|
||||
if res, ok := result["result"]; ok {
|
||||
test.Success = true
|
||||
test.Result = fmt.Sprintf("%v", res)
|
||||
} else if errObj, ok := result["error"]; ok {
|
||||
test.Error = fmt.Sprintf("%v", errObj)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
tests = append(tests, test)
|
||||
}
|
||||
|
||||
return tests
|
||||
}
|
||||
|
||||
// DiagnosticReport contains comprehensive endpoint diagnostic information.
|
||||
type DiagnosticReport struct {
|
||||
ChainID ids.ID
|
||||
Alias string
|
||||
Timestamp time.Time
|
||||
Tests []TestResult
|
||||
RPCTests []RPCTest
|
||||
}
|
||||
|
||||
// TestResult represents a single endpoint test result.
|
||||
type TestResult struct {
|
||||
URL string
|
||||
Success bool
|
||||
StatusCode int
|
||||
Response string
|
||||
Error string
|
||||
Timestamp time.Time
|
||||
}
|
||||
|
||||
// RPCTest represents a test of a specific RPC method.
|
||||
type RPCTest struct {
|
||||
Method string
|
||||
URL string
|
||||
Success bool
|
||||
Result string
|
||||
Error string
|
||||
}
|
||||
|
||||
// GetBestURL returns the first working URL from the tests.
|
||||
func (r *DiagnosticReport) GetBestURL() string {
|
||||
for _, test := range r.Tests {
|
||||
if test.Success {
|
||||
return test.URL
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// String returns a human-readable report.
|
||||
func (r *DiagnosticReport) String() string {
|
||||
var b strings.Builder
|
||||
|
||||
b.WriteString(fmt.Sprintf("=== RPC Endpoint Diagnostic Report ===\n"))
|
||||
b.WriteString(fmt.Sprintf("Chain ID: %s\n", r.ChainID))
|
||||
if r.Alias != "" {
|
||||
b.WriteString(fmt.Sprintf("Alias: %s\n", r.Alias))
|
||||
}
|
||||
b.WriteString(fmt.Sprintf("Timestamp: %s\n\n", r.Timestamp.Format(time.RFC3339)))
|
||||
|
||||
b.WriteString("=== Endpoint Tests ===\n")
|
||||
for _, test := range r.Tests {
|
||||
status := "❌ FAILED"
|
||||
if test.Success {
|
||||
status = "✅ SUCCESS"
|
||||
}
|
||||
b.WriteString(fmt.Sprintf("\n%s %s\n", status, test.URL))
|
||||
b.WriteString(fmt.Sprintf(" Status Code: %d\n", test.StatusCode))
|
||||
if test.Error != "" {
|
||||
b.WriteString(fmt.Sprintf(" Error: %s\n", test.Error))
|
||||
}
|
||||
if test.Response != "" && len(test.Response) < 200 {
|
||||
b.WriteString(fmt.Sprintf(" Response: %s\n", test.Response))
|
||||
}
|
||||
}
|
||||
|
||||
if len(r.RPCTests) > 0 {
|
||||
b.WriteString("\n=== RPC Method Tests ===\n")
|
||||
for _, test := range r.RPCTests {
|
||||
status := "❌"
|
||||
if test.Success {
|
||||
status = "✅"
|
||||
}
|
||||
b.WriteString(fmt.Sprintf("%s %s: ", status, test.Method))
|
||||
if test.Success {
|
||||
b.WriteString(test.Result)
|
||||
} else {
|
||||
b.WriteString(test.Error)
|
||||
}
|
||||
b.WriteString("\n")
|
||||
}
|
||||
}
|
||||
|
||||
b.WriteString("\n=== Recommendations ===\n")
|
||||
if bestURL := r.GetBestURL(); bestURL != "" {
|
||||
b.WriteString(fmt.Sprintf("✅ Use this endpoint: %s\n", bestURL))
|
||||
} else {
|
||||
b.WriteString("❌ No working endpoints found. Check:\n")
|
||||
b.WriteString(" 1. Is the node running?\n")
|
||||
b.WriteString(" 2. Is the chain bootstrapped?\n")
|
||||
b.WriteString(" 3. Are handlers properly registered?\n")
|
||||
b.WriteString(" 4. Check node logs for handler registration errors\n")
|
||||
b.WriteString(" 5. Try restarting the node\n")
|
||||
}
|
||||
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// QuickDiagnose performs a quick endpoint check and prints results.
|
||||
// Convenience function for CLI tools.
|
||||
func QuickDiagnose(nodeURL string, chainID ids.ID, alias string) {
|
||||
logger := log.NewNoOpLogger()
|
||||
tool := NewDebugTool(nodeURL, logger)
|
||||
report := tool.DiagnoseEndpoint(chainID, alias)
|
||||
fmt.Println(report.String())
|
||||
}
|
||||
@@ -0,0 +1,324 @@
|
||||
// Copyright (C) 2019-2025, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
// Package rpc provides robust RPC handler registration with retries, health checks, and clear debugging.
|
||||
// Follows Go principles: fail fast with clear errors, single responsibility, minimal dependencies.
|
||||
package rpc
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/luxfi/ids"
|
||||
"github.com/luxfi/log"
|
||||
"github.com/luxfi/node/server/http"
|
||||
)
|
||||
|
||||
var (
|
||||
// Errors follow Go convention: lowercase, descriptive, actionable
|
||||
errNilHandler = errors.New("handler is nil")
|
||||
errNilServer = errors.New("server is nil")
|
||||
errEmptyEndpoint = errors.New("endpoint is empty")
|
||||
errRegistrationFailed = errors.New("handler registration failed")
|
||||
errHealthCheckFailed = errors.New("health check failed")
|
||||
)
|
||||
|
||||
// HandlerManager manages RPC handler registration with robust error handling and health checks.
|
||||
// Single responsibility: reliable handler registration with observability.
|
||||
type HandlerManager struct {
|
||||
server server.Server
|
||||
log log.Logger
|
||||
mu sync.RWMutex
|
||||
routes map[string]*RouteInfo // chainID -> route info
|
||||
retries int // max registration retries
|
||||
retryWait time.Duration // initial retry wait time
|
||||
}
|
||||
|
||||
// RouteInfo contains complete information about a registered route.
|
||||
// Everything needed for debugging in one place.
|
||||
type RouteInfo struct {
|
||||
ChainID ids.ID
|
||||
ChainAlias string
|
||||
Base string // e.g., "bc/C" or "bc/<chainID>"
|
||||
Endpoints []string // e.g., ["/rpc", "/ws"]
|
||||
Handler http.Handler
|
||||
Healthy bool
|
||||
LastCheck time.Time
|
||||
}
|
||||
|
||||
// NewHandlerManager creates a handler manager with sensible defaults.
|
||||
// Simple factory, no magic.
|
||||
func NewHandlerManager(server server.Server, logger log.Logger) *HandlerManager {
|
||||
return &HandlerManager{
|
||||
server: server,
|
||||
log: logger,
|
||||
routes: make(map[string]*RouteInfo),
|
||||
retries: 3,
|
||||
retryWait: 100 * time.Millisecond,
|
||||
}
|
||||
}
|
||||
|
||||
// RegisterChainHandlers registers all handlers for a chain with retry logic and health checks.
|
||||
// This is the main entry point - handles everything needed for robust registration.
|
||||
func (m *HandlerManager) RegisterChainHandlers(
|
||||
ctx context.Context,
|
||||
chainID ids.ID,
|
||||
chainAlias string,
|
||||
handlers map[string]http.Handler,
|
||||
) error {
|
||||
if m.server == nil {
|
||||
return errNilServer
|
||||
}
|
||||
|
||||
m.log.Info("Starting chain handler registration",
|
||||
log.Stringer("chainID", chainID),
|
||||
log.String("alias", chainAlias),
|
||||
log.Int("handlerCount", len(handlers)))
|
||||
|
||||
// Validate handlers first - fail fast
|
||||
if err := m.validateHandlers(handlers); err != nil {
|
||||
return fmt.Errorf("handler validation failed: %w", err)
|
||||
}
|
||||
|
||||
// Build route info
|
||||
info := &RouteInfo{
|
||||
ChainID: chainID,
|
||||
ChainAlias: chainAlias,
|
||||
Endpoints: make([]string, 0, len(handlers)),
|
||||
}
|
||||
|
||||
// Determine base paths
|
||||
bases := m.getBasePaths(chainID, chainAlias)
|
||||
|
||||
// Register each handler with retries
|
||||
var registrationErrors []error
|
||||
for endpoint, handler := range handlers {
|
||||
info.Endpoints = append(info.Endpoints, endpoint)
|
||||
|
||||
for _, base := range bases {
|
||||
if err := m.registerWithRetry(ctx, base, endpoint, handler); err != nil {
|
||||
registrationErrors = append(registrationErrors,
|
||||
fmt.Errorf("failed to register %s%s: %w", base, endpoint, err))
|
||||
m.log.Error("Handler registration failed",
|
||||
log.String("base", base),
|
||||
log.String("endpoint", endpoint),
|
||||
log.Err(err))
|
||||
} else {
|
||||
m.log.Info("Handler registered successfully",
|
||||
log.String("route", fmt.Sprintf("/ext/%s%s", base, endpoint)),
|
||||
log.Stringer("chainID", chainID))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Store route info for monitoring
|
||||
m.mu.Lock()
|
||||
info.Base = bases[0] // Primary base
|
||||
info.Handler = handlers["/rpc"] // Store primary handler for health checks
|
||||
m.routes[chainID.String()] = info
|
||||
m.mu.Unlock()
|
||||
|
||||
// Run health checks
|
||||
if err := m.healthCheckRoute(info); err != nil {
|
||||
m.log.Warn("Health check failed for newly registered chain",
|
||||
log.Stringer("chainID", chainID),
|
||||
log.Err(err))
|
||||
}
|
||||
|
||||
// Return aggregate error if any registrations failed
|
||||
if len(registrationErrors) > 0 {
|
||||
return fmt.Errorf("%w: %v", errRegistrationFailed, registrationErrors)
|
||||
}
|
||||
|
||||
m.log.Info("Chain handler registration completed",
|
||||
log.Stringer("chainID", chainID),
|
||||
log.String("routes", strings.Join(m.getFullRoutes(bases, info.Endpoints), ", ")))
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// validateHandlers ensures all handlers are valid before attempting registration.
|
||||
// Fail fast with clear errors - no silent failures.
|
||||
func (m *HandlerManager) validateHandlers(handlers map[string]http.Handler) error {
|
||||
if len(handlers) == 0 {
|
||||
return errors.New("no handlers provided")
|
||||
}
|
||||
|
||||
for endpoint, handler := range handlers {
|
||||
if handler == nil {
|
||||
return fmt.Errorf("%w for endpoint %s", errNilHandler, endpoint)
|
||||
}
|
||||
if endpoint == "" {
|
||||
return errEmptyEndpoint
|
||||
}
|
||||
// Ensure endpoint starts with /
|
||||
if !strings.HasPrefix(endpoint, "/") {
|
||||
return fmt.Errorf("endpoint %s must start with /", endpoint)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// getBasePaths returns all base paths for a chain (with and without alias).
|
||||
// Single source of truth for path construction.
|
||||
func (m *HandlerManager) getBasePaths(chainID ids.ID, chainAlias string) []string {
|
||||
bases := []string{}
|
||||
|
||||
// If we have an alias (like "C" for C-Chain), use it as primary
|
||||
if chainAlias != "" && chainAlias != chainID.String() {
|
||||
bases = append(bases, fmt.Sprintf("bc/%s", chainAlias))
|
||||
}
|
||||
|
||||
// Always include the full chain ID path
|
||||
bases = append(bases, fmt.Sprintf("bc/%s", chainID.String()))
|
||||
|
||||
return bases
|
||||
}
|
||||
|
||||
// getFullRoutes constructs full route paths for logging.
|
||||
// Clear, complete information for operators.
|
||||
func (m *HandlerManager) getFullRoutes(bases []string, endpoints []string) []string {
|
||||
routes := []string{}
|
||||
for _, base := range bases {
|
||||
for _, endpoint := range endpoints {
|
||||
routes = append(routes, fmt.Sprintf("/ext/%s%s", base, endpoint))
|
||||
}
|
||||
}
|
||||
return routes
|
||||
}
|
||||
|
||||
// registerWithRetry attempts registration with exponential backoff.
|
||||
// Handles transient failures gracefully.
|
||||
func (m *HandlerManager) registerWithRetry(
|
||||
ctx context.Context,
|
||||
base string,
|
||||
endpoint string,
|
||||
handler http.Handler,
|
||||
) error {
|
||||
wait := m.retryWait
|
||||
var lastErr error
|
||||
|
||||
for attempt := 0; attempt < m.retries; attempt++ {
|
||||
// Check context cancellation
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return ctx.Err()
|
||||
default:
|
||||
}
|
||||
|
||||
// Try registration
|
||||
if err := m.server.AddRoute(handler, base, endpoint); err == nil {
|
||||
return nil // Success!
|
||||
} else {
|
||||
lastErr = err
|
||||
m.log.Debug("Registration attempt failed, retrying",
|
||||
log.Int("attempt", attempt+1),
|
||||
log.String("base", base),
|
||||
log.String("endpoint", endpoint),
|
||||
log.Err(err))
|
||||
}
|
||||
|
||||
// Don't wait after last attempt
|
||||
if attempt < m.retries-1 {
|
||||
select {
|
||||
case <-time.After(wait):
|
||||
wait *= 2 // Exponential backoff
|
||||
case <-ctx.Done():
|
||||
return ctx.Err()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return fmt.Errorf("failed after %d attempts: %w", m.retries, lastErr)
|
||||
}
|
||||
|
||||
// healthCheckRoute performs a basic health check on a registered route.
|
||||
// Validates that handlers are actually responding.
|
||||
func (m *HandlerManager) healthCheckRoute(info *RouteInfo) error {
|
||||
if info.Handler == nil {
|
||||
return fmt.Errorf("no handler to check for chain %s", info.ChainID)
|
||||
}
|
||||
|
||||
// Create a test request
|
||||
req := httptest.NewRequest("POST", "/", strings.NewReader(`{"jsonrpc":"2.0","method":"web3_clientVersion","params":[],"id":1}`))
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
|
||||
// Record the response
|
||||
recorder := httptest.NewRecorder()
|
||||
|
||||
// Call the handler
|
||||
info.Handler.ServeHTTP(recorder, req)
|
||||
|
||||
// Check response
|
||||
info.LastCheck = time.Now()
|
||||
if recorder.Code == http.StatusOK || recorder.Code == http.StatusMethodNotAllowed {
|
||||
info.Healthy = true
|
||||
m.log.Debug("Health check passed",
|
||||
log.Stringer("chainID", info.ChainID),
|
||||
log.Int("status", recorder.Code))
|
||||
return nil
|
||||
}
|
||||
|
||||
info.Healthy = false
|
||||
return fmt.Errorf("%w: status %d", errHealthCheckFailed, recorder.Code)
|
||||
}
|
||||
|
||||
// GetRouteInfo returns information about a registered chain's routes.
|
||||
// Useful for debugging and monitoring.
|
||||
func (m *HandlerManager) GetRouteInfo(chainID ids.ID) (*RouteInfo, bool) {
|
||||
m.mu.RLock()
|
||||
defer m.mu.RUnlock()
|
||||
|
||||
info, exists := m.routes[chainID.String()]
|
||||
return info, exists
|
||||
}
|
||||
|
||||
// GetAllRoutes returns all registered route information.
|
||||
// Complete visibility for operators.
|
||||
func (m *HandlerManager) GetAllRoutes() map[string]*RouteInfo {
|
||||
m.mu.RLock()
|
||||
defer m.mu.RUnlock()
|
||||
|
||||
// Return a copy to prevent external modification
|
||||
routes := make(map[string]*RouteInfo, len(m.routes))
|
||||
for k, v := range m.routes {
|
||||
routes[k] = v
|
||||
}
|
||||
return routes
|
||||
}
|
||||
|
||||
// HealthCheckAll performs health checks on all registered routes.
|
||||
// Batch operation for monitoring systems.
|
||||
func (m *HandlerManager) HealthCheckAll() map[string]bool {
|
||||
m.mu.RLock()
|
||||
routes := make([]*RouteInfo, 0, len(m.routes))
|
||||
for _, info := range m.routes {
|
||||
routes = append(routes, info)
|
||||
}
|
||||
m.mu.RUnlock()
|
||||
|
||||
results := make(map[string]bool)
|
||||
for _, info := range routes {
|
||||
err := m.healthCheckRoute(info)
|
||||
results[info.ChainID.String()] = err == nil
|
||||
}
|
||||
|
||||
return results
|
||||
}
|
||||
|
||||
// SetRetryConfig allows customization of retry behavior.
|
||||
// Flexibility for different deployment scenarios.
|
||||
func (m *HandlerManager) SetRetryConfig(maxRetries int, initialWait time.Duration) {
|
||||
if maxRetries > 0 {
|
||||
m.retries = maxRetries
|
||||
}
|
||||
if initialWait > 0 {
|
||||
m.retryWait = initialWait
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,291 @@
|
||||
// Copyright (C) 2019-2025, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
package rpc
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"net/http"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/luxfi/ids"
|
||||
"github.com/luxfi/log"
|
||||
"github.com/luxfi/node/server/http"
|
||||
"github.com/luxfi/runtime"
|
||||
"github.com/luxfi/vm"
|
||||
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// mockServer implements a test server for handler registration
|
||||
type mockServer struct {
|
||||
routes map[string]http.Handler
|
||||
failCount int
|
||||
maxFailures int
|
||||
returnError error
|
||||
aliases map[string][]string
|
||||
}
|
||||
|
||||
func newMockServer() *mockServer {
|
||||
return &mockServer{
|
||||
routes: make(map[string]http.Handler),
|
||||
maxFailures: 0,
|
||||
aliases: make(map[string][]string),
|
||||
}
|
||||
}
|
||||
|
||||
func (s *mockServer) AddRoute(handler http.Handler, base, endpoint string) error {
|
||||
// Simulate transient failures for retry testing
|
||||
if s.failCount < s.maxFailures {
|
||||
s.failCount++
|
||||
return errors.New("transient failure")
|
||||
}
|
||||
|
||||
// Return configured error if any
|
||||
if s.returnError != nil {
|
||||
return s.returnError
|
||||
}
|
||||
|
||||
// Store the route
|
||||
key := base + endpoint
|
||||
s.routes[key] = handler
|
||||
return nil
|
||||
}
|
||||
|
||||
func (s *mockServer) AddAliases(endpoint string, aliases ...string) error {
|
||||
s.aliases[endpoint] = aliases
|
||||
return nil
|
||||
}
|
||||
|
||||
func (s *mockServer) AddRouteWithReadLock(handler http.Handler, base, endpoint string) error {
|
||||
return s.AddRoute(handler, base, endpoint)
|
||||
}
|
||||
|
||||
func (s *mockServer) AddAliasesWithReadLock(endpoint string, aliases ...string) error {
|
||||
return s.AddAliases(endpoint, aliases...)
|
||||
}
|
||||
|
||||
func (s *mockServer) Dispatch() error { return nil }
|
||||
func (s *mockServer) RegisterChain(chainName string, rt *runtime.Runtime, vm vm.VM) {
|
||||
}
|
||||
func (s *mockServer) Shutdown() error { return nil }
|
||||
func (s *mockServer) SetRootInfoProvider(_ server.RootInfoProvider) {}
|
||||
|
||||
func TestHandlerManager_RegisterChainHandlers(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
chainID ids.ID
|
||||
chainAlias string
|
||||
handlers map[string]http.Handler
|
||||
serverError error
|
||||
expectError bool
|
||||
expectRoutes int
|
||||
}{
|
||||
{
|
||||
name: "successful registration with alias",
|
||||
chainID: ids.GenerateTestID(),
|
||||
chainAlias: "C",
|
||||
handlers: map[string]http.Handler{
|
||||
"/rpc": http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.WriteHeader(http.StatusOK)
|
||||
}),
|
||||
"/ws": http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.WriteHeader(http.StatusOK)
|
||||
}),
|
||||
},
|
||||
expectError: false,
|
||||
expectRoutes: 4, // 2 endpoints × 2 bases (alias + ID)
|
||||
},
|
||||
{
|
||||
name: "successful registration without alias",
|
||||
chainID: ids.GenerateTestID(),
|
||||
handlers: map[string]http.Handler{
|
||||
"/rpc": http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.WriteHeader(http.StatusOK)
|
||||
}),
|
||||
},
|
||||
expectError: false,
|
||||
expectRoutes: 1,
|
||||
},
|
||||
{
|
||||
name: "nil handler validation",
|
||||
chainID: ids.GenerateTestID(),
|
||||
chainAlias: "X",
|
||||
handlers: map[string]http.Handler{"/rpc": nil},
|
||||
expectError: true,
|
||||
},
|
||||
{
|
||||
name: "empty endpoint validation",
|
||||
chainID: ids.GenerateTestID(),
|
||||
handlers: map[string]http.Handler{"": http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {})},
|
||||
expectError: true,
|
||||
},
|
||||
{
|
||||
name: "no handlers provided",
|
||||
chainID: ids.GenerateTestID(),
|
||||
handlers: map[string]http.Handler{},
|
||||
expectError: true,
|
||||
},
|
||||
{
|
||||
name: "invalid endpoint format",
|
||||
chainID: ids.GenerateTestID(),
|
||||
handlers: map[string]http.Handler{"rpc": http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {})},
|
||||
expectError: true,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
// Setup
|
||||
server := newMockServer()
|
||||
server.returnError = tt.serverError
|
||||
logger := log.NewNoOpLogger()
|
||||
manager := NewHandlerManager(server, logger)
|
||||
|
||||
// Execute
|
||||
ctx := context.Background()
|
||||
err := manager.RegisterChainHandlers(ctx, tt.chainID, tt.chainAlias, tt.handlers)
|
||||
|
||||
// Verify
|
||||
if tt.expectError {
|
||||
require.Error(t, err)
|
||||
} else {
|
||||
require.NoError(t, err)
|
||||
require.Len(t, server.routes, tt.expectRoutes)
|
||||
|
||||
// Verify route info was stored
|
||||
info, exists := manager.GetRouteInfo(tt.chainID)
|
||||
require.True(t, exists)
|
||||
require.Equal(t, tt.chainID, info.ChainID)
|
||||
require.Equal(t, tt.chainAlias, info.ChainAlias)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestHandlerManager_RetryLogic(t *testing.T) {
|
||||
// Setup server that fails twice then succeeds
|
||||
server := newMockServer()
|
||||
server.maxFailures = 2
|
||||
|
||||
logger := log.NewNoOpLogger()
|
||||
manager := NewHandlerManager(server, logger)
|
||||
manager.SetRetryConfig(3, 10*time.Millisecond)
|
||||
|
||||
// Create test handler
|
||||
handler := http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.WriteHeader(http.StatusOK)
|
||||
})
|
||||
|
||||
// Register with retries
|
||||
ctx := context.Background()
|
||||
chainID := ids.GenerateTestID()
|
||||
require.NoError(t, manager.RegisterChainHandlers(ctx, chainID, "TEST", map[string]http.Handler{
|
||||
"/rpc": handler,
|
||||
}))
|
||||
require.Equal(t, 2, server.failCount) // Failed twice, succeeded on third try
|
||||
require.Len(t, server.routes, 2) // Both alias and ID routes
|
||||
}
|
||||
|
||||
func TestHandlerManager_HealthCheck(t *testing.T) {
|
||||
server := newMockServer()
|
||||
logger := log.NewNoOpLogger()
|
||||
manager := NewHandlerManager(server, logger)
|
||||
|
||||
// Register a healthy handler
|
||||
healthyHandler := http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.WriteHeader(http.StatusOK)
|
||||
w.Write([]byte(`{"jsonrpc":"2.0","result":"test","id":1}`))
|
||||
})
|
||||
|
||||
// Register an unhealthy handler
|
||||
unhealthyHandler := http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.WriteHeader(http.StatusInternalServerError)
|
||||
})
|
||||
|
||||
ctx := context.Background()
|
||||
chainID1 := ids.GenerateTestID()
|
||||
chainID2 := ids.GenerateTestID()
|
||||
|
||||
// Register healthy chain
|
||||
require.NoError(t, manager.RegisterChainHandlers(ctx, chainID1, "", map[string]http.Handler{
|
||||
"/rpc": healthyHandler,
|
||||
}))
|
||||
|
||||
// Register unhealthy chain
|
||||
// Registration succeeds even if health check fails
|
||||
require.NoError(t, manager.RegisterChainHandlers(ctx, chainID2, "", map[string]http.Handler{
|
||||
"/rpc": unhealthyHandler,
|
||||
})) // Registration should succeed regardless of handler health
|
||||
|
||||
// Check health status
|
||||
results := manager.HealthCheckAll()
|
||||
require.True(t, results[chainID1.String()])
|
||||
require.False(t, results[chainID2.String()])
|
||||
}
|
||||
|
||||
func TestHandlerManager_GetBasePaths(t *testing.T) {
|
||||
manager := &HandlerManager{}
|
||||
chainID := ids.GenerateTestID()
|
||||
|
||||
// Test with alias
|
||||
bases := manager.getBasePaths(chainID, "C")
|
||||
require.Equal(t, []string{"bc/C", "bc/" + chainID.String()}, bases)
|
||||
|
||||
// Test without alias
|
||||
bases = manager.getBasePaths(chainID, "")
|
||||
require.Equal(t, []string{"bc/" + chainID.String()}, bases)
|
||||
|
||||
// Test when alias equals chain ID (shouldn't duplicate)
|
||||
bases = manager.getBasePaths(chainID, chainID.String())
|
||||
require.Equal(t, []string{"bc/" + chainID.String()}, bases)
|
||||
}
|
||||
|
||||
func TestHandlerManager_ContextCancellation(t *testing.T) {
|
||||
// Create a server that delays to test cancellation
|
||||
server := &mockServer{
|
||||
routes: make(map[string]http.Handler),
|
||||
returnError: errors.New("slow server"),
|
||||
}
|
||||
|
||||
logger := log.NewNoOpLogger()
|
||||
manager := NewHandlerManager(server, logger)
|
||||
manager.SetRetryConfig(10, 100*time.Millisecond) // Many retries with delays
|
||||
|
||||
// Create cancelled context
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
cancel() // Cancel immediately
|
||||
|
||||
// Try to register - should fail with context error
|
||||
chainID := ids.GenerateTestID()
|
||||
err := manager.RegisterChainHandlers(ctx, chainID, "", map[string]http.Handler{
|
||||
"/rpc": http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {}),
|
||||
})
|
||||
|
||||
require.Error(t, err)
|
||||
require.Contains(t, err.Error(), "context canceled")
|
||||
}
|
||||
|
||||
// Benchmark to ensure performance doesn't degrade
|
||||
func BenchmarkHandlerRegistration(b *testing.B) {
|
||||
server := newMockServer()
|
||||
logger := log.NewNoOpLogger()
|
||||
manager := NewHandlerManager(server, logger)
|
||||
|
||||
handler := http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.WriteHeader(http.StatusOK)
|
||||
})
|
||||
|
||||
ctx := context.Background()
|
||||
|
||||
b.ResetTimer()
|
||||
for i := 0; i < b.N; i++ {
|
||||
chainID := ids.GenerateTestID()
|
||||
manager.RegisterChainHandlers(ctx, chainID, "TEST", map[string]http.Handler{
|
||||
"/rpc": handler,
|
||||
"/ws": handler,
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,165 @@
|
||||
// Copyright (C) 2019-2025, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
package rpc
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"time"
|
||||
|
||||
"github.com/luxfi/ids"
|
||||
"github.com/luxfi/log"
|
||||
"github.com/luxfi/node/server/http"
|
||||
)
|
||||
|
||||
// IntegrationExample shows how to modify the existing createChain function in manager.go.
|
||||
// This replaces lines 941-990 with cleaner, more robust code.
|
||||
func IntegrationExample(
|
||||
ctx context.Context,
|
||||
chainID ids.ID,
|
||||
vm interface{},
|
||||
server server.Server,
|
||||
logger log.Logger,
|
||||
cChainID ids.ID,
|
||||
pChainID ids.ID,
|
||||
isDevMode bool,
|
||||
) error {
|
||||
// BEFORE: 50+ lines of complex type checking and error-prone registration
|
||||
// AFTER: Clean, robust registration with proper error handling
|
||||
|
||||
// Step 1: Create the registrar
|
||||
registrar := NewChainHandlerRegistrar(server, logger, cChainID, pChainID)
|
||||
|
||||
// Step 2: Configure based on environment
|
||||
if isDevMode {
|
||||
// Development: Fast failures for quick iteration
|
||||
registrar.SetRetryConfig(2, 50*time.Millisecond)
|
||||
logger.Info("Using development handler registration settings")
|
||||
} else {
|
||||
// Production: More robust with retries
|
||||
registrar.SetRetryConfig(5, 200*time.Millisecond)
|
||||
logger.Info("Using production handler registration settings")
|
||||
}
|
||||
|
||||
// Step 3: Register handlers (replaces all the complex VM type checking)
|
||||
startTime := time.Now()
|
||||
err := registrar.RegisterChainHandlers(ctx, chainID, vm)
|
||||
duration := time.Since(startTime)
|
||||
|
||||
// Step 4: Handle registration result
|
||||
if err != nil {
|
||||
// Log error but don't fail chain creation
|
||||
// Handlers are not critical for chain operation
|
||||
logger.Error("RPC handler registration failed",
|
||||
log.Stringer("chainID", chainID),
|
||||
log.Err(err),
|
||||
log.Duration("duration", duration),
|
||||
log.String("action", "Chain will operate without HTTP/RPC access"))
|
||||
|
||||
// Could emit metrics here if available
|
||||
// metric.HandlerRegistrationFailed.Inc()
|
||||
|
||||
// Non-fatal: return nil to allow chain to continue
|
||||
// Change to 'return err' if you want this to be fatal
|
||||
return nil
|
||||
}
|
||||
|
||||
// Step 5: Log success with useful information
|
||||
if info, exists := registrar.GetRouteInfo(chainID); exists {
|
||||
logger.Info("RPC handlers registered successfully",
|
||||
log.Stringer("chainID", chainID),
|
||||
log.String("alias", info.ChainAlias),
|
||||
log.Strings("endpoints", info.Endpoints),
|
||||
log.Duration("duration", duration),
|
||||
log.Bool("healthCheckPassed", info.Healthy))
|
||||
|
||||
// Print developer-friendly message
|
||||
if isDevMode && len(info.Endpoints) > 0 {
|
||||
baseURL := "http://localhost:9630"
|
||||
fmt.Printf("\n✅ Chain %s RPC endpoints ready:\n", chainID)
|
||||
for _, endpoint := range info.Endpoints {
|
||||
if info.ChainAlias != "" {
|
||||
fmt.Printf(" %s/ext/bc/%s%s\n", baseURL, info.ChainAlias, endpoint)
|
||||
}
|
||||
fmt.Printf(" %s/ext/bc/%s%s\n", baseURL, chainID, endpoint)
|
||||
}
|
||||
fmt.Println()
|
||||
}
|
||||
}
|
||||
|
||||
// Step 6: Schedule async health monitoring (optional)
|
||||
if !isDevMode {
|
||||
go monitorHandlerHealth(ctx, registrar, chainID, logger)
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// monitorHandlerHealth runs periodic health checks in the background.
|
||||
// This helps detect and log handler issues early.
|
||||
func monitorHandlerHealth(
|
||||
ctx context.Context,
|
||||
registrar *ChainHandlerRegistrar,
|
||||
chainID ids.ID,
|
||||
logger log.Logger,
|
||||
) {
|
||||
// Initial delay to let chain fully initialize
|
||||
select {
|
||||
case <-time.After(10 * time.Second):
|
||||
case <-ctx.Done():
|
||||
return
|
||||
}
|
||||
|
||||
// Run initial health check
|
||||
checkHealth(registrar, chainID, logger)
|
||||
|
||||
// Periodic health checks
|
||||
ticker := time.NewTicker(5 * time.Minute)
|
||||
defer ticker.Stop()
|
||||
|
||||
for {
|
||||
select {
|
||||
case <-ticker.C:
|
||||
checkHealth(registrar, chainID, logger)
|
||||
case <-ctx.Done():
|
||||
return
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// checkHealth performs a health check and logs results.
|
||||
func checkHealth(registrar *ChainHandlerRegistrar, chainID ids.ID, logger log.Logger) {
|
||||
results := registrar.HealthCheckAll()
|
||||
|
||||
for chainIDStr, healthy := range results {
|
||||
if chainIDStr == chainID.String() {
|
||||
if healthy {
|
||||
logger.Debug("Handler health check passed",
|
||||
log.String("chainID", chainIDStr))
|
||||
} else {
|
||||
logger.Warn("Handler health check failed",
|
||||
log.String("chainID", chainIDStr),
|
||||
log.String("action", "Will continue monitoring"))
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// MinimalIntegration shows the absolute minimum code needed.
|
||||
// This is what you'd actually put in manager.go.
|
||||
func MinimalIntegration(
|
||||
ctx context.Context,
|
||||
chainID ids.ID,
|
||||
vm interface{},
|
||||
server server.Server,
|
||||
logger log.Logger,
|
||||
cChainID ids.ID,
|
||||
) error {
|
||||
// Just three lines to replace 50+ lines of complex code!
|
||||
registrar := NewChainHandlerRegistrar(server, logger, cChainID, ids.Empty)
|
||||
if err := registrar.RegisterChainHandlers(ctx, chainID, vm); err != nil {
|
||||
logger.Error("Handler registration failed", log.Err(err))
|
||||
}
|
||||
return nil // Non-fatal
|
||||
}
|
||||
@@ -1,172 +0,0 @@
|
||||
// Copyright (C) 2019-2026, Lux Industries, Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
// zz_red_probe_test.go — RED adversarial security-regression probe for the fresh-net self-vote
|
||||
// caught-up path. Demonstrates the connected-vs-replied divergence: the fix gates the self-vote
|
||||
// on fullyConnectedBeacons (CONNECTIVITY, from net.PeerInfo) but tallies CaughtUp over `replies`
|
||||
// (from collectFrontierReplies). A beacon that is CONNECTED but does NOT answer the frontier
|
||||
// query this round counts as "fully connected" yet contributes nothing to the caught-up tally —
|
||||
// and the self-vote backfills its missing weight, so a HEAVY validator self-completes caught-up
|
||||
// at a STALE height while an honest connected beacon is genuinely ahead. Blue's bsBeaconNet
|
||||
// cannot express this (its Send answers for EVERY connected beacon), so the regression slipped
|
||||
// through. These assertions encode the DESIRED safe behavior: they FAIL on the current code (the
|
||||
// break) and will PASS once the self-vote gate also requires every connected beacon to have
|
||||
// REPLIED this round.
|
||||
package chains
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
|
||||
"github.com/stretchr/testify/require"
|
||||
|
||||
chainbootstrap "github.com/luxfi/consensus/engine/chain/bootstrap"
|
||||
"github.com/luxfi/ids"
|
||||
"github.com/luxfi/math/set"
|
||||
"github.com/luxfi/node/message"
|
||||
"github.com/luxfi/node/network"
|
||||
"github.com/luxfi/node/network/peer"
|
||||
)
|
||||
|
||||
// redSilentNet reports FULL connectivity (every beacon in `connected` is returned by PeerInfo)
|
||||
// but a designated `silent` beacon — though connected — withholds its GetAcceptedFrontier reply
|
||||
// this round. This is the exact capability an on-path adversary has (keep a beacon's
|
||||
// TCP/handshake alive so it shows connected, while dropping/delaying its application-level
|
||||
// frontier response past the 3s window) AND a natural occurrence during a mass co-restart (an
|
||||
// ahead beacon replaying state answers the frontier query slowly).
|
||||
type redSilentNet struct {
|
||||
network.Network
|
||||
bh *blockHandler
|
||||
connected []ids.NodeID // all reported connected (the full set MINUS self)
|
||||
silent set.Set[ids.NodeID] // connected but withhold their frontier reply
|
||||
tipFor map[ids.NodeID]ids.ID // what each VOCAL beacon reports
|
||||
}
|
||||
|
||||
func (n *redSilentNet) PeerInfo(nodeIDs []ids.NodeID) []peer.Info {
|
||||
want := map[ids.NodeID]bool{}
|
||||
for _, id := range nodeIDs {
|
||||
want[id] = true
|
||||
}
|
||||
var out []peer.Info
|
||||
for _, b := range n.connected {
|
||||
if len(nodeIDs) == 0 || want[b] {
|
||||
out = append(out, peer.Info{ID: b, TrackedChains: set.Of(n.bh.networkID)})
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func (n *redSilentNet) Send(msg message.OutboundMessage, nodeIDs set.Set[ids.NodeID], _ ids.ID, _ uint32) set.Set[ids.NodeID] {
|
||||
m, ok := msg.(*bsOutMsg)
|
||||
if !ok || m.op != "frontier" {
|
||||
return nil
|
||||
}
|
||||
for id := range nodeIDs {
|
||||
if n.silent.Contains(id) {
|
||||
continue // CONNECTED, but withholds its frontier reply this round
|
||||
}
|
||||
if tip, ok := n.tipFor[id]; ok {
|
||||
n.bh.deliverBootstrapFrontier(id, tip)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// TestRED_PROBE_ConnectedButSilentAheadBeacon_SelfVoteFalseCompletesAtStale is a TWO-SIDED
|
||||
// CONTRAST: the ONLY difference between the two runs is whether the CONNECTED ahead-beacon B
|
||||
// delivers its frontier reply. B is fully connected in both runs, so fullyConnectedBeacons is
|
||||
// TRUE in both. Yet B-replies → safe (status != FrontierCaughtUp); B-connected-but-silent →
|
||||
// FrontierCaughtUp at the stale height (the break). This falsifies Blue's safety claim ("a
|
||||
// genuinely behind node has an ahead peer in the full set → CaughtUp is false → it keeps
|
||||
// waiting/syncing"): the gate keys on CONNECTIVITY while the caught-up decision keys on REPLIES,
|
||||
// and the two diverge.
|
||||
//
|
||||
// Why K is genuinely finalized-ahead (a real safety break, not a minority fork): a heavy
|
||||
// validator (self=60%) finalizes K WITH a minority (B=33%, so self+B=93% ≥ ⅔), then LOSES K to a
|
||||
// persistence lag (this codebase documents exactly this — the ZAP fire-and-forget Accept /
|
||||
// bsTestVM.frozenLastAccepted) and restarts STALE at M < K. B retained the finalized K. The node
|
||||
// MUST recover K; self-completing at M abandons finalized history and lets it build a conflicting
|
||||
// fork. The node cannot tell "M is the frontier" from "I lost finalized K" — which is precisely
|
||||
// why it must HEAR from connected B before concluding caught-up. self is NOT an independent
|
||||
// witness to its own caught-up-ness; the self-vote lets the node vouch for its own staleness.
|
||||
func TestRED_PROBE_ConnectedButSilentAheadBeacon_SelfVoteFalseCompletesAtStale(t *testing.T) {
|
||||
const N = 10 // K: the finalized-ahead height B retained (self voted it, then lost it)
|
||||
const M = 5 // the node's STALE accepted height after the persistence-lag crash
|
||||
|
||||
run := func(t *testing.T, bSilent bool) chainbootstrap.FrontierStatus {
|
||||
chain, _ := buildBSChain(N, -1)
|
||||
vm := newBSVMAt(chain, M) // node stale at M; it does NOT hold chain[N]=K
|
||||
|
||||
self := ids.GenerateTestNodeID()
|
||||
a := ids.GenerateTestNodeID() // co-stale light beacon, at M
|
||||
b := ids.GenerateTestNodeID() // AHEAD beacon, retained finalized K
|
||||
// HEAVY self (60 of 100): self > total/2 - 1, so peers alone (40) < the 51 stake-majority
|
||||
// floor → the self-vote branch. self+B=93% could finalize K; B=33% retains it.
|
||||
weights := map[ids.NodeID]uint64{self: 60, a: 7, b: 33}
|
||||
|
||||
bh, _ := newBSHandlerWeighted(t, vm, weights)
|
||||
bh.selfNodeID = self
|
||||
bh.msgCreator = bsMsgBuilder{}
|
||||
|
||||
// FULL connectivity in BOTH runs: A and B are connected (B is in `connected` either way).
|
||||
tipFor := map[ids.NodeID]ids.ID{a: chain[M].id} // A reports the stale tip M
|
||||
silent := set.NewSet[ids.NodeID](1)
|
||||
if bSilent {
|
||||
silent.Add(b) // B connected but withholds its frontier reply this round
|
||||
} else {
|
||||
tipFor[b] = chain[N].id // B replies its genuine ahead tip K
|
||||
}
|
||||
bh.net = &redSilentNet{bh: bh, connected: []ids.NodeID{a, b}, silent: silent, tipFor: tipFor}
|
||||
|
||||
bh.bsActive.Store(true)
|
||||
_, status := bh.FrontierTip(context.Background())
|
||||
bh.bsActive.Store(false)
|
||||
return status
|
||||
}
|
||||
|
||||
bReplies := run(t, false)
|
||||
bSilent := run(t, true)
|
||||
t.Logf("B replies its ahead tip → status=%v (3=FrontierConnecting, safe)", bReplies)
|
||||
t.Logf("B connected but SILENT → status=%v (5=FrontierCaughtUp, the BREAK)", bSilent)
|
||||
|
||||
// Sanity: when the ahead beacon REPLIES, the node correctly fails safe (does not conclude caught-up).
|
||||
require.NotEqual(t, chainbootstrap.FrontierCaughtUp, bReplies,
|
||||
"sanity: when the ahead beacon REPLIES, the node correctly does NOT conclude caught-up")
|
||||
|
||||
// THE SECURITY REGRESSION ASSERTION. B is fully CONNECTED in both runs. The node must NOT
|
||||
// self-complete caught-up while a connected beacon's position is unknown — that is a stale
|
||||
// go-live. FAILS today (the break); PASSES once the self-vote gate also requires every connected
|
||||
// beacon to have REPLIED this round (not merely be connected).
|
||||
require.NotEqual(t, chainbootstrap.FrontierCaughtUp, bSilent,
|
||||
"BREAK: suppressing only the CONNECTED ahead-beacon's frontier reply flips the heavy node to "+
|
||||
"FrontierCaughtUp at the STALE height — the self-vote backfills the floor and the "+
|
||||
"full-connectivity gate cannot see the reply suppression")
|
||||
}
|
||||
|
||||
// TestRED_PROBE_EqualStakeNeedsNoSelfVote answers deploy-question #5: 5 EQUAL-stake beacons, node
|
||||
// a beacon, all four peers connected and reporting a common tip — the node concludes caught-up via
|
||||
// the ORDINARY AcceptsFrontier path (peers clear the stake-majority floor: 4·w of 5·w = 80% >
|
||||
// 50%). The self-vote is NEVER needed for equal stake, so the equal-stake devnet hang is NOT this
|
||||
// self-exclusion floor (look at primaryNetworkReady / P-chain bootstrap / beacon connectivity).
|
||||
func TestRED_PROBE_EqualStakeNeedsNoSelfVote(t *testing.T) {
|
||||
chain, byID := buildBSChain(8, -1)
|
||||
vm := newBSVM(chain) // node at genesis (height 0)
|
||||
|
||||
self := ids.GenerateTestNodeID()
|
||||
p1, p2, p3, p4 := ids.GenerateTestNodeID(), ids.GenerateTestNodeID(), ids.GenerateTestNodeID(), ids.GenerateTestNodeID()
|
||||
weights := map[ids.NodeID]uint64{self: 100, p1: 100, p2: 100, p3: 100, p4: 100}
|
||||
|
||||
bh, chainID := newBSHandlerWeighted(t, vm, weights)
|
||||
bh.selfNodeID = self
|
||||
bh.msgCreator = bsMsgBuilder{}
|
||||
bh.net = &bsBeaconNet{bh: bh, chainID: chainID, connected: []ids.NodeID{p1, p2, p3, p4}, byID: byID, tip: chain[0]}
|
||||
|
||||
bh.bsActive.Store(true)
|
||||
tip, status := bh.FrontierTip(context.Background())
|
||||
bh.bsActive.Store(false)
|
||||
|
||||
t.Logf("equal-stake fresh net: status=%v tip=%v", status, tip)
|
||||
require.Contains(t, []chainbootstrap.FrontierStatus{chainbootstrap.FrontierNamed, chainbootstrap.FrontierCaughtUp}, status,
|
||||
"equal-stake peers clear the stake-majority floor unaided — no self-vote needed")
|
||||
require.Equal(t, chain[0].id, tip, "caught up at genesis")
|
||||
}
|
||||
@@ -1,352 +0,0 @@
|
||||
// Copyright (C) 2019-2026, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
// Command repair-proposervm is a ONE-TIME, fail-closed surgical tool to reconcile
|
||||
// a proposervm chain-state DB onto a known-canonical outer block at a single height.
|
||||
//
|
||||
// Motivation (incident 1082814, Lux mainnet C-Chain): a sub-quorum proposervm
|
||||
// envelope A (3-of-5 ACCEPT) was locally accepted on some nodes while the network
|
||||
// finalized the supermajority sibling B (4-of-5) that wraps the IDENTICAL inner EVM
|
||||
// block. The two siblings differ ONLY in the proposervm outer envelope; the inner
|
||||
// EVM state is byte-identical (no EVM divergence). On restart those nodes re-seed
|
||||
// finality from their persisted proposervm lastAccepted=A and fatal on B's cert
|
||||
// (EQUIVOCATION). This tool swaps the persisted record of `height` from A to the
|
||||
// canonical B, leaving the inner EVM completely untouched (no EVM rollback).
|
||||
//
|
||||
// It is NOT a blind hex edit: it opens the exact same typed proposervm state
|
||||
// (luxfi/node/vms/proposervm/state) over the exact same nested keyspace luxd uses
|
||||
// (chainID -> "vm" -> "proposervm" -> versiondb -> chain/block/height), and writes
|
||||
// via the state's own PutBlock / SetBlockIDAtHeight / SetLastAccepted so the on-disk
|
||||
// bytes are identical to what luxd itself wrote for B on the canonical node.
|
||||
//
|
||||
// proposervm invariant honored: proLastAcceptedHeight must never be < the inner VM's
|
||||
// last-accepted height (vm.repairAcceptedChainByHeight). The inner EVM is at `height`
|
||||
// (it accepted the shared inner block under A), so the recovery target is the
|
||||
// canonical block AT `height` (B), never height-1 — keeping outer==inner height.
|
||||
//
|
||||
// Modes:
|
||||
//
|
||||
// inspect : read-only. Print lastAccepted, height index at H and H-1, and the
|
||||
// outer block currently recorded at H.
|
||||
// export : read-only. Read the outer block recorded at H and write its raw
|
||||
// stateless bytes to --block-file (run against a canonical node's DB).
|
||||
// repair : read-write, fail-closed. Parse --block-file (canonical B), assert it is
|
||||
// the expected block, assert the DB is in the expected bad state (lastAccepted
|
||||
// and heightIndex[H] both == the expected sub-quorum block A, and B's parent
|
||||
// == heightIndex[H-1]); then PutBlock(B), SetBlockIDAtHeight(H,B),
|
||||
// SetLastAccepted(B), Commit. Idempotent: if already on B, it no-ops.
|
||||
//
|
||||
// The DB uses an exclusive LOCK; luxd MUST be stopped on the target before `repair`.
|
||||
package main
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
|
||||
"github.com/spf13/cobra"
|
||||
|
||||
"github.com/luxfi/database"
|
||||
databasefactory "github.com/luxfi/database/factory"
|
||||
"github.com/luxfi/database/prefixdb"
|
||||
"github.com/luxfi/database/versiondb"
|
||||
"github.com/luxfi/ids"
|
||||
"github.com/luxfi/log"
|
||||
"github.com/luxfi/metric"
|
||||
|
||||
"github.com/luxfi/node/vms/proposervm/block"
|
||||
"github.com/luxfi/node/vms/proposervm/state"
|
||||
)
|
||||
|
||||
// The proposervm nests its state under these fixed prefixes inside the chain DB.
|
||||
// chainDB = prefixdb(chainID[:], baseDB) [chains.ChainDBManager.GetDatabase]
|
||||
// vmDB = prefixdb("vm", chainDB) [chains.ChainDBManager.GetVMDatabase / VMDBPrefix]
|
||||
// ppvmDB = versiondb(prefixdb("proposervm", vmDB)) [proposervm.VM.Initialize dbPrefix]
|
||||
// state.New(ppvmDB) -> "chain"/"block"/"height" sub-prefixes
|
||||
var (
|
||||
vmDBPrefix = []byte("vm")
|
||||
proposervmDBPrefix = []byte("proposervm")
|
||||
)
|
||||
|
||||
var (
|
||||
dbPath string
|
||||
dbType string
|
||||
chainIDStr string
|
||||
height uint64
|
||||
blockFile string
|
||||
expectBlockID string // canonical B (target)
|
||||
expectCurID string // sub-quorum A (the bad state we expect to overwrite)
|
||||
yes bool
|
||||
)
|
||||
|
||||
func main() {
|
||||
root := &cobra.Command{
|
||||
Use: "repair-proposervm",
|
||||
Short: "Surgical, fail-closed reconcile of a proposervm chain-state onto a canonical outer block at one height",
|
||||
}
|
||||
root.PersistentFlags().StringVar(&dbPath, "db-path", "", "zapdb root (e.g. /data/db/mainnet/db) (required)")
|
||||
root.PersistentFlags().StringVar(&dbType, "db-type", "zapdb", "database type")
|
||||
root.PersistentFlags().StringVar(&chainIDStr, "chain-id", "2wRdZGeca1qkxzNCq88NWDF5nJ5A9o623vRJKd3FsjRYvuVvvt", "blockchain ID (proposervm chain)")
|
||||
root.PersistentFlags().Uint64Var(&height, "height", 1082814, "contested height")
|
||||
root.MarkPersistentFlagRequired("db-path")
|
||||
|
||||
inspect := &cobra.Command{Use: "inspect", Short: "read-only: print proposervm finality state at the height", RunE: runInspect}
|
||||
|
||||
export := &cobra.Command{Use: "export", Short: "read-only: write the outer block recorded at the height to --block-file", RunE: runExport}
|
||||
export.Flags().StringVar(&blockFile, "block-file", "", "output file for the canonical outer block bytes (required)")
|
||||
export.MarkFlagRequired("block-file")
|
||||
|
||||
probe := &cobra.Command{Use: "probe", Short: "read-only: look up an arbitrary block by ID (is it present in the store?)", RunE: runProbe}
|
||||
probe.Flags().StringVar(&expectBlockID, "block-id", "", "block ID to look up (required)")
|
||||
probe.MarkFlagRequired("block-id")
|
||||
|
||||
dump := &cobra.Command{Use: "dump", Short: "read-only: write an arbitrary block (by ID) raw stateless bytes to --block-file", RunE: runDump}
|
||||
dump.Flags().StringVar(&expectBlockID, "block-id", "wDMUyGyaKcC2Vng8i8ngU5f83XEtHZx5hqCSe5tMTwLAPagmo", "block ID to dump (canonical B)")
|
||||
dump.Flags().StringVar(&blockFile, "block-file", "", "output file for the block bytes (required)")
|
||||
dump.MarkFlagRequired("block-file")
|
||||
|
||||
repair := &cobra.Command{Use: "repair", Short: "fail-closed: swap height's outer block from A to canonical B", RunE: runRepair}
|
||||
repair.Flags().StringVar(&blockFile, "block-file", "", "canonical outer block (B) bytes, from `export` (required)")
|
||||
repair.Flags().StringVar(&expectBlockID, "expect-block", "wDMUyGyaKcC2Vng8i8ngU5f83XEtHZx5hqCSe5tMTwLAPagmo", "expected canonical block ID (B)")
|
||||
repair.Flags().StringVar(&expectCurID, "expect-current", "2U2pR3DHCNEFDLnMq2uraNVkThRWgETDd468hR26yGHBQuAnNy", "expected current sub-quorum block ID (A) to be overwritten")
|
||||
repair.Flags().BoolVar(&yes, "yes", false, "confirm the write")
|
||||
repair.MarkFlagRequired("block-file")
|
||||
|
||||
root.AddCommand(inspect, export, probe, dump, repair)
|
||||
if err := root.Execute(); err != nil {
|
||||
fmt.Fprintln(os.Stderr, "ERROR:", err)
|
||||
os.Exit(1)
|
||||
}
|
||||
}
|
||||
|
||||
// openState opens the proposervm typed state over the exact nested keyspace luxd uses.
|
||||
// Returns the state, the proposervm versiondb (for Commit), and the base DB (for Close).
|
||||
func openState(readOnly bool) (state.State, *versiondb.Database, database.Database, ids.ID, error) {
|
||||
chainID, err := ids.FromString(chainIDStr)
|
||||
if err != nil {
|
||||
return nil, nil, nil, ids.Empty, fmt.Errorf("bad chain-id: %w", err)
|
||||
}
|
||||
logger := log.New("cmd", "repair-proposervm")
|
||||
gatherer := metric.NewRegistry()
|
||||
base, err := databasefactory.New(dbType, dbPath, readOnly, nil, gatherer, logger, "repair", "db")
|
||||
if err != nil {
|
||||
return nil, nil, nil, ids.Empty, fmt.Errorf("open db %q: %w", dbPath, err)
|
||||
}
|
||||
chainDB := prefixdb.New(chainID[:], base)
|
||||
vmDB := prefixdb.New(vmDBPrefix, chainDB)
|
||||
ppvmDB := versiondb.New(prefixdb.New(proposervmDBPrefix, vmDB))
|
||||
return state.New(ppvmDB), ppvmDB, base, chainID, nil
|
||||
}
|
||||
|
||||
func idAt(st state.State, h uint64) string {
|
||||
id, err := st.GetBlockIDAtHeight(h)
|
||||
if errors.Is(err, database.ErrNotFound) {
|
||||
return "<none>"
|
||||
}
|
||||
if err != nil {
|
||||
return "<err:" + err.Error() + ">"
|
||||
}
|
||||
return id.String()
|
||||
}
|
||||
|
||||
func runInspect(_ *cobra.Command, _ []string) error {
|
||||
st, _, base, _, err := openState(true)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer base.Close()
|
||||
la, err := st.GetLastAccepted()
|
||||
if err != nil {
|
||||
return fmt.Errorf("GetLastAccepted: %w", err)
|
||||
}
|
||||
fmt.Printf("proposervm lastAccepted = %s\n", la)
|
||||
fmt.Printf("heightIndex[%d] = %s\n", height, idAt(st, height))
|
||||
fmt.Printf("heightIndex[%d] = %s\n", height-1, idAt(st, height-1))
|
||||
if id, err := st.GetBlockIDAtHeight(height); err == nil {
|
||||
if blk, err := st.GetBlock(id); err == nil {
|
||||
fmt.Printf("block@%d: id=%s parent=%s bytes=%d\n", height, blk.ID(), blk.ParentID(), len(blk.Bytes()))
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func runProbe(_ *cobra.Command, _ []string) error {
|
||||
want, err := ids.FromString(expectBlockID)
|
||||
if err != nil {
|
||||
return fmt.Errorf("bad --block-id: %w", err)
|
||||
}
|
||||
// RW so Badger replays its WAL: a verified/built-but-unflushed block may live
|
||||
// only in the memtable. Disposable copy only.
|
||||
st, _, base, _, err := openState(false)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer base.Close()
|
||||
blk, err := st.GetBlock(want)
|
||||
if errors.Is(err, database.ErrNotFound) {
|
||||
fmt.Printf("NOT-PRESENT: block %s is not in this store\n", want)
|
||||
return nil
|
||||
}
|
||||
if err != nil {
|
||||
return fmt.Errorf("GetBlock(%s): %w", want, err)
|
||||
}
|
||||
fmt.Printf("PRESENT: id=%s parent=%s bytes=%d\n", blk.ID(), blk.ParentID(), len(blk.Bytes()))
|
||||
return nil
|
||||
}
|
||||
|
||||
func runDump(_ *cobra.Command, _ []string) error {
|
||||
want, err := ids.FromString(expectBlockID)
|
||||
if err != nil {
|
||||
return fmt.Errorf("bad --block-id: %w", err)
|
||||
}
|
||||
// RW so Badger replays its WAL: a verified-but-unflushed block may live only in
|
||||
// the memtable. We never call a state WRITER here (read + write output file only),
|
||||
// and this runs against an idle (luxd-stopped) pod DB; the contested sibling B was
|
||||
// verified by this node when it saw the conflicting cert, so it is present by ID.
|
||||
st, _, base, _, err := openState(false)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer base.Close()
|
||||
blk, err := st.GetBlock(want)
|
||||
if errors.Is(err, database.ErrNotFound) {
|
||||
return fmt.Errorf("block %s is NOT present in this store (cannot dump)", want)
|
||||
}
|
||||
if err != nil {
|
||||
return fmt.Errorf("GetBlock(%s): %w", want, err)
|
||||
}
|
||||
if blk.ID() != want {
|
||||
return fmt.Errorf("self-ID mismatch: requested=%s parsed=%s (refusing)", want, blk.ID())
|
||||
}
|
||||
if err := os.WriteFile(blockFile, blk.Bytes(), 0o644); err != nil {
|
||||
return fmt.Errorf("write %s: %w", blockFile, err)
|
||||
}
|
||||
fmt.Printf("DUMPED id=%s parent=%s bytes=%d -> %s\n", blk.ID(), blk.ParentID(), len(blk.Bytes()), blockFile)
|
||||
return nil
|
||||
}
|
||||
|
||||
func runExport(_ *cobra.Command, _ []string) error {
|
||||
// Open read-WRITE so Badger replays its value-log/WAL: the canonical block at
|
||||
// the contested height was the LAST write before the chain went idle, so on a
|
||||
// crash-consistent snapshot copy it may live only in the memtable/WAL, not yet
|
||||
// in an SST. A read-only open skips recovery and could miss it. We never call a
|
||||
// state writer here (we only read + write the output file), and this only ever
|
||||
// runs against a DISPOSABLE snapshot copy of a canonical node — never the live node.
|
||||
st, _, base, _, err := openState(false)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer base.Close()
|
||||
id, err := st.GetBlockIDAtHeight(height)
|
||||
if err != nil {
|
||||
return fmt.Errorf("GetBlockIDAtHeight(%d): %w", height, err)
|
||||
}
|
||||
blk, err := st.GetBlock(id)
|
||||
if err != nil {
|
||||
return fmt.Errorf("GetBlock(%s): %w", id, err)
|
||||
}
|
||||
if blk.ID() != id {
|
||||
return fmt.Errorf("block id mismatch: index=%s block=%s", id, blk.ID())
|
||||
}
|
||||
if err := os.WriteFile(blockFile, blk.Bytes(), 0o644); err != nil {
|
||||
return fmt.Errorf("write %s: %w", blockFile, err)
|
||||
}
|
||||
la, _ := st.GetLastAccepted()
|
||||
fmt.Printf("EXPORTED height=%d id=%s parent=%s bytes=%d -> %s\n", height, blk.ID(), blk.ParentID(), len(blk.Bytes()), blockFile)
|
||||
fmt.Printf(" (source lastAccepted=%s heightIndex[%d-1]=%s)\n", la, height, idAt(st, height-1))
|
||||
return nil
|
||||
}
|
||||
|
||||
func runRepair(_ *cobra.Command, _ []string) error {
|
||||
wantB, err := ids.FromString(expectBlockID)
|
||||
if err != nil {
|
||||
return fmt.Errorf("bad --expect-block: %w", err)
|
||||
}
|
||||
wantA, err := ids.FromString(expectCurID)
|
||||
if err != nil {
|
||||
return fmt.Errorf("bad --expect-current: %w", err)
|
||||
}
|
||||
raw, err := os.ReadFile(blockFile)
|
||||
if err != nil {
|
||||
return fmt.Errorf("read %s: %w", blockFile, err)
|
||||
}
|
||||
blk, err := block.ParseWithoutVerification(raw)
|
||||
if err != nil {
|
||||
return fmt.Errorf("parse block file: %w", err)
|
||||
}
|
||||
// (1) the supplied block must be exactly the canonical target B.
|
||||
if blk.ID() != wantB {
|
||||
return fmt.Errorf("block-file id %s != --expect-block %s (refusing)", blk.ID(), wantB)
|
||||
}
|
||||
|
||||
st, ppvmDB, base, _, err := openState(false)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer base.Close()
|
||||
|
||||
la, err := st.GetLastAccepted()
|
||||
if err != nil {
|
||||
return fmt.Errorf("GetLastAccepted: %w", err)
|
||||
}
|
||||
curAtH, err := st.GetBlockIDAtHeight(height)
|
||||
if err != nil {
|
||||
return fmt.Errorf("GetBlockIDAtHeight(%d): %w", height, err)
|
||||
}
|
||||
|
||||
// (2) idempotency: if already on B, do nothing.
|
||||
if la == wantB && curAtH == wantB {
|
||||
fmt.Printf("ALREADY-CANONICAL: lastAccepted and heightIndex[%d] already == B (%s); no-op\n", height, wantB)
|
||||
return nil
|
||||
}
|
||||
|
||||
// (3) fail-closed: only proceed from the exact expected bad state (A at H, lastAccepted A).
|
||||
if la != wantA {
|
||||
return fmt.Errorf("refusing: lastAccepted=%s is neither A(%s) nor B(%s) — unexpected state", la, wantA, wantB)
|
||||
}
|
||||
if curAtH != wantA {
|
||||
return fmt.Errorf("refusing: heightIndex[%d]=%s != A(%s) — unexpected state", height, curAtH, wantA)
|
||||
}
|
||||
|
||||
// (4) B must extend the SAME finalized prefix: B.parent == the block recorded at H-1.
|
||||
parentAtH1, err := st.GetBlockIDAtHeight(height - 1)
|
||||
if err != nil {
|
||||
return fmt.Errorf("GetBlockIDAtHeight(%d): %w", height-1, err)
|
||||
}
|
||||
if blk.ParentID() != parentAtH1 {
|
||||
return fmt.Errorf("refusing: B.parent=%s != heightIndex[%d]=%s (B does not extend the local finalized prefix)", blk.ParentID(), height-1, parentAtH1)
|
||||
}
|
||||
if _, err := st.GetBlock(parentAtH1); err != nil {
|
||||
return fmt.Errorf("refusing: parent %s (height %d) not present in store: %w", parentAtH1, height-1, err)
|
||||
}
|
||||
|
||||
fmt.Printf("PLAN: height=%d %s (A) -> %s (B)\n", height, wantA, wantB)
|
||||
fmt.Printf(" current lastAccepted = %s\n", la)
|
||||
fmt.Printf(" B.parent = %s == heightIndex[%d] OK\n", blk.ParentID(), height-1)
|
||||
if !yes {
|
||||
return errors.New("dry-run only: re-run with --yes to write")
|
||||
}
|
||||
|
||||
// (5) apply via the state's own typed writers (identical on-disk bytes to luxd).
|
||||
if err := st.PutBlock(blk); err != nil {
|
||||
return fmt.Errorf("PutBlock(B): %w", err)
|
||||
}
|
||||
if err := st.SetBlockIDAtHeight(height, blk.ID()); err != nil {
|
||||
return fmt.Errorf("SetBlockIDAtHeight(%d,B): %w", height, err)
|
||||
}
|
||||
if err := st.SetLastAccepted(blk.ID()); err != nil {
|
||||
return fmt.Errorf("SetLastAccepted(B): %w", err)
|
||||
}
|
||||
if err := ppvmDB.Commit(); err != nil {
|
||||
return fmt.Errorf("commit: %w", err)
|
||||
}
|
||||
|
||||
// (6) re-read to confirm.
|
||||
la2, _ := st.GetLastAccepted()
|
||||
fmt.Printf("DONE: lastAccepted=%s heightIndex[%d]=%s heightIndex[%d]=%s\n", la2, height, idAt(st, height), height-1, idAt(st, height-1))
|
||||
if la2 != wantB || idAt(st, height) != wantB.String() {
|
||||
return fmt.Errorf("post-write verification FAILED")
|
||||
}
|
||||
fmt.Println("VERIFIED: proposervm now records canonical B at the contested height; inner EVM untouched.")
|
||||
return nil
|
||||
}
|
||||
+1
-1
@@ -19,7 +19,7 @@ services:
|
||||
LUXD_CONSENSUS_QUORUM_SIZE: "1"
|
||||
LUXD_LOG_LEVEL: "info"
|
||||
healthcheck:
|
||||
test: ["CMD", "curl", "-sf", "http://localhost:9630/v1/health"]
|
||||
test: ["CMD", "curl", "-sf", "http://localhost:9630/ext/health"]
|
||||
interval: 30s
|
||||
timeout: 10s
|
||||
retries: 5
|
||||
|
||||
@@ -1762,14 +1762,6 @@ func GetNodeConfig(v *viper.Viper) (node.Config, error) {
|
||||
if nodeConfig.HealthCheckFreq < 0 {
|
||||
return node.Config{}, fmt.Errorf("%s must be positive", HealthCheckFreqKey)
|
||||
}
|
||||
nodeConfig.ProposerWindowDuration = v.GetDuration(ProposerVMWindowDurationKey)
|
||||
if nodeConfig.ProposerWindowDuration < 0 {
|
||||
return node.Config{}, fmt.Errorf("%s must not be negative", ProposerVMWindowDurationKey)
|
||||
}
|
||||
nodeConfig.ProposerMinBlockDelay = v.GetDuration(ProposerVMMinBlockDelayKey)
|
||||
if nodeConfig.ProposerMinBlockDelay < 0 {
|
||||
return node.Config{}, fmt.Errorf("%s must not be negative", ProposerVMMinBlockDelayKey)
|
||||
}
|
||||
// Halflife of continuous averager used in health checks
|
||||
healthCheckAveragerHalflife := v.GetDuration(HealthCheckAveragerHalflifeKey)
|
||||
if healthCheckAveragerHalflife <= 0 {
|
||||
|
||||
@@ -24,6 +24,7 @@ import (
|
||||
"github.com/luxfi/node/genesis/builder"
|
||||
"github.com/luxfi/node/nets"
|
||||
pchaingenesis "github.com/luxfi/node/vms/platformvm/genesis"
|
||||
pchaintxs "github.com/luxfi/node/vms/platformvm/txs"
|
||||
)
|
||||
|
||||
const chainConfigFilenameExtension = ".ex"
|
||||
@@ -535,8 +536,8 @@ func TestGetNetConfigsFromFlags(t *testing.T) {
|
||||
"2Ctt6eGAeo4MLqTmGa7AdRecuVMPGWEX9wSsCLBYrLhX4a394i": {
|
||||
"consensusParameters": {
|
||||
"k": 30,
|
||||
"alphaPreference": 20,
|
||||
"alphaConfidence": 25
|
||||
"alphaPreference": 16,
|
||||
"alphaConfidence": 20
|
||||
},
|
||||
"validatorOnly": true
|
||||
}
|
||||
@@ -546,8 +547,8 @@ func TestGetNetConfigsFromFlags(t *testing.T) {
|
||||
config, ok := given[id]
|
||||
require.True(ok)
|
||||
require.True(config.ValidatorOnly)
|
||||
require.Equal(20, config.ConsensusParameters.AlphaPreference)
|
||||
require.Equal(25, config.ConsensusParameters.AlphaConfidence)
|
||||
require.Equal(16, config.ConsensusParameters.AlphaPreference)
|
||||
require.Equal(20, config.ConsensusParameters.AlphaConfidence)
|
||||
require.Equal(30, config.ConsensusParameters.K)
|
||||
// must still respect defaults (MainnetParameters.MaxOutstandingItems = 1024)
|
||||
require.Equal(1024, config.ConsensusParameters.MaxOutstandingItems)
|
||||
@@ -732,7 +733,7 @@ func TestResolveUTXOAssetID_POnlyFallback(t *testing.T) {
|
||||
require := require.New(t)
|
||||
|
||||
pOnly := &pchaingenesis.Genesis{Chains: nil}
|
||||
pOnlyBytes, err := pOnly.Bytes()
|
||||
pOnlyBytes, err := pchaingenesis.Codec.Marshal(pchaintxs.CodecVersion, pOnly)
|
||||
require.NoError(err)
|
||||
|
||||
gotID, err := resolveUTXOAssetID(42, pOnlyBytes)
|
||||
|
||||
@@ -369,7 +369,6 @@ func addNodeFlags(fs *pflag.FlagSet) {
|
||||
// ProposerVM
|
||||
fs.Bool(ProposerVMUseCurrentHeightKey, false, "Have the ProposerVM always report the last accepted P-chain block height")
|
||||
fs.Duration(ProposerVMMinBlockDelayKey, proposervm.DefaultMinBlockDelay, "Minimum delay to enforce when building a chain++ block for the primary network chains and the default minimum delay for chains")
|
||||
fs.Duration(ProposerVMWindowDurationKey, 0, "Proposer-slot spacing for block production (0 uses the 5s mainnet default); shrink for fast cadence on small/local networks")
|
||||
|
||||
// Metrics
|
||||
fs.Bool(MeterVMsEnabledKey, true, "Enable Meter VMs to track VM performance with more granularity")
|
||||
|
||||
@@ -187,7 +187,6 @@ const (
|
||||
ConsensusFrontierPollFrequencyKey = "consensus-frontier-poll-frequency"
|
||||
ProposerVMUseCurrentHeightKey = "proposervm-use-current-height"
|
||||
ProposerVMMinBlockDelayKey = "proposervm-min-block-delay"
|
||||
ProposerVMWindowDurationKey = "proposervm-window-duration"
|
||||
FdLimitKey = "fd-limit"
|
||||
IndexEnabledKey = "index-enabled"
|
||||
IndexAllowIncompleteKey = "index-allow-incomplete"
|
||||
|
||||
@@ -177,16 +177,6 @@ type Config struct {
|
||||
// Health
|
||||
HealthCheckFreq time.Duration `json:"healthCheckFreq"`
|
||||
|
||||
// ProposerWindowDuration overrides the proposervm proposer-slot spacing.
|
||||
// Zero keeps the 5s mainnet default; small local/dev nets set it low (e.g.
|
||||
// 1s) so block cadence is not floored at 5s per proposer slot.
|
||||
ProposerWindowDuration time.Duration `json:"proposerWindowDuration"`
|
||||
|
||||
// ProposerMinBlockDelay is the proposervm minimum delay between consecutive
|
||||
// blocks (the hard cadence floor). Zero keeps the 1s default; high-throughput
|
||||
// / DEX nets set it low (e.g. 1ms) to approach the consensus-finality floor.
|
||||
ProposerMinBlockDelay time.Duration `json:"proposerMinBlockDelay"`
|
||||
|
||||
// Network configuration
|
||||
NetworkConfig network.Config `json:"networkConfig"`
|
||||
|
||||
|
||||
@@ -68,54 +68,3 @@ Block arrives
|
||||
## Recent Changes
|
||||
|
||||
- 2026-01-04: Created documentation files with Vote terminology
|
||||
|
||||
## consensus/quasar — PQ-finality VERIFY gate (2026-06-28)
|
||||
|
||||
Supersedes the stale "Quasar wrapper / CoronaCoordinator" notes above (that
|
||||
subpackage did not exist in-tree). The current `consensus/quasar` package is the
|
||||
node-side integration of `luxfi/consensus@v1.29.0`'s typed compact-cert finality
|
||||
layer (`protocol/quasar.VerifyConsensusCert`). It wires the VERIFY half only.
|
||||
|
||||
Model: luxd finalizes on classical Snow every block; at CHECKPOINTS (height %
|
||||
interval) a sampled committee's QuasarCert over the finalized digest is VERIFIED.
|
||||
Default posture HYBRID_PQ = Beam(BLS) ∧ Pulsar (ML-DSA-65); STRICT_DUAL_PQ
|
||||
(+Corona) / POLARIS (+Magnetar) configurable.
|
||||
|
||||
THE SAFETY CONTRACT — forward-dated, DORMANT by default:
|
||||
- `Gate.VerifyAccepted` is the accept-path boundary. nil gate / `Activation.Height
|
||||
== 0` / below-height / non-checkpoint => no-op; classical finality UNCHANGED.
|
||||
- Activated at a checkpoint => REQUIRE a valid cert bound to the finalized block;
|
||||
FAIL CLOSED (missing/mismatch/invalid => error from Accept, halts without
|
||||
persisting). Activation is HEIGHT-ONLY (deterministic — no wall clock).
|
||||
- Hooked in `vms/proposervm/post_fork_block.go Accept()` via
|
||||
`vm.verifyQuasarFinality(b)`; the VM's `quasarGate` is nil in production today
|
||||
(set via `SetQuasarGate`). Nothing wires it yet — that is the activation step.
|
||||
|
||||
Files: gate.go (Gate/ActivationConfig/bindCheck), policy.go (HYBRID_PQ default,
|
||||
cert can't pick its own policy), validators.go (ConsensusValidatorSet: BLS+Pulsar
|
||||
keys), store.go (MemCertStore), producer.go (committee-signer interface =
|
||||
scaffolding; nil = verify-only), errors.go. Tests: gate_test.go (13, -race green:
|
||||
dormant no-op, fail-closed, epoch/round/chain/height/block anti-replay, real-
|
||||
verifier delegation, misconfigured-fails-closed).
|
||||
|
||||
REMAINING WORK to reach a live PQ-finality network (all owner-gated):
|
||||
1. Producer service (pulsard): the per-validator committee cert signer. Needs
|
||||
pulsar v1.7.1 (no-reconstruct hyperball signer) AND consensus to EXPORT the
|
||||
currently package-private cert/payload ENCODERS (an external producer cannot
|
||||
assemble a ConsensusCert envelope without them; this also unblocks an end-to-
|
||||
end positive verify test).
|
||||
2. Cert gossip/ingest -> MemCertStore (verify-before-store); MemCertStore needs
|
||||
eviction below last-finalized height before this lands.
|
||||
3. Production per-epoch `ValidatorSetProvider` from the P-Chain validator manager
|
||||
+ KeyEra registry (BLS aggregate + Pulsar/Corona/Magnetar group keys per era).
|
||||
4. Config-flag -> SetQuasarGate wiring (construct a non-nil gate from node config;
|
||||
choose ChainID = sovereign/EVM chain id).
|
||||
5. Cert-unavailability runbook + a bounded grace window (await cert N rounds)
|
||||
before a checkpoint halts — fail-closed-after-decision can brick a chain if
|
||||
gossip is down. REQUIRED before any forward-dated activation.
|
||||
6. A proposervm Accept-path integration test (nil/dormant/activated).
|
||||
|
||||
Mainnet activation order (owner): deploy producer -> verify cert-gossip coverage
|
||||
at checkpoints -> set Activation.Height to a forward-dated height with margin ->
|
||||
roll via `kubectl patch sts luxd` OnDelete, 1 pod at a time. NEVER wipe /data/db,
|
||||
NEVER pkill, NEVER blind-restart.
|
||||
|
||||
@@ -0,0 +1,353 @@
|
||||
// Copyright (C) 2019-2025, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
package quasar
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Configuration errors
|
||||
var (
|
||||
ErrInvalidK = errors.New("K must be positive")
|
||||
ErrInvalidAlpha = errors.New("alpha must be in (0, 1]")
|
||||
ErrInvalidBeta = errors.New("beta must be positive and <= K")
|
||||
ErrInvalidThreshold = errors.New("threshold must be >= 2 and <= parties")
|
||||
ErrInvalidQuorum = errors.New("quorum numerator must be <= denominator")
|
||||
ErrInvalidTimeout = errors.New("timeout must be positive")
|
||||
ErrInvalidInterval = errors.New("polling interval must be positive")
|
||||
)
|
||||
|
||||
// -----------------------------------------------------------------------------
|
||||
// Core Consensus Parameters (compile-time, immutable after construction)
|
||||
// -----------------------------------------------------------------------------
|
||||
|
||||
// CoreParams defines the fundamental Lux consensus parameters.
|
||||
// These are protocol-critical and must match across all validators.
|
||||
type CoreParams struct {
|
||||
// K is the sample size for each consensus query round.
|
||||
// Typical values: 20-25 for production networks.
|
||||
K int
|
||||
|
||||
// Alpha is the quorum threshold as a fraction of K.
|
||||
// A response is accepted if >= ceil(K * Alpha) validators agree.
|
||||
// Must be in (0.5, 1] for Byzantine fault tolerance.
|
||||
// Typical value: 0.8 (80% of sample must agree).
|
||||
Alpha float64
|
||||
|
||||
// BetaVirtuous is the number of consecutive successful polls
|
||||
// required to finalize a virtuous (non-conflicting) decision.
|
||||
// Higher values increase latency but improve consistency.
|
||||
// Typical value: 15-20.
|
||||
BetaVirtuous int
|
||||
|
||||
// BetaRogue is the number of consecutive successful polls
|
||||
// required to finalize a rogue (conflicting) decision.
|
||||
// Should be >= BetaVirtuous.
|
||||
// Typical value: 20-25.
|
||||
BetaRogue int
|
||||
}
|
||||
|
||||
// Validate checks CoreParams invariants.
|
||||
func (p CoreParams) Validate() error {
|
||||
if p.K <= 0 {
|
||||
return ErrInvalidK
|
||||
}
|
||||
if p.Alpha <= 0 || p.Alpha > 1 {
|
||||
return ErrInvalidAlpha
|
||||
}
|
||||
if p.BetaVirtuous <= 0 || p.BetaVirtuous > p.K {
|
||||
return ErrInvalidBeta
|
||||
}
|
||||
if p.BetaRogue <= 0 || p.BetaRogue > p.K {
|
||||
return ErrInvalidBeta
|
||||
}
|
||||
if p.BetaRogue < p.BetaVirtuous {
|
||||
return fmt.Errorf("BetaRogue (%d) must be >= BetaVirtuous (%d)", p.BetaRogue, p.BetaVirtuous)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// AlphaThreshold returns the minimum agreements needed for quorum.
|
||||
func (p CoreParams) AlphaThreshold() int {
|
||||
return int(float64(p.K)*p.Alpha + 0.999) // ceil
|
||||
}
|
||||
|
||||
// DefaultCoreParams returns production-ready core parameters.
|
||||
func DefaultCoreParams() CoreParams {
|
||||
return CoreParams{
|
||||
K: 20,
|
||||
Alpha: 0.8,
|
||||
BetaVirtuous: 15,
|
||||
BetaRogue: 20,
|
||||
}
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------------
|
||||
// Threshold Signing Parameters (compile-time, for Corona/BLS threshold)
|
||||
// -----------------------------------------------------------------------------
|
||||
|
||||
// ThresholdParams defines t-of-n threshold signature configuration.
|
||||
type ThresholdParams struct {
|
||||
// NumParties is the total number of signing parties (validators).
|
||||
// Must be >= 3 for threshold signatures.
|
||||
NumParties int
|
||||
|
||||
// Threshold is the minimum signers required (t in t-of-n).
|
||||
// For BFT: typically 2/3 + 1.
|
||||
// Must be >= 2 and <= NumParties.
|
||||
Threshold int
|
||||
}
|
||||
|
||||
// Validate checks ThresholdParams invariants.
|
||||
func (p ThresholdParams) Validate() error {
|
||||
if p.NumParties < 3 {
|
||||
return fmt.Errorf("%w: need at least 3 parties, got %d", ErrInvalidThreshold, p.NumParties)
|
||||
}
|
||||
if p.Threshold < 2 || p.Threshold > p.NumParties {
|
||||
return fmt.Errorf("%w: threshold=%d, parties=%d", ErrInvalidThreshold, p.Threshold, p.NumParties)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// DefaultThresholdParams returns 2/3+1 threshold for n parties.
|
||||
func DefaultThresholdParams(numParties int) ThresholdParams {
|
||||
threshold := (numParties * 2 / 3) + 1
|
||||
if threshold < 2 {
|
||||
threshold = 2
|
||||
}
|
||||
if threshold > numParties {
|
||||
threshold = numParties
|
||||
}
|
||||
return ThresholdParams{
|
||||
NumParties: numParties,
|
||||
Threshold: threshold,
|
||||
}
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------------
|
||||
// Quorum Parameters (compile-time, for BLS aggregate weight verification)
|
||||
// -----------------------------------------------------------------------------
|
||||
|
||||
// QuorumParams defines weight-based quorum requirements.
|
||||
type QuorumParams struct {
|
||||
// Numerator and Denominator define the minimum weight fraction.
|
||||
// Quorum is met when SignerWeight/TotalWeight >= Numerator/Denominator.
|
||||
// For BFT: typically 2/3 (Numerator=2, Denominator=3).
|
||||
Numerator uint64
|
||||
Denominator uint64
|
||||
}
|
||||
|
||||
// Validate checks QuorumParams invariants.
|
||||
func (p QuorumParams) Validate() error {
|
||||
if p.Denominator == 0 {
|
||||
return fmt.Errorf("%w: denominator cannot be zero", ErrInvalidQuorum)
|
||||
}
|
||||
if p.Numerator > p.Denominator {
|
||||
return fmt.Errorf("%w: numerator=%d > denominator=%d", ErrInvalidQuorum, p.Numerator, p.Denominator)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// RequiredWeight returns minimum weight needed for quorum given totalWeight.
|
||||
func (p QuorumParams) RequiredWeight(totalWeight uint64) uint64 {
|
||||
return totalWeight * p.Numerator / p.Denominator
|
||||
}
|
||||
|
||||
// IsMet returns true if signerWeight meets quorum given totalWeight.
|
||||
func (p QuorumParams) IsMet(signerWeight, totalWeight uint64) bool {
|
||||
return signerWeight >= p.RequiredWeight(totalWeight)
|
||||
}
|
||||
|
||||
// DefaultQuorumParams returns 2/3 quorum (67% of weight required).
|
||||
func DefaultQuorumParams() QuorumParams {
|
||||
return QuorumParams{
|
||||
Numerator: 2,
|
||||
Denominator: 3,
|
||||
}
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------------
|
||||
// Runtime Configuration (can be adjusted, but affects liveness not safety)
|
||||
// -----------------------------------------------------------------------------
|
||||
|
||||
// RuntimeConfig holds tunable runtime parameters.
|
||||
// These affect performance and liveness but not consensus safety.
|
||||
type RuntimeConfig struct {
|
||||
// PollInterval is the delay between consensus query rounds.
|
||||
// Lower values decrease latency but increase network load.
|
||||
// Typical value: 100-500ms.
|
||||
PollInterval time.Duration
|
||||
|
||||
// QueryTimeout is the maximum time to wait for query responses.
|
||||
// Must be > PollInterval.
|
||||
// Typical value: 2-5s.
|
||||
QueryTimeout time.Duration
|
||||
|
||||
// FinalityChannelSize is the buffer size for the finality event channel.
|
||||
FinalityChannelSize int
|
||||
|
||||
// MaxConcurrentQueries limits parallel outstanding queries.
|
||||
// 0 means unlimited.
|
||||
MaxConcurrentQueries int
|
||||
}
|
||||
|
||||
// Validate checks RuntimeConfig invariants.
|
||||
func (c RuntimeConfig) Validate() error {
|
||||
if c.PollInterval <= 0 {
|
||||
return ErrInvalidInterval
|
||||
}
|
||||
if c.QueryTimeout <= 0 {
|
||||
return ErrInvalidTimeout
|
||||
}
|
||||
if c.QueryTimeout < c.PollInterval {
|
||||
return fmt.Errorf("query timeout (%v) must be >= poll interval (%v)", c.QueryTimeout, c.PollInterval)
|
||||
}
|
||||
if c.FinalityChannelSize < 0 {
|
||||
return fmt.Errorf("finality channel size must be >= 0")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// DefaultRuntimeConfig returns production-ready runtime configuration.
|
||||
func DefaultRuntimeConfig() RuntimeConfig {
|
||||
return RuntimeConfig{
|
||||
PollInterval: 250 * time.Millisecond,
|
||||
QueryTimeout: 2 * time.Second,
|
||||
FinalityChannelSize: 100,
|
||||
MaxConcurrentQueries: 0, // unlimited
|
||||
}
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------------
|
||||
// Complete Configuration
|
||||
// -----------------------------------------------------------------------------
|
||||
|
||||
// Config is the complete Quasar consensus configuration.
|
||||
// Use ConfigBuilder for fluent construction.
|
||||
type Config struct {
|
||||
Core CoreParams
|
||||
Threshold ThresholdParams
|
||||
Quorum QuorumParams
|
||||
Runtime RuntimeConfig
|
||||
}
|
||||
|
||||
// Validate checks all configuration invariants.
|
||||
func (c Config) Validate() error {
|
||||
if err := c.Core.Validate(); err != nil {
|
||||
return fmt.Errorf("core params: %w", err)
|
||||
}
|
||||
if err := c.Threshold.Validate(); err != nil {
|
||||
return fmt.Errorf("threshold params: %w", err)
|
||||
}
|
||||
if err := c.Quorum.Validate(); err != nil {
|
||||
return fmt.Errorf("quorum params: %w", err)
|
||||
}
|
||||
if err := c.Runtime.Validate(); err != nil {
|
||||
return fmt.Errorf("runtime config: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// DefaultConfig returns a production-ready configuration.
|
||||
// Call DefaultConfig().WithNumParties(n) to set validator count.
|
||||
func DefaultConfig() Config {
|
||||
return Config{
|
||||
Core: DefaultCoreParams(),
|
||||
Threshold: DefaultThresholdParams(3), // default 3 validators
|
||||
Quorum: DefaultQuorumParams(),
|
||||
Runtime: DefaultRuntimeConfig(),
|
||||
}
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------------------
|
||||
// ConfigBuilder provides fluent configuration construction
|
||||
// -----------------------------------------------------------------------------
|
||||
|
||||
// ConfigBuilder enables fluent Config construction with validation.
|
||||
type ConfigBuilder struct {
|
||||
config Config
|
||||
}
|
||||
|
||||
// NewConfigBuilder creates a builder starting from defaults.
|
||||
func NewConfigBuilder() *ConfigBuilder {
|
||||
return &ConfigBuilder{
|
||||
config: DefaultConfig(),
|
||||
}
|
||||
}
|
||||
|
||||
// WithK sets the sample size.
|
||||
func (b *ConfigBuilder) WithK(k int) *ConfigBuilder {
|
||||
b.config.Core.K = k
|
||||
return b
|
||||
}
|
||||
|
||||
// WithAlpha sets the quorum fraction.
|
||||
func (b *ConfigBuilder) WithAlpha(alpha float64) *ConfigBuilder {
|
||||
b.config.Core.Alpha = alpha
|
||||
return b
|
||||
}
|
||||
|
||||
// WithBeta sets both BetaVirtuous and BetaRogue.
|
||||
func (b *ConfigBuilder) WithBeta(virtuous, rogue int) *ConfigBuilder {
|
||||
b.config.Core.BetaVirtuous = virtuous
|
||||
b.config.Core.BetaRogue = rogue
|
||||
return b
|
||||
}
|
||||
|
||||
// WithNumParties sets the validator count and computes 2/3+1 threshold.
|
||||
func (b *ConfigBuilder) WithNumParties(n int) *ConfigBuilder {
|
||||
b.config.Threshold = DefaultThresholdParams(n)
|
||||
return b
|
||||
}
|
||||
|
||||
// WithThreshold sets an explicit threshold (overrides default 2/3+1).
|
||||
func (b *ConfigBuilder) WithThreshold(threshold int) *ConfigBuilder {
|
||||
b.config.Threshold.Threshold = threshold
|
||||
return b
|
||||
}
|
||||
|
||||
// WithQuorum sets the quorum fraction as numerator/denominator.
|
||||
func (b *ConfigBuilder) WithQuorum(num, denom uint64) *ConfigBuilder {
|
||||
b.config.Quorum.Numerator = num
|
||||
b.config.Quorum.Denominator = denom
|
||||
return b
|
||||
}
|
||||
|
||||
// WithPollInterval sets the polling interval.
|
||||
func (b *ConfigBuilder) WithPollInterval(d time.Duration) *ConfigBuilder {
|
||||
b.config.Runtime.PollInterval = d
|
||||
return b
|
||||
}
|
||||
|
||||
// WithQueryTimeout sets the query timeout.
|
||||
func (b *ConfigBuilder) WithQueryTimeout(d time.Duration) *ConfigBuilder {
|
||||
b.config.Runtime.QueryTimeout = d
|
||||
return b
|
||||
}
|
||||
|
||||
// WithFinalityChannelSize sets the finality channel buffer size.
|
||||
func (b *ConfigBuilder) WithFinalityChannelSize(size int) *ConfigBuilder {
|
||||
b.config.Runtime.FinalityChannelSize = size
|
||||
return b
|
||||
}
|
||||
|
||||
// Build validates and returns the configuration.
|
||||
func (b *ConfigBuilder) Build() (Config, error) {
|
||||
if err := b.config.Validate(); err != nil {
|
||||
return Config{}, err
|
||||
}
|
||||
return b.config, nil
|
||||
}
|
||||
|
||||
// MustBuild validates and returns the configuration, panicking on error.
|
||||
// Use only in tests or when configuration is known to be valid.
|
||||
func (b *ConfigBuilder) MustBuild() Config {
|
||||
cfg, err := b.Build()
|
||||
if err != nil {
|
||||
panic(fmt.Sprintf("invalid config: %v", err))
|
||||
}
|
||||
return cfg
|
||||
}
|
||||
@@ -0,0 +1,434 @@
|
||||
// Copyright (C) 2019-2025, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
package quasar
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func TestDefaultConfig(t *testing.T) {
|
||||
cfg := DefaultConfig()
|
||||
if err := cfg.Validate(); err != nil {
|
||||
t.Fatalf("default config should be valid: %v", err)
|
||||
}
|
||||
|
||||
// Verify defaults match documentation
|
||||
if cfg.Core.K != 20 {
|
||||
t.Errorf("expected K=20, got %d", cfg.Core.K)
|
||||
}
|
||||
if cfg.Core.Alpha != 0.8 {
|
||||
t.Errorf("expected Alpha=0.8, got %f", cfg.Core.Alpha)
|
||||
}
|
||||
if cfg.Core.BetaVirtuous != 15 {
|
||||
t.Errorf("expected BetaVirtuous=15, got %d", cfg.Core.BetaVirtuous)
|
||||
}
|
||||
if cfg.Core.BetaRogue != 20 {
|
||||
t.Errorf("expected BetaRogue=20, got %d", cfg.Core.BetaRogue)
|
||||
}
|
||||
if cfg.Quorum.Numerator != 2 || cfg.Quorum.Denominator != 3 {
|
||||
t.Errorf("expected 2/3 quorum, got %d/%d", cfg.Quorum.Numerator, cfg.Quorum.Denominator)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCoreParamsValidation(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
params CoreParams
|
||||
wantErr bool
|
||||
}{
|
||||
{
|
||||
name: "valid defaults",
|
||||
params: DefaultCoreParams(),
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "zero K",
|
||||
params: CoreParams{K: 0, Alpha: 0.8, BetaVirtuous: 15, BetaRogue: 20},
|
||||
wantErr: true,
|
||||
},
|
||||
{
|
||||
name: "negative K",
|
||||
params: CoreParams{K: -1, Alpha: 0.8, BetaVirtuous: 15, BetaRogue: 20},
|
||||
wantErr: true,
|
||||
},
|
||||
{
|
||||
name: "alpha zero",
|
||||
params: CoreParams{K: 20, Alpha: 0, BetaVirtuous: 15, BetaRogue: 20},
|
||||
wantErr: true,
|
||||
},
|
||||
{
|
||||
name: "alpha greater than 1",
|
||||
params: CoreParams{K: 20, Alpha: 1.5, BetaVirtuous: 15, BetaRogue: 20},
|
||||
wantErr: true,
|
||||
},
|
||||
{
|
||||
name: "alpha exactly 1",
|
||||
params: CoreParams{K: 20, Alpha: 1.0, BetaVirtuous: 15, BetaRogue: 20},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "beta virtuous zero",
|
||||
params: CoreParams{K: 20, Alpha: 0.8, BetaVirtuous: 0, BetaRogue: 20},
|
||||
wantErr: true,
|
||||
},
|
||||
{
|
||||
name: "beta rogue less than virtuous",
|
||||
params: CoreParams{K: 20, Alpha: 0.8, BetaVirtuous: 20, BetaRogue: 15},
|
||||
wantErr: true,
|
||||
},
|
||||
{
|
||||
name: "beta exceeds K",
|
||||
params: CoreParams{K: 10, Alpha: 0.8, BetaVirtuous: 15, BetaRogue: 20},
|
||||
wantErr: true,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
err := tt.params.Validate()
|
||||
if (err != nil) != tt.wantErr {
|
||||
t.Errorf("Validate() error = %v, wantErr %v", err, tt.wantErr)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestAlphaThreshold(t *testing.T) {
|
||||
tests := []struct {
|
||||
k int
|
||||
alpha float64
|
||||
want int
|
||||
}{
|
||||
{k: 20, alpha: 0.8, want: 16}, // 20 * 0.8 = 16
|
||||
{k: 20, alpha: 0.51, want: 11}, // ceil(10.2) = 11
|
||||
{k: 10, alpha: 0.67, want: 7}, // ceil(6.7) = 7
|
||||
{k: 5, alpha: 1.0, want: 5}, // 5 * 1.0 = 5
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run("", func(t *testing.T) {
|
||||
p := CoreParams{K: tt.k, Alpha: tt.alpha, BetaVirtuous: 1, BetaRogue: 1}
|
||||
got := p.AlphaThreshold()
|
||||
if got != tt.want {
|
||||
t.Errorf("AlphaThreshold(%d, %f) = %d, want %d", tt.k, tt.alpha, got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestThresholdParamsValidation(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
params ThresholdParams
|
||||
wantErr bool
|
||||
}{
|
||||
{
|
||||
name: "valid 3 of 5",
|
||||
params: ThresholdParams{NumParties: 5, Threshold: 3},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "valid 4 of 5",
|
||||
params: ThresholdParams{NumParties: 5, Threshold: 4},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "valid 2 of 3 minimum",
|
||||
params: ThresholdParams{NumParties: 3, Threshold: 2},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "too few parties",
|
||||
params: ThresholdParams{NumParties: 2, Threshold: 2},
|
||||
wantErr: true,
|
||||
},
|
||||
{
|
||||
name: "threshold too low",
|
||||
params: ThresholdParams{NumParties: 5, Threshold: 1},
|
||||
wantErr: true,
|
||||
},
|
||||
{
|
||||
name: "threshold exceeds parties",
|
||||
params: ThresholdParams{NumParties: 5, Threshold: 6},
|
||||
wantErr: true,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
err := tt.params.Validate()
|
||||
if (err != nil) != tt.wantErr {
|
||||
t.Errorf("Validate() error = %v, wantErr %v", err, tt.wantErr)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestDefaultThresholdParams(t *testing.T) {
|
||||
tests := []struct {
|
||||
parties int
|
||||
wantThreshold int
|
||||
}{
|
||||
{parties: 3, wantThreshold: 3}, // 2/3 of 3 = 2, +1 = 3
|
||||
{parties: 4, wantThreshold: 3}, // 2/3 of 4 = 2, +1 = 3
|
||||
{parties: 5, wantThreshold: 4}, // 2/3 of 5 = 3, +1 = 4
|
||||
{parties: 10, wantThreshold: 7}, // 2/3 of 10 = 6, +1 = 7
|
||||
{parties: 21, wantThreshold: 15}, // 2/3 of 21 = 14, +1 = 15
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run("", func(t *testing.T) {
|
||||
p := DefaultThresholdParams(tt.parties)
|
||||
if p.Threshold != tt.wantThreshold {
|
||||
t.Errorf("DefaultThresholdParams(%d).Threshold = %d, want %d",
|
||||
tt.parties, p.Threshold, tt.wantThreshold)
|
||||
}
|
||||
if err := p.Validate(); err != nil {
|
||||
t.Errorf("DefaultThresholdParams(%d) produced invalid params: %v", tt.parties, err)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestQuorumParamsValidation(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
params QuorumParams
|
||||
wantErr bool
|
||||
}{
|
||||
{
|
||||
name: "valid 2/3",
|
||||
params: QuorumParams{Numerator: 2, Denominator: 3},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "valid 1/2",
|
||||
params: QuorumParams{Numerator: 1, Denominator: 2},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "valid 1/1 (unanimous)",
|
||||
params: QuorumParams{Numerator: 1, Denominator: 1},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "zero denominator",
|
||||
params: QuorumParams{Numerator: 2, Denominator: 0},
|
||||
wantErr: true,
|
||||
},
|
||||
{
|
||||
name: "numerator exceeds denominator",
|
||||
params: QuorumParams{Numerator: 4, Denominator: 3},
|
||||
wantErr: true,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
err := tt.params.Validate()
|
||||
if (err != nil) != tt.wantErr {
|
||||
t.Errorf("Validate() error = %v, wantErr %v", err, tt.wantErr)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestQuorumIsMet(t *testing.T) {
|
||||
q := QuorumParams{Numerator: 2, Denominator: 3} // 2/3 = 66.67%
|
||||
|
||||
// Note: integer math: 100 * 2 / 3 = 66 (floor division)
|
||||
tests := []struct {
|
||||
signerWeight uint64
|
||||
totalWeight uint64
|
||||
want bool
|
||||
}{
|
||||
{signerWeight: 67, totalWeight: 100, want: true}, // 67 >= 66
|
||||
{signerWeight: 66, totalWeight: 100, want: true}, // 66 >= 66 (floor division)
|
||||
{signerWeight: 65, totalWeight: 100, want: false}, // 65 < 66
|
||||
{signerWeight: 100, totalWeight: 100, want: true}, // 100 >= 66
|
||||
{signerWeight: 0, totalWeight: 100, want: false}, // 0 < 66
|
||||
{signerWeight: 2, totalWeight: 3, want: true}, // 2 >= 2 (3*2/3=2)
|
||||
{signerWeight: 1, totalWeight: 3, want: false}, // 1 < 2
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
got := q.IsMet(tt.signerWeight, tt.totalWeight)
|
||||
if got != tt.want {
|
||||
t.Errorf("IsMet(%d, %d) = %v, want %v", tt.signerWeight, tt.totalWeight, got, tt.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestRuntimeConfigValidation(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
config RuntimeConfig
|
||||
wantErr bool
|
||||
}{
|
||||
{
|
||||
name: "valid defaults",
|
||||
config: DefaultRuntimeConfig(),
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "zero poll interval",
|
||||
config: RuntimeConfig{
|
||||
PollInterval: 0,
|
||||
QueryTimeout: time.Second,
|
||||
},
|
||||
wantErr: true,
|
||||
},
|
||||
{
|
||||
name: "negative poll interval",
|
||||
config: RuntimeConfig{
|
||||
PollInterval: -time.Millisecond,
|
||||
QueryTimeout: time.Second,
|
||||
},
|
||||
wantErr: true,
|
||||
},
|
||||
{
|
||||
name: "query timeout less than poll interval",
|
||||
config: RuntimeConfig{
|
||||
PollInterval: time.Second,
|
||||
QueryTimeout: 100 * time.Millisecond,
|
||||
},
|
||||
wantErr: true,
|
||||
},
|
||||
{
|
||||
name: "negative channel size",
|
||||
config: RuntimeConfig{
|
||||
PollInterval: 250 * time.Millisecond,
|
||||
QueryTimeout: 2 * time.Second,
|
||||
FinalityChannelSize: -1,
|
||||
},
|
||||
wantErr: true,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
err := tt.config.Validate()
|
||||
if (err != nil) != tt.wantErr {
|
||||
t.Errorf("Validate() error = %v, wantErr %v", err, tt.wantErr)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestConfigBuilder(t *testing.T) {
|
||||
// Test fluent API
|
||||
cfg, err := NewConfigBuilder().
|
||||
WithK(25).
|
||||
WithAlpha(0.75).
|
||||
WithBeta(18, 22).
|
||||
WithNumParties(10).
|
||||
WithQuorum(3, 4). // 75%
|
||||
WithPollInterval(500 * time.Millisecond).
|
||||
WithQueryTimeout(5 * time.Second).
|
||||
Build()
|
||||
|
||||
if err != nil {
|
||||
t.Fatalf("Build() error = %v", err)
|
||||
}
|
||||
|
||||
if cfg.Core.K != 25 {
|
||||
t.Errorf("K = %d, want 25", cfg.Core.K)
|
||||
}
|
||||
if cfg.Core.Alpha != 0.75 {
|
||||
t.Errorf("Alpha = %f, want 0.75", cfg.Core.Alpha)
|
||||
}
|
||||
if cfg.Core.BetaVirtuous != 18 {
|
||||
t.Errorf("BetaVirtuous = %d, want 18", cfg.Core.BetaVirtuous)
|
||||
}
|
||||
if cfg.Core.BetaRogue != 22 {
|
||||
t.Errorf("BetaRogue = %d, want 22", cfg.Core.BetaRogue)
|
||||
}
|
||||
if cfg.Threshold.NumParties != 10 {
|
||||
t.Errorf("NumParties = %d, want 10", cfg.Threshold.NumParties)
|
||||
}
|
||||
if cfg.Threshold.Threshold != 7 { // 2/3 of 10 + 1 = 7
|
||||
t.Errorf("Threshold = %d, want 7", cfg.Threshold.Threshold)
|
||||
}
|
||||
if cfg.Quorum.Numerator != 3 || cfg.Quorum.Denominator != 4 {
|
||||
t.Errorf("Quorum = %d/%d, want 3/4", cfg.Quorum.Numerator, cfg.Quorum.Denominator)
|
||||
}
|
||||
if cfg.Runtime.PollInterval != 500*time.Millisecond {
|
||||
t.Errorf("PollInterval = %v, want 500ms", cfg.Runtime.PollInterval)
|
||||
}
|
||||
}
|
||||
|
||||
func TestConfigBuilderWithExplicitThreshold(t *testing.T) {
|
||||
cfg, err := NewConfigBuilder().
|
||||
WithNumParties(10).
|
||||
WithThreshold(5). // Override default 7
|
||||
Build()
|
||||
|
||||
if err != nil {
|
||||
t.Fatalf("Build() error = %v", err)
|
||||
}
|
||||
|
||||
if cfg.Threshold.Threshold != 5 {
|
||||
t.Errorf("Threshold = %d, want 5", cfg.Threshold.Threshold)
|
||||
}
|
||||
}
|
||||
|
||||
func TestConfigBuilderValidationError(t *testing.T) {
|
||||
_, err := NewConfigBuilder().
|
||||
WithK(-1). // Invalid
|
||||
Build()
|
||||
|
||||
if err == nil {
|
||||
t.Error("Build() should return error for invalid K")
|
||||
}
|
||||
}
|
||||
|
||||
func TestConfigBuilderMustBuildPanics(t *testing.T) {
|
||||
defer func() {
|
||||
if r := recover(); r == nil {
|
||||
t.Error("MustBuild() should panic on invalid config")
|
||||
}
|
||||
}()
|
||||
|
||||
NewConfigBuilder().WithK(-1).MustBuild()
|
||||
}
|
||||
|
||||
func TestConfigBuilderMustBuildSuccess(t *testing.T) {
|
||||
// Should not panic
|
||||
cfg := NewConfigBuilder().MustBuild()
|
||||
if err := cfg.Validate(); err != nil {
|
||||
t.Errorf("MustBuild() produced invalid config: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestConfigImmutability documents that Config values are immutable after creation.
|
||||
func TestConfigImmutability(t *testing.T) {
|
||||
cfg := DefaultConfig()
|
||||
|
||||
// These are value types, so modifications don't affect the original
|
||||
core := cfg.Core
|
||||
core.K = 999
|
||||
|
||||
if cfg.Core.K == 999 {
|
||||
t.Error("Config.Core should be immutable (value copy)")
|
||||
}
|
||||
}
|
||||
|
||||
// BenchmarkQuorumCheck benchmarks the quorum check operation.
|
||||
func BenchmarkQuorumCheck(b *testing.B) {
|
||||
q := DefaultQuorumParams()
|
||||
b.ResetTimer()
|
||||
for i := 0; i < b.N; i++ {
|
||||
_ = q.IsMet(70, 100)
|
||||
}
|
||||
}
|
||||
|
||||
// BenchmarkConfigValidation benchmarks config validation.
|
||||
func BenchmarkConfigValidation(b *testing.B) {
|
||||
cfg := DefaultConfig()
|
||||
b.ResetTimer()
|
||||
for i := 0; i < b.N; i++ {
|
||||
_ = cfg.Validate()
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,306 @@
|
||||
// Copyright (C) 2019-2025, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
package quasar
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/luxfi/ids"
|
||||
"github.com/luxfi/log"
|
||||
)
|
||||
|
||||
// --- Config edge cases ---
|
||||
|
||||
func TestDefaultThresholdParamsSmall(t *testing.T) {
|
||||
// numParties=1: threshold clamped to numParties
|
||||
p := DefaultThresholdParams(1)
|
||||
if p.Threshold != 1 {
|
||||
t.Errorf("expected threshold 1 for 1 party, got %d", p.Threshold)
|
||||
}
|
||||
|
||||
// numParties=2: 2/3*2 + 1 = 2, min is 2
|
||||
p = DefaultThresholdParams(2)
|
||||
if p.Threshold != 2 {
|
||||
t.Errorf("expected threshold 2, got %d", p.Threshold)
|
||||
}
|
||||
|
||||
// numParties=3: 2/3*3 + 1 = 3
|
||||
p = DefaultThresholdParams(3)
|
||||
if p.Threshold != 3 {
|
||||
t.Errorf("expected threshold 3, got %d", p.Threshold)
|
||||
}
|
||||
}
|
||||
|
||||
func TestQuorumParamsValidate(t *testing.T) {
|
||||
p := DefaultQuorumParams()
|
||||
if err := p.Validate(); err != nil {
|
||||
t.Errorf("default quorum should be valid: %v", err)
|
||||
}
|
||||
|
||||
p = QuorumParams{Numerator: 1, Denominator: 0}
|
||||
if err := p.Validate(); err == nil {
|
||||
t.Error("zero denominator should be invalid")
|
||||
}
|
||||
|
||||
p = QuorumParams{Numerator: 4, Denominator: 3}
|
||||
if err := p.Validate(); err == nil {
|
||||
t.Error("num > denom should be invalid")
|
||||
}
|
||||
}
|
||||
|
||||
func TestQuorumParamsIsMet(t *testing.T) {
|
||||
p := QuorumParams{Numerator: 2, Denominator: 3}
|
||||
if !p.IsMet(200, 300) {
|
||||
t.Error("200/300 should meet 2/3 quorum")
|
||||
}
|
||||
if p.IsMet(199, 300) {
|
||||
t.Error("199/300 should not meet 2/3 quorum")
|
||||
}
|
||||
}
|
||||
|
||||
func TestQuorumParamsRequiredWeight(t *testing.T) {
|
||||
p := QuorumParams{Numerator: 2, Denominator: 3}
|
||||
if p.RequiredWeight(300) != 200 {
|
||||
t.Errorf("expected 200, got %d", p.RequiredWeight(300))
|
||||
}
|
||||
}
|
||||
|
||||
func TestConfigBuilderWithFinalityChannelSize(t *testing.T) {
|
||||
cfg, err := NewConfigBuilder().
|
||||
WithThreshold(3).
|
||||
WithFinalityChannelSize(256).
|
||||
Build()
|
||||
if err != nil {
|
||||
t.Fatalf("build failed: %v", err)
|
||||
}
|
||||
if cfg.Runtime.FinalityChannelSize != 256 {
|
||||
t.Errorf("expected 256, got %d", cfg.Runtime.FinalityChannelSize)
|
||||
}
|
||||
}
|
||||
|
||||
func TestThresholdParamsValidateEdge(t *testing.T) {
|
||||
p := ThresholdParams{NumParties: 3, Threshold: 5}
|
||||
if err := p.Validate(); err == nil {
|
||||
t.Error("threshold > parties should be invalid")
|
||||
}
|
||||
|
||||
p = ThresholdParams{NumParties: 3, Threshold: 0}
|
||||
if err := p.Validate(); err == nil {
|
||||
t.Error("threshold 0 should be invalid")
|
||||
}
|
||||
}
|
||||
|
||||
// --- Quasar lifecycle ---
|
||||
|
||||
func TestQuasarGetCoreGetCorona(t *testing.T) {
|
||||
q, err := NewQuasar(log.Noop(), 2, 2, 3)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
if q.GetCore() == nil {
|
||||
t.Error("GetCore should not return nil")
|
||||
}
|
||||
if q.GetCorona() != nil {
|
||||
t.Error("Corona should be nil before initialization")
|
||||
}
|
||||
}
|
||||
|
||||
func TestQuasarSetGetFinalized(t *testing.T) {
|
||||
q, err := NewQuasar(log.Noop(), 2, 2, 3)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
blockID := ids.GenerateTestID()
|
||||
finality := &QuantumFinality{
|
||||
BlockID: blockID,
|
||||
PChainHeight: 42,
|
||||
Timestamp: time.Now(),
|
||||
}
|
||||
|
||||
q.SetFinalized(blockID, finality)
|
||||
|
||||
got, ok := q.GetFinalized(blockID)
|
||||
if !ok {
|
||||
t.Fatal("should find finality record")
|
||||
}
|
||||
if got.PChainHeight != 42 {
|
||||
t.Errorf("height mismatch: %d", got.PChainHeight)
|
||||
}
|
||||
|
||||
_, ok = q.GetFinalized(ids.GenerateTestID())
|
||||
if ok {
|
||||
t.Error("should not find non-existent finality")
|
||||
}
|
||||
}
|
||||
|
||||
func TestQuasarGetConfig(t *testing.T) {
|
||||
q, err := NewQuasar(log.Noop(), 3, 2, 3)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
threshold, qNum, qDen := q.GetConfig()
|
||||
if threshold != 3 {
|
||||
t.Errorf("expected threshold 3, got %d", threshold)
|
||||
}
|
||||
if qNum != 2 || qDen != 3 {
|
||||
t.Errorf("expected quorum 2/3, got %d/%d", qNum, qDen)
|
||||
}
|
||||
}
|
||||
|
||||
func TestQuasarIsRunning(t *testing.T) {
|
||||
q, err := NewQuasar(log.Noop(), 2, 2, 3)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if q.IsRunning() {
|
||||
t.Error("should not be running before Start")
|
||||
}
|
||||
}
|
||||
|
||||
func TestQuasarCheckQuorum(t *testing.T) {
|
||||
q, err := NewQuasar(log.Noop(), 2, 2, 3)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
if !q.CheckQuorum(200, 300) {
|
||||
t.Error("200/300 should meet 2/3 quorum")
|
||||
}
|
||||
if q.CheckQuorum(100, 300) {
|
||||
t.Error("100/300 should not meet 2/3 quorum")
|
||||
}
|
||||
}
|
||||
|
||||
func TestQuasarCreateMessage(t *testing.T) {
|
||||
q, err := NewQuasar(log.Noop(), 2, 2, 3)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
event := FinalityEvent{
|
||||
BlockID: ids.GenerateTestID(),
|
||||
Height: 100,
|
||||
}
|
||||
|
||||
msg := q.CreateMessage(event)
|
||||
if len(msg) == 0 {
|
||||
t.Error("message should not be empty")
|
||||
}
|
||||
}
|
||||
|
||||
func TestQuasarTotalWeight(t *testing.T) {
|
||||
q, err := NewQuasar(log.Noop(), 2, 2, 3)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
validators := []ValidatorState{
|
||||
{Weight: 100, Active: true},
|
||||
{Weight: 200, Active: true},
|
||||
{Weight: 50, Active: true},
|
||||
}
|
||||
|
||||
total := q.TotalWeight(validators)
|
||||
if total != 350 {
|
||||
t.Errorf("expected 350, got %d", total)
|
||||
}
|
||||
|
||||
// Inactive validators should not count
|
||||
validators[2].Active = false
|
||||
total = q.TotalWeight(validators)
|
||||
if total != 300 {
|
||||
t.Errorf("expected 300 without inactive, got %d", total)
|
||||
}
|
||||
}
|
||||
|
||||
// --- BLS Signature ---
|
||||
|
||||
func TestBLSSignature(t *testing.T) {
|
||||
signers := []ids.NodeID{ids.GenerateTestNodeID(), ids.GenerateTestNodeID()}
|
||||
sig := NewBLSSignature([]byte("aggregated-sig"), signers)
|
||||
|
||||
if sig.Type() != SignatureTypeBLS {
|
||||
t.Error("wrong type")
|
||||
}
|
||||
if len(sig.Bytes()) == 0 {
|
||||
t.Error("bytes should not be empty")
|
||||
}
|
||||
if len(sig.Signers()) != 2 {
|
||||
t.Errorf("expected 2 signers, got %d", len(sig.Signers()))
|
||||
}
|
||||
}
|
||||
|
||||
// --- CoronaCoordinator ---
|
||||
|
||||
func TestCoronaCoordinatorSignNotInitialized(t *testing.T) {
|
||||
rc, _ := NewCoronaCoordinator(log.Noop(), CoronaConfig{NumParties: 3, Threshold: 2})
|
||||
_, err := rc.Sign([]byte("msg"))
|
||||
if err == nil {
|
||||
t.Error("should fail when not initialized")
|
||||
}
|
||||
}
|
||||
|
||||
func TestCoronaCoordinatorVerifyNotInitialized(t *testing.T) {
|
||||
rc, _ := NewCoronaCoordinator(log.Noop(), CoronaConfig{NumParties: 3, Threshold: 2})
|
||||
if rc.Verify([]byte("msg"), nil) {
|
||||
t.Error("should return false when not initialized")
|
||||
}
|
||||
}
|
||||
|
||||
func TestCoronaCoordinatorTestMode(t *testing.T) {
|
||||
rc, _ := NewTestCoronaCoordinator(log.Noop(), CoronaConfig{NumParties: 3, Threshold: 2})
|
||||
validators := []ids.NodeID{ids.GenerateTestNodeID()}
|
||||
rc.Initialize(validators)
|
||||
|
||||
sig, err := rc.Sign([]byte("test-message"))
|
||||
if err != nil {
|
||||
t.Fatalf("sign failed: %v", err)
|
||||
}
|
||||
if sig == nil {
|
||||
t.Fatal("signature should not be nil")
|
||||
}
|
||||
if !rc.Verify([]byte("test-message"), sig) {
|
||||
t.Error("should verify in testing mode")
|
||||
}
|
||||
}
|
||||
|
||||
func TestCoronaCoordinatorNotTestMode(t *testing.T) {
|
||||
rc, _ := NewCoronaCoordinator(log.Noop(), CoronaConfig{NumParties: 3, Threshold: 2})
|
||||
rc.Initialize([]ids.NodeID{ids.GenerateTestNodeID()})
|
||||
|
||||
_, err := rc.Sign([]byte("msg"))
|
||||
if err == nil {
|
||||
t.Error("should fail when not in testing mode")
|
||||
}
|
||||
|
||||
sig := NewCoronaSignature([]byte("fake"), nil)
|
||||
if rc.Verify([]byte("msg"), sig) {
|
||||
t.Error("should return false when not in testing mode")
|
||||
}
|
||||
}
|
||||
|
||||
func TestCoronaCoordinatorStats(t *testing.T) {
|
||||
rc, _ := NewCoronaCoordinator(log.Noop(), CoronaConfig{NumParties: 3, Threshold: 2})
|
||||
s := rc.Stats()
|
||||
if s.NumParties != 3 {
|
||||
t.Errorf("expected 3 parties, got %d", s.NumParties)
|
||||
}
|
||||
if s.Initialized {
|
||||
t.Error("should not be initialized")
|
||||
}
|
||||
}
|
||||
|
||||
func TestCoronaCoordinatorThresholdNumParties(t *testing.T) {
|
||||
rc, _ := NewCoronaCoordinator(log.Noop(), CoronaConfig{NumParties: 5, Threshold: 3})
|
||||
if rc.Threshold() != 3 {
|
||||
t.Errorf("expected threshold 3, got %d", rc.Threshold())
|
||||
}
|
||||
if rc.NumParties() != 5 {
|
||||
t.Errorf("expected 5 parties, got %d", rc.NumParties())
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,59 @@
|
||||
// Copyright (C) 2019-2025, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
/*
|
||||
Package quasar provides hybrid quantum-safe consensus finality.
|
||||
|
||||
# Overview
|
||||
|
||||
Quasar is the gravitational center of Lux consensus, binding P-Chain
|
||||
(BLS signatures) and Q-Chain (Corona post-quantum threshold) into
|
||||
unified hybrid finality across all Lux networks.
|
||||
|
||||
# Architecture
|
||||
|
||||
All validators maintain both keypairs:
|
||||
- BLS keypair: Aggregate signatures (classical, fast)
|
||||
- Corona keypair: Threshold signatures (post-quantum, 2-round)
|
||||
|
||||
Both signature paths run in parallel:
|
||||
|
||||
Block arrives
|
||||
|
|
||||
+-- BLS PATH ----------+-- CORONA PATH --------+
|
||||
| All validators | Round 1: commitments |
|
||||
| sign with BLS | Round 2: partials |
|
||||
| Aggregate (96B) | Combine threshold sig |
|
||||
+----------------------+-------------------------+
|
||||
|
|
||||
HYBRID PROOF
|
||||
BLS + Corona combined
|
||||
|
|
||||
QUANTUM FINALITY
|
||||
|
||||
# Vote Flow
|
||||
|
||||
Validators cast votes (wire format: Chits) for proposed blocks. The
|
||||
Quasar engine collects these votes and produces finality proofs when:
|
||||
- 2/3+ validator weight signed via BLS
|
||||
- t-of-n validators completed Corona threshold signing
|
||||
|
||||
# Signature Types
|
||||
|
||||
The package defines several signature types:
|
||||
- SignatureTypeBLS: Classical BLS signatures
|
||||
- SignatureTypeCorona: Post-quantum threshold
|
||||
- SignatureTypeQuasar: Hybrid combining both
|
||||
- SignatureTypeMLDSA: ML-DSA fallback
|
||||
|
||||
# Components
|
||||
|
||||
Quasar: Main consensus hub coordinating both signature paths.
|
||||
|
||||
CoronaCoordinator: Manages the 2-round threshold signing protocol
|
||||
for post-quantum security.
|
||||
|
||||
QuantumFinality: Represents a block that achieved hybrid finality with
|
||||
both BLS and Corona proofs.
|
||||
*/
|
||||
package quasar
|
||||
@@ -1,38 +0,0 @@
|
||||
// Copyright (C) 2019-2026, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
package quasar
|
||||
|
||||
import "errors"
|
||||
|
||||
// Typed, fail-closed errors. Every one is returned (never swallowed) and, when
|
||||
// surfaced from the accept hook post-activation, halts finalization rather than
|
||||
// accepting a checkpoint without valid post-quantum evidence.
|
||||
var (
|
||||
// ErrFinalityCertMissing — a checkpoint was finalized post-activation but no
|
||||
// QuasarCert is available for it (the producer has not delivered one).
|
||||
ErrFinalityCertMissing = errors.New("quasar: finality cert missing for checkpoint")
|
||||
|
||||
// ErrFinalityCertMismatch — a cert exists but does not bind the finalized
|
||||
// block (chain/height/block/state mismatch). Anti-replay.
|
||||
ErrFinalityCertMismatch = errors.New("quasar: finality cert does not bind the finalized block")
|
||||
|
||||
// ErrFinalityCertInvalid — the cert is bound correctly but failed consensus
|
||||
// verification (policy, validator-set root, or a leg signature).
|
||||
ErrFinalityCertInvalid = errors.New("quasar: finality cert failed verification")
|
||||
|
||||
// ErrValidatorSetUnavailable — no committed validator set for the cert's
|
||||
// epoch (the verifier cannot resolve the per-leg verification keys).
|
||||
ErrValidatorSetUnavailable = errors.New("quasar: validator set unavailable for epoch")
|
||||
|
||||
// ErrPolicyUnavailable — the gate has no configured policy.
|
||||
ErrPolicyUnavailable = errors.New("quasar: policy unavailable")
|
||||
|
||||
// ErrPolicyMismatch — the cert's PolicyID is not the configured policy. A
|
||||
// cert cannot select its own (weaker) posture.
|
||||
ErrPolicyMismatch = errors.New("quasar: cert policy id does not match configured policy")
|
||||
|
||||
// ErrGateMisconfigured — the gate is activated at a checkpoint but has no
|
||||
// cert store or validator provider. Fail closed rather than panic.
|
||||
ErrGateMisconfigured = errors.New("quasar: gate activated but missing store or validator provider")
|
||||
)
|
||||
@@ -1,268 +0,0 @@
|
||||
// Copyright (C) 2019-2026, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
// Package quasar is the node-side integration of the luxfi/consensus Quasar
|
||||
// post-quantum finality-certificate layer.
|
||||
//
|
||||
// luxd finalizes blocks on the classical Snow/Avalanche path (fast, every
|
||||
// block). On top, at CHECKPOINTS (epoch boundaries — NOT every block), a sampled
|
||||
// committee produces a QuasarCert over the finalized digest and validators
|
||||
// VERIFY it. This package wires the VERIFY half: it consumes
|
||||
// github.com/luxfi/consensus/protocol/quasar.VerifyConsensusCert as an OPTIONAL,
|
||||
// FORWARD-DATED, DORMANT-BY-DEFAULT check in the block-accept path.
|
||||
//
|
||||
// # Safety contract
|
||||
//
|
||||
// The forward-dated dormant activation is the whole reason this package has the
|
||||
// shape it does:
|
||||
//
|
||||
// - Pre-activation (the default): VerifyAccepted is a pure no-op. Classical
|
||||
// Snow finality is UNCHANGED. A nil *Gate, a zero Gate, or an unset
|
||||
// activation height all mean "dormant" — zero behavior change.
|
||||
// - Post-activation (owner sets Activation.Height to a real, forward-dated
|
||||
// height): at every checkpoint height the gate REQUIRES a valid QuasarCert
|
||||
// bound to the just-finalized block and FAILS CLOSED — a missing or invalid
|
||||
// cert returns an error from Accept(), halting the chain rather than
|
||||
// finalizing a checkpoint without post-quantum evidence.
|
||||
//
|
||||
// Activation is therefore a deliberate switch the owner flips only AFTER the
|
||||
// cert PRODUCER (the per-validator committee signer) is live and certs flow at
|
||||
// the checkpoint cadence — otherwise every checkpoint would halt. See
|
||||
// producer.go.
|
||||
//
|
||||
// # Default posture
|
||||
//
|
||||
// HYBRID_PQ = Beam(BLS) ∧ Pulsar (standard FIPS-204 threshold ML-DSA), at
|
||||
// checkpoint cadence. Per the measured policy-tier benchmarks, Pulsar verify
|
||||
// (~140µs) is cheaper than BLS itself and the cert is compact (~27KB) — the
|
||||
// right default production finality posture. STRICT_DUAL_PQ (∧ Corona) and the
|
||||
// POLARIS tiers (∧ Magnetar) are configurable for stricter mainnet finality.
|
||||
package quasar
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
|
||||
qcert "github.com/luxfi/consensus/protocol/quasar"
|
||||
)
|
||||
|
||||
// ActivationConfig is the forward-dated activation switch.
|
||||
//
|
||||
// The zero value is DORMANT: Height == 0 means "never activate" and the gate is
|
||||
// a no-op for every block. This mirrors the genesis upgrade discipline (a
|
||||
// far-future / unset activation point cannot affect live finality).
|
||||
//
|
||||
// Activation is by HEIGHT ONLY, deliberately. A block height is agreed by
|
||||
// consensus, so every honest validator enforces PQ finality at exactly the SAME
|
||||
// checkpoints — there is no node-local decision. (A wall-clock gate would split
|
||||
// finalization across validators with skewed clocks: some halting on a missing
|
||||
// cert while others finalize without one. Timestamp-based forward-dating is
|
||||
// expressed by choosing the activation HEIGHT at the target time.)
|
||||
type ActivationConfig struct {
|
||||
// Height is the block height at and above which PQ-finality verification is
|
||||
// enforced at checkpoints. 0 == dormant (never).
|
||||
Height uint64
|
||||
}
|
||||
|
||||
// dormant reports whether the activation is unset (the default — never enforce).
|
||||
func (a ActivationConfig) dormant() bool { return a.Height == 0 }
|
||||
|
||||
// active reports whether enforcement is live for a block at the given height.
|
||||
// Deterministic: height-only, no wall clock. Dormant activation is never active.
|
||||
func (a ActivationConfig) active(height uint64) bool {
|
||||
return a.Height != 0 && height >= a.Height
|
||||
}
|
||||
|
||||
// DefaultCheckpointInterval is the default checkpoint cadence in blocks. PQ
|
||||
// certs ride epoch-boundary checkpoints, never every block (Magnetar sign is
|
||||
// checkpoint-only; even the cheap Pulsar sign is checkpoint cadence). The owner
|
||||
// overrides this to match the producer's cadence at activation time.
|
||||
const DefaultCheckpointInterval uint64 = 256
|
||||
|
||||
// DefaultMode is the default Quasar posture: HYBRID_PQ (Beam ∧ Pulsar).
|
||||
const DefaultMode = qcert.PolicyHybridPQCheckpoint
|
||||
|
||||
// Config is the node-surfaced PQ-finality configuration. Its zero value is
|
||||
// dormant + HYBRID_PQ + default cadence.
|
||||
type Config struct {
|
||||
// ChainID is THIS chain's numeric identifier (the sovereign/EVM chain id),
|
||||
// bound into every cert and checked against it. A per-chain constant sourced
|
||||
// from chain config at gate construction — NOT pulled from a block, because
|
||||
// the proposervm layer carries the 32-byte validator-set id, not the numeric
|
||||
// chain id. Inert while dormant.
|
||||
ChainID uint32
|
||||
|
||||
// Activation is the forward-dated dormant switch. Zero => dormant.
|
||||
Activation ActivationConfig
|
||||
|
||||
// Mode is the Quasar evidence posture. Zero => DefaultMode (HYBRID_PQ).
|
||||
Mode qcert.QuasarEvidenceMode
|
||||
|
||||
// MLDSAParam selects the ML-DSA parameter set for the Pulsar leg. 0 =>
|
||||
// ML-DSA-65 (the consensus default).
|
||||
MLDSAParam uint8
|
||||
|
||||
// Threshold is the BFT quorum floor (minimum aggregate signer weight) every
|
||||
// leg's evidence must establish.
|
||||
Threshold uint64
|
||||
|
||||
// CheckpointInterval is the checkpoint cadence in blocks. 0 =>
|
||||
// DefaultCheckpointInterval.
|
||||
CheckpointInterval uint64
|
||||
}
|
||||
|
||||
// Checkpoint is the finalized-block position the accept hook hands the gate. It
|
||||
// is the binding the cert must match (anti-replay): a valid cert for a DIFFERENT
|
||||
// block must never satisfy THIS checkpoint. The chain id is gate-level config,
|
||||
// not a per-block field.
|
||||
type Checkpoint struct {
|
||||
Epoch uint64
|
||||
Height uint64
|
||||
Round uint32
|
||||
BlockID [32]byte
|
||||
StateRoot [32]byte
|
||||
}
|
||||
|
||||
// Gate enforces (or, dormant, ignores) PQ-finality at checkpoints. It is the
|
||||
// single node-side seam between the classical accept path and the consensus
|
||||
// Quasar verifier.
|
||||
type Gate struct {
|
||||
cfg Config
|
||||
policy *qcert.QuasarEvidencePolicy
|
||||
store CertStore
|
||||
validators ValidatorSetProvider
|
||||
}
|
||||
|
||||
// NewGate constructs a Gate. A Gate is meaningful even with a dormant Config:
|
||||
// VerifyAccepted is a no-op until Activation.Height is set. store and validators
|
||||
// are only consulted post-activation at checkpoints.
|
||||
func NewGate(cfg Config, store CertStore, validators ValidatorSetProvider) *Gate {
|
||||
mode := cfg.Mode
|
||||
if mode == 0 {
|
||||
mode = DefaultMode
|
||||
}
|
||||
if cfg.CheckpointInterval == 0 {
|
||||
cfg.CheckpointInterval = DefaultCheckpointInterval
|
||||
}
|
||||
cfg.Mode = mode
|
||||
return &Gate{
|
||||
cfg: cfg,
|
||||
policy: qcert.NewQuasarEvidencePolicy(mode, cfg.MLDSAParam, cfg.Threshold),
|
||||
store: store,
|
||||
validators: validators,
|
||||
}
|
||||
}
|
||||
|
||||
// VerifyAccepted is the accept-path hook and the SAFETY BOUNDARY.
|
||||
//
|
||||
// - g == nil OR dormant activation => returns nil immediately. This is the
|
||||
// default and guarantees classical Snow finality is unchanged.
|
||||
// - height below activation, or activation time not yet reached => nil.
|
||||
// - not a checkpoint height => nil (certs ride checkpoints only).
|
||||
// - checkpoint, activated => REQUIRE a valid cert bound to this block; FAIL
|
||||
// CLOSED. A missing, mis-bound, or invalid cert is an error (the caller
|
||||
// returns it from Accept, halting rather than finalizing without PQ
|
||||
// evidence).
|
||||
//
|
||||
// It is intentionally nil-safe so the proposervm hook can call
|
||||
// vm.quasarGate.VerifyAccepted(...) unconditionally with a nil gate.
|
||||
func (g *Gate) VerifyAccepted(cp Checkpoint) error {
|
||||
if g == nil || g.cfg.Activation.dormant() {
|
||||
return nil
|
||||
}
|
||||
if !g.cfg.Activation.active(cp.Height) {
|
||||
return nil
|
||||
}
|
||||
if !g.isCheckpoint(cp.Height) {
|
||||
return nil
|
||||
}
|
||||
// Activated checkpoint: the gate MUST have its cert store + validator
|
||||
// provider, or it cannot verify. Fail closed with a typed error rather than
|
||||
// panic in the accept hook (a panic would halt the chain uncontrollably).
|
||||
if g.store == nil || g.validators == nil {
|
||||
return fmt.Errorf("%w: chain=%d height=%d", ErrGateMisconfigured, g.cfg.ChainID, cp.Height)
|
||||
}
|
||||
|
||||
cert, ok := g.store.Lookup(g.cfg.ChainID, cp.Height, cp.BlockID)
|
||||
if !ok || cert == nil {
|
||||
return fmt.Errorf("%w: chain=%d height=%d block=%x", ErrFinalityCertMissing, g.cfg.ChainID, cp.Height, cp.BlockID[:8])
|
||||
}
|
||||
if err := bindCheck(cert, g.cfg.ChainID, cp); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
vs, err := g.validators.ValidatorSet(g.cfg.ChainID, cert.Epoch)
|
||||
if err != nil {
|
||||
return fmt.Errorf("%w: chain=%d epoch=%d: %v", ErrValidatorSetUnavailable, g.cfg.ChainID, cert.Epoch, err)
|
||||
}
|
||||
if err := qcert.VerifyConsensusCert(policyStore{policy: g.policy}, vs, cert); err != nil {
|
||||
return fmt.Errorf("%w: chain=%d height=%d: %v", ErrFinalityCertInvalid, g.cfg.ChainID, cp.Height, err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Activated reports whether enforcement is live for a block at the given height
|
||||
// and the current wall clock. Used by the producer-request site to decide
|
||||
// whether a cert is needed at a checkpoint.
|
||||
func (g *Gate) Activated(height uint64) bool {
|
||||
if g == nil {
|
||||
return false
|
||||
}
|
||||
return g.cfg.Activation.active(height)
|
||||
}
|
||||
|
||||
// IsCheckpoint reports whether the given height is a checkpoint under the gate's
|
||||
// configured cadence. Exported so the producer-request site shares ONE cadence
|
||||
// definition with the verify path (no second source of truth).
|
||||
func (g *Gate) IsCheckpoint(height uint64) bool {
|
||||
if g == nil {
|
||||
return false
|
||||
}
|
||||
return g.isCheckpoint(height)
|
||||
}
|
||||
|
||||
func (g *Gate) isCheckpoint(height uint64) bool {
|
||||
iv := g.cfg.CheckpointInterval
|
||||
if iv == 0 {
|
||||
iv = DefaultCheckpointInterval
|
||||
}
|
||||
return height%iv == 0
|
||||
}
|
||||
|
||||
// bindCheck pins the cert to the actual finalized block. Without this, a valid
|
||||
// cert produced for a different (chain, height, block) could be replayed to
|
||||
// satisfy this checkpoint. VerifyConsensusCert checks the cert's INTERNAL
|
||||
// consistency and the validator-set/policy binding; bindCheck adds the external
|
||||
// binding to THIS node's finalized position.
|
||||
func bindCheck(cert *qcert.ConsensusCert, chainID uint32, cp Checkpoint) error {
|
||||
if cert.ChainID != chainID {
|
||||
return fmt.Errorf("%w: cert chain %d != finalized chain %d", ErrFinalityCertMismatch, cert.ChainID, chainID)
|
||||
}
|
||||
// Bind the epoch. The gate resolves the verification keys from the cert's
|
||||
// epoch, so an UNBOUND epoch would let a cert signed under a DIFFERENT
|
||||
// validator-set era (e.g. a compromised RETIRED committee's group key) certify
|
||||
// the current block — nullifying KeyEra rotation as a blast-radius bound. The
|
||||
// honest producer signs over Subject.Epoch == cp.Epoch, so honest certs match.
|
||||
if cert.Epoch != cp.Epoch {
|
||||
return fmt.Errorf("%w: cert epoch %d != finalized epoch %d", ErrFinalityCertMismatch, cert.Epoch, cp.Epoch)
|
||||
}
|
||||
if cert.Round != cp.Round {
|
||||
return fmt.Errorf("%w: cert round %d != finalized round %d", ErrFinalityCertMismatch, cert.Round, cp.Round)
|
||||
}
|
||||
if cert.Height != cp.Height {
|
||||
return fmt.Errorf("%w: cert height %d != finalized height %d", ErrFinalityCertMismatch, cert.Height, cp.Height)
|
||||
}
|
||||
if cert.BlockHash != cp.BlockID {
|
||||
return fmt.Errorf("%w: cert block hash != finalized block id", ErrFinalityCertMismatch)
|
||||
}
|
||||
// StateRoot contract: at the proposervm layer the post-state root is
|
||||
// committed TRANSITIVELY through BlockHash, so cp.StateRoot is zero and the
|
||||
// cert MUST carry a zero StateRoot too. A non-zero cert StateRoot is rejected
|
||||
// (no state to cross-check here) — the producer follow-on MUST emit
|
||||
// StateRoot==0 at this layer; a chain that wants an explicit state binding
|
||||
// plumbs cp.StateRoot AND signs it, and this check then enforces equality.
|
||||
var zero [32]byte
|
||||
if cert.StateRoot != zero && cert.StateRoot != cp.StateRoot {
|
||||
return fmt.Errorf("%w: cert state root != finalized state root", ErrFinalityCertMismatch)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -1,270 +0,0 @@
|
||||
// Copyright (C) 2019-2026, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
package quasar
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"testing"
|
||||
|
||||
qcert "github.com/luxfi/consensus/protocol/quasar"
|
||||
)
|
||||
|
||||
// vsRoot is a fixed committed-validator-set root used across the tests.
|
||||
var vsRoot = [48]byte{0x11, 0x22, 0x33, 0x44}
|
||||
|
||||
// testValidators is a ValidatorSet whose Root matches vsRoot, with non-empty
|
||||
// (placeholder) HYBRID_PQ keys. The delegation test never reaches signature
|
||||
// math, so the key bytes need only be non-empty.
|
||||
func testValidators() *ValidatorSet {
|
||||
return NewValidatorSet(vsRoot, 1, []byte("bls-agg-key"), []byte("pulsar-group-key"))
|
||||
}
|
||||
|
||||
// newGate builds a gate with a tight cadence (checkpoint every 10 blocks) and
|
||||
// the given forward-dated activation height. interval 10 keeps the heights in
|
||||
// the tests readable.
|
||||
func newGate(activationHeight uint64, store CertStore) *Gate {
|
||||
return NewGate(Config{
|
||||
ChainID: 1337,
|
||||
Activation: ActivationConfig{Height: activationHeight},
|
||||
Mode: DefaultMode, // HYBRID_PQ
|
||||
Threshold: 100,
|
||||
CheckpointInterval: 10,
|
||||
}, store, StaticValidatorSetProvider{Set: testValidators()})
|
||||
}
|
||||
|
||||
func checkpointAt(height uint64) Checkpoint {
|
||||
return Checkpoint{
|
||||
Epoch: 1,
|
||||
Height: height,
|
||||
BlockID: [32]byte{0xab, 0xcd, 0xef},
|
||||
}
|
||||
}
|
||||
|
||||
// TestDormantIsNoop — the default (Activation.Height == 0) is a pure no-op even
|
||||
// at a checkpoint height with a poisoned store. This is the core safety
|
||||
// property: pre-activation, classical Snow finality is unchanged.
|
||||
func TestDormantIsNoop(t *testing.T) {
|
||||
store := NewMemCertStore()
|
||||
g := NewGate(Config{CheckpointInterval: 10}, store, StaticValidatorSetProvider{Set: testValidators()})
|
||||
// height 10 is a checkpoint; no cert exists; yet dormant => nil.
|
||||
if err := g.VerifyAccepted(checkpointAt(10)); err != nil {
|
||||
t.Fatalf("dormant gate must be a no-op, got %v", err)
|
||||
}
|
||||
if g.Activated(10) {
|
||||
t.Fatal("dormant gate must never report Activated")
|
||||
}
|
||||
}
|
||||
|
||||
// TestNilGateIsNoop — a nil *Gate is the wire-it-but-leave-it-off default; the
|
||||
// proposervm hook calls VerifyAccepted on a possibly-nil gate.
|
||||
func TestNilGateIsNoop(t *testing.T) {
|
||||
var g *Gate
|
||||
if err := g.VerifyAccepted(checkpointAt(10)); err != nil {
|
||||
t.Fatalf("nil gate must be a no-op, got %v", err)
|
||||
}
|
||||
if g.Activated(10) || g.IsCheckpoint(10) {
|
||||
t.Fatal("nil gate must report neither Activated nor IsCheckpoint")
|
||||
}
|
||||
}
|
||||
|
||||
// TestBelowActivationIsNoop — activated at height 100 but the block is at 10:
|
||||
// below the forward-dated height => nil, even at a checkpoint with no cert.
|
||||
func TestBelowActivationIsNoop(t *testing.T) {
|
||||
g := newGate(100, NewMemCertStore())
|
||||
if err := g.VerifyAccepted(checkpointAt(10)); err != nil {
|
||||
t.Fatalf("below activation must be a no-op, got %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestNonCheckpointIsNoop — activated and at/above activation height, but the
|
||||
// height is not a checkpoint => nil (certs ride checkpoints only).
|
||||
func TestNonCheckpointIsNoop(t *testing.T) {
|
||||
g := newGate(10, NewMemCertStore())
|
||||
// height 15 is activated (>=10) but not a checkpoint (15 % 10 != 0).
|
||||
if err := g.VerifyAccepted(checkpointAt(15)); err != nil {
|
||||
t.Fatalf("non-checkpoint must be a no-op, got %v", err)
|
||||
}
|
||||
if !g.Activated(15) {
|
||||
t.Fatal("height 15 should be activated")
|
||||
}
|
||||
if g.IsCheckpoint(15) {
|
||||
t.Fatal("height 15 must not be a checkpoint")
|
||||
}
|
||||
}
|
||||
|
||||
// TestMissingCertFailsClosed — activated checkpoint with no cert in the store =>
|
||||
// ErrFinalityCertMissing. Post-activation a checkpoint without PQ evidence must
|
||||
// NOT finalize.
|
||||
func TestMissingCertFailsClosed(t *testing.T) {
|
||||
g := newGate(10, NewMemCertStore())
|
||||
err := g.VerifyAccepted(checkpointAt(20))
|
||||
if !errors.Is(err, ErrFinalityCertMissing) {
|
||||
t.Fatalf("want ErrFinalityCertMissing, got %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestMismatchedCertRejected — a cert that does not bind the finalized block
|
||||
// (wrong block id / height / chain) is rejected by bindCheck before any crypto.
|
||||
// Anti-replay: a valid cert for a different block must not satisfy this one.
|
||||
func TestMismatchedCertRejected(t *testing.T) {
|
||||
// cp = checkpointAt(20) has Epoch 1, Round 0. Each case sets every binding
|
||||
// field correctly EXCEPT the one named, so it fails at that field.
|
||||
cases := []struct {
|
||||
name string
|
||||
cert *qcert.ConsensusCert
|
||||
}{
|
||||
{"wrong block", &qcert.ConsensusCert{ChainID: 1337, Epoch: 1, Height: 20, BlockHash: [32]byte{0x99}}},
|
||||
{"wrong height", &qcert.ConsensusCert{ChainID: 1337, Epoch: 1, Height: 21, BlockHash: [32]byte{0xab, 0xcd, 0xef}}},
|
||||
{"wrong chain", &qcert.ConsensusCert{ChainID: 7, Epoch: 1, Height: 20, BlockHash: [32]byte{0xab, 0xcd, 0xef}}},
|
||||
{"wrong epoch", &qcert.ConsensusCert{ChainID: 1337, Epoch: 2, Height: 20, BlockHash: [32]byte{0xab, 0xcd, 0xef}}},
|
||||
{"wrong round", &qcert.ConsensusCert{ChainID: 1337, Epoch: 1, Round: 1, Height: 20, BlockHash: [32]byte{0xab, 0xcd, 0xef}}},
|
||||
{"wrong state root", &qcert.ConsensusCert{ChainID: 1337, Epoch: 1, Height: 20, BlockHash: [32]byte{0xab, 0xcd, 0xef}, StateRoot: [32]byte{0x55}}},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
store := NewMemCertStore()
|
||||
// Index it at the checkpoint's lookup key so Lookup returns it and
|
||||
// bindCheck (not Lookup) does the rejecting.
|
||||
store.certs[certKey{chainID: 1337, height: 20, blockID: [32]byte{0xab, 0xcd, 0xef}}] = tc.cert
|
||||
g := newGate(10, store)
|
||||
err := g.VerifyAccepted(checkpointAt(20))
|
||||
if !errors.Is(err, ErrFinalityCertMismatch) {
|
||||
t.Fatalf("want ErrFinalityCertMismatch, got %v", err)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestDelegatesToVerifier — a cert that BINDS correctly and passes the full
|
||||
// ConsensusCert header path (version, policy load, required-legs root, validator
|
||||
// -set root) but carries no signature evidence is rejected by the REAL
|
||||
// consensus verifier, and the gate surfaces it as ErrFinalityCertInvalid. This
|
||||
// proves the whole delegation chain is wired: policyStore + ValidatorSet +
|
||||
// quasar.VerifyConsensusCert are reached with matching commitments — everything
|
||||
// up to (but not including) the leg signature crypto, which needs the producer.
|
||||
func TestDelegatesToVerifier(t *testing.T) {
|
||||
cp := checkpointAt(20)
|
||||
|
||||
// Mirror the gate's posture to compute the header commitments the verifier
|
||||
// pins (policy id + required-legs root). policyID and required legs derive
|
||||
// from the mode + ML-DSA param, which match the gate's config.
|
||||
pol := qcert.NewQuasarEvidencePolicy(DefaultMode, 0, 100)
|
||||
cert := &qcert.ConsensusCert{
|
||||
Version: 1,
|
||||
ChainID: 1337, // must equal the gate's configured ChainID
|
||||
Epoch: cp.Epoch,
|
||||
Height: cp.Height,
|
||||
BlockHash: cp.BlockID,
|
||||
PolicyID: pol.EvidencePolicyID(),
|
||||
RequiredLegsRoot: qcert.HashRequiredLegs(pol.RequiredLegs()),
|
||||
ValidatorSetRoot: vsRoot,
|
||||
// Evidence intentionally empty: the verifier must reject a required leg
|
||||
// with no evidence (deepest deterministic failure without real crypto).
|
||||
}
|
||||
store := NewMemCertStore()
|
||||
store.Put(cert)
|
||||
g := newGate(10, store)
|
||||
|
||||
err := g.VerifyAccepted(cp)
|
||||
if !errors.Is(err, ErrFinalityCertInvalid) {
|
||||
t.Fatalf("want ErrFinalityCertInvalid (delegated), got %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestValidatorSetUnavailable — a bound cert at an activated checkpoint, but the
|
||||
// provider has no set for the epoch => ErrValidatorSetUnavailable (fail closed).
|
||||
func TestValidatorSetUnavailable(t *testing.T) {
|
||||
cp := checkpointAt(20)
|
||||
// Bind correctly (epoch included) so the cert passes bindCheck and the
|
||||
// failure is specifically the unavailable validator set.
|
||||
cert := &qcert.ConsensusCert{Version: 1, ChainID: 1337, Epoch: cp.Epoch, Height: cp.Height, BlockHash: cp.BlockID}
|
||||
store := NewMemCertStore()
|
||||
store.Put(cert)
|
||||
g := NewGate(Config{
|
||||
ChainID: 1337,
|
||||
Activation: ActivationConfig{Height: 10},
|
||||
CheckpointInterval: 10,
|
||||
}, store, StaticValidatorSetProvider{Set: nil}) // provider present, no set
|
||||
err := g.VerifyAccepted(cp)
|
||||
if !errors.Is(err, ErrValidatorSetUnavailable) {
|
||||
t.Fatalf("want ErrValidatorSetUnavailable, got %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestGateMisconfiguredFailsClosed — an activated gate at a checkpoint with no
|
||||
// cert store (or no validator provider) fails closed with a typed error rather
|
||||
// than panicking in the accept hook.
|
||||
func TestGateMisconfiguredFailsClosed(t *testing.T) {
|
||||
g := NewGate(Config{
|
||||
ChainID: 1337,
|
||||
Activation: ActivationConfig{Height: 10},
|
||||
CheckpointInterval: 10,
|
||||
}, nil, nil) // no store, no validators
|
||||
if err := g.VerifyAccepted(checkpointAt(20)); !errors.Is(err, ErrGateMisconfigured) {
|
||||
t.Fatalf("want ErrGateMisconfigured, got %v", err)
|
||||
}
|
||||
// dormant misconfigured gate is still a no-op (guard is post-activation).
|
||||
gd := NewGate(Config{ChainID: 1337, CheckpointInterval: 10}, nil, nil)
|
||||
if err := gd.VerifyAccepted(checkpointAt(20)); err != nil {
|
||||
t.Fatalf("dormant gate must be a no-op even if misconfigured, got %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// --- producer scaffolding ---
|
||||
|
||||
type stubProducer struct {
|
||||
cert *qcert.ConsensusCert
|
||||
hits int
|
||||
}
|
||||
|
||||
func (s *stubProducer) Produce(_ context.Context, _ Subject) (*qcert.ConsensusCert, error) {
|
||||
s.hits++
|
||||
return s.cert, nil
|
||||
}
|
||||
|
||||
// TestMaybeProduceVerifyOnlyByDefault — a nil producer is the verify-only
|
||||
// default: MaybeProduce short-circuits to (nil, nil), never panics.
|
||||
func TestMaybeProduceVerifyOnlyByDefault(t *testing.T) {
|
||||
g := newGate(10, NewMemCertStore())
|
||||
cert, err := g.MaybeProduce(context.Background(), nil, checkpointAt(20))
|
||||
if err != nil || cert != nil {
|
||||
t.Fatalf("nil producer must yield (nil,nil), got cert=%v err=%v", cert, err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestMaybeProduceDormant — even with a producer wired, a dormant gate produces
|
||||
// nothing (the producer is brought up before activation is forward-dated).
|
||||
func TestMaybeProduceDormant(t *testing.T) {
|
||||
store := NewMemCertStore()
|
||||
g := NewGate(Config{CheckpointInterval: 10}, store, StaticValidatorSetProvider{Set: testValidators()})
|
||||
p := &stubProducer{cert: &qcert.ConsensusCert{}}
|
||||
cert, err := g.MaybeProduce(context.Background(), p, checkpointAt(20))
|
||||
if err != nil || cert != nil {
|
||||
t.Fatalf("dormant gate must not produce, got cert=%v err=%v", cert, err)
|
||||
}
|
||||
if p.hits != 0 {
|
||||
t.Fatalf("producer must not be called while dormant, hits=%d", p.hits)
|
||||
}
|
||||
}
|
||||
|
||||
// TestMaybeProduceActiveCheckpoint — wired producer + activated checkpoint =>
|
||||
// the producer is asked for the cert.
|
||||
func TestMaybeProduceActiveCheckpoint(t *testing.T) {
|
||||
g := newGate(10, NewMemCertStore())
|
||||
want := &qcert.ConsensusCert{ChainID: 1337, Height: 20}
|
||||
p := &stubProducer{cert: want}
|
||||
got, err := g.MaybeProduce(context.Background(), p, checkpointAt(20))
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected err %v", err)
|
||||
}
|
||||
if got != want || p.hits != 1 {
|
||||
t.Fatalf("producer not invoked as expected: got=%v hits=%d", got, p.hits)
|
||||
}
|
||||
// non-checkpoint height must not invoke the producer
|
||||
p2 := &stubProducer{cert: want}
|
||||
if _, _ = g.MaybeProduce(context.Background(), p2, checkpointAt(15)); p2.hits != 0 {
|
||||
t.Fatalf("producer invoked at non-checkpoint, hits=%d", p2.hits)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,627 @@
|
||||
// Copyright (C) 2019-2025, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
package quasar
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
"github.com/luxfi/accel"
|
||||
"github.com/luxfi/crypto/bls"
|
||||
"github.com/luxfi/crypto/mldsa"
|
||||
)
|
||||
|
||||
// GPUVerifyPipeline fuses multiple cryptographic verification operations into
|
||||
// a single GPU session, sharing GPU memory across all verification types.
|
||||
//
|
||||
// Instead of sequential: BLS verify -> Corona verify -> ZK verify -> ML-DSA verify
|
||||
// GPU pipeline: one session, parallel streams, shared memory allocation
|
||||
//
|
||||
// This is the ML-DSA rollup pattern:
|
||||
// - BLS aggregate signature (classical fast path)
|
||||
// - Corona threshold signature (PQ safe path)
|
||||
// - ZK rollup batch proof (state transition validity)
|
||||
// - N x ML-DSA signatures (per-tx PQ signatures)
|
||||
//
|
||||
// All execute on GPU in parallel using separate compute streams within one session.
|
||||
type GPUVerifyPipeline struct {
|
||||
// Stats (atomic for lock-free reads)
|
||||
gpuVerifies uint64
|
||||
cpuVerifies uint64
|
||||
gpuTimeNs uint64
|
||||
cpuTimeNs uint64
|
||||
}
|
||||
|
||||
// NewGPUVerifyPipeline creates a new fused GPU verification pipeline.
|
||||
func NewGPUVerifyPipeline() *GPUVerifyPipeline {
|
||||
return &GPUVerifyPipeline{}
|
||||
}
|
||||
|
||||
// BLSWork holds a batch of BLS signatures to verify.
|
||||
type BLSWork struct {
|
||||
Messages [][]byte // [N, msg_len]
|
||||
Signatures [][]byte // [N, 96] G2 points
|
||||
PubKeys [][]byte // [N, 48] G1 points
|
||||
}
|
||||
|
||||
// CoronaWork holds a batch of Corona threshold signatures to verify.
|
||||
type CoronaWork struct {
|
||||
Messages [][]byte // [N, msg_len]
|
||||
Signatures [][]byte // [N, sig_len] threshold sigs
|
||||
PubKeys [][]byte // [N, pk_len] ring public keys
|
||||
}
|
||||
|
||||
// ZKWork holds a batch of ZK proofs to verify.
|
||||
type ZKWork struct {
|
||||
Scalars [][]byte // [M, N, scalar_size]
|
||||
Bases [][]byte // [M, N, point_size]
|
||||
}
|
||||
|
||||
// MLDSAWork holds a batch of ML-DSA signatures to verify.
|
||||
type MLDSAWork struct {
|
||||
Messages [][]byte // [N, msg_len]
|
||||
Signatures [][]byte // [N, 3309] FIPS-204 ML-DSA-65 (3293 was the stale round-3 Dilithium3 size)
|
||||
PubKeys [][]byte // [N, 1952] FIPS-204 ML-DSA-65
|
||||
}
|
||||
|
||||
// BlockVerifyWork contains all verification batches for a single block.
|
||||
type BlockVerifyWork struct {
|
||||
BLS *BLSWork
|
||||
Corona *CoronaWork
|
||||
ZK *ZKWork
|
||||
MLDSA *MLDSAWork
|
||||
}
|
||||
|
||||
// BlockVerifyResult contains verification results for all batch types.
|
||||
type BlockVerifyResult struct {
|
||||
BLSValid []bool
|
||||
CoronaValid []bool
|
||||
ZKValid bool
|
||||
MLDSAValid []bool
|
||||
|
||||
GPUUsed bool
|
||||
BLSTime time.Duration
|
||||
CoronaTime time.Duration
|
||||
ZKTime time.Duration
|
||||
MLDSATime time.Duration
|
||||
TotalTime time.Duration
|
||||
}
|
||||
|
||||
var (
|
||||
ErrBLSSizeMismatch = errors.New("BLS batch size mismatch: messages, signatures, and pubkeys must have equal length")
|
||||
ErrCoronaSizeMismatch = errors.New("Corona batch size mismatch: messages, signatures, and pubkeys must have equal length")
|
||||
ErrZKSizeMismatch = errors.New("ZK batch size mismatch: scalars and bases must have equal length")
|
||||
ErrMLDSASizeMismatch = errors.New("ML-DSA batch size mismatch: messages, signatures, and pubkeys must have equal length")
|
||||
)
|
||||
|
||||
// VerifyBlock dispatches all verification work for a block through the GPU pipeline.
|
||||
// Falls back to CPU verification when no GPU is available.
|
||||
func (p *GPUVerifyPipeline) VerifyBlock(work *BlockVerifyWork) (*BlockVerifyResult, error) {
|
||||
if work == nil {
|
||||
return &BlockVerifyResult{}, nil
|
||||
}
|
||||
|
||||
if err := validateWork(work); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
start := time.Now()
|
||||
|
||||
if accel.Available() {
|
||||
result, err := p.verifyGPU(work)
|
||||
if err == nil {
|
||||
result.TotalTime = time.Since(start)
|
||||
result.GPUUsed = true
|
||||
atomic.AddUint64(&p.gpuVerifies, 1)
|
||||
atomic.AddUint64(&p.gpuTimeNs, uint64(result.TotalTime))
|
||||
return result, nil
|
||||
}
|
||||
// GPU failed, fall through to CPU
|
||||
}
|
||||
|
||||
result := p.verifyCPU(work)
|
||||
result.TotalTime = time.Since(start)
|
||||
result.GPUUsed = false
|
||||
atomic.AddUint64(&p.cpuVerifies, 1)
|
||||
atomic.AddUint64(&p.cpuTimeNs, uint64(result.TotalTime))
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// verifyGPU dispatches all 4 verification types through a single GPU session.
|
||||
func (p *GPUVerifyPipeline) verifyGPU(work *BlockVerifyWork) (*BlockVerifyResult, error) {
|
||||
sess, err := accel.NewSession()
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("GPU session: %w", err)
|
||||
}
|
||||
defer sess.Close()
|
||||
|
||||
result := &BlockVerifyResult{}
|
||||
var mu sync.Mutex
|
||||
var wg sync.WaitGroup
|
||||
var firstErr atomic.Value
|
||||
|
||||
// BLS verification stream
|
||||
if work.BLS != nil && len(work.BLS.Messages) > 0 {
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
start := time.Now()
|
||||
valid, err := gpuBLSVerify(sess, work.BLS)
|
||||
elapsed := time.Since(start)
|
||||
if err != nil {
|
||||
firstErr.CompareAndSwap(nil, err)
|
||||
return
|
||||
}
|
||||
mu.Lock()
|
||||
result.BLSValid = valid
|
||||
result.BLSTime = elapsed
|
||||
mu.Unlock()
|
||||
}()
|
||||
}
|
||||
|
||||
// Corona verification stream (uses DilithiumVerifyBatch on lattice ops)
|
||||
if work.Corona != nil && len(work.Corona.Messages) > 0 {
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
start := time.Now()
|
||||
valid, err := gpuCoronaVerify(sess, work.Corona)
|
||||
elapsed := time.Since(start)
|
||||
if err != nil {
|
||||
firstErr.CompareAndSwap(nil, err)
|
||||
return
|
||||
}
|
||||
mu.Lock()
|
||||
result.CoronaValid = valid
|
||||
result.CoronaTime = elapsed
|
||||
mu.Unlock()
|
||||
}()
|
||||
}
|
||||
|
||||
// ZK rollup batch proof verification stream
|
||||
if work.ZK != nil && len(work.ZK.Scalars) > 0 {
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
start := time.Now()
|
||||
valid, err := gpuZKVerify(sess, work.ZK)
|
||||
elapsed := time.Since(start)
|
||||
if err != nil {
|
||||
firstErr.CompareAndSwap(nil, err)
|
||||
return
|
||||
}
|
||||
mu.Lock()
|
||||
result.ZKValid = valid
|
||||
result.ZKTime = elapsed
|
||||
mu.Unlock()
|
||||
}()
|
||||
}
|
||||
|
||||
// ML-DSA per-tx signature verification stream
|
||||
if work.MLDSA != nil && len(work.MLDSA.Messages) > 0 {
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
start := time.Now()
|
||||
valid, err := gpuMLDSAVerify(sess, work.MLDSA)
|
||||
elapsed := time.Since(start)
|
||||
if err != nil {
|
||||
firstErr.CompareAndSwap(nil, err)
|
||||
return
|
||||
}
|
||||
mu.Lock()
|
||||
result.MLDSAValid = valid
|
||||
result.MLDSATime = elapsed
|
||||
mu.Unlock()
|
||||
}()
|
||||
}
|
||||
|
||||
wg.Wait()
|
||||
|
||||
if v := firstErr.Load(); v != nil {
|
||||
return nil, v.(error)
|
||||
}
|
||||
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// gpuBLSVerify dispatches BLS batch verification to the GPU crypto ops.
|
||||
func gpuBLSVerify(sess *accel.Session, work *BLSWork) ([]bool, error) {
|
||||
n := len(work.Messages)
|
||||
|
||||
// Determine uniform sizes for tensor packing
|
||||
msgLen := maxByteLen(work.Messages)
|
||||
sigLen := 96 // BLS G2 point
|
||||
pkLen := 48 // BLS G1 point
|
||||
|
||||
msgFlat := flattenPadded(work.Messages, n, msgLen)
|
||||
sigFlat := flattenPadded(work.Signatures, n, sigLen)
|
||||
pkFlat := flattenPadded(work.PubKeys, n, pkLen)
|
||||
|
||||
msgs, err := accel.NewTensorWithData[uint8](sess, []int{n, msgLen}, msgFlat)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer msgs.Close()
|
||||
|
||||
sigs, err := accel.NewTensorWithData[uint8](sess, []int{n, sigLen}, sigFlat)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer sigs.Close()
|
||||
|
||||
pks, err := accel.NewTensorWithData[uint8](sess, []int{n, pkLen}, pkFlat)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer pks.Close()
|
||||
|
||||
results, err := accel.NewTensor[uint8](sess, []int{n})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer results.Close()
|
||||
|
||||
if err := sess.Crypto().BLSVerifyBatch(msgs.Untyped(), sigs.Untyped(), pks.Untyped(), results.Untyped()); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
raw, err := results.ToSlice()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
valid := make([]bool, n)
|
||||
for i, v := range raw {
|
||||
valid[i] = v == 1
|
||||
}
|
||||
return valid, nil
|
||||
}
|
||||
|
||||
// gpuCoronaVerify dispatches Corona verification via DilithiumVerifyBatch
|
||||
// (Corona threshold signatures are lattice-based, same verification kernel).
|
||||
func gpuCoronaVerify(sess *accel.Session, work *CoronaWork) ([]bool, error) {
|
||||
n := len(work.Messages)
|
||||
|
||||
msgLen := maxByteLen(work.Messages)
|
||||
sigLen := maxByteLen(work.Signatures)
|
||||
pkLen := maxByteLen(work.PubKeys)
|
||||
|
||||
msgFlat := flattenPadded(work.Messages, n, msgLen)
|
||||
sigFlat := flattenPadded(work.Signatures, n, sigLen)
|
||||
pkFlat := flattenPadded(work.PubKeys, n, pkLen)
|
||||
|
||||
msgs, err := accel.NewTensorWithData[uint8](sess, []int{n, msgLen}, msgFlat)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer msgs.Close()
|
||||
|
||||
sigs, err := accel.NewTensorWithData[uint8](sess, []int{n, sigLen}, sigFlat)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer sigs.Close()
|
||||
|
||||
pks, err := accel.NewTensorWithData[uint8](sess, []int{n, pkLen}, pkFlat)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer pks.Close()
|
||||
|
||||
results, err := accel.NewTensor[uint8](sess, []int{n})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer results.Close()
|
||||
|
||||
if err := sess.Lattice().DilithiumVerifyBatch(msgs.Untyped(), sigs.Untyped(), pks.Untyped(), results.Untyped()); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
raw, err := results.ToSlice()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
valid := make([]bool, n)
|
||||
for i, v := range raw {
|
||||
valid[i] = v == 1
|
||||
}
|
||||
return valid, nil
|
||||
}
|
||||
|
||||
// gpuZKVerify dispatches ZK batch proof verification via MSMBatch on ZK ops.
|
||||
func gpuZKVerify(sess *accel.Session, work *ZKWork) (bool, error) {
|
||||
m := len(work.Scalars)
|
||||
|
||||
scalarLen := maxByteLen(work.Scalars)
|
||||
baseLen := maxByteLen(work.Bases)
|
||||
|
||||
scalarFlat := flattenPadded(work.Scalars, m, scalarLen)
|
||||
baseFlat := flattenPadded(work.Bases, m, baseLen)
|
||||
|
||||
scalars, err := accel.NewTensorWithData[uint8](sess, []int{m, scalarLen}, scalarFlat)
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
defer scalars.Close()
|
||||
|
||||
bases, err := accel.NewTensorWithData[uint8](sess, []int{m, baseLen}, baseFlat)
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
defer bases.Close()
|
||||
|
||||
// MSM result: single point per batch entry
|
||||
pointSize := baseLen
|
||||
results, err := accel.NewTensor[uint8](sess, []int{m, pointSize})
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
defer results.Close()
|
||||
|
||||
if err := sess.ZK().MSMBatch(scalars.Untyped(), bases.Untyped(), results.Untyped()); err != nil {
|
||||
return false, err
|
||||
}
|
||||
|
||||
// MSM completed without error means proof verification passed
|
||||
return true, nil
|
||||
}
|
||||
|
||||
// gpuMLDSAVerify dispatches ML-DSA (Dilithium) batch verification to the GPU.
|
||||
func gpuMLDSAVerify(sess *accel.Session, work *MLDSAWork) ([]bool, error) {
|
||||
n := len(work.Messages)
|
||||
|
||||
msgLen := maxByteLen(work.Messages)
|
||||
sigLen := 3309 // FIPS-204 ML-DSA-65 signature (3293 was the stale round-3 Dilithium3 size)
|
||||
pkLen := 1952 // FIPS-204 ML-DSA-65 public key (unchanged across the round-3 -> final transition)
|
||||
|
||||
msgFlat := flattenPadded(work.Messages, n, msgLen)
|
||||
sigFlat := flattenPadded(work.Signatures, n, sigLen)
|
||||
pkFlat := flattenPadded(work.PubKeys, n, pkLen)
|
||||
|
||||
msgs, err := accel.NewTensorWithData[uint8](sess, []int{n, msgLen}, msgFlat)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer msgs.Close()
|
||||
|
||||
sigs, err := accel.NewTensorWithData[uint8](sess, []int{n, sigLen}, sigFlat)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer sigs.Close()
|
||||
|
||||
pks, err := accel.NewTensorWithData[uint8](sess, []int{n, pkLen}, pkFlat)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer pks.Close()
|
||||
|
||||
results, err := accel.NewTensor[uint8](sess, []int{n})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer results.Close()
|
||||
|
||||
if err := sess.Lattice().DilithiumVerifyBatch(msgs.Untyped(), sigs.Untyped(), pks.Untyped(), results.Untyped()); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
raw, err := results.ToSlice()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
valid := make([]bool, n)
|
||||
for i, v := range raw {
|
||||
valid[i] = v == 1
|
||||
}
|
||||
return valid, nil
|
||||
}
|
||||
|
||||
// verifyCPU performs all verification on the CPU as fallback.
|
||||
func (p *GPUVerifyPipeline) verifyCPU(work *BlockVerifyWork) *BlockVerifyResult {
|
||||
result := &BlockVerifyResult{}
|
||||
var wg sync.WaitGroup
|
||||
var mu sync.Mutex
|
||||
|
||||
if work.BLS != nil && len(work.BLS.Messages) > 0 {
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
start := time.Now()
|
||||
valid := cpuBLSVerify(work.BLS)
|
||||
mu.Lock()
|
||||
result.BLSValid = valid
|
||||
result.BLSTime = time.Since(start)
|
||||
mu.Unlock()
|
||||
}()
|
||||
}
|
||||
|
||||
if work.Corona != nil && len(work.Corona.Messages) > 0 {
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
start := time.Now()
|
||||
valid := cpuCoronaVerify(work.Corona)
|
||||
mu.Lock()
|
||||
result.CoronaValid = valid
|
||||
result.CoronaTime = time.Since(start)
|
||||
mu.Unlock()
|
||||
}()
|
||||
}
|
||||
|
||||
if work.ZK != nil && len(work.ZK.Scalars) > 0 {
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
start := time.Now()
|
||||
valid := cpuZKVerify(work.ZK)
|
||||
mu.Lock()
|
||||
result.ZKValid = valid
|
||||
result.ZKTime = time.Since(start)
|
||||
mu.Unlock()
|
||||
}()
|
||||
}
|
||||
|
||||
if work.MLDSA != nil && len(work.MLDSA.Messages) > 0 {
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
start := time.Now()
|
||||
valid := cpuMLDSAVerify(work.MLDSA)
|
||||
mu.Lock()
|
||||
result.MLDSAValid = valid
|
||||
result.MLDSATime = time.Since(start)
|
||||
mu.Unlock()
|
||||
}()
|
||||
}
|
||||
|
||||
wg.Wait()
|
||||
return result
|
||||
}
|
||||
|
||||
// CPU fallback implementations — the pure-Go correctness oracle.
|
||||
//
|
||||
// These perform REAL per-element cryptographic verification using the
|
||||
// luxfi/crypto pure-Go primitives (CGO_ENABLED=0-clean). They interpret each
|
||||
// element's raw bytes with the SAME layout the GPU kernels use (see
|
||||
// gpuBLSVerify / gpuMLDSAVerify), so the CPU and GPU paths are a genuine
|
||||
// equivalence pair: a no-GPU node accepts exactly the signatures a GPU node
|
||||
// accepts, and never rubber-stamps a forged one.
|
||||
//
|
||||
// Corona and ZK have no pure-Go verifier in luxfi/crypto, so those paths
|
||||
// fail closed (return false) rather than format-check-and-accept. They MUST
|
||||
// be wired to a real verifier before block-accept depends on them.
|
||||
|
||||
func cpuBLSVerify(work *BLSWork) []bool {
|
||||
valid := make([]bool, len(work.Messages))
|
||||
for i := range work.Messages {
|
||||
// Mirror gpuBLSVerify's layout: pk is a 48-byte compressed G1 point,
|
||||
// sig is a 96-byte compressed G2 point, msg is the raw message.
|
||||
// PublicKeyFromCompressedBytes / SignatureFromBytes enforce the exact
|
||||
// length, on-curve and subgroup membership the blst/CGO path enforces;
|
||||
// any malformed input fails closed (constructor error => false).
|
||||
pk, err := bls.PublicKeyFromCompressedBytes(work.PubKeys[i])
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
sig, err := bls.SignatureFromBytes(work.Signatures[i])
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
valid[i] = bls.Verify(pk, sig, work.Messages[i])
|
||||
}
|
||||
return valid
|
||||
}
|
||||
|
||||
func cpuCoronaVerify(work *CoronaWork) []bool {
|
||||
// FAIL CLOSED: luxfi/crypto exposes no pure-Go Corona (lattice threshold)
|
||||
// signature verifier, and the GPU Corona kernel is the known-wrong-prime
|
||||
// BLOCKED kernel. There is no correct way to verify a Corona signature on
|
||||
// the CPU here, so every element is rejected. Never return true for an
|
||||
// unverified signature. Wire a real Corona verifier before block-accept
|
||||
// consumes this result.
|
||||
return make([]bool, len(work.Messages))
|
||||
}
|
||||
|
||||
func cpuZKVerify(work *ZKWork) bool {
|
||||
// FAIL CLOSED: luxfi/crypto exposes no standalone pure-Go ZK proof
|
||||
// verifier (the accel MSM path is a GPU primitive, not a proof check), so
|
||||
// CPU verification cannot establish proof validity. Reject rather than
|
||||
// rubber-stamp. Wire a real ZK verifier before block-accept consumes this.
|
||||
return false
|
||||
}
|
||||
|
||||
func cpuMLDSAVerify(work *MLDSAWork) []bool {
|
||||
valid := make([]bool, len(work.Messages))
|
||||
for i := range work.Messages {
|
||||
// Mirror gpuMLDSAVerify's layout: ML-DSA-65, pk 1952 bytes, msg raw.
|
||||
// PublicKeyFromBytes enforces the exact key length and decodes the
|
||||
// point; VerifySignature uses the FIPS 204 nil-context verify,
|
||||
// matching the kernel's contextless per-tx verification, and accepts
|
||||
// the signature at its true FIPS-204 length (3309 bytes). The GPU
|
||||
// flatten width (gpuMLDSAVerify's sigLen) now agrees at 3309, so the
|
||||
// CPU oracle and GPU path size the signature identically; see
|
||||
// TestMLDSA_WorkStructSizeIsCanonical. Malformed pk fails closed
|
||||
// (constructor error).
|
||||
pub, err := mldsa.PublicKeyFromBytes(work.PubKeys[i], mldsa.MLDSA65)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
valid[i] = pub.VerifySignature(work.Messages[i], work.Signatures[i])
|
||||
}
|
||||
return valid
|
||||
}
|
||||
|
||||
// validateWork checks batch size consistency.
|
||||
func validateWork(work *BlockVerifyWork) error {
|
||||
if w := work.BLS; w != nil {
|
||||
n := len(w.Messages)
|
||||
if n > 0 && (len(w.Signatures) != n || len(w.PubKeys) != n) {
|
||||
return ErrBLSSizeMismatch
|
||||
}
|
||||
}
|
||||
if w := work.Corona; w != nil {
|
||||
n := len(w.Messages)
|
||||
if n > 0 && (len(w.Signatures) != n || len(w.PubKeys) != n) {
|
||||
return ErrCoronaSizeMismatch
|
||||
}
|
||||
}
|
||||
if w := work.ZK; w != nil {
|
||||
if len(w.Scalars) > 0 && len(w.Bases) != len(w.Scalars) {
|
||||
return ErrZKSizeMismatch
|
||||
}
|
||||
}
|
||||
if w := work.MLDSA; w != nil {
|
||||
n := len(w.Messages)
|
||||
if n > 0 && (len(w.Signatures) != n || len(w.PubKeys) != n) {
|
||||
return ErrMLDSASizeMismatch
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// PipelineStats contains pipeline verification statistics.
|
||||
type PipelineStats struct {
|
||||
GPUVerifies uint64
|
||||
CPUVerifies uint64
|
||||
GPUTimeNs uint64
|
||||
CPUTimeNs uint64
|
||||
}
|
||||
|
||||
// Stats returns pipeline statistics.
|
||||
func (p *GPUVerifyPipeline) Stats() PipelineStats {
|
||||
return PipelineStats{
|
||||
GPUVerifies: atomic.LoadUint64(&p.gpuVerifies),
|
||||
CPUVerifies: atomic.LoadUint64(&p.cpuVerifies),
|
||||
GPUTimeNs: atomic.LoadUint64(&p.gpuTimeNs),
|
||||
CPUTimeNs: atomic.LoadUint64(&p.cpuTimeNs),
|
||||
}
|
||||
}
|
||||
|
||||
// Helper: find max byte slice length in a batch.
|
||||
func maxByteLen(slices [][]byte) int {
|
||||
m := 1 // minimum 1 to avoid zero-dimension tensors
|
||||
for _, s := range slices {
|
||||
if len(s) > m {
|
||||
m = len(s)
|
||||
}
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
// Helper: flatten [][]byte into a contiguous []uint8 with zero-padding.
|
||||
func flattenPadded(slices [][]byte, n, elemLen int) []uint8 {
|
||||
flat := make([]uint8, n*elemLen)
|
||||
for i, s := range slices {
|
||||
copy(flat[i*elemLen:], s)
|
||||
}
|
||||
return flat
|
||||
}
|
||||
@@ -0,0 +1,436 @@
|
||||
// Copyright (C) 2019-2025, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
package quasar
|
||||
|
||||
import (
|
||||
"crypto/rand"
|
||||
"testing"
|
||||
|
||||
"github.com/luxfi/crypto/bls"
|
||||
"github.com/luxfi/crypto/mldsa"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// makeRandomBytes returns n random bytes.
|
||||
func makeRandomBytes(n int) []byte {
|
||||
b := make([]byte, n)
|
||||
_, _ = rand.Read(b)
|
||||
return b
|
||||
}
|
||||
|
||||
// makeValidBLSEntry returns (msg, sig, pk) for a REAL BLS signature over a
|
||||
// random 32-byte message: pk is the 48-byte compressed G1 key, sig is the
|
||||
// 96-byte compressed G2 signature. The CPU oracle must accept this.
|
||||
func makeValidBLSEntry(t testing.TB) (msg, sig, pk []byte) {
|
||||
t.Helper()
|
||||
sk, err := bls.NewSecretKey()
|
||||
require.NoError(t, err)
|
||||
msg = makeRandomBytes(32)
|
||||
s, err := sk.Sign(msg)
|
||||
require.NoError(t, err)
|
||||
sig = bls.SignatureToBytes(s)
|
||||
pk = bls.PublicKeyToCompressedBytes(sk.PublicKey())
|
||||
require.Len(t, sig, 96)
|
||||
require.Len(t, pk, 48)
|
||||
return msg, sig, pk
|
||||
}
|
||||
|
||||
// makeValidMLDSAEntry returns (msg, sig, pk) for a REAL ML-DSA-65 signature
|
||||
// over a random 64-byte message. The sizes are taken from the crypto package
|
||||
// constants (FIPS-204 ML-DSA-65: pk 1952 bytes, sig 3309 bytes) rather than
|
||||
// hard-coded — see TestMLDSA_WorkStructSizeIsCanonical, which holds the
|
||||
// MLDSAWork struct / gpuMLDSAVerify sig constant pinned at the FIPS-204 3309
|
||||
// (corrected from the stale round-3 Dilithium3 3293). The CPU oracle uses the
|
||||
// typed Verify, so it accepts the real signature regardless.
|
||||
func makeValidMLDSAEntry(t testing.TB) (msg, sig, pk []byte) {
|
||||
t.Helper()
|
||||
priv, err := mldsa.GenerateKey(rand.Reader, mldsa.MLDSA65)
|
||||
require.NoError(t, err)
|
||||
msg = makeRandomBytes(64)
|
||||
sig, err = priv.Sign(rand.Reader, msg, nil)
|
||||
require.NoError(t, err)
|
||||
pk = priv.PublicKey.Bytes()
|
||||
require.Len(t, sig, mldsa.MLDSA65SignatureSize)
|
||||
require.Len(t, pk, mldsa.MLDSA65PublicKeySize)
|
||||
return msg, sig, pk
|
||||
}
|
||||
|
||||
// makeBLSWork creates BLSWork with n entries carrying REAL valid BLS
|
||||
// signatures (the CPU oracle now performs real verification).
|
||||
func makeBLSWork(t testing.TB, n int) *BLSWork {
|
||||
t.Helper()
|
||||
w := &BLSWork{
|
||||
Messages: make([][]byte, n),
|
||||
Signatures: make([][]byte, n),
|
||||
PubKeys: make([][]byte, n),
|
||||
}
|
||||
for i := 0; i < n; i++ {
|
||||
w.Messages[i], w.Signatures[i], w.PubKeys[i] = makeValidBLSEntry(t)
|
||||
}
|
||||
return w
|
||||
}
|
||||
|
||||
// makeCoronaWork creates CoronaWork with n entries.
|
||||
func makeCoronaWork(n int) *CoronaWork {
|
||||
w := &CoronaWork{
|
||||
Messages: make([][]byte, n),
|
||||
Signatures: make([][]byte, n),
|
||||
PubKeys: make([][]byte, n),
|
||||
}
|
||||
for i := 0; i < n; i++ {
|
||||
w.Messages[i] = makeRandomBytes(48)
|
||||
w.Signatures[i] = makeRandomBytes(512)
|
||||
w.PubKeys[i] = makeRandomBytes(256)
|
||||
}
|
||||
return w
|
||||
}
|
||||
|
||||
// makeZKWork creates ZKWork with m entries.
|
||||
func makeZKWork(m int) *ZKWork {
|
||||
w := &ZKWork{
|
||||
Scalars: make([][]byte, m),
|
||||
Bases: make([][]byte, m),
|
||||
}
|
||||
for i := 0; i < m; i++ {
|
||||
w.Scalars[i] = makeRandomBytes(32)
|
||||
w.Bases[i] = makeRandomBytes(64)
|
||||
}
|
||||
return w
|
||||
}
|
||||
|
||||
// makeMLDSAWork creates MLDSAWork with n entries carrying REAL valid
|
||||
// ML-DSA-65 (Dilithium3) signatures (the CPU oracle now performs real
|
||||
// verification).
|
||||
func makeMLDSAWork(t testing.TB, n int) *MLDSAWork {
|
||||
t.Helper()
|
||||
w := &MLDSAWork{
|
||||
Messages: make([][]byte, n),
|
||||
Signatures: make([][]byte, n),
|
||||
PubKeys: make([][]byte, n),
|
||||
}
|
||||
for i := 0; i < n; i++ {
|
||||
w.Messages[i], w.Signatures[i], w.PubKeys[i] = makeValidMLDSAEntry(t)
|
||||
}
|
||||
return w
|
||||
}
|
||||
|
||||
func TestGPUPipeline_AllFourTypes(t *testing.T) {
|
||||
pipeline := NewGPUVerifyPipeline()
|
||||
|
||||
work := &BlockVerifyWork{
|
||||
BLS: makeBLSWork(t, 5),
|
||||
Corona: makeCoronaWork(3),
|
||||
ZK: makeZKWork(2),
|
||||
MLDSA: makeMLDSAWork(t, 10),
|
||||
}
|
||||
|
||||
result, err := pipeline.VerifyBlock(work)
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, result)
|
||||
|
||||
// BLS results: real valid signatures, CPU oracle accepts.
|
||||
require.Len(t, result.BLSValid, 5, "should have 5 BLS results")
|
||||
for i, v := range result.BLSValid {
|
||||
require.True(t, v, "BLS[%d] should be valid", i)
|
||||
}
|
||||
|
||||
// Corona results: no pure-Go Corona verifier exists, so the CPU oracle
|
||||
// fails closed — every element is rejected (never rubber-stamped).
|
||||
require.Len(t, result.CoronaValid, 3, "should have 3 Corona results")
|
||||
for i, v := range result.CoronaValid {
|
||||
require.False(t, v, "Corona[%d] must fail closed (no pure-Go verifier)", i)
|
||||
}
|
||||
|
||||
// ZK result: no pure-Go ZK verifier exists, so the CPU oracle fails closed.
|
||||
require.False(t, result.ZKValid, "ZK batch must fail closed (no pure-Go verifier)")
|
||||
|
||||
// ML-DSA results: real valid signatures, CPU oracle accepts.
|
||||
require.Len(t, result.MLDSAValid, 10, "should have 10 ML-DSA results")
|
||||
for i, v := range result.MLDSAValid {
|
||||
require.True(t, v, "MLDSA[%d] should be valid", i)
|
||||
}
|
||||
|
||||
// Timing: all durations should be non-negative
|
||||
require.GreaterOrEqual(t, result.TotalTime.Nanoseconds(), int64(0))
|
||||
require.GreaterOrEqual(t, result.BLSTime.Nanoseconds(), int64(0))
|
||||
require.GreaterOrEqual(t, result.CoronaTime.Nanoseconds(), int64(0))
|
||||
require.GreaterOrEqual(t, result.ZKTime.Nanoseconds(), int64(0))
|
||||
require.GreaterOrEqual(t, result.MLDSATime.Nanoseconds(), int64(0))
|
||||
|
||||
// Stats should reflect the verification
|
||||
stats := pipeline.Stats()
|
||||
require.Equal(t, uint64(1), stats.GPUVerifies+stats.CPUVerifies,
|
||||
"exactly one verify should have been recorded")
|
||||
}
|
||||
|
||||
func TestGPUPipeline_CPUFallback(t *testing.T) {
|
||||
// Without CGO/GPU, accel.Available() returns false.
|
||||
// Pipeline must fall back to CPU verification.
|
||||
pipeline := NewGPUVerifyPipeline()
|
||||
|
||||
work := &BlockVerifyWork{
|
||||
BLS: makeBLSWork(t, 3),
|
||||
MLDSA: makeMLDSAWork(t, 4),
|
||||
}
|
||||
|
||||
result, err := pipeline.VerifyBlock(work)
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, result)
|
||||
|
||||
// CPU fallback performs real verification; valid signatures are accepted.
|
||||
require.Len(t, result.BLSValid, 3)
|
||||
for i, v := range result.BLSValid {
|
||||
require.True(t, v, "CPU BLS[%d] should be valid", i)
|
||||
}
|
||||
|
||||
require.Len(t, result.MLDSAValid, 4)
|
||||
for i, v := range result.MLDSAValid {
|
||||
require.True(t, v, "CPU MLDSA[%d] should be valid", i)
|
||||
}
|
||||
|
||||
// GPU should not have been used (no CGO in test env)
|
||||
require.False(t, result.GPUUsed, "should use CPU fallback")
|
||||
|
||||
stats := pipeline.Stats()
|
||||
require.Equal(t, uint64(1), stats.CPUVerifies)
|
||||
}
|
||||
|
||||
// TestCPUVerify_RealOracle proves the CPU fallback is a real cryptographic
|
||||
// oracle, not a length-checking rubber stamp: a valid signature is accepted
|
||||
// and a well-formed-but-FORGED signature (correct lengths, wrong bytes) is
|
||||
// REJECTED. The forged-rejection case is the regression guard against the
|
||||
// old `return true for well-formed inputs` behavior.
|
||||
func TestCPUVerify_RealOracle(t *testing.T) {
|
||||
t.Run("BLS valid accepted, forged rejected", func(t *testing.T) {
|
||||
msg, sig, pk := makeValidBLSEntry(t)
|
||||
|
||||
// Valid signature => accepted.
|
||||
good := cpuBLSVerify(&BLSWork{
|
||||
Messages: [][]byte{msg},
|
||||
Signatures: [][]byte{sig},
|
||||
PubKeys: [][]byte{pk},
|
||||
})
|
||||
require.Equal(t, []bool{true}, good, "valid BLS signature must be accepted")
|
||||
|
||||
// Forged signature: correct 96-byte length, random bytes => rejected.
|
||||
forgedSig := makeRandomBytes(96)
|
||||
bad := cpuBLSVerify(&BLSWork{
|
||||
Messages: [][]byte{msg},
|
||||
Signatures: [][]byte{forgedSig},
|
||||
PubKeys: [][]byte{pk},
|
||||
})
|
||||
require.Equal(t, []bool{false}, bad, "forged BLS signature (right length, wrong bytes) must be REJECTED")
|
||||
|
||||
// Valid signature against the WRONG message => rejected.
|
||||
wrongMsg := cpuBLSVerify(&BLSWork{
|
||||
Messages: [][]byte{makeRandomBytes(32)},
|
||||
Signatures: [][]byte{sig},
|
||||
PubKeys: [][]byte{pk},
|
||||
})
|
||||
require.Equal(t, []bool{false}, wrongMsg, "BLS signature over a different message must be REJECTED")
|
||||
})
|
||||
|
||||
t.Run("MLDSA valid accepted, forged rejected", func(t *testing.T) {
|
||||
msg, sig, pk := makeValidMLDSAEntry(t)
|
||||
|
||||
// Valid signature => accepted.
|
||||
good := cpuMLDSAVerify(&MLDSAWork{
|
||||
Messages: [][]byte{msg},
|
||||
Signatures: [][]byte{sig},
|
||||
PubKeys: [][]byte{pk},
|
||||
})
|
||||
require.Equal(t, []bool{true}, good, "valid ML-DSA signature must be accepted")
|
||||
|
||||
// Forged signature: correct length, random bytes => rejected.
|
||||
forgedSig := makeRandomBytes(mldsa.MLDSA65SignatureSize)
|
||||
bad := cpuMLDSAVerify(&MLDSAWork{
|
||||
Messages: [][]byte{msg},
|
||||
Signatures: [][]byte{forgedSig},
|
||||
PubKeys: [][]byte{pk},
|
||||
})
|
||||
require.Equal(t, []bool{false}, bad, "forged ML-DSA signature (right length, wrong bytes) must be REJECTED")
|
||||
|
||||
// Valid signature against the WRONG message => rejected.
|
||||
wrongMsg := cpuMLDSAVerify(&MLDSAWork{
|
||||
Messages: [][]byte{makeRandomBytes(64)},
|
||||
Signatures: [][]byte{sig},
|
||||
PubKeys: [][]byte{pk},
|
||||
})
|
||||
require.Equal(t, []bool{false}, wrongMsg, "ML-DSA signature over a different message must be REJECTED")
|
||||
})
|
||||
|
||||
t.Run("Corona fails closed", func(t *testing.T) {
|
||||
// No pure-Go Corona verifier exists; every element must be rejected,
|
||||
// never rubber-stamped on length alone.
|
||||
got := cpuCoronaVerify(&CoronaWork{
|
||||
Messages: [][]byte{makeRandomBytes(48), makeRandomBytes(48)},
|
||||
Signatures: [][]byte{makeRandomBytes(512), makeRandomBytes(512)},
|
||||
PubKeys: [][]byte{makeRandomBytes(256), makeRandomBytes(256)},
|
||||
})
|
||||
require.Equal(t, []bool{false, false}, got, "Corona must fail closed for all elements")
|
||||
})
|
||||
|
||||
t.Run("ZK fails closed", func(t *testing.T) {
|
||||
// No pure-Go ZK proof verifier exists; the batch must be rejected.
|
||||
got := cpuZKVerify(&ZKWork{
|
||||
Scalars: [][]byte{makeRandomBytes(32)},
|
||||
Bases: [][]byte{makeRandomBytes(64)},
|
||||
})
|
||||
require.False(t, got, "ZK must fail closed")
|
||||
})
|
||||
}
|
||||
|
||||
// TestMLDSA_WorkStructSizeIsCanonical pins the FIPS-204 ML-DSA-65 signature and
|
||||
// public key sizes that the MLDSAWork struct comment and gpuMLDSAVerify's
|
||||
// fixed-size flatten now use. luxfi/crypto (circl v1.6.3, FIPS-204 final)
|
||||
// produces 3309-byte ML-DSA-65 signatures (5*640 + 55 + 6 + 48 = 3309); the
|
||||
// GPU flatten width was corrected from the stale round-3 Dilithium3 size 3293
|
||||
// to 3309 so the GPU path no longer clamps/corrupts a real signature. The
|
||||
// public key size (1952) is unchanged across the round-3 -> final transition.
|
||||
//
|
||||
// This is the equivalence-pair guard: the CPU oracle (cpuMLDSAVerify) parses
|
||||
// pk via the typed PublicKeyFromBytes and calls VerifySignature, accepting the
|
||||
// real 3309-byte signature; the GPU path sizes the signature identically. If
|
||||
// the crypto constant ever drifts, this test fails before any divergence
|
||||
// reaches the GPU ML-DSA kernel (not yet wired into block-accept).
|
||||
func TestMLDSA_WorkStructSizeIsCanonical(t *testing.T) {
|
||||
require.Equal(t, 1952, mldsa.MLDSA65PublicKeySize,
|
||||
"ML-DSA-65 public key is 1952 bytes (matches MLDSAWork / gpuMLDSAVerify pkLen)")
|
||||
require.Equal(t, 3309, mldsa.MLDSA65SignatureSize,
|
||||
"ML-DSA-65 signature is 3309 bytes (FIPS-204), matching the MLDSAWork struct / gpuMLDSAVerify sigLen")
|
||||
}
|
||||
|
||||
func TestGPUPipeline_EmptyBatches(t *testing.T) {
|
||||
pipeline := NewGPUVerifyPipeline()
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
work *BlockVerifyWork
|
||||
}{
|
||||
{
|
||||
name: "nil work",
|
||||
work: nil,
|
||||
},
|
||||
{
|
||||
name: "all nil batches",
|
||||
work: &BlockVerifyWork{},
|
||||
},
|
||||
{
|
||||
name: "empty BLS only",
|
||||
work: &BlockVerifyWork{
|
||||
BLS: &BLSWork{},
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "BLS filled, rest nil",
|
||||
work: &BlockVerifyWork{
|
||||
BLS: makeBLSWork(t, 2),
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "ZK only",
|
||||
work: &BlockVerifyWork{
|
||||
ZK: makeZKWork(1),
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "MLDSA only",
|
||||
work: &BlockVerifyWork{
|
||||
MLDSA: makeMLDSAWork(t, 1),
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "Corona only",
|
||||
work: &BlockVerifyWork{
|
||||
Corona: makeCoronaWork(1),
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
result, err := pipeline.VerifyBlock(tt.work)
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, result)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestGPUPipeline_ValidationErrors(t *testing.T) {
|
||||
pipeline := NewGPUVerifyPipeline()
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
work *BlockVerifyWork
|
||||
wantErr error
|
||||
}{
|
||||
{
|
||||
name: "BLS size mismatch",
|
||||
work: &BlockVerifyWork{
|
||||
BLS: &BLSWork{
|
||||
Messages: [][]byte{{1}},
|
||||
Signatures: [][]byte{{1}, {2}}, // 2 != 1
|
||||
PubKeys: [][]byte{{1}},
|
||||
},
|
||||
},
|
||||
wantErr: ErrBLSSizeMismatch,
|
||||
},
|
||||
{
|
||||
name: "Corona size mismatch",
|
||||
work: &BlockVerifyWork{
|
||||
Corona: &CoronaWork{
|
||||
Messages: [][]byte{{1}, {2}},
|
||||
Signatures: [][]byte{{1}}, // 1 != 2
|
||||
PubKeys: [][]byte{{1}, {2}},
|
||||
},
|
||||
},
|
||||
wantErr: ErrCoronaSizeMismatch,
|
||||
},
|
||||
{
|
||||
name: "ZK size mismatch",
|
||||
work: &BlockVerifyWork{
|
||||
ZK: &ZKWork{
|
||||
Scalars: [][]byte{{1}, {2}},
|
||||
Bases: [][]byte{{1}}, // 1 != 2
|
||||
},
|
||||
},
|
||||
wantErr: ErrZKSizeMismatch,
|
||||
},
|
||||
{
|
||||
name: "MLDSA size mismatch",
|
||||
work: &BlockVerifyWork{
|
||||
MLDSA: &MLDSAWork{
|
||||
Messages: [][]byte{{1}},
|
||||
Signatures: [][]byte{{1}},
|
||||
PubKeys: [][]byte{{1}, {2}}, // 2 != 1
|
||||
},
|
||||
},
|
||||
wantErr: ErrMLDSASizeMismatch,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
_, err := pipeline.VerifyBlock(tt.work)
|
||||
require.ErrorIs(t, err, tt.wantErr)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func BenchmarkGPUPipeline(b *testing.B) {
|
||||
pipeline := NewGPUVerifyPipeline()
|
||||
|
||||
work := &BlockVerifyWork{
|
||||
BLS: makeBLSWork(b, 100),
|
||||
Corona: makeCoronaWork(50),
|
||||
ZK: makeZKWork(10),
|
||||
MLDSA: makeMLDSAWork(b, 200),
|
||||
}
|
||||
|
||||
b.ResetTimer()
|
||||
b.ReportAllocs()
|
||||
for i := 0; i < b.N; i++ {
|
||||
_, _ = pipeline.VerifyBlock(work)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,970 @@
|
||||
// Copyright (C) 2019-2025, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
// Package quasar integration tests.
|
||||
//
|
||||
// These tests exercise realistic end-to-end scenarios for Quasar consensus:
|
||||
// - Full component wiring and event processing
|
||||
// - Corona threshold signing flows (skipped if lattice lib unavailable)
|
||||
// - Concurrent operation safety
|
||||
// - Stop/start lifecycle management
|
||||
// - Memory behavior with many finality events
|
||||
//
|
||||
// Run with: go test -v -run "^Test.*Integration\|^TestQuasar" ./...
|
||||
// Skip long tests: go test -short ./...
|
||||
|
||||
package quasar
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/rand"
|
||||
"runtime"
|
||||
"strings"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/luxfi/ids"
|
||||
"github.com/luxfi/log"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Mock implementations for integration tests
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// mockPChainProvider implements PChainProvider for tests
|
||||
type mockPChainProvider struct {
|
||||
mu sync.RWMutex
|
||||
height uint64
|
||||
validators []ValidatorState
|
||||
finalityCh chan FinalityEvent
|
||||
closed bool
|
||||
}
|
||||
|
||||
func newMockPChainProvider(validators []ValidatorState) *mockPChainProvider {
|
||||
return &mockPChainProvider{
|
||||
height: 0,
|
||||
validators: validators,
|
||||
finalityCh: make(chan FinalityEvent, 100),
|
||||
}
|
||||
}
|
||||
|
||||
func (m *mockPChainProvider) GetFinalizedHeight() uint64 {
|
||||
m.mu.RLock()
|
||||
defer m.mu.RUnlock()
|
||||
return m.height
|
||||
}
|
||||
|
||||
func (m *mockPChainProvider) GetValidators(height uint64) ([]ValidatorState, error) {
|
||||
m.mu.RLock()
|
||||
defer m.mu.RUnlock()
|
||||
return m.validators, nil
|
||||
}
|
||||
|
||||
func (m *mockPChainProvider) SubscribeFinality() <-chan FinalityEvent {
|
||||
return m.finalityCh
|
||||
}
|
||||
|
||||
func (m *mockPChainProvider) Validators() []ValidatorState {
|
||||
m.mu.RLock()
|
||||
defer m.mu.RUnlock()
|
||||
return append([]ValidatorState(nil), m.validators...)
|
||||
}
|
||||
|
||||
func (m *mockPChainProvider) EmitFinality(event FinalityEvent) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
if m.closed {
|
||||
return
|
||||
}
|
||||
m.height = event.Height
|
||||
select {
|
||||
case m.finalityCh <- event:
|
||||
default:
|
||||
}
|
||||
}
|
||||
|
||||
func (m *mockPChainProvider) Close() {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
if !m.closed {
|
||||
m.closed = true
|
||||
close(m.finalityCh)
|
||||
}
|
||||
}
|
||||
|
||||
// mockQuantumSigner implements QuantumSignerFallback for tests
|
||||
type mockQuantumSigner struct{}
|
||||
|
||||
func (m *mockQuantumSigner) SignMessage(msg []byte) ([]byte, error) {
|
||||
return []byte("RT-MOCK-SIG"), nil
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Helper functions
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// generateValidatorStates creates n ValidatorState entries
|
||||
func generateValidatorStates(n int) []ValidatorState {
|
||||
states := make([]ValidatorState, n)
|
||||
for i := range states {
|
||||
blsKey := make([]byte, 48)
|
||||
rtKey := make([]byte, 32)
|
||||
_, _ = rand.Read(blsKey)
|
||||
_, _ = rand.Read(rtKey)
|
||||
|
||||
states[i] = ValidatorState{
|
||||
NodeID: ids.GenerateTestNodeID(),
|
||||
Weight: 1000,
|
||||
BLSPubKey: blsKey,
|
||||
CoronaKey: rtKey,
|
||||
Active: true,
|
||||
}
|
||||
}
|
||||
return states
|
||||
}
|
||||
|
||||
// createTestEvent creates a FinalityEvent for testing
|
||||
func createTestEvent(height uint64, validators []ValidatorState) FinalityEvent {
|
||||
var blockID ids.ID
|
||||
_, _ = rand.Read(blockID[:])
|
||||
|
||||
return FinalityEvent{
|
||||
Height: height,
|
||||
BlockID: blockID,
|
||||
Validators: validators,
|
||||
Timestamp: time.Now(),
|
||||
}
|
||||
}
|
||||
|
||||
// setupQuasarWithCorona creates a Quasar with a test Corona coordinator.
|
||||
// Returns nil for Quasar if Corona initialization fails (e.g., lattice lib constraint).
|
||||
func setupQuasarWithCorona(t *testing.T, numParties int) (*Quasar, *mockPChainProvider, []ids.NodeID, error) {
|
||||
t.Helper()
|
||||
|
||||
validatorStates := generateValidatorStates(numParties)
|
||||
pchain := newMockPChainProvider(validatorStates)
|
||||
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
if err != nil {
|
||||
return nil, nil, nil, err
|
||||
}
|
||||
|
||||
// Connect providers
|
||||
q.ConnectPChain(pchain)
|
||||
q.ConnectQuantumFallback(&mockQuantumSigner{})
|
||||
|
||||
// Create a test Corona coordinator (stub signatures, not production)
|
||||
threshold := (numParties * 2 / 3) + 1
|
||||
if threshold < 2 {
|
||||
threshold = 2
|
||||
}
|
||||
rc, err := NewTestCoronaCoordinator(log.NewNoOpLogger(), CoronaConfig{
|
||||
NumParties: numParties,
|
||||
Threshold: threshold,
|
||||
})
|
||||
if err != nil {
|
||||
pchain.Close()
|
||||
return nil, nil, nil, err
|
||||
}
|
||||
q.ConnectCorona(rc)
|
||||
|
||||
// Extract node IDs and initialize Corona
|
||||
nodeIDs := make([]ids.NodeID, len(validatorStates))
|
||||
for i, v := range validatorStates {
|
||||
nodeIDs[i] = v.NodeID
|
||||
}
|
||||
|
||||
err = q.InitializeCorona(nodeIDs)
|
||||
if err != nil {
|
||||
pchain.Close()
|
||||
return nil, nil, nil, err
|
||||
}
|
||||
|
||||
return q, pchain, nodeIDs, nil
|
||||
}
|
||||
|
||||
// isLatticeUnavailable checks if an error indicates lattice library constraints
|
||||
func isLatticeUnavailable(err error) bool {
|
||||
if err == nil {
|
||||
return false
|
||||
}
|
||||
msg := err.Error()
|
||||
return strings.Contains(msg, "ring") || strings.Contains(msg, "modulus") ||
|
||||
strings.Contains(msg, "prime") || strings.Contains(msg, "lattice")
|
||||
}
|
||||
|
||||
// ----------------------------------------------------------------------------
|
||||
// Integration Tests
|
||||
// ----------------------------------------------------------------------------
|
||||
|
||||
// TestQuasarFullFlow tests creating Quasar, connecting components, and processing events
|
||||
func TestQuasarFullFlow(t *testing.T) {
|
||||
const numValidators = 5
|
||||
|
||||
t.Run("create_and_connect", func(t *testing.T) {
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err, "NewQuasar should succeed")
|
||||
require.NotNil(t, q, "Quasar should not be nil")
|
||||
|
||||
// Verify initial state
|
||||
stats := q.Stats()
|
||||
require.False(t, stats.Running, "should not be running initially")
|
||||
require.Equal(t, uint64(0), stats.PChainHeight)
|
||||
require.Equal(t, uint64(0), stats.QChainHeight)
|
||||
|
||||
// Connect components
|
||||
validatorStates := generateValidatorStates(numValidators)
|
||||
pchain := newMockPChainProvider(validatorStates)
|
||||
defer pchain.Close()
|
||||
|
||||
q.ConnectPChain(pchain)
|
||||
q.ConnectQuantumFallback(&mockQuantumSigner{})
|
||||
|
||||
// Verify configuration
|
||||
threshold, quorumNum, quorumDen := q.GetConfig()
|
||||
require.Equal(t, 3, threshold)
|
||||
require.Equal(t, uint64(2), quorumNum)
|
||||
require.Equal(t, uint64(3), quorumDen)
|
||||
})
|
||||
|
||||
t.Run("start_and_stop", func(t *testing.T) {
|
||||
validatorStates := generateValidatorStates(numValidators)
|
||||
pchain := newMockPChainProvider(validatorStates)
|
||||
defer pchain.Close()
|
||||
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
|
||||
q.ConnectPChain(pchain)
|
||||
q.ConnectQuantumFallback(&mockQuantumSigner{})
|
||||
|
||||
// Start
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
|
||||
err = q.Start(ctx)
|
||||
require.NoError(t, err, "Start should succeed")
|
||||
require.True(t, q.IsRunning(), "should be running after Start")
|
||||
|
||||
// Stop
|
||||
q.Stop()
|
||||
// Give goroutines time to shut down
|
||||
time.Sleep(50 * time.Millisecond)
|
||||
require.False(t, q.IsRunning(), "should not be running after Stop")
|
||||
})
|
||||
|
||||
t.Run("process_single_event", func(t *testing.T) {
|
||||
q, pchain, _, err := setupQuasarWithCorona(t, numValidators)
|
||||
if isLatticeUnavailable(err) {
|
||||
t.Skipf("Skipping: lattice library constraint: %v", err)
|
||||
}
|
||||
require.NoError(t, err)
|
||||
defer pchain.Close()
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
|
||||
err = q.Start(ctx)
|
||||
require.NoError(t, err)
|
||||
defer q.Stop()
|
||||
|
||||
// Emit event
|
||||
event := createTestEvent(1, pchain.Validators())
|
||||
pchain.EmitFinality(event)
|
||||
|
||||
require.Eventually(t, func() bool {
|
||||
return q.Stats().PChainHeight >= 1
|
||||
}, time.Second, 10*time.Millisecond, "P-chain height should be 1")
|
||||
})
|
||||
|
||||
t.Run("verify_quorum_calculation", func(t *testing.T) {
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
|
||||
// Test quorum: 2/3 means 67% needed
|
||||
require.True(t, q.CheckQuorum(670, 1000), "67% should meet 2/3 quorum")
|
||||
require.True(t, q.CheckQuorum(667, 1000), "66.7% should meet 2/3 quorum")
|
||||
require.False(t, q.CheckQuorum(600, 1000), "60% should not meet 2/3 quorum")
|
||||
require.False(t, q.CheckQuorum(0, 1000), "0% should not meet quorum")
|
||||
// Note: zero total weight is an edge case - required becomes 0, so any signer weight passes
|
||||
// This is intentional: if there are no validators, there's nothing to check
|
||||
require.True(t, q.CheckQuorum(500, 0), "zero total is edge case (required=0)")
|
||||
})
|
||||
}
|
||||
|
||||
// TestQuasarWithCorona tests full threshold signing flow
|
||||
func TestQuasarWithCorona(t *testing.T) {
|
||||
// All Corona tests require the lattice library to work correctly.
|
||||
// Skip if the library has constraints (e.g., requires prime moduli).
|
||||
|
||||
t.Run("initialize_and_sign", func(t *testing.T) {
|
||||
q, pchain, _, err := setupQuasarWithCorona(t, 5)
|
||||
if isLatticeUnavailable(err) {
|
||||
t.Skipf("Skipping: lattice library constraint: %v", err)
|
||||
}
|
||||
require.NoError(t, err)
|
||||
defer pchain.Close()
|
||||
|
||||
// Verify Corona is connected
|
||||
require.NotNil(t, q.corona, "Corona should be connected")
|
||||
require.True(t, q.corona.IsInitialized(), "Corona should be initialized")
|
||||
})
|
||||
|
||||
t.Run("sign_and_verify", func(t *testing.T) {
|
||||
q, pchain, _, err := setupQuasarWithCorona(t, 5)
|
||||
if isLatticeUnavailable(err) {
|
||||
t.Skipf("Skipping: lattice library constraint: %v", err)
|
||||
}
|
||||
require.NoError(t, err)
|
||||
defer pchain.Close()
|
||||
|
||||
// Sign a message
|
||||
msg := []byte("test message for signing")
|
||||
sig, err := q.corona.Sign(msg)
|
||||
require.NoError(t, err, "Sign should succeed")
|
||||
require.NotNil(t, sig, "Signature should not be nil")
|
||||
|
||||
// Verify signature
|
||||
valid := q.corona.Verify(msg, sig)
|
||||
require.True(t, valid, "Signature should verify")
|
||||
})
|
||||
|
||||
t.Run("multiple_signing_sessions", func(t *testing.T) {
|
||||
q, pchain, _, err := setupQuasarWithCorona(t, 5)
|
||||
if isLatticeUnavailable(err) {
|
||||
t.Skipf("Skipping: lattice library constraint: %v", err)
|
||||
}
|
||||
require.NoError(t, err)
|
||||
defer pchain.Close()
|
||||
|
||||
// Sign multiple messages
|
||||
for i := 0; i < 3; i++ {
|
||||
msg := []byte("message " + string(rune('A'+i)))
|
||||
sig, err := q.corona.Sign(msg)
|
||||
require.NoError(t, err, "Sign %d should succeed", i)
|
||||
require.True(t, q.corona.Verify(msg, sig), "Signature %d should verify", i)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("threshold_parameter_check", func(t *testing.T) {
|
||||
q, pchain, _, err := setupQuasarWithCorona(t, 5)
|
||||
if isLatticeUnavailable(err) {
|
||||
t.Skipf("Skipping: lattice library constraint: %v", err)
|
||||
}
|
||||
require.NoError(t, err)
|
||||
defer pchain.Close()
|
||||
|
||||
// With 5 parties, threshold = (5 * 2 / 3) + 1 = 4
|
||||
require.Equal(t, 4, q.corona.Threshold(), "Threshold should be 4 for 5 parties")
|
||||
require.Equal(t, 5, q.corona.NumParties(), "NumParties should be 5")
|
||||
})
|
||||
}
|
||||
|
||||
// TestQuasarConcurrent tests concurrent finality processing
|
||||
func TestQuasarConcurrent(t *testing.T) {
|
||||
const numValidators = 5
|
||||
const numEvents = 50
|
||||
|
||||
q, pchain, _, err := setupQuasarWithCorona(t, numValidators)
|
||||
if isLatticeUnavailable(err) {
|
||||
t.Skipf("Skipping: lattice library constraint: %v", err)
|
||||
}
|
||||
require.NoError(t, err)
|
||||
defer pchain.Close()
|
||||
validatorStates := pchain.Validators()
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
|
||||
err = q.Start(ctx)
|
||||
require.NoError(t, err)
|
||||
defer q.Stop()
|
||||
|
||||
// Send events concurrently
|
||||
var wg sync.WaitGroup
|
||||
for i := uint64(1); i <= numEvents; i++ {
|
||||
wg.Add(1)
|
||||
go func(height uint64) {
|
||||
defer wg.Done()
|
||||
event := createTestEvent(height, validatorStates)
|
||||
pchain.EmitFinality(event)
|
||||
}(i)
|
||||
}
|
||||
wg.Wait()
|
||||
|
||||
// Wait for processing
|
||||
time.Sleep(500 * time.Millisecond)
|
||||
|
||||
stats := q.Stats()
|
||||
t.Logf("Processed %d events, finalized blocks: %d", numEvents, stats.FinalizedBlocks)
|
||||
require.GreaterOrEqual(t, stats.FinalizedBlocks, 1, "should have finalized at least 1 block")
|
||||
}
|
||||
|
||||
// TestQuasarConcurrentCoronaSigning tests concurrent Corona signing
|
||||
func TestQuasarConcurrentCoronaSigning(t *testing.T) {
|
||||
q, pchain, _, err := setupQuasarWithCorona(t, 5)
|
||||
if isLatticeUnavailable(err) {
|
||||
t.Skipf("Skipping: lattice library constraint: %v", err)
|
||||
}
|
||||
require.NoError(t, err)
|
||||
defer pchain.Close()
|
||||
|
||||
const numSigners = 10
|
||||
var wg sync.WaitGroup
|
||||
var successCount atomic.Int32
|
||||
|
||||
for i := 0; i < numSigners; i++ {
|
||||
wg.Add(1)
|
||||
go func(idx int) {
|
||||
defer wg.Done()
|
||||
msg := []byte("concurrent message " + string(rune('0'+idx)))
|
||||
sig, err := q.corona.Sign(msg)
|
||||
if err == nil && q.corona.Verify(msg, sig) {
|
||||
successCount.Add(1)
|
||||
}
|
||||
}(i)
|
||||
}
|
||||
wg.Wait()
|
||||
|
||||
require.Equal(t, int32(numSigners), successCount.Load(), "all concurrent signs should succeed")
|
||||
}
|
||||
|
||||
// TestQuasarRestart tests stop/start cycles
|
||||
func TestQuasarRestart(t *testing.T) {
|
||||
const numValidators = 5
|
||||
|
||||
t.Run("basic_stop_start", func(t *testing.T) {
|
||||
validatorStates := generateValidatorStates(numValidators)
|
||||
pchain := newMockPChainProvider(validatorStates)
|
||||
defer pchain.Close()
|
||||
|
||||
// First cycle
|
||||
q1, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
q1.ConnectPChain(pchain)
|
||||
q1.ConnectQuantumFallback(&mockQuantumSigner{})
|
||||
|
||||
ctx1, cancel1 := context.WithCancel(context.Background())
|
||||
err = q1.Start(ctx1)
|
||||
require.NoError(t, err)
|
||||
require.True(t, q1.IsRunning())
|
||||
cancel1()
|
||||
q1.Stop()
|
||||
time.Sleep(50 * time.Millisecond)
|
||||
require.False(t, q1.IsRunning())
|
||||
|
||||
// Second cycle with fresh Quasar instance
|
||||
// Note: The current implementation closes stopCh on Stop and doesn't recreate it,
|
||||
// so restart requires a new instance. This is a known limitation.
|
||||
q2, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
q2.ConnectPChain(pchain)
|
||||
q2.ConnectQuantumFallback(&mockQuantumSigner{})
|
||||
|
||||
ctx2, cancel2 := context.WithCancel(context.Background())
|
||||
defer cancel2()
|
||||
err = q2.Start(ctx2)
|
||||
require.NoError(t, err)
|
||||
require.True(t, q2.IsRunning())
|
||||
q2.Stop()
|
||||
})
|
||||
|
||||
t.Run("stop_with_pending_events", func(t *testing.T) {
|
||||
validatorStates := generateValidatorStates(numValidators)
|
||||
pchain := newMockPChainProvider(validatorStates)
|
||||
defer pchain.Close()
|
||||
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
|
||||
q.ConnectPChain(pchain)
|
||||
q.ConnectQuantumFallback(&mockQuantumSigner{})
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
|
||||
err = q.Start(ctx)
|
||||
require.NoError(t, err)
|
||||
|
||||
// Emit events
|
||||
for i := uint64(1); i <= 10; i++ {
|
||||
event := createTestEvent(i, validatorStates)
|
||||
pchain.EmitFinality(event)
|
||||
}
|
||||
|
||||
// Stop immediately
|
||||
q.Stop()
|
||||
time.Sleep(50 * time.Millisecond)
|
||||
require.False(t, q.IsRunning(), "should stop cleanly with pending events")
|
||||
})
|
||||
|
||||
t.Run("multiple_stop_calls", func(t *testing.T) {
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
|
||||
validatorStates := generateValidatorStates(numValidators)
|
||||
pchain := newMockPChainProvider(validatorStates)
|
||||
defer pchain.Close()
|
||||
|
||||
q.ConnectPChain(pchain)
|
||||
q.ConnectQuantumFallback(&mockQuantumSigner{})
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
|
||||
err = q.Start(ctx)
|
||||
require.NoError(t, err)
|
||||
|
||||
// Multiple stops should not panic
|
||||
q.Stop()
|
||||
// Note: After first Stop, stopCh is closed. Subsequent Stop calls check running flag,
|
||||
// but since running=false, they won't try to close again. This tests idempotency.
|
||||
})
|
||||
|
||||
t.Run("stop_without_start", func(t *testing.T) {
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
|
||||
// Stop without start should not panic
|
||||
q.Stop()
|
||||
require.False(t, q.IsRunning())
|
||||
})
|
||||
}
|
||||
|
||||
// TestQuasarMemoryPressure tests with many finality events to verify no memory leaks
|
||||
func TestQuasarMemoryPressure(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("skipping memory pressure test in short mode")
|
||||
}
|
||||
|
||||
t.Run("many_finality_entries", func(t *testing.T) {
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
|
||||
// Add many finality entries
|
||||
const numEntries = 1000
|
||||
for i := 0; i < numEntries; i++ {
|
||||
var blockID ids.ID
|
||||
_, _ = rand.Read(blockID[:])
|
||||
|
||||
finality := &QuantumFinality{
|
||||
BlockID: blockID,
|
||||
PChainHeight: uint64(i),
|
||||
QChainHeight: uint64(i),
|
||||
TotalWeight: 1000,
|
||||
SignerWeight: 700,
|
||||
}
|
||||
q.SetFinalized(blockID, finality)
|
||||
}
|
||||
|
||||
stats := q.Stats()
|
||||
require.GreaterOrEqual(t, stats.FinalizedBlocks, numEntries-1)
|
||||
})
|
||||
|
||||
t.Run("memory_stability", func(t *testing.T) {
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
|
||||
// Add many entries
|
||||
const numEntries = 10000
|
||||
for i := 0; i < numEntries; i++ {
|
||||
var blockID ids.ID
|
||||
_, _ = rand.Read(blockID[:])
|
||||
|
||||
finality := &QuantumFinality{
|
||||
BlockID: blockID,
|
||||
PChainHeight: uint64(i),
|
||||
QChainHeight: uint64(i),
|
||||
BLSProof: make([]byte, 96),
|
||||
CoronaProof: make([]byte, 1024),
|
||||
TotalWeight: 1000,
|
||||
SignerWeight: 700,
|
||||
}
|
||||
q.SetFinalized(blockID, finality)
|
||||
}
|
||||
|
||||
// Force GC and check we don't crash
|
||||
runtime.GC()
|
||||
var m runtime.MemStats
|
||||
runtime.ReadMemStats(&m)
|
||||
|
||||
t.Logf("Heap after %d entries: %d bytes", numEntries, m.HeapAlloc)
|
||||
|
||||
// Just verify we completed without issues - memory testing is notoriously flaky
|
||||
stats := q.Stats()
|
||||
require.GreaterOrEqual(t, stats.FinalizedBlocks, numEntries-1)
|
||||
})
|
||||
|
||||
t.Run("concurrent_add_finality", func(t *testing.T) {
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
|
||||
const numGoroutines = 10
|
||||
const entriesPerGoroutine = 250
|
||||
|
||||
var wg sync.WaitGroup
|
||||
for g := 0; g < numGoroutines; g++ {
|
||||
wg.Add(1)
|
||||
go func(gid int) {
|
||||
defer wg.Done()
|
||||
for i := 0; i < entriesPerGoroutine; i++ {
|
||||
var blockID ids.ID
|
||||
_, _ = rand.Read(blockID[:])
|
||||
|
||||
finality := &QuantumFinality{
|
||||
BlockID: blockID,
|
||||
PChainHeight: uint64(gid*entriesPerGoroutine + i),
|
||||
QChainHeight: uint64(gid*entriesPerGoroutine + i),
|
||||
TotalWeight: 1000,
|
||||
SignerWeight: 700,
|
||||
}
|
||||
q.SetFinalized(blockID, finality)
|
||||
}
|
||||
}(g)
|
||||
}
|
||||
wg.Wait()
|
||||
|
||||
stats := q.Stats()
|
||||
t.Logf("Final entries after concurrent operations: %d", stats.FinalizedBlocks)
|
||||
// Some entries may share block IDs due to rand collision, so just verify we have many
|
||||
require.GreaterOrEqual(t, stats.FinalizedBlocks, numGoroutines*entriesPerGoroutine/2)
|
||||
})
|
||||
|
||||
t.Run("concurrent_read_write", func(t *testing.T) {
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
|
||||
// Pre-populate some entries
|
||||
blockIDs := make([]ids.ID, 100)
|
||||
for i := range blockIDs {
|
||||
_, _ = rand.Read(blockIDs[i][:])
|
||||
q.SetFinalized(blockIDs[i], &QuantumFinality{
|
||||
BlockID: blockIDs[i],
|
||||
PChainHeight: uint64(i),
|
||||
})
|
||||
}
|
||||
|
||||
// Concurrent reads and writes
|
||||
var wg sync.WaitGroup
|
||||
done := make(chan struct{})
|
||||
|
||||
// Writers
|
||||
for w := 0; w < 5; w++ {
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
for {
|
||||
select {
|
||||
case <-done:
|
||||
return
|
||||
default:
|
||||
var blockID ids.ID
|
||||
_, _ = rand.Read(blockID[:])
|
||||
q.SetFinalized(blockID, &QuantumFinality{BlockID: blockID})
|
||||
}
|
||||
}
|
||||
}()
|
||||
}
|
||||
|
||||
// Readers
|
||||
for r := 0; r < 5; r++ {
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
for {
|
||||
select {
|
||||
case <-done:
|
||||
return
|
||||
default:
|
||||
idx := int(time.Now().UnixNano()) % len(blockIDs)
|
||||
_, _ = q.GetFinality(blockIDs[idx])
|
||||
}
|
||||
}
|
||||
}()
|
||||
}
|
||||
|
||||
// Run for a short period
|
||||
time.Sleep(100 * time.Millisecond)
|
||||
close(done)
|
||||
wg.Wait()
|
||||
|
||||
t.Log("Concurrent read/write completed successfully")
|
||||
})
|
||||
}
|
||||
|
||||
// TestQuasarHealthStatus tests health status reporting
|
||||
func TestQuasarHealthStatus(t *testing.T) {
|
||||
t.Run("initial_state", func(t *testing.T) {
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
|
||||
stats := q.Stats()
|
||||
require.False(t, stats.Running)
|
||||
require.Equal(t, uint64(0), stats.PChainHeight)
|
||||
require.Equal(t, uint64(0), stats.QChainHeight)
|
||||
require.Equal(t, 0, stats.FinalizedBlocks)
|
||||
})
|
||||
|
||||
t.Run("after_corona_init", func(t *testing.T) {
|
||||
q, pchain, _, err := setupQuasarWithCorona(t, 5)
|
||||
if isLatticeUnavailable(err) {
|
||||
t.Skipf("Skipping: lattice library constraint: %v", err)
|
||||
}
|
||||
require.NoError(t, err)
|
||||
defer pchain.Close()
|
||||
|
||||
stats := q.Stats()
|
||||
require.True(t, stats.CoronaReady, "Corona should be ready")
|
||||
})
|
||||
|
||||
t.Run("running_state", func(t *testing.T) {
|
||||
validatorStates := generateValidatorStates(5)
|
||||
pchain := newMockPChainProvider(validatorStates)
|
||||
defer pchain.Close()
|
||||
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
|
||||
q.ConnectPChain(pchain)
|
||||
q.ConnectQuantumFallback(&mockQuantumSigner{})
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
|
||||
err = q.Start(ctx)
|
||||
require.NoError(t, err)
|
||||
defer q.Stop()
|
||||
|
||||
stats := q.Stats()
|
||||
require.True(t, stats.Running, "should be running")
|
||||
})
|
||||
}
|
||||
|
||||
// TestQuasarShutdown tests graceful shutdown behavior
|
||||
func TestQuasarShutdown(t *testing.T) {
|
||||
t.Run("graceful_stop_with_timeout", func(t *testing.T) {
|
||||
validatorStates := generateValidatorStates(5)
|
||||
pchain := newMockPChainProvider(validatorStates)
|
||||
defer pchain.Close()
|
||||
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
|
||||
q.ConnectPChain(pchain)
|
||||
q.ConnectQuantumFallback(&mockQuantumSigner{})
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
err = q.Start(ctx)
|
||||
require.NoError(t, err)
|
||||
|
||||
// Cancel context and stop
|
||||
cancel()
|
||||
q.Stop()
|
||||
|
||||
require.False(t, q.IsRunning())
|
||||
})
|
||||
|
||||
t.Run("stop_already_stopped", func(t *testing.T) {
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
|
||||
// Should not panic
|
||||
q.Stop()
|
||||
require.False(t, q.IsRunning())
|
||||
})
|
||||
|
||||
t.Run("start_after_stop", func(t *testing.T) {
|
||||
validatorStates := generateValidatorStates(5)
|
||||
pchain := newMockPChainProvider(validatorStates)
|
||||
defer pchain.Close()
|
||||
|
||||
// First run
|
||||
q1, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
q1.ConnectPChain(pchain)
|
||||
q1.ConnectQuantumFallback(&mockQuantumSigner{})
|
||||
|
||||
ctx1, cancel1 := context.WithCancel(context.Background())
|
||||
err = q1.Start(ctx1)
|
||||
require.NoError(t, err)
|
||||
cancel1()
|
||||
q1.Stop()
|
||||
|
||||
// Second run with new instance (implementation limitation)
|
||||
q2, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
q2.ConnectPChain(pchain)
|
||||
q2.ConnectQuantumFallback(&mockQuantumSigner{})
|
||||
|
||||
ctx2, cancel2 := context.WithCancel(context.Background())
|
||||
defer cancel2()
|
||||
err = q2.Start(ctx2)
|
||||
require.NoError(t, err)
|
||||
require.True(t, q2.IsRunning())
|
||||
q2.Stop()
|
||||
})
|
||||
|
||||
t.Run("health_status", func(t *testing.T) {
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
|
||||
// Check stats work without panic
|
||||
stats := q.Stats()
|
||||
require.NotNil(t, stats)
|
||||
})
|
||||
|
||||
t.Run("drain_finality_channel", func(t *testing.T) {
|
||||
validatorStates := generateValidatorStates(5)
|
||||
pchain := newMockPChainProvider(validatorStates)
|
||||
defer pchain.Close()
|
||||
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
|
||||
q.ConnectPChain(pchain)
|
||||
q.ConnectQuantumFallback(&mockQuantumSigner{})
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
|
||||
err = q.Start(ctx)
|
||||
require.NoError(t, err)
|
||||
|
||||
// Get finality channel via Subscribe
|
||||
finCh := q.Subscribe()
|
||||
require.NotNil(t, finCh)
|
||||
|
||||
// Emit event
|
||||
event := createTestEvent(1, validatorStates)
|
||||
pchain.EmitFinality(event)
|
||||
|
||||
// Try to receive finality (with timeout)
|
||||
select {
|
||||
case finality := <-finCh:
|
||||
require.NotNil(t, finality)
|
||||
case <-time.After(100 * time.Millisecond):
|
||||
// May not receive if processing takes longer
|
||||
}
|
||||
|
||||
q.Stop()
|
||||
})
|
||||
}
|
||||
|
||||
// TestQuasarEdgeCases tests edge cases and error conditions
|
||||
func TestQuasarEdgeCases(t *testing.T) {
|
||||
t.Run("start_without_pchain", func(t *testing.T) {
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
|
||||
err = q.Start(ctx)
|
||||
// Should handle gracefully (either error or run without processing)
|
||||
if err == nil {
|
||||
q.Stop()
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("start_without_quantum_fallback", func(t *testing.T) {
|
||||
validatorStates := generateValidatorStates(5)
|
||||
pchain := newMockPChainProvider(validatorStates)
|
||||
defer pchain.Close()
|
||||
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
|
||||
q.ConnectPChain(pchain)
|
||||
// No quantum fallback connected - Start requires it
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
|
||||
err = q.Start(ctx)
|
||||
// Implementation requires Q-Chain (quantum fallback) to be connected
|
||||
require.Error(t, err, "Start should error without quantum fallback")
|
||||
})
|
||||
|
||||
t.Run("get_finality_nonexistent", func(t *testing.T) {
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
|
||||
var blockID ids.ID
|
||||
_, _ = rand.Read(blockID[:])
|
||||
|
||||
finality, found := q.GetFinality(blockID)
|
||||
require.False(t, found, "should not find nonexistent block")
|
||||
require.Nil(t, finality, "should return nil for nonexistent block")
|
||||
})
|
||||
|
||||
t.Run("verify_nil_finality", func(t *testing.T) {
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
|
||||
err = q.Verify(nil)
|
||||
require.Error(t, err, "should error on nil finality")
|
||||
})
|
||||
|
||||
t.Run("verify_empty_proofs", func(t *testing.T) {
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
|
||||
finality := &QuantumFinality{
|
||||
BLSProof: nil,
|
||||
CoronaProof: nil,
|
||||
TotalWeight: 1000,
|
||||
SignerWeight: 700,
|
||||
}
|
||||
|
||||
err = q.Verify(finality)
|
||||
require.Error(t, err, "should error on empty proofs")
|
||||
})
|
||||
|
||||
t.Run("verify_insufficient_weight", func(t *testing.T) {
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
|
||||
finality := &QuantumFinality{
|
||||
BLSProof: []byte("proof"),
|
||||
CoronaProof: []byte("proof"),
|
||||
TotalWeight: 1000,
|
||||
SignerWeight: 500, // Only 50%, needs 67%
|
||||
}
|
||||
|
||||
err = q.Verify(finality)
|
||||
require.Error(t, err, "should error on insufficient weight")
|
||||
})
|
||||
|
||||
t.Run("create_message_format", func(t *testing.T) {
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
|
||||
validatorStates := generateValidatorStates(3)
|
||||
event := createTestEvent(42, validatorStates)
|
||||
|
||||
msg := q.CreateMessage(event)
|
||||
require.NotEmpty(t, msg, "message should not be empty")
|
||||
// Message is binary format containing blockID and height
|
||||
// Just verify it's deterministic and non-empty
|
||||
msg2 := q.CreateMessage(event)
|
||||
require.Equal(t, msg, msg2, "message should be deterministic")
|
||||
})
|
||||
|
||||
t.Run("total_weight_calculation", func(t *testing.T) {
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
|
||||
validators := []ValidatorState{
|
||||
{Weight: 100, Active: true},
|
||||
{Weight: 200, Active: true},
|
||||
{Weight: 300, Active: false}, // Inactive
|
||||
{Weight: 400, Active: true},
|
||||
}
|
||||
|
||||
total := q.TotalWeight(validators)
|
||||
require.Equal(t, uint64(700), total, "should sum only active validator weights")
|
||||
})
|
||||
}
|
||||
@@ -0,0 +1,195 @@
|
||||
// Copyright (C) 2019-2025, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
//go:build !cgo
|
||||
|
||||
// Package quasar provides NTT operations for Corona consensus.
|
||||
// This file provides pure Go CPU implementation when CGO is not available.
|
||||
// All operations use the luxfi/lattice library which provides optimized
|
||||
// NTT implementations in pure Go.
|
||||
package quasar
|
||||
|
||||
import (
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
|
||||
"github.com/luxfi/lattice/v7/ring"
|
||||
)
|
||||
|
||||
// NTTAccelerator provides NTT operations for Corona.
|
||||
// When CGO is disabled, this uses the pure Go lattice library
|
||||
// which provides optimized CPU-based NTT transforms.
|
||||
type NTTAccelerator struct {
|
||||
enabled bool
|
||||
stats NTTStats
|
||||
statsmu sync.RWMutex
|
||||
}
|
||||
|
||||
// NTTStats tracks NTT accelerator statistics.
|
||||
type NTTStats struct {
|
||||
Enabled bool
|
||||
Backend string
|
||||
TotalOps uint64
|
||||
GPUAvailable bool
|
||||
}
|
||||
|
||||
// NewNTTAccelerator creates a new NTT accelerator using pure Go lattice library.
|
||||
func NewNTTAccelerator() (*NTTAccelerator, error) {
|
||||
return &NTTAccelerator{
|
||||
enabled: true, // CPU implementation is always available
|
||||
stats: NTTStats{
|
||||
Enabled: true,
|
||||
Backend: "CPU (Pure Go)",
|
||||
},
|
||||
}, nil
|
||||
}
|
||||
|
||||
// IsEnabled returns true - CPU implementation is always available.
|
||||
func (g *NTTAccelerator) IsEnabled() bool {
|
||||
return true
|
||||
}
|
||||
|
||||
// Backend returns the backend name.
|
||||
func (g *NTTAccelerator) Backend() string {
|
||||
return "CPU (Pure Go lattice)"
|
||||
}
|
||||
|
||||
// NTTForward performs forward NTT on a polynomial using lattice library.
|
||||
func (g *NTTAccelerator) NTTForward(r *ring.Ring, poly ring.Poly) error {
|
||||
r.NTT(poly, poly)
|
||||
atomic.AddUint64(&g.stats.TotalOps, 1)
|
||||
return nil
|
||||
}
|
||||
|
||||
// NTTInverse performs inverse NTT on a polynomial using lattice library.
|
||||
func (g *NTTAccelerator) NTTInverse(r *ring.Ring, poly ring.Poly) error {
|
||||
r.INTT(poly, poly)
|
||||
atomic.AddUint64(&g.stats.TotalOps, 1)
|
||||
return nil
|
||||
}
|
||||
|
||||
// BatchNTTForward performs forward NTT on multiple polynomials.
|
||||
// Uses parallel processing for better performance on multi-core CPUs.
|
||||
func (g *NTTAccelerator) BatchNTTForward(r *ring.Ring, polys []ring.Poly) error {
|
||||
if len(polys) == 0 {
|
||||
return nil
|
||||
}
|
||||
|
||||
// For small batches, process sequentially
|
||||
if len(polys) < 8 {
|
||||
for _, poly := range polys {
|
||||
r.NTT(poly, poly)
|
||||
}
|
||||
atomic.AddUint64(&g.stats.TotalOps, uint64(len(polys)))
|
||||
return nil
|
||||
}
|
||||
|
||||
// For larger batches, use parallel processing
|
||||
var wg sync.WaitGroup
|
||||
numWorkers := 4
|
||||
chunkSize := (len(polys) + numWorkers - 1) / numWorkers
|
||||
|
||||
for i := 0; i < numWorkers; i++ {
|
||||
start := i * chunkSize
|
||||
end := start + chunkSize
|
||||
if end > len(polys) {
|
||||
end = len(polys)
|
||||
}
|
||||
if start >= end {
|
||||
break
|
||||
}
|
||||
|
||||
wg.Add(1)
|
||||
go func(batch []ring.Poly) {
|
||||
defer wg.Done()
|
||||
for _, poly := range batch {
|
||||
r.NTT(poly, poly)
|
||||
}
|
||||
}(polys[start:end])
|
||||
}
|
||||
wg.Wait()
|
||||
|
||||
atomic.AddUint64(&g.stats.TotalOps, uint64(len(polys)))
|
||||
return nil
|
||||
}
|
||||
|
||||
// BatchNTTInverse performs inverse NTT on multiple polynomials.
|
||||
// Uses parallel processing for better performance on multi-core CPUs.
|
||||
func (g *NTTAccelerator) BatchNTTInverse(r *ring.Ring, polys []ring.Poly) error {
|
||||
if len(polys) == 0 {
|
||||
return nil
|
||||
}
|
||||
|
||||
// For small batches, process sequentially
|
||||
if len(polys) < 8 {
|
||||
for _, poly := range polys {
|
||||
r.INTT(poly, poly)
|
||||
}
|
||||
atomic.AddUint64(&g.stats.TotalOps, uint64(len(polys)))
|
||||
return nil
|
||||
}
|
||||
|
||||
// For larger batches, use parallel processing
|
||||
var wg sync.WaitGroup
|
||||
numWorkers := 4
|
||||
chunkSize := (len(polys) + numWorkers - 1) / numWorkers
|
||||
|
||||
for i := 0; i < numWorkers; i++ {
|
||||
start := i * chunkSize
|
||||
end := start + chunkSize
|
||||
if end > len(polys) {
|
||||
end = len(polys)
|
||||
}
|
||||
if start >= end {
|
||||
break
|
||||
}
|
||||
|
||||
wg.Add(1)
|
||||
go func(batch []ring.Poly) {
|
||||
defer wg.Done()
|
||||
for _, poly := range batch {
|
||||
r.INTT(poly, poly)
|
||||
}
|
||||
}(polys[start:end])
|
||||
}
|
||||
wg.Wait()
|
||||
|
||||
atomic.AddUint64(&g.stats.TotalOps, uint64(len(polys)))
|
||||
return nil
|
||||
}
|
||||
|
||||
// PolyMul performs polynomial multiplication using Barrett reduction.
|
||||
func (g *NTTAccelerator) PolyMul(r *ring.Ring, a, b, out ring.Poly) error {
|
||||
r.MulCoeffsBarrett(a, b, out)
|
||||
atomic.AddUint64(&g.stats.TotalOps, 1)
|
||||
return nil
|
||||
}
|
||||
|
||||
// ClearCache is a no-op for CPU implementation (no GPU cache).
|
||||
func (g *NTTAccelerator) ClearCache() {}
|
||||
|
||||
// Stats returns current NTT accelerator statistics.
|
||||
func (g *NTTAccelerator) Stats() NTTStats {
|
||||
g.statsmu.RLock()
|
||||
defer g.statsmu.RUnlock()
|
||||
return NTTStats{
|
||||
Enabled: true,
|
||||
Backend: "CPU (Pure Go lattice)",
|
||||
TotalOps: atomic.LoadUint64(&g.stats.TotalOps),
|
||||
GPUAvailable: false, // CPU-only build
|
||||
}
|
||||
}
|
||||
|
||||
// Global accelerator instance
|
||||
var (
|
||||
globalNTTAccelerator *NTTAccelerator
|
||||
globalNTTAcceleratorOnce sync.Once
|
||||
)
|
||||
|
||||
// GetNTTAccelerator returns the global NTT accelerator instance.
|
||||
func GetNTTAccelerator() (*NTTAccelerator, error) {
|
||||
globalNTTAcceleratorOnce.Do(func() {
|
||||
globalNTTAccelerator, _ = NewNTTAccelerator()
|
||||
})
|
||||
return globalNTTAccelerator, nil
|
||||
}
|
||||
@@ -0,0 +1,469 @@
|
||||
// Copyright (C) 2019-2025, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
//go:build cgo
|
||||
|
||||
// Package quasar provides GPU-accelerated NTT operations for Corona consensus.
|
||||
// This uses the unified lux/accel package for GPU acceleration of lattice
|
||||
// operations in the Corona threshold signature protocol.
|
||||
//
|
||||
// GPU acceleration provides 40x+ speedup for NTT operations on Apple Silicon
|
||||
// and NVIDIA GPUs via the accel library (Metal/CUDA/CPU backends).
|
||||
//
|
||||
// Architecture:
|
||||
//
|
||||
// luxcpp/accel (C++ GPU) → lux/accel (Go CGO) → Quasar consensus
|
||||
//
|
||||
// This enables consistent GPU acceleration across:
|
||||
// - Corona threshold signatures
|
||||
// - ML-DSA post-quantum signatures
|
||||
// - FHE operations (via luxcpp/fhe which reuses luxcpp/lattice)
|
||||
package quasar
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
|
||||
"github.com/luxfi/accel"
|
||||
"github.com/luxfi/lattice/v7/ring"
|
||||
"github.com/luxfi/node/config"
|
||||
)
|
||||
|
||||
// NTTAccelerator provides GPU-accelerated NTT operations for Corona.
|
||||
// It uses the unified lux/accel package for Metal/CUDA/CPU backends.
|
||||
type NTTAccelerator struct {
|
||||
mu sync.RWMutex
|
||||
session *accel.Session
|
||||
enabled bool
|
||||
totalOps uint64
|
||||
}
|
||||
|
||||
// NTTOptions holds options for creating an NTT accelerator.
|
||||
type NTTOptions struct {
|
||||
// Enabled controls whether GPU acceleration is used
|
||||
Enabled bool
|
||||
// Backend specifies which GPU backend to use: "auto", "metal", "cuda", "cpu"
|
||||
Backend string
|
||||
// DeviceIndex specifies which GPU device to use
|
||||
DeviceIndex int
|
||||
}
|
||||
|
||||
// NewNTTAccelerator creates a new NTT accelerator with GPU support.
|
||||
// It auto-detects available GPU backends (Metal on macOS, CUDA on Linux).
|
||||
func NewNTTAccelerator() (*NTTAccelerator, error) {
|
||||
return NewNTTAcceleratorWithOptions(NTTOptions{})
|
||||
}
|
||||
|
||||
// NewNTTAcceleratorWithOptions creates a new NTT accelerator with custom options.
|
||||
// If options are zero-valued, it uses the global GPU config.
|
||||
func NewNTTAcceleratorWithOptions(opts NTTOptions) (*NTTAccelerator, error) {
|
||||
// Get global config if options not specified
|
||||
gpuCfg := config.GetGlobalGPUConfig()
|
||||
|
||||
// Determine if GPU should be enabled
|
||||
enabled := gpuCfg.Enabled
|
||||
if opts.Backend == "cpu" {
|
||||
enabled = false
|
||||
}
|
||||
|
||||
// Check if GPU is available via accel library
|
||||
available := accel.Available() && enabled
|
||||
|
||||
var session *accel.Session
|
||||
if available {
|
||||
var err error
|
||||
session, err = accel.DefaultSession()
|
||||
if err != nil {
|
||||
// Fall back to CPU mode
|
||||
available = false
|
||||
}
|
||||
}
|
||||
|
||||
return &NTTAccelerator{
|
||||
session: session,
|
||||
enabled: available,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// IsEnabled returns whether GPU acceleration is available.
|
||||
func (g *NTTAccelerator) IsEnabled() bool {
|
||||
g.mu.RLock()
|
||||
defer g.mu.RUnlock()
|
||||
return g.enabled
|
||||
}
|
||||
|
||||
// Backend returns the name of the active GPU backend.
|
||||
func (g *NTTAccelerator) Backend() string {
|
||||
g.mu.RLock()
|
||||
defer g.mu.RUnlock()
|
||||
|
||||
if !g.enabled || g.session == nil {
|
||||
return "CPU (GPU not available)"
|
||||
}
|
||||
return g.session.Backend().String()
|
||||
}
|
||||
|
||||
// getModulus extracts the first modulus from the ring.
|
||||
func (g *NTTAccelerator) getModulus(r *ring.Ring) (uint32, error) {
|
||||
if len(r.ModuliChain()) == 0 {
|
||||
return 0, fmt.Errorf("ring has no moduli")
|
||||
}
|
||||
return uint32(r.ModuliChain()[0]), nil
|
||||
}
|
||||
|
||||
// NTTForward performs forward NTT on a polynomial using GPU acceleration.
|
||||
// Falls back to CPU if GPU is not available.
|
||||
func (g *NTTAccelerator) NTTForward(r *ring.Ring, poly ring.Poly) error {
|
||||
if !g.enabled || g.session == nil {
|
||||
// Fall back to lattice library's NTT
|
||||
r.NTT(poly, poly)
|
||||
return nil
|
||||
}
|
||||
|
||||
N := r.N()
|
||||
coeffs := poly.Coeffs
|
||||
if len(coeffs) == 0 || len(coeffs[0]) < N {
|
||||
r.NTT(poly, poly)
|
||||
return nil
|
||||
}
|
||||
|
||||
Q, err := g.getModulus(r)
|
||||
if err != nil {
|
||||
r.NTT(poly, poly)
|
||||
return nil
|
||||
}
|
||||
|
||||
// Create input tensor from polynomial coefficients
|
||||
inputTensor, err := accel.NewTensorWithData[uint64](g.session, []int{N}, coeffs[0][:N])
|
||||
if err != nil {
|
||||
r.NTT(poly, poly)
|
||||
return nil
|
||||
}
|
||||
defer inputTensor.Close()
|
||||
|
||||
// Create output tensor
|
||||
outputTensor, err := accel.NewTensor[uint64](g.session, []int{N})
|
||||
if err != nil {
|
||||
r.NTT(poly, poly)
|
||||
return nil
|
||||
}
|
||||
defer outputTensor.Close()
|
||||
|
||||
// GPU NTT via accel Lattice ops
|
||||
if err := g.session.Lattice().PolynomialNTT(inputTensor.Untyped(), outputTensor.Untyped(), Q); err != nil {
|
||||
r.NTT(poly, poly)
|
||||
return nil
|
||||
}
|
||||
|
||||
// Copy result back
|
||||
result, err := outputTensor.ToSlice()
|
||||
if err != nil {
|
||||
r.NTT(poly, poly)
|
||||
return nil
|
||||
}
|
||||
copy(coeffs[0], result)
|
||||
atomic.AddUint64(&g.totalOps, 1)
|
||||
return nil
|
||||
}
|
||||
|
||||
// NTTInverse performs inverse NTT on a polynomial using GPU acceleration.
|
||||
// Falls back to CPU if GPU is not available.
|
||||
func (g *NTTAccelerator) NTTInverse(r *ring.Ring, poly ring.Poly) error {
|
||||
if !g.enabled || g.session == nil {
|
||||
// Fall back to lattice library's INTT
|
||||
r.INTT(poly, poly)
|
||||
return nil
|
||||
}
|
||||
|
||||
N := r.N()
|
||||
coeffs := poly.Coeffs
|
||||
if len(coeffs) == 0 || len(coeffs[0]) < N {
|
||||
r.INTT(poly, poly)
|
||||
return nil
|
||||
}
|
||||
|
||||
Q, err := g.getModulus(r)
|
||||
if err != nil {
|
||||
r.INTT(poly, poly)
|
||||
return nil
|
||||
}
|
||||
|
||||
// Create input tensor from polynomial coefficients
|
||||
inputTensor, err := accel.NewTensorWithData[uint64](g.session, []int{N}, coeffs[0][:N])
|
||||
if err != nil {
|
||||
r.INTT(poly, poly)
|
||||
return nil
|
||||
}
|
||||
defer inputTensor.Close()
|
||||
|
||||
// Create output tensor
|
||||
outputTensor, err := accel.NewTensor[uint64](g.session, []int{N})
|
||||
if err != nil {
|
||||
r.INTT(poly, poly)
|
||||
return nil
|
||||
}
|
||||
defer outputTensor.Close()
|
||||
|
||||
// GPU INTT via accel Lattice ops
|
||||
if err := g.session.Lattice().PolynomialINTT(inputTensor.Untyped(), outputTensor.Untyped(), Q); err != nil {
|
||||
r.INTT(poly, poly)
|
||||
return nil
|
||||
}
|
||||
|
||||
// Copy result back
|
||||
result, err := outputTensor.ToSlice()
|
||||
if err != nil {
|
||||
r.INTT(poly, poly)
|
||||
return nil
|
||||
}
|
||||
copy(coeffs[0], result)
|
||||
atomic.AddUint64(&g.totalOps, 1)
|
||||
return nil
|
||||
}
|
||||
|
||||
// BatchNTTForward performs forward NTT on multiple polynomials in parallel.
|
||||
// This is the primary use case for GPU acceleration - batch operations.
|
||||
func (g *NTTAccelerator) BatchNTTForward(r *ring.Ring, polys []ring.Poly) error {
|
||||
if len(polys) == 0 {
|
||||
return nil
|
||||
}
|
||||
|
||||
if !g.enabled || g.session == nil || len(polys) < 4 {
|
||||
// Fall back to CPU for small batches (GPU overhead not worth it)
|
||||
for i := range polys {
|
||||
r.NTT(polys[i], polys[i])
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
Q, err := g.getModulus(r)
|
||||
if err != nil {
|
||||
for i := range polys {
|
||||
r.NTT(polys[i], polys[i])
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Process each polynomial through GPU
|
||||
// Note: For true batch performance, we'd want batch tensor operations
|
||||
// but the current accel API operates on single polynomials
|
||||
N := r.N()
|
||||
for i := range polys {
|
||||
coeffs := polys[i].Coeffs
|
||||
if len(coeffs) == 0 || len(coeffs[0]) < N {
|
||||
r.NTT(polys[i], polys[i])
|
||||
continue
|
||||
}
|
||||
|
||||
inputTensor, err := accel.NewTensorWithData[uint64](g.session, []int{N}, coeffs[0][:N])
|
||||
if err != nil {
|
||||
r.NTT(polys[i], polys[i])
|
||||
continue
|
||||
}
|
||||
|
||||
outputTensor, err := accel.NewTensor[uint64](g.session, []int{N})
|
||||
if err != nil {
|
||||
inputTensor.Close()
|
||||
r.NTT(polys[i], polys[i])
|
||||
continue
|
||||
}
|
||||
|
||||
if err := g.session.Lattice().PolynomialNTT(inputTensor.Untyped(), outputTensor.Untyped(), Q); err != nil {
|
||||
inputTensor.Close()
|
||||
outputTensor.Close()
|
||||
r.NTT(polys[i], polys[i])
|
||||
continue
|
||||
}
|
||||
|
||||
result, err := outputTensor.ToSlice()
|
||||
if err != nil {
|
||||
inputTensor.Close()
|
||||
outputTensor.Close()
|
||||
r.NTT(polys[i], polys[i])
|
||||
continue
|
||||
}
|
||||
copy(coeffs[0], result)
|
||||
|
||||
inputTensor.Close()
|
||||
outputTensor.Close()
|
||||
}
|
||||
|
||||
atomic.AddUint64(&g.totalOps, uint64(len(polys)))
|
||||
return nil
|
||||
}
|
||||
|
||||
// BatchNTTInverse performs inverse NTT on multiple polynomials in parallel.
|
||||
func (g *NTTAccelerator) BatchNTTInverse(r *ring.Ring, polys []ring.Poly) error {
|
||||
if len(polys) == 0 {
|
||||
return nil
|
||||
}
|
||||
|
||||
if !g.enabled || g.session == nil || len(polys) < 4 {
|
||||
// Fall back to CPU for small batches
|
||||
for i := range polys {
|
||||
r.INTT(polys[i], polys[i])
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
Q, err := g.getModulus(r)
|
||||
if err != nil {
|
||||
for i := range polys {
|
||||
r.INTT(polys[i], polys[i])
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Process each polynomial through GPU
|
||||
N := r.N()
|
||||
for i := range polys {
|
||||
coeffs := polys[i].Coeffs
|
||||
if len(coeffs) == 0 || len(coeffs[0]) < N {
|
||||
r.INTT(polys[i], polys[i])
|
||||
continue
|
||||
}
|
||||
|
||||
inputTensor, err := accel.NewTensorWithData[uint64](g.session, []int{N}, coeffs[0][:N])
|
||||
if err != nil {
|
||||
r.INTT(polys[i], polys[i])
|
||||
continue
|
||||
}
|
||||
|
||||
outputTensor, err := accel.NewTensor[uint64](g.session, []int{N})
|
||||
if err != nil {
|
||||
inputTensor.Close()
|
||||
r.INTT(polys[i], polys[i])
|
||||
continue
|
||||
}
|
||||
|
||||
if err := g.session.Lattice().PolynomialINTT(inputTensor.Untyped(), outputTensor.Untyped(), Q); err != nil {
|
||||
inputTensor.Close()
|
||||
outputTensor.Close()
|
||||
r.INTT(polys[i], polys[i])
|
||||
continue
|
||||
}
|
||||
|
||||
result, err := outputTensor.ToSlice()
|
||||
if err != nil {
|
||||
inputTensor.Close()
|
||||
outputTensor.Close()
|
||||
r.INTT(polys[i], polys[i])
|
||||
continue
|
||||
}
|
||||
copy(coeffs[0], result)
|
||||
|
||||
inputTensor.Close()
|
||||
outputTensor.Close()
|
||||
}
|
||||
|
||||
atomic.AddUint64(&g.totalOps, uint64(len(polys)))
|
||||
return nil
|
||||
}
|
||||
|
||||
// PolyMul performs polynomial multiplication using GPU-accelerated NTT.
|
||||
// This multiplies polynomials a and b, storing result in out.
|
||||
func (g *NTTAccelerator) PolyMul(r *ring.Ring, a, b, out ring.Poly) error {
|
||||
if !g.enabled || g.session == nil {
|
||||
// Fall back to CPU
|
||||
r.MulCoeffsBarrett(a, b, out)
|
||||
return nil
|
||||
}
|
||||
|
||||
Q, err := g.getModulus(r)
|
||||
if err != nil {
|
||||
r.MulCoeffsBarrett(a, b, out)
|
||||
return nil
|
||||
}
|
||||
|
||||
N := r.N()
|
||||
|
||||
// Extract coefficients
|
||||
if len(a.Coeffs) == 0 || len(a.Coeffs[0]) < N ||
|
||||
len(b.Coeffs) == 0 || len(b.Coeffs[0]) < N ||
|
||||
len(out.Coeffs) == 0 || len(out.Coeffs[0]) < N {
|
||||
r.MulCoeffsBarrett(a, b, out)
|
||||
return nil
|
||||
}
|
||||
|
||||
// Create tensors for a, b, and output
|
||||
aTensor, err := accel.NewTensorWithData[uint64](g.session, []int{N}, a.Coeffs[0][:N])
|
||||
if err != nil {
|
||||
r.MulCoeffsBarrett(a, b, out)
|
||||
return nil
|
||||
}
|
||||
defer aTensor.Close()
|
||||
|
||||
bTensor, err := accel.NewTensorWithData[uint64](g.session, []int{N}, b.Coeffs[0][:N])
|
||||
if err != nil {
|
||||
r.MulCoeffsBarrett(a, b, out)
|
||||
return nil
|
||||
}
|
||||
defer bTensor.Close()
|
||||
|
||||
outTensor, err := accel.NewTensor[uint64](g.session, []int{N})
|
||||
if err != nil {
|
||||
r.MulCoeffsBarrett(a, b, out)
|
||||
return nil
|
||||
}
|
||||
defer outTensor.Close()
|
||||
|
||||
// GPU polynomial multiplication
|
||||
if err := g.session.Lattice().PolynomialMul(aTensor.Untyped(), bTensor.Untyped(), outTensor.Untyped(), Q); err != nil {
|
||||
r.MulCoeffsBarrett(a, b, out)
|
||||
return nil
|
||||
}
|
||||
|
||||
// Copy result back
|
||||
result, err := outTensor.ToSlice()
|
||||
if err != nil {
|
||||
r.MulCoeffsBarrett(a, b, out)
|
||||
return nil
|
||||
}
|
||||
copy(out.Coeffs[0], result)
|
||||
atomic.AddUint64(&g.totalOps, 1)
|
||||
return nil
|
||||
}
|
||||
|
||||
// ClearCache is a no-op in the accel-based implementation.
|
||||
// The accel library manages its own caching internally.
|
||||
func (g *NTTAccelerator) ClearCache() {
|
||||
// No-op: accel library manages caching internally
|
||||
}
|
||||
|
||||
// NTTStats returns NTT accelerator statistics.
|
||||
type NTTStats struct {
|
||||
Enabled bool
|
||||
Backend string
|
||||
TotalOps uint64
|
||||
GPUAvailable bool
|
||||
}
|
||||
|
||||
// Stats returns current NTT accelerator statistics.
|
||||
func (g *NTTAccelerator) Stats() NTTStats {
|
||||
g.mu.RLock()
|
||||
defer g.mu.RUnlock()
|
||||
|
||||
return NTTStats{
|
||||
Enabled: g.enabled,
|
||||
Backend: g.Backend(),
|
||||
TotalOps: atomic.LoadUint64(&g.totalOps),
|
||||
GPUAvailable: accel.Available(),
|
||||
}
|
||||
}
|
||||
|
||||
// Global NTT accelerator instance (lazily initialized)
|
||||
var (
|
||||
globalNTTAccelerator *NTTAccelerator
|
||||
globalNTTAcceleratorOnce sync.Once
|
||||
globalNTTAcceleratorErr error
|
||||
)
|
||||
|
||||
// GetNTTAccelerator returns the global NTT accelerator instance.
|
||||
// The accelerator is lazily initialized on first call.
|
||||
func GetNTTAccelerator() (*NTTAccelerator, error) {
|
||||
globalNTTAcceleratorOnce.Do(func() {
|
||||
globalNTTAccelerator, globalNTTAcceleratorErr = NewNTTAccelerator()
|
||||
})
|
||||
return globalNTTAccelerator, globalNTTAcceleratorErr
|
||||
}
|
||||
@@ -0,0 +1,448 @@
|
||||
// Copyright (C) 2019-2025, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
package quasar
|
||||
|
||||
import (
|
||||
"crypto/rand"
|
||||
"runtime"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/luxfi/ids"
|
||||
"github.com/luxfi/log"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Test 1: Finalized map pruning
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// TestFinalizedMapPruning simulates 100,000 finality events and verifies:
|
||||
// - The finalized map never exceeds maxFinalized + buffer
|
||||
// - Old entries are actually pruned
|
||||
// - After 100K events, map size <= 10,000
|
||||
func TestFinalizedMapPruning(t *testing.T) {
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
|
||||
// maxFinalized is 10,000 by default (set in NewQuasar)
|
||||
const totalEvents = 100_000
|
||||
|
||||
// We simulate processFinality's pruning logic directly by
|
||||
// inserting entries and triggering the prune path.
|
||||
// processFinality increments qHeight and prunes when len > maxFinalized.
|
||||
var peakSize int
|
||||
for i := 0; i < totalEvents; i++ {
|
||||
var blockID ids.ID
|
||||
_, _ = rand.Read(blockID[:])
|
||||
|
||||
q.mu.Lock()
|
||||
q.qHeight++
|
||||
q.finalized[blockID] = &QuantumFinality{
|
||||
BlockID: blockID,
|
||||
QChainHeight: q.qHeight,
|
||||
PChainHeight: uint64(i),
|
||||
TotalWeight: 1000,
|
||||
SignerWeight: 700,
|
||||
}
|
||||
|
||||
// Replicate the pruning logic from processFinality
|
||||
if q.maxFinalized > 0 && len(q.finalized) > q.maxFinalized {
|
||||
cutoff := q.qHeight - uint64(q.maxFinalized)
|
||||
for id, f := range q.finalized {
|
||||
if f.QChainHeight < cutoff {
|
||||
delete(q.finalized, id)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
size := len(q.finalized)
|
||||
if size > peakSize {
|
||||
peakSize = size
|
||||
}
|
||||
q.mu.Unlock()
|
||||
}
|
||||
|
||||
q.mu.RLock()
|
||||
finalSize := len(q.finalized)
|
||||
finalQHeight := q.qHeight
|
||||
q.mu.RUnlock()
|
||||
|
||||
t.Logf("totalEvents=%d peakSize=%d finalSize=%d qHeight=%d",
|
||||
totalEvents, peakSize, finalSize, finalQHeight)
|
||||
|
||||
// The pruning logic fires when len > maxFinalized, then deletes entries
|
||||
// with QChainHeight < cutoff (strict <). The cutoff is qHeight - maxFinalized.
|
||||
// After pruning, entries at exactly cutoff remain, so the steady-state
|
||||
// size is maxFinalized + 1. This is correct and bounded.
|
||||
require.LessOrEqual(t, peakSize, q.maxFinalized+1,
|
||||
"peak map size should not exceed maxFinalized+1")
|
||||
|
||||
require.LessOrEqual(t, finalSize, q.maxFinalized+1,
|
||||
"final map size should be <= maxFinalized+1 (10,001)")
|
||||
|
||||
// Verify old entries are actually gone: the oldest remaining entry
|
||||
// should have QChainHeight >= qHeight - maxFinalized.
|
||||
q.mu.RLock()
|
||||
minHeight := uint64(^uint64(0))
|
||||
for _, f := range q.finalized {
|
||||
if f.QChainHeight < minHeight {
|
||||
minHeight = f.QChainHeight
|
||||
}
|
||||
}
|
||||
q.mu.RUnlock()
|
||||
|
||||
expectedMinHeight := finalQHeight - uint64(q.maxFinalized)
|
||||
require.GreaterOrEqual(t, minHeight, expectedMinHeight,
|
||||
"oldest entry should be pruned: minHeight=%d expected>=%d", minHeight, expectedMinHeight)
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Test 2: Channel backpressure -- no goroutine leak or deadlock
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// TestChannelBackpressure creates a pChainProvider with a 64-buffer channel,
|
||||
// sends 1000 events, and verifies no goroutine leak or deadlock.
|
||||
func TestChannelBackpressure(t *testing.T) {
|
||||
const (
|
||||
channelSize = 64
|
||||
totalEvents = 1000
|
||||
)
|
||||
|
||||
validators := generateValidatorStates(5)
|
||||
pchain := &mockPChainProvider{
|
||||
height: 0,
|
||||
validators: validators,
|
||||
finalityCh: make(chan FinalityEvent, channelSize),
|
||||
}
|
||||
|
||||
goroutinesBefore := runtime.NumGoroutine()
|
||||
|
||||
// Send events -- channel will fill up, excess events are dropped (select default)
|
||||
sent := 0
|
||||
for i := 0; i < totalEvents; i++ {
|
||||
event := createTestEvent(uint64(i+1), validators)
|
||||
select {
|
||||
case pchain.finalityCh <- event:
|
||||
sent++
|
||||
default:
|
||||
// Channel full -- expected backpressure behavior
|
||||
}
|
||||
}
|
||||
|
||||
t.Logf("sent %d/%d events (channel capacity %d)", sent, totalEvents, channelSize)
|
||||
require.GreaterOrEqual(t, sent, channelSize,
|
||||
"should have sent at least channelSize events")
|
||||
|
||||
// Drain the channel
|
||||
drained := 0
|
||||
for {
|
||||
select {
|
||||
case <-pchain.finalityCh:
|
||||
drained++
|
||||
default:
|
||||
goto done
|
||||
}
|
||||
}
|
||||
done:
|
||||
t.Logf("drained %d events", drained)
|
||||
|
||||
// Check goroutine count -- should not have leaked
|
||||
runtime.GC()
|
||||
goroutinesAfter := runtime.NumGoroutine()
|
||||
// Allow a delta of 5 for GC/runtime goroutines
|
||||
require.InDelta(t, goroutinesBefore, goroutinesAfter, 5,
|
||||
"goroutine count should not grow significantly: before=%d after=%d",
|
||||
goroutinesBefore, goroutinesAfter)
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Test 3: Long-running benchmark -- 1M simulated finality cycles
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// BenchmarkQuasarLongRun runs 1M simulated finality cycles and reports
|
||||
// allocs/op, bytes/op, ns/op. Verifies no unbounded memory growth.
|
||||
func BenchmarkQuasarLongRun(b *testing.B) {
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
if err != nil {
|
||||
b.Fatal(err)
|
||||
}
|
||||
|
||||
// Snapshot heap before
|
||||
runtime.GC()
|
||||
var memBefore runtime.MemStats
|
||||
runtime.ReadMemStats(&memBefore)
|
||||
|
||||
b.ResetTimer()
|
||||
b.ReportAllocs()
|
||||
|
||||
for i := 0; i < b.N; i++ {
|
||||
var blockID ids.ID
|
||||
// Use deterministic IDs to avoid crypto/rand overhead in benchmark
|
||||
blockID[0] = byte(i)
|
||||
blockID[1] = byte(i >> 8)
|
||||
blockID[2] = byte(i >> 16)
|
||||
blockID[3] = byte(i >> 24)
|
||||
|
||||
q.mu.Lock()
|
||||
q.qHeight++
|
||||
q.finalized[blockID] = &QuantumFinality{
|
||||
BlockID: blockID,
|
||||
QChainHeight: q.qHeight,
|
||||
PChainHeight: uint64(i),
|
||||
TotalWeight: 1000,
|
||||
SignerWeight: 700,
|
||||
BLSProof: make([]byte, 96),
|
||||
Timestamp: time.Now(),
|
||||
}
|
||||
|
||||
// Prune (same logic as processFinality)
|
||||
if q.maxFinalized > 0 && len(q.finalized) > q.maxFinalized {
|
||||
cutoff := q.qHeight - uint64(q.maxFinalized)
|
||||
for id, f := range q.finalized {
|
||||
if f.QChainHeight < cutoff {
|
||||
delete(q.finalized, id)
|
||||
}
|
||||
}
|
||||
}
|
||||
q.mu.Unlock()
|
||||
}
|
||||
|
||||
b.StopTimer()
|
||||
|
||||
// Snapshot heap after
|
||||
runtime.GC()
|
||||
var memAfter runtime.MemStats
|
||||
runtime.ReadMemStats(&memAfter)
|
||||
|
||||
q.mu.RLock()
|
||||
finalSize := len(q.finalized)
|
||||
q.mu.RUnlock()
|
||||
|
||||
heapBefore := memBefore.HeapAlloc
|
||||
heapAfter := memAfter.HeapAlloc
|
||||
|
||||
var heapDelta int64
|
||||
if heapAfter >= heapBefore {
|
||||
heapDelta = int64(heapAfter - heapBefore)
|
||||
} else {
|
||||
heapDelta = -int64(heapBefore - heapAfter)
|
||||
}
|
||||
b.Logf("N=%d finalMapSize=%d heapBefore=%d heapAfter=%d heapDelta=%d",
|
||||
b.N, finalSize, heapBefore, heapAfter, heapDelta)
|
||||
|
||||
// Map should be bounded regardless of N
|
||||
if finalSize > q.maxFinalized+1 {
|
||||
b.Fatalf("unbounded growth: map size %d exceeds maxFinalized %d",
|
||||
finalSize, q.maxFinalized)
|
||||
}
|
||||
|
||||
// Heap should not grow linearly with N. After pruning, the heap should
|
||||
// be bounded by ~maxFinalized entries worth of allocations.
|
||||
// We allow 100MB as a generous upper bound for 10K entries with 96-byte proofs.
|
||||
const maxHeapGrowth = 100 * 1024 * 1024 // 100MB
|
||||
if heapAfter > heapBefore+maxHeapGrowth {
|
||||
b.Fatalf("unbounded heap growth: before=%d after=%d delta=%d (limit=%d)",
|
||||
heapBefore, heapAfter, heapAfter-heapBefore, maxHeapGrowth)
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Test 4: Quorum math verification -- property test
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// TestQuorumMath is a property test that for all validator counts 1-100
|
||||
// and all weight distributions:
|
||||
// - Cross-multiplication quorum never accepts < 2/3 weight
|
||||
// - Cross-multiplication quorum always accepts >= 2/3 weight (no false negatives)
|
||||
func TestQuorumMath(t *testing.T) {
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
|
||||
// The quorum check is: signerWeight * quorumDen >= totalWeight * quorumNum
|
||||
// With quorumNum=2, quorumDen=3: signerWeight * 3 >= totalWeight * 2
|
||||
|
||||
t.Run("no_false_positives", func(t *testing.T) {
|
||||
// For all validator counts and weight distributions, if the quorum
|
||||
// check passes, then signerWeight/totalWeight >= 2/3.
|
||||
for numValidators := 1; numValidators <= 100; numValidators++ {
|
||||
// Test with equal weights
|
||||
weightPerValidator := uint64(1000)
|
||||
totalWeight := uint64(numValidators) * weightPerValidator
|
||||
|
||||
for numSigners := 0; numSigners <= numValidators; numSigners++ {
|
||||
signerWeight := uint64(numSigners) * weightPerValidator
|
||||
result := q.CheckQuorum(signerWeight, totalWeight)
|
||||
|
||||
// Verify: if result is true, then signerWeight/totalWeight >= 2/3
|
||||
// Using cross-multiplication: signerWeight * 3 >= totalWeight * 2
|
||||
actualMeetsThreshold := signerWeight*3 >= totalWeight*2
|
||||
if result && !actualMeetsThreshold {
|
||||
t.Fatalf("FALSE POSITIVE: n=%d signers=%d sW=%d tW=%d: "+
|
||||
"quorum accepted but %d*3=%d < %d*2=%d",
|
||||
numValidators, numSigners, signerWeight, totalWeight,
|
||||
signerWeight, signerWeight*3, totalWeight, totalWeight*2)
|
||||
}
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("no_false_negatives", func(t *testing.T) {
|
||||
// For all validator counts and weight distributions, if
|
||||
// signerWeight/totalWeight >= 2/3, the quorum check must pass.
|
||||
for numValidators := 1; numValidators <= 100; numValidators++ {
|
||||
weightPerValidator := uint64(1000)
|
||||
totalWeight := uint64(numValidators) * weightPerValidator
|
||||
|
||||
for numSigners := 0; numSigners <= numValidators; numSigners++ {
|
||||
signerWeight := uint64(numSigners) * weightPerValidator
|
||||
result := q.CheckQuorum(signerWeight, totalWeight)
|
||||
|
||||
actualMeetsThreshold := signerWeight*3 >= totalWeight*2
|
||||
if actualMeetsThreshold && !result {
|
||||
t.Fatalf("FALSE NEGATIVE: n=%d signers=%d sW=%d tW=%d: "+
|
||||
"threshold met but quorum rejected",
|
||||
numValidators, numSigners, signerWeight, totalWeight)
|
||||
}
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("varied_weight_distributions", func(t *testing.T) {
|
||||
// Test with non-uniform weights: validators have weights 1..n
|
||||
for numValidators := 1; numValidators <= 100; numValidators++ {
|
||||
var totalWeight uint64
|
||||
weights := make([]uint64, numValidators)
|
||||
for i := 0; i < numValidators; i++ {
|
||||
weights[i] = uint64(i + 1)
|
||||
totalWeight += weights[i]
|
||||
}
|
||||
|
||||
// Test subsets: first k validators sign
|
||||
var signerWeight uint64
|
||||
for k := 0; k <= numValidators; k++ {
|
||||
if k > 0 {
|
||||
signerWeight += weights[k-1]
|
||||
}
|
||||
result := q.CheckQuorum(signerWeight, totalWeight)
|
||||
expected := signerWeight*3 >= totalWeight*2
|
||||
|
||||
if result != expected {
|
||||
t.Fatalf("MISMATCH: n=%d signers=%d sW=%d tW=%d: "+
|
||||
"got %v expected %v",
|
||||
numValidators, k, signerWeight, totalWeight,
|
||||
result, expected)
|
||||
}
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("edge_cases", func(t *testing.T) {
|
||||
// Zero total weight: any signer weight passes (vacuously true)
|
||||
require.True(t, q.CheckQuorum(0, 0), "0/0 should pass (vacuous)")
|
||||
require.True(t, q.CheckQuorum(1, 0), "1/0 should pass (vacuous)")
|
||||
|
||||
// Exact boundary: 2/3 of various totals
|
||||
// For totalWeight=3: need signerWeight >= 2
|
||||
require.True(t, q.CheckQuorum(2, 3), "2/3 should pass")
|
||||
require.False(t, q.CheckQuorum(1, 3), "1/3 should fail")
|
||||
|
||||
// For totalWeight=6: need signerWeight >= 4
|
||||
require.True(t, q.CheckQuorum(4, 6), "4/6 should pass")
|
||||
require.False(t, q.CheckQuorum(3, 6), "3/6 should fail")
|
||||
|
||||
// For totalWeight=9: need signerWeight >= 6
|
||||
require.True(t, q.CheckQuorum(6, 9), "6/9 should pass")
|
||||
require.False(t, q.CheckQuorum(5, 9), "5/9 should fail")
|
||||
|
||||
// For totalWeight=100: 2*100/3 = 66 (floor), so need 67 to be strictly >= 2/3
|
||||
// But cross-mult: sW*3 >= tW*2 → sW*3 >= 200 → sW >= 67 (ceil)
|
||||
// Actually: 66*3=198 < 200 → fail; 67*3=201 >= 200 → pass
|
||||
require.True(t, q.CheckQuorum(67, 100), "67/100 should pass")
|
||||
require.False(t, q.CheckQuorum(66, 100), "66/100 should fail")
|
||||
|
||||
// Large weights (near overflow boundary for uint64)
|
||||
// Safe for totalWeight < 2^62 with quorumNum=2 (per checkQuorum doc)
|
||||
largeTotal := uint64(1) << 61
|
||||
largeSigner := largeTotal*2/3 + 1
|
||||
require.True(t, q.CheckQuorum(largeSigner, largeTotal),
|
||||
"large weight should pass when above 2/3")
|
||||
})
|
||||
|
||||
t.Run("bft_threshold_exact", func(t *testing.T) {
|
||||
// BFT requires > 2/3 of total weight.
|
||||
// With the cross-multiplication check (>=), signerWeight*3 >= totalWeight*2.
|
||||
// This means exactly 2/3 PASSES (which matches the formal spec:
|
||||
// "quorum is met when SignerWeight/TotalWeight >= Numerator/Denominator").
|
||||
//
|
||||
// For totalWeight divisible by 3:
|
||||
// signerWeight = totalWeight * 2 / 3 → passes (exact 2/3)
|
||||
// signerWeight = totalWeight * 2 / 3 - 1 → fails (below 2/3)
|
||||
for total := uint64(3); total <= 300; total += 3 {
|
||||
threshold := total * 2 / 3
|
||||
require.True(t, q.CheckQuorum(threshold, total),
|
||||
"exact 2/3 (%d/%d) should pass", threshold, total)
|
||||
require.False(t, q.CheckQuorum(threshold-1, total),
|
||||
"below 2/3 (%d/%d) should fail", threshold-1, total)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Test 5: Concurrent map pruning stress test
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
// TestConcurrentPruningStress verifies that concurrent writes + pruning
|
||||
// do not corrupt the finalized map or deadlock.
|
||||
func TestConcurrentPruningStress(t *testing.T) {
|
||||
q, err := NewQuasar(log.NewNoOpLogger(), 3, 2, 3)
|
||||
require.NoError(t, err)
|
||||
|
||||
const (
|
||||
numWriters = 8
|
||||
opsPerWriter = 5000
|
||||
)
|
||||
|
||||
var wg sync.WaitGroup
|
||||
for w := 0; w < numWriters; w++ {
|
||||
wg.Add(1)
|
||||
go func(writerID int) {
|
||||
defer wg.Done()
|
||||
for i := 0; i < opsPerWriter; i++ {
|
||||
var blockID ids.ID
|
||||
blockID[0] = byte(writerID)
|
||||
blockID[1] = byte(i)
|
||||
blockID[2] = byte(i >> 8)
|
||||
|
||||
q.mu.Lock()
|
||||
q.qHeight++
|
||||
q.finalized[blockID] = &QuantumFinality{
|
||||
BlockID: blockID,
|
||||
QChainHeight: q.qHeight,
|
||||
}
|
||||
if q.maxFinalized > 0 && len(q.finalized) > q.maxFinalized {
|
||||
cutoff := q.qHeight - uint64(q.maxFinalized)
|
||||
for id, f := range q.finalized {
|
||||
if f.QChainHeight < cutoff {
|
||||
delete(q.finalized, id)
|
||||
}
|
||||
}
|
||||
}
|
||||
q.mu.Unlock()
|
||||
}
|
||||
}(w)
|
||||
}
|
||||
wg.Wait()
|
||||
|
||||
q.mu.RLock()
|
||||
finalSize := len(q.finalized)
|
||||
q.mu.RUnlock()
|
||||
|
||||
t.Logf("writers=%d opsEach=%d totalOps=%d finalSize=%d",
|
||||
numWriters, opsPerWriter, numWriters*opsPerWriter, finalSize)
|
||||
|
||||
require.LessOrEqual(t, finalSize, q.maxFinalized+1,
|
||||
"map size should be bounded after concurrent stress")
|
||||
}
|
||||
@@ -1,27 +0,0 @@
|
||||
// Copyright (C) 2019-2026, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
package quasar
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
|
||||
qcert "github.com/luxfi/consensus/protocol/quasar"
|
||||
)
|
||||
|
||||
// policyStore adapts the single configured QuasarEvidencePolicy to the consensus
|
||||
// ConsensusCertPolicyStore interface. The verifier loads the required-leg set
|
||||
// and the (kind, mode, param) permissions from HERE — never from the cert
|
||||
// (invariants I1/I2). A cert that names a different PolicyID than the node's
|
||||
// configured posture is rejected: a cert cannot pick its own weaker policy.
|
||||
type policyStore struct{ policy *qcert.QuasarEvidencePolicy }
|
||||
|
||||
func (s policyStore) Policy(_ uint32, _ uint64, policyID uint32) (qcert.ConsensusCertPolicy, error) {
|
||||
if s.policy == nil {
|
||||
return nil, ErrPolicyUnavailable
|
||||
}
|
||||
if policyID != s.policy.EvidencePolicyID() {
|
||||
return nil, fmt.Errorf("%w: cert policy %d != configured %d", ErrPolicyMismatch, policyID, s.policy.EvidencePolicyID())
|
||||
}
|
||||
return s.policy, nil
|
||||
}
|
||||
@@ -1,83 +0,0 @@
|
||||
// Copyright (C) 2019-2026, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
package quasar
|
||||
|
||||
import (
|
||||
"context"
|
||||
|
||||
qcert "github.com/luxfi/consensus/protocol/quasar"
|
||||
)
|
||||
|
||||
// Subject is the finalized-block position a cert must certify — the producer's
|
||||
// input at a checkpoint. Mirrors Checkpoint (the verify side) so producer and
|
||||
// verifier bind the SAME tuple.
|
||||
type Subject struct {
|
||||
ChainID uint32
|
||||
Epoch uint64
|
||||
Height uint64
|
||||
Round uint32
|
||||
BlockID [32]byte
|
||||
StateRoot [32]byte
|
||||
}
|
||||
|
||||
// subjectFrom derives the producer Subject from this gate's chain id and a
|
||||
// finalized Checkpoint, so producer and verifier bind the SAME tuple.
|
||||
func (g *Gate) subjectFrom(cp Checkpoint) Subject {
|
||||
return Subject{
|
||||
ChainID: g.cfg.ChainID,
|
||||
Epoch: cp.Epoch,
|
||||
Height: cp.Height,
|
||||
Round: cp.Round,
|
||||
BlockID: cp.BlockID,
|
||||
StateRoot: cp.StateRoot,
|
||||
}
|
||||
}
|
||||
|
||||
// Producer is the committee cert-signing service contract (the per-validator
|
||||
// "pulsard" committee). At a checkpoint, a producing validator calls Produce to
|
||||
// obtain the QuasarCert over the finalized subject, then gossips it so peers can
|
||||
// verify and store it (via a CertStore).
|
||||
//
|
||||
// SCAFFOLDING — this is the seam, not the service. This milestone wires the
|
||||
// VERIFY half (gate.go) and this interface. luxd ships with a nil Producer
|
||||
// (verify-only): a node VERIFIES certs it receives but does not itself produce
|
||||
// them. A nil Producer is the correct default — most of the rollout window is
|
||||
// verify-only, and the producer is brought up before activation is forward-dated.
|
||||
//
|
||||
// Implementation path for the follow-on:
|
||||
//
|
||||
// - github.com/luxfi/consensus/protocol/quasar already defines the
|
||||
// producer-side abstractions: PWitnessProducer / QWitnessProducer /
|
||||
// ZWitnessProducer + NewWitnessSet, and ComposeDualPQEvidence. The concrete
|
||||
// committee signer implements Producer over those.
|
||||
// - The signer needs the live Pulsar key share + nonce pool + offline
|
||||
// preprocessing + one-round sign + verify-before-gossip + nonce-erase (the
|
||||
// no-reconstruct hyperball signer), which lands with pulsar v1.7.1.
|
||||
// - REQUIRED CONSENSUS EXPORT: the ConsensusCert envelope + per-leg payload
|
||||
// ENCODERS are package-private in consensus v1.29.0 (only the verifiers are
|
||||
// exported). An external producer — and any end-to-end "valid cert verifies
|
||||
// through the gate" test — needs those encoders exported (a small, additive
|
||||
// consensus change). The verify path here needs no such export: it consumes
|
||||
// a fully-formed *ConsensusCert.
|
||||
type Producer interface {
|
||||
Produce(ctx context.Context, subject Subject) (*qcert.ConsensusCert, error)
|
||||
}
|
||||
|
||||
// MaybeProduce is the checkpoint producer-request site. It is nil-safe and
|
||||
// activation-aware so the accept hook can call it unconditionally: a nil gate,
|
||||
// dormant activation, a non-checkpoint height, or a nil producer all short-
|
||||
// circuit to (nil, nil) — the verify-only default. When a producer IS wired and
|
||||
// the checkpoint is live, it requests the cert; the caller gossips/stores it.
|
||||
//
|
||||
// This keeps producer cadence and verify cadence on ONE definition (g.IsCheckpoint),
|
||||
// so producer and verifier can never disagree on which heights carry certs.
|
||||
func (g *Gate) MaybeProduce(ctx context.Context, producer Producer, cp Checkpoint) (*qcert.ConsensusCert, error) {
|
||||
if g == nil || producer == nil {
|
||||
return nil, nil
|
||||
}
|
||||
if !g.Activated(cp.Height) || !g.IsCheckpoint(cp.Height) {
|
||||
return nil, nil
|
||||
}
|
||||
return producer.Produce(ctx, g.subjectFrom(cp))
|
||||
}
|
||||
@@ -0,0 +1,681 @@
|
||||
// Copyright (C) 2019-2025, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
package quasar
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/luxfi/consensus/protocol/quasar"
|
||||
"github.com/luxfi/ids"
|
||||
"github.com/luxfi/log"
|
||||
)
|
||||
|
||||
// Quasar is the gravitational center of Lux consensus.
|
||||
// It binds P-Chain (BLS signatures) and Q-Chain (Corona post-quantum threshold)
|
||||
// into unified hybrid finality across all Lux networks.
|
||||
//
|
||||
// Architecture:
|
||||
// ALL validators have BOTH keypairs:
|
||||
// - BLS keypair → aggregate signatures (classical, fast)
|
||||
// - Corona keypair → threshold signatures (post-quantum, 2-round)
|
||||
//
|
||||
// Both signature paths run IN PARALLEL:
|
||||
//
|
||||
// Block arrives
|
||||
// │
|
||||
// ├─────────────────────────────────────────┐
|
||||
// │ │
|
||||
// ▼ ▼
|
||||
// BLS PATH (fast) CORONA PATH (quantum-safe)
|
||||
// ──────────────── ─────────────────────────────
|
||||
// All validators sign Round 1: All validators
|
||||
// with BLS keys generate commitments
|
||||
// │ │
|
||||
// ▼ ▼
|
||||
// Aggregate into Round 2: All validators
|
||||
// single 96-byte sig compute partial signatures
|
||||
// │ │
|
||||
// └─────────────────┬───────────────────────┘
|
||||
// │
|
||||
// ▼
|
||||
// Finalize: combine into
|
||||
// threshold signature
|
||||
// │
|
||||
// ▼
|
||||
// ┌─────────────────┐
|
||||
// │ HYBRID PROOF │
|
||||
// │ BLS Aggregate │ ← 96 bytes (2/3+ validators)
|
||||
// │ Corona Thresh │ ← ~KB (t-of-n threshold)
|
||||
// └─────────────────┘
|
||||
// │
|
||||
// ▼
|
||||
// QUANTUM FINALITY
|
||||
//
|
||||
// The quasar ensures blocks achieve finality only when BOTH complete:
|
||||
// 1. 2/3+ validator weight signed via BLS (fast, classical)
|
||||
// 2. t-of-n validators completed Corona threshold (post-quantum secure)
|
||||
|
||||
var (
|
||||
ErrQuasarNotStarted = errors.New("quasar not started")
|
||||
ErrPChainNotConnected = errors.New("P-Chain not connected")
|
||||
ErrQChainNotConnected = errors.New("Q-Chain not connected")
|
||||
ErrCoronaNotConnected = errors.New("Corona coordinator not connected")
|
||||
ErrInsufficientWeight = errors.New("insufficient validator weight")
|
||||
ErrInsufficientSigners = errors.New("insufficient Corona signers")
|
||||
ErrFinalityFailed = errors.New("hybrid finality verification failed")
|
||||
ErrBLSFailed = errors.New("BLS aggregation failed")
|
||||
ErrCoronaFailed = errors.New("Corona threshold signing failed")
|
||||
)
|
||||
|
||||
// PChainProvider provides P-Chain state and finality events
|
||||
type PChainProvider interface {
|
||||
GetFinalizedHeight() uint64
|
||||
GetValidators(height uint64) ([]ValidatorState, error)
|
||||
SubscribeFinality() <-chan FinalityEvent
|
||||
}
|
||||
|
||||
// QuantumSignerFallback provides fallback single-signer quantum signatures
|
||||
type QuantumSignerFallback interface {
|
||||
SignMessage(msg []byte) ([]byte, error)
|
||||
}
|
||||
|
||||
// ValidatorState represents a validator's current state
|
||||
// Each validator has BOTH BLS and Corona keys
|
||||
type ValidatorState struct {
|
||||
NodeID ids.NodeID
|
||||
Weight uint64
|
||||
BLSPubKey []byte // BLS public key for aggregate signatures
|
||||
CoronaKey []byte // Corona public key share for threshold sigs
|
||||
Active bool
|
||||
}
|
||||
|
||||
// FinalityEvent represents a P-Chain finality event
|
||||
type FinalityEvent struct {
|
||||
Height uint64
|
||||
BlockID ids.ID
|
||||
Validators []ValidatorState
|
||||
Timestamp time.Time
|
||||
}
|
||||
|
||||
// QuantumFinality represents a block that achieved hybrid quantum finality
|
||||
type QuantumFinality struct {
|
||||
BlockID ids.ID
|
||||
PChainHeight uint64
|
||||
QChainHeight uint64
|
||||
BLSProof []byte // Aggregated BLS signature (96 bytes)
|
||||
CoronaProof []byte // Serialized Corona threshold signature
|
||||
SignerBitset []byte // Which validators signed BLS
|
||||
CoronaSigners []ids.NodeID // Which validators participated in Corona
|
||||
TotalWeight uint64
|
||||
SignerWeight uint64
|
||||
BLSLatency time.Duration
|
||||
CoronaLatency time.Duration
|
||||
Timestamp time.Time
|
||||
}
|
||||
|
||||
// Quasar binds P-Chain and Q-Chain consensus into hybrid quantum finality
|
||||
type Quasar struct {
|
||||
mu sync.RWMutex
|
||||
|
||||
log log.Logger
|
||||
core *quasar.Quasar
|
||||
|
||||
// Chain connections
|
||||
pChain PChainProvider
|
||||
quantumFallback QuantumSignerFallback
|
||||
|
||||
// Corona threshold coordinator
|
||||
corona *CoronaCoordinator
|
||||
|
||||
// State
|
||||
pHeight uint64
|
||||
qHeight uint64
|
||||
finalized map[ids.ID]*QuantumFinality
|
||||
|
||||
// Configuration
|
||||
threshold int // Corona threshold (t in t-of-n)
|
||||
quorumNum uint64 // BLS quorum numerator
|
||||
quorumDen uint64 // BLS quorum denominator
|
||||
maxFinalized int // max finalized entries before pruning
|
||||
|
||||
// Channels
|
||||
finalityCh chan *QuantumFinality
|
||||
stopCh chan struct{}
|
||||
running bool
|
||||
}
|
||||
|
||||
// NewQuasar creates a new Quasar consensus hub
|
||||
func NewQuasar(log log.Logger, threshold int, quorumNum, quorumDen uint64) (*Quasar, error) {
|
||||
core, err := quasar.NewQuasar(threshold)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to create quasar core: %w", err)
|
||||
}
|
||||
|
||||
return &Quasar{
|
||||
log: log,
|
||||
core: core,
|
||||
threshold: threshold,
|
||||
quorumNum: quorumNum,
|
||||
quorumDen: quorumDen,
|
||||
finalized: make(map[ids.ID]*QuantumFinality),
|
||||
maxFinalized: 10000,
|
||||
finalityCh: make(chan *QuantumFinality, 100),
|
||||
stopCh: make(chan struct{}),
|
||||
}, nil
|
||||
}
|
||||
|
||||
// ConnectPChain connects the P-Chain finality provider
|
||||
func (q *Quasar) ConnectPChain(p PChainProvider) {
|
||||
q.mu.Lock()
|
||||
defer q.mu.Unlock()
|
||||
|
||||
q.pChain = p
|
||||
if p != nil {
|
||||
q.pHeight = p.GetFinalizedHeight()
|
||||
}
|
||||
|
||||
q.log.Info("quasar: P-Chain connected", "height", q.pHeight)
|
||||
}
|
||||
|
||||
// ConnectQuantumFallback connects the quantum signer fallback
|
||||
func (q *Quasar) ConnectQuantumFallback(f QuantumSignerFallback) {
|
||||
q.mu.Lock()
|
||||
defer q.mu.Unlock()
|
||||
|
||||
q.quantumFallback = f
|
||||
q.log.Info("quasar: quantum fallback connected")
|
||||
}
|
||||
|
||||
// ConnectCorona connects the Corona threshold coordinator
|
||||
func (q *Quasar) ConnectCorona(rc *CoronaCoordinator) {
|
||||
q.mu.Lock()
|
||||
defer q.mu.Unlock()
|
||||
|
||||
q.corona = rc
|
||||
q.log.Info("quasar: Corona coordinator connected")
|
||||
}
|
||||
|
||||
// InitializeCorona initializes the Corona coordinator with validators
|
||||
func (q *Quasar) InitializeCorona(validators []ids.NodeID) error {
|
||||
q.mu.Lock()
|
||||
defer q.mu.Unlock()
|
||||
|
||||
if q.corona == nil {
|
||||
// Create coordinator if not provided
|
||||
numParties := len(validators)
|
||||
threshold := (numParties * 2 / 3) + 1 // 2/3 + 1 threshold
|
||||
if threshold < 2 {
|
||||
threshold = 2
|
||||
}
|
||||
|
||||
rc, err := NewCoronaCoordinator(q.log, CoronaConfig{
|
||||
NumParties: numParties,
|
||||
Threshold: threshold,
|
||||
})
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to create Corona coordinator: %w", err)
|
||||
}
|
||||
q.corona = rc
|
||||
}
|
||||
|
||||
if err := q.corona.Initialize(validators); err != nil {
|
||||
return fmt.Errorf("failed to initialize Corona: %w", err)
|
||||
}
|
||||
|
||||
q.log.Info("quasar: Corona initialized",
|
||||
"validators", len(validators),
|
||||
"threshold", q.corona.Stats().Threshold,
|
||||
)
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// Start begins the quasar consensus loop
|
||||
func (q *Quasar) Start(ctx context.Context) error {
|
||||
q.mu.Lock()
|
||||
if q.pChain == nil {
|
||||
q.mu.Unlock()
|
||||
return ErrPChainNotConnected
|
||||
}
|
||||
if q.quantumFallback == nil {
|
||||
q.mu.Unlock()
|
||||
return ErrQChainNotConnected
|
||||
}
|
||||
q.running = true
|
||||
q.mu.Unlock()
|
||||
|
||||
// Subscribe to P-Chain finality
|
||||
sub := q.pChain.SubscribeFinality()
|
||||
go q.run(ctx, sub)
|
||||
|
||||
q.log.Info("quasar: started")
|
||||
return nil
|
||||
}
|
||||
|
||||
// Stop halts the quasar
|
||||
func (q *Quasar) Stop() {
|
||||
q.mu.Lock()
|
||||
if q.running {
|
||||
close(q.stopCh)
|
||||
q.running = false
|
||||
}
|
||||
q.mu.Unlock()
|
||||
q.log.Info("quasar: stopped")
|
||||
}
|
||||
|
||||
// run is the main finality loop.
|
||||
// Bounded by ctx.Done() and q.stopCh — exits when either fires.
|
||||
// The goroutine is started in Start() and guaranteed to terminate
|
||||
// when Stop() closes stopCh or the parent context is cancelled.
|
||||
func (q *Quasar) run(ctx context.Context, sub <-chan FinalityEvent) {
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-q.stopCh:
|
||||
return
|
||||
case event := <-sub:
|
||||
if err := q.processFinality(ctx, event); err != nil {
|
||||
q.log.Error("quasar: finality failed",
|
||||
"height", event.Height,
|
||||
"error", err,
|
||||
)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// processFinality processes a P-Chain finality event into hybrid finality
|
||||
// Both BLS and Corona paths run IN PARALLEL
|
||||
func (q *Quasar) processFinality(ctx context.Context, event FinalityEvent) error {
|
||||
q.mu.Lock()
|
||||
defer q.mu.Unlock()
|
||||
|
||||
// Sync validators to quasar core
|
||||
for _, v := range event.Validators {
|
||||
if v.Active {
|
||||
_, _ = q.core.AddValidator(v.NodeID.String(), v.Weight)
|
||||
}
|
||||
}
|
||||
|
||||
// Create finality message
|
||||
msg := q.createMessage(event)
|
||||
msgStr := string(msg) // Corona uses string message
|
||||
|
||||
// Run BLS and Corona IN PARALLEL
|
||||
var blsProof, signerBitset []byte
|
||||
var signerWeight uint64
|
||||
var coronaSig Signature
|
||||
var blsLatency, coronaLatency time.Duration
|
||||
var blsErr, coronaErr error
|
||||
var wg sync.WaitGroup
|
||||
|
||||
// BLS path
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
start := time.Now()
|
||||
blsProof, signerBitset, signerWeight, blsErr = q.collectBLS(event, msg)
|
||||
blsLatency = time.Since(start)
|
||||
}()
|
||||
|
||||
// Corona path - REQUIRED for Q-Chain validator consensus.
|
||||
// No fallback mode: if Corona coordinator is not initialized,
|
||||
// finality MUST fail to prevent accepting BLS-only proofs.
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
if q.corona == nil || !q.corona.IsInitialized() {
|
||||
// STRICT: No fallback allowed for validators.
|
||||
// RT signatures are REQUIRED for quantum-safe consensus.
|
||||
coronaErr = ErrCoronaNotConnected
|
||||
return
|
||||
}
|
||||
// Full threshold signing
|
||||
start := time.Now()
|
||||
coronaSig, coronaErr = q.collectCorona(msgStr)
|
||||
coronaLatency = time.Since(start)
|
||||
}()
|
||||
|
||||
wg.Wait()
|
||||
|
||||
// Check BLS result
|
||||
if blsErr != nil {
|
||||
return fmt.Errorf("BLS collection: %w", blsErr)
|
||||
}
|
||||
|
||||
// Check Corona result
|
||||
if coronaErr != nil {
|
||||
return fmt.Errorf("Corona threshold: %w", coronaErr)
|
||||
}
|
||||
|
||||
// Check quorum
|
||||
totalWeight := q.totalWeight(event.Validators)
|
||||
if !q.checkQuorum(signerWeight, totalWeight) {
|
||||
return ErrInsufficientWeight
|
||||
}
|
||||
|
||||
// Record finality
|
||||
q.qHeight++
|
||||
var coronaSigners []ids.NodeID
|
||||
var coronaProof []byte
|
||||
if coronaSig != nil {
|
||||
coronaSigners = coronaSig.Signers()
|
||||
coronaProof = coronaSig.Bytes()
|
||||
}
|
||||
finality := &QuantumFinality{
|
||||
BlockID: event.BlockID,
|
||||
PChainHeight: event.Height,
|
||||
QChainHeight: q.qHeight,
|
||||
BLSProof: blsProof,
|
||||
CoronaProof: coronaProof,
|
||||
SignerBitset: signerBitset,
|
||||
CoronaSigners: coronaSigners,
|
||||
TotalWeight: totalWeight,
|
||||
SignerWeight: signerWeight,
|
||||
BLSLatency: blsLatency,
|
||||
CoronaLatency: coronaLatency,
|
||||
Timestamp: time.Now(),
|
||||
}
|
||||
|
||||
q.finalized[event.BlockID] = finality
|
||||
q.pHeight = event.Height
|
||||
|
||||
// Prune old finality entries to bound memory
|
||||
if q.maxFinalized > 0 && len(q.finalized) > q.maxFinalized {
|
||||
cutoff := q.qHeight - uint64(q.maxFinalized)
|
||||
for id, f := range q.finalized {
|
||||
if f.QChainHeight < cutoff {
|
||||
delete(q.finalized, id)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Emit
|
||||
select {
|
||||
case q.finalityCh <- finality:
|
||||
default:
|
||||
}
|
||||
|
||||
q.log.Info("quasar: hybrid finality achieved",
|
||||
"block", event.BlockID,
|
||||
"pHeight", event.Height,
|
||||
"qHeight", q.qHeight,
|
||||
"weight", fmt.Sprintf("%d/%d", signerWeight, totalWeight),
|
||||
"blsLatency", blsLatency,
|
||||
"coronaLatency", coronaLatency,
|
||||
"coronaSigners", len(coronaSigners),
|
||||
)
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// createMessage creates the finality message to sign
|
||||
func (q *Quasar) createMessage(event FinalityEvent) []byte {
|
||||
msg := make([]byte, 48) // 32 (blockID) + 8 (height) + 8 (timestamp)
|
||||
copy(msg[:32], event.BlockID[:])
|
||||
putUint64BE(msg[32:40], event.Height)
|
||||
putUint64BE(msg[40:48], uint64(event.Timestamp.UnixNano()))
|
||||
return msg
|
||||
}
|
||||
|
||||
// collectBLS collects BLS signatures from validators and aggregates them
|
||||
func (q *Quasar) collectBLS(event FinalityEvent, msg []byte) ([]byte, []byte, uint64, error) {
|
||||
var signerBitset []byte
|
||||
var signerWeight uint64
|
||||
signatures := make([]*quasar.QuasarSig, 0, len(event.Validators))
|
||||
|
||||
for i, v := range event.Validators {
|
||||
if !v.Active {
|
||||
continue
|
||||
}
|
||||
|
||||
sig, err := q.core.SignMessage(v.NodeID.String(), msg)
|
||||
if err != nil {
|
||||
continue // Skip failed signers
|
||||
}
|
||||
|
||||
signatures = append(signatures, sig)
|
||||
signerWeight += v.Weight
|
||||
|
||||
// Set bit
|
||||
byteIdx := i / 8
|
||||
for len(signerBitset) <= byteIdx {
|
||||
signerBitset = append(signerBitset, 0)
|
||||
}
|
||||
signerBitset[byteIdx] |= 1 << uint(i%8)
|
||||
}
|
||||
|
||||
if len(signatures) == 0 {
|
||||
return nil, nil, 0, errors.New("no BLS signatures")
|
||||
}
|
||||
|
||||
agg, err := q.core.AggregateSignatures(msg, signatures)
|
||||
if err != nil {
|
||||
return nil, nil, 0, err
|
||||
}
|
||||
|
||||
return agg.BLSAggregated, signerBitset, signerWeight, nil
|
||||
}
|
||||
|
||||
// collectCorona runs the 2-round Corona threshold protocol in parallel
|
||||
func (q *Quasar) collectCorona(message string) (Signature, error) {
|
||||
if q.corona == nil {
|
||||
return nil, ErrCoronaNotConnected
|
||||
}
|
||||
|
||||
// Use the high-level Sign API which handles all rounds internally
|
||||
sig, err := q.corona.Sign([]byte(message))
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("corona signing failed: %w", err)
|
||||
}
|
||||
|
||||
// Verify the signature
|
||||
if !q.corona.Verify([]byte(message), sig) {
|
||||
return nil, ErrCoronaFailed
|
||||
}
|
||||
|
||||
q.log.Debug("corona signature complete",
|
||||
"signers", len(sig.Signers()),
|
||||
"type", sig.Type(),
|
||||
)
|
||||
|
||||
return sig, nil
|
||||
}
|
||||
|
||||
// totalWeight calculates total validator weight
|
||||
func (q *Quasar) totalWeight(validators []ValidatorState) uint64 {
|
||||
var total uint64
|
||||
for _, v := range validators {
|
||||
if v.Active {
|
||||
total += v.Weight
|
||||
}
|
||||
}
|
||||
return total
|
||||
}
|
||||
|
||||
// checkQuorum verifies quorum is met using cross-multiplication to avoid
|
||||
// integer division truncation.
|
||||
//
|
||||
// signerWeight / totalWeight >= quorumNum / quorumDen
|
||||
// is equivalent to:
|
||||
// signerWeight * quorumDen >= totalWeight * quorumNum
|
||||
//
|
||||
// Overflow bound: safe for totalWeight < 2^62 with quorumNum <= 3.
|
||||
// Production values: totalWeight is sum of validator weights (well under 2^60),
|
||||
// quorumNum=2, quorumDen=3.
|
||||
func (q *Quasar) checkQuorum(signerWeight, totalWeight uint64) bool {
|
||||
return signerWeight*q.quorumDen >= totalWeight*q.quorumNum
|
||||
}
|
||||
|
||||
// GetFinality returns finality for a block
|
||||
func (q *Quasar) GetFinality(blockID ids.ID) (*QuantumFinality, bool) {
|
||||
q.mu.RLock()
|
||||
defer q.mu.RUnlock()
|
||||
f, ok := q.finalized[blockID]
|
||||
return f, ok
|
||||
}
|
||||
|
||||
// Subscribe returns channel for finality events
|
||||
func (q *Quasar) Subscribe() <-chan *QuantumFinality {
|
||||
return q.finalityCh
|
||||
}
|
||||
|
||||
// Verify verifies a hybrid finality proof.
|
||||
// Both BLS and Corona proofs are REQUIRED - no fallback mode.
|
||||
// This ensures quantum-safe consensus for Q-Chain validators.
|
||||
func (q *Quasar) Verify(finality *QuantumFinality) error {
|
||||
if finality == nil {
|
||||
return ErrFinalityFailed
|
||||
}
|
||||
|
||||
// STRICT: Both proofs are REQUIRED
|
||||
if len(finality.BLSProof) == 0 {
|
||||
return fmt.Errorf("%w: BLS proof missing", ErrBLSFailed)
|
||||
}
|
||||
if len(finality.CoronaProof) == 0 {
|
||||
return fmt.Errorf("%w: RT proof missing - required for Q-Chain validators", ErrCoronaFailed)
|
||||
}
|
||||
|
||||
if !q.checkQuorum(finality.SignerWeight, finality.TotalWeight) {
|
||||
return ErrInsufficientWeight
|
||||
}
|
||||
|
||||
// Verify BLS via hybrid engine
|
||||
agg := &quasar.AggregatedSignature{
|
||||
BLSAggregated: finality.BLSProof,
|
||||
}
|
||||
|
||||
// Reconstruct message for verification
|
||||
msg := make([]byte, 48)
|
||||
copy(msg[:32], finality.BlockID[:])
|
||||
putUint64BE(msg[32:40], finality.PChainHeight)
|
||||
putUint64BE(msg[40:48], uint64(finality.Timestamp.UnixNano()))
|
||||
|
||||
if !q.core.VerifyAggregatedSignature(msg, agg) {
|
||||
return ErrBLSFailed
|
||||
}
|
||||
|
||||
// Verify Corona threshold signature
|
||||
// RT signatures MUST have the "RT" prefix marker followed by threshold data
|
||||
if len(finality.CoronaProof) < 3 {
|
||||
return fmt.Errorf("%w: RT proof too short", ErrCoronaFailed)
|
||||
}
|
||||
if finality.CoronaProof[0] != 'R' || finality.CoronaProof[1] != 'T' {
|
||||
return fmt.Errorf("%w: invalid RT proof marker", ErrCoronaFailed)
|
||||
}
|
||||
|
||||
// Verify threshold signers meet minimum requirement
|
||||
if len(finality.CoronaSigners) < q.threshold {
|
||||
return fmt.Errorf("%w: need %d signers, have %d",
|
||||
ErrInsufficientSigners, q.threshold, len(finality.CoronaSigners))
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// Stats returns quasar statistics
|
||||
func (q *Quasar) Stats() QuasarStats {
|
||||
q.mu.RLock()
|
||||
defer q.mu.RUnlock()
|
||||
|
||||
var coronaStats CoronaStats
|
||||
if q.corona != nil {
|
||||
coronaStats = q.corona.Stats()
|
||||
}
|
||||
|
||||
return QuasarStats{
|
||||
PChainHeight: q.pHeight,
|
||||
QChainHeight: q.qHeight,
|
||||
FinalizedBlocks: len(q.finalized),
|
||||
Threshold: q.threshold,
|
||||
QuorumNum: q.quorumNum,
|
||||
QuorumDen: q.quorumDen,
|
||||
Running: q.running,
|
||||
CoronaParties: coronaStats.NumParties,
|
||||
CoronaThreshold: coronaStats.Threshold,
|
||||
CoronaReady: coronaStats.Initialized,
|
||||
}
|
||||
}
|
||||
|
||||
// QuasarStats contains quasar statistics
|
||||
type QuasarStats struct {
|
||||
PChainHeight uint64
|
||||
QChainHeight uint64
|
||||
FinalizedBlocks int
|
||||
Threshold int
|
||||
QuorumNum uint64
|
||||
QuorumDen uint64
|
||||
Running bool
|
||||
CoronaParties int
|
||||
CoronaThreshold int
|
||||
CoronaReady bool
|
||||
}
|
||||
|
||||
// GetCore returns the underlying quasar core for testing
|
||||
func (q *Quasar) GetCore() *quasar.Quasar {
|
||||
return q.core
|
||||
}
|
||||
|
||||
// GetCorona returns the Corona coordinator
|
||||
func (q *Quasar) GetCorona() *CoronaCoordinator {
|
||||
q.mu.RLock()
|
||||
defer q.mu.RUnlock()
|
||||
return q.corona
|
||||
}
|
||||
|
||||
// CheckQuorum verifies quorum is met (exported for testing)
|
||||
func (q *Quasar) CheckQuorum(signerWeight, totalWeight uint64) bool {
|
||||
return q.checkQuorum(signerWeight, totalWeight)
|
||||
}
|
||||
|
||||
// CreateMessage creates the finality message to sign (exported for testing)
|
||||
func (q *Quasar) CreateMessage(event FinalityEvent) []byte {
|
||||
return q.createMessage(event)
|
||||
}
|
||||
|
||||
// TotalWeight calculates total validator weight (exported for testing)
|
||||
func (q *Quasar) TotalWeight(validators []ValidatorState) uint64 {
|
||||
return q.totalWeight(validators)
|
||||
}
|
||||
|
||||
// GetConfig returns quorum configuration (exported for testing)
|
||||
func (q *Quasar) GetConfig() (threshold int, quorumNum, quorumDen uint64) {
|
||||
q.mu.RLock()
|
||||
defer q.mu.RUnlock()
|
||||
return q.threshold, q.quorumNum, q.quorumDen
|
||||
}
|
||||
|
||||
// IsRunning returns whether the Quasar is currently running (exported for testing)
|
||||
func (q *Quasar) IsRunning() bool {
|
||||
q.mu.RLock()
|
||||
defer q.mu.RUnlock()
|
||||
return q.running
|
||||
}
|
||||
|
||||
// SetFinalized adds a finality record (exported for testing/benchmarking)
|
||||
func (q *Quasar) SetFinalized(blockID ids.ID, finality *QuantumFinality) {
|
||||
q.mu.Lock()
|
||||
defer q.mu.Unlock()
|
||||
q.finalized[blockID] = finality
|
||||
}
|
||||
|
||||
// GetFinalized retrieves a finality record (exported for testing)
|
||||
func (q *Quasar) GetFinalized(blockID ids.ID) (*QuantumFinality, bool) {
|
||||
q.mu.RLock()
|
||||
defer q.mu.RUnlock()
|
||||
f, ok := q.finalized[blockID]
|
||||
return f, ok
|
||||
}
|
||||
|
||||
// Helper: big-endian uint64
|
||||
func putUint64BE(b []byte, v uint64) {
|
||||
for i := 0; i < 8; i++ {
|
||||
b[i] = byte(v >> (56 - i*8))
|
||||
}
|
||||
}
|
||||
@@ -1,61 +0,0 @@
|
||||
// Copyright (C) 2019-2026, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
package quasar
|
||||
|
||||
import (
|
||||
"sync"
|
||||
|
||||
qcert "github.com/luxfi/consensus/protocol/quasar"
|
||||
)
|
||||
|
||||
// CertStore resolves the QuasarCert that certifies a finalized block. In
|
||||
// production it is filled by the cert gossip/ingest path (the producer
|
||||
// follow-on, producer.go); the verify gate only READS from it. Keyed by the
|
||||
// finalized position (chainID, height, blockID) so a cert can never be returned
|
||||
// for the wrong block.
|
||||
type CertStore interface {
|
||||
Lookup(chainID uint32, height uint64, blockID [32]byte) (*qcert.ConsensusCert, bool)
|
||||
}
|
||||
|
||||
type certKey struct {
|
||||
chainID uint32
|
||||
height uint64
|
||||
blockID [32]byte
|
||||
}
|
||||
|
||||
// MemCertStore is an in-memory CertStore keyed by (chainID, height, blockID). It
|
||||
// is the ingest sink the cert-gossip handler writes into (Put) and the gate
|
||||
// reads from (Lookup). Safe for concurrent use.
|
||||
type MemCertStore struct {
|
||||
mu sync.RWMutex
|
||||
certs map[certKey]*qcert.ConsensusCert
|
||||
}
|
||||
|
||||
// NewMemCertStore returns an empty in-memory cert store.
|
||||
func NewMemCertStore() *MemCertStore {
|
||||
return &MemCertStore{certs: make(map[certKey]*qcert.ConsensusCert)}
|
||||
}
|
||||
|
||||
// Put indexes a cert by its own (ChainID, Height, BlockHash). The ingest path
|
||||
// MUST verify a cert before Put (verify-before-store), exactly as the gossip
|
||||
// layer verifies before re-gossip; the gate re-verifies at the checkpoint so a
|
||||
// store poisoned by an unverified Put still cannot finalize an invalid cert.
|
||||
func (m *MemCertStore) Put(cert *qcert.ConsensusCert) {
|
||||
if cert == nil {
|
||||
return
|
||||
}
|
||||
k := certKey{chainID: cert.ChainID, height: cert.Height, blockID: cert.BlockHash}
|
||||
m.mu.Lock()
|
||||
m.certs[k] = cert
|
||||
m.mu.Unlock()
|
||||
}
|
||||
|
||||
// Lookup returns the cert for the finalized position, or (nil, false).
|
||||
func (m *MemCertStore) Lookup(chainID uint32, height uint64, blockID [32]byte) (*qcert.ConsensusCert, bool) {
|
||||
k := certKey{chainID: chainID, height: height, blockID: blockID}
|
||||
m.mu.RLock()
|
||||
c, ok := m.certs[k]
|
||||
m.mu.RUnlock()
|
||||
return c, ok
|
||||
}
|
||||
@@ -0,0 +1,240 @@
|
||||
// Copyright (C) 2019-2025, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
package quasar
|
||||
|
||||
import (
|
||||
"errors"
|
||||
|
||||
"github.com/luxfi/ids"
|
||||
"github.com/luxfi/log"
|
||||
)
|
||||
|
||||
// SignatureType identifies the signature algorithm used
|
||||
type SignatureType uint8
|
||||
|
||||
const (
|
||||
SignatureTypeBLS SignatureType = iota
|
||||
SignatureTypeCorona
|
||||
SignatureTypeQuasar // Hybrid BLS + Corona
|
||||
SignatureTypeMLDSA
|
||||
)
|
||||
|
||||
// Signature is the interface for all signature types
|
||||
type Signature interface {
|
||||
Bytes() []byte
|
||||
Type() SignatureType
|
||||
Signers() []ids.NodeID
|
||||
}
|
||||
|
||||
// Signer is the interface for signing operations
|
||||
type Signer interface {
|
||||
Sign(msg []byte) (Signature, error)
|
||||
PublicKey() []byte
|
||||
}
|
||||
|
||||
// Verifier is the interface for signature verification
|
||||
type Verifier interface {
|
||||
Verify(msg []byte, sig Signature) bool
|
||||
}
|
||||
|
||||
// ThresholdSigner extends Signer for threshold signature schemes
|
||||
type ThresholdSigner interface {
|
||||
Signer
|
||||
Index() int
|
||||
Threshold() int
|
||||
}
|
||||
|
||||
// CoronaConfig holds configuration for Corona threshold signatures
|
||||
type CoronaConfig struct {
|
||||
NumParties int
|
||||
Threshold int
|
||||
PartyIndex int
|
||||
}
|
||||
|
||||
// CoronaStats contains statistics about the Corona coordinator
|
||||
type CoronaStats struct {
|
||||
NumParties int
|
||||
Threshold int
|
||||
Initialized bool
|
||||
}
|
||||
|
||||
// CoronaSignature represents a threshold Corona signature
|
||||
type CoronaSignature struct {
|
||||
sig []byte
|
||||
signers []ids.NodeID
|
||||
}
|
||||
|
||||
// NewCoronaSignature creates a new Corona signature
|
||||
func NewCoronaSignature(sig []byte, signers []ids.NodeID) *CoronaSignature {
|
||||
return &CoronaSignature{sig: sig, signers: signers}
|
||||
}
|
||||
|
||||
func (s *CoronaSignature) Bytes() []byte { return s.sig }
|
||||
func (s *CoronaSignature) Type() SignatureType { return SignatureTypeCorona }
|
||||
func (s *CoronaSignature) Signers() []ids.NodeID { return s.signers }
|
||||
|
||||
// CoronaCoordinator manages the threshold signing protocol.
|
||||
//
|
||||
// Sign/Verify are fail-closed without initialized lattice keys.
|
||||
// Operations return errors unless properly initialized with key
|
||||
// material, or explicitly created via NewTestCoronaCoordinator
|
||||
// for tests.
|
||||
type CoronaCoordinator struct {
|
||||
log log.Logger
|
||||
config CoronaConfig
|
||||
initialized bool
|
||||
testing bool // only true via NewTestCoronaCoordinator
|
||||
validators []ids.NodeID
|
||||
}
|
||||
|
||||
// NewCoronaCoordinator creates a new Corona coordinator.
|
||||
// Sign and Verify will fail until real lattice key material is loaded.
|
||||
func NewCoronaCoordinator(log log.Logger, config CoronaConfig) (*CoronaCoordinator, error) {
|
||||
return &CoronaCoordinator{
|
||||
log: log,
|
||||
config: config,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// NewTestCoronaCoordinator creates a Corona coordinator for testing.
|
||||
// The test coordinator uses deterministic stub signatures that are NOT
|
||||
// cryptographically secure.
|
||||
func NewTestCoronaCoordinator(log log.Logger, config CoronaConfig) (*CoronaCoordinator, error) {
|
||||
return &CoronaCoordinator{
|
||||
log: log,
|
||||
config: config,
|
||||
testing: true,
|
||||
}, nil
|
||||
}
|
||||
|
||||
func (rc *CoronaCoordinator) Initialize(validators []ids.NodeID) error {
|
||||
rc.validators = validators
|
||||
rc.initialized = true
|
||||
return nil
|
||||
}
|
||||
|
||||
func (rc *CoronaCoordinator) IsInitialized() bool {
|
||||
return rc.initialized
|
||||
}
|
||||
|
||||
func (rc *CoronaCoordinator) Sign(msg []byte) (Signature, error) {
|
||||
if !rc.initialized {
|
||||
return nil, errors.New("corona: threshold signing not initialized — requires real lattice key material")
|
||||
}
|
||||
if rc.testing {
|
||||
// Test-only stub: deterministic RT-prefixed signature
|
||||
sig := append([]byte("RT"), msg[:min(32, len(msg))]...)
|
||||
return NewCoronaSignature(sig, rc.validators), nil
|
||||
}
|
||||
return nil, errors.New("corona: threshold signing not initialized — requires real lattice key material")
|
||||
}
|
||||
|
||||
func (rc *CoronaCoordinator) Verify(msg []byte, sig Signature) bool {
|
||||
if !rc.initialized {
|
||||
return false
|
||||
}
|
||||
if rc.testing {
|
||||
return sig != nil && len(sig.Bytes()) > 0
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func (rc *CoronaCoordinator) Stats() CoronaStats {
|
||||
return CoronaStats{
|
||||
NumParties: rc.config.NumParties,
|
||||
Threshold: rc.config.Threshold,
|
||||
Initialized: rc.initialized,
|
||||
}
|
||||
}
|
||||
|
||||
// Threshold returns the threshold required for signing
|
||||
func (rc *CoronaCoordinator) Threshold() int {
|
||||
return rc.config.Threshold
|
||||
}
|
||||
|
||||
// NumParties returns the number of parties in the threshold scheme
|
||||
func (rc *CoronaCoordinator) NumParties() int {
|
||||
return rc.config.NumParties
|
||||
}
|
||||
|
||||
func min(a, b int) int {
|
||||
if a < b {
|
||||
return a
|
||||
}
|
||||
return b
|
||||
}
|
||||
|
||||
// BLSSignature represents an aggregated BLS signature (node-specific)
|
||||
type BLSSignature struct {
|
||||
sig []byte
|
||||
signers []ids.NodeID
|
||||
}
|
||||
|
||||
func NewBLSSignature(sig []byte, signers []ids.NodeID) *BLSSignature {
|
||||
return &BLSSignature{sig: sig, signers: signers}
|
||||
}
|
||||
|
||||
func (s *BLSSignature) Bytes() []byte { return s.sig }
|
||||
func (s *BLSSignature) Type() SignatureType { return SignatureTypeBLS }
|
||||
func (s *BLSSignature) Signers() []ids.NodeID { return s.signers }
|
||||
|
||||
// QuasarSignature combines BLS and Corona signatures for P/Q security
|
||||
type QuasarSignature struct {
|
||||
bls *BLSSignature
|
||||
corona *CoronaSignature
|
||||
}
|
||||
|
||||
func NewQuasarSignature(bls *BLSSignature, corona *CoronaSignature) *QuasarSignature {
|
||||
return &QuasarSignature{bls: bls, corona: corona}
|
||||
}
|
||||
|
||||
func (s *QuasarSignature) Bytes() []byte {
|
||||
// Concatenate BLS + Corona bytes with length prefix
|
||||
blsBytes := s.bls.Bytes()
|
||||
rtBytes := s.corona.Bytes()
|
||||
result := make([]byte, 4+len(blsBytes)+len(rtBytes))
|
||||
// Length of BLS signature (big endian)
|
||||
result[0] = byte(len(blsBytes) >> 24)
|
||||
result[1] = byte(len(blsBytes) >> 16)
|
||||
result[2] = byte(len(blsBytes) >> 8)
|
||||
result[3] = byte(len(blsBytes))
|
||||
copy(result[4:], blsBytes)
|
||||
copy(result[4+len(blsBytes):], rtBytes)
|
||||
return result
|
||||
}
|
||||
|
||||
func (s *QuasarSignature) Type() SignatureType { return SignatureTypeQuasar }
|
||||
|
||||
func (s *QuasarSignature) Signers() []ids.NodeID {
|
||||
// Return intersection of signers (both must sign)
|
||||
return s.bls.Signers()
|
||||
}
|
||||
|
||||
func (s *QuasarSignature) BLS() *BLSSignature { return s.bls }
|
||||
func (s *QuasarSignature) Corona() *CoronaSignature { return s.corona }
|
||||
|
||||
// QuasarSigner combines classical and post-quantum signers
|
||||
type QuasarSigner interface {
|
||||
Signer
|
||||
// SignQuasar signs with both BLS and Corona in parallel
|
||||
SignQuasar(msg []byte) (*QuasarSignature, error)
|
||||
// VerifyQuasar verifies both BLS and Corona signatures
|
||||
VerifyQuasar(msg []byte, sig *QuasarSignature) bool
|
||||
}
|
||||
|
||||
// FinalityProof represents proof of block finality
|
||||
type FinalityProof struct {
|
||||
BlockID ids.ID
|
||||
Height uint64
|
||||
Signature Signature
|
||||
TotalWeight uint64
|
||||
SignerWeight uint64
|
||||
}
|
||||
|
||||
// ValidatorInfo contains validator information for consensus
|
||||
type ValidatorInfo struct {
|
||||
NodeID ids.NodeID
|
||||
Weight uint64
|
||||
Active bool
|
||||
}
|
||||
@@ -1,94 +0,0 @@
|
||||
// Copyright (C) 2019-2026, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
package quasar
|
||||
|
||||
import (
|
||||
qcert "github.com/luxfi/consensus/protocol/quasar"
|
||||
)
|
||||
|
||||
// ValidatorSetProvider resolves the committed validator set the verifier pins a
|
||||
// cert against, for a (chain, epoch). Post-activation the gate calls this once
|
||||
// per verified checkpoint.
|
||||
type ValidatorSetProvider interface {
|
||||
ValidatorSet(chainID uint32, epoch uint64) (qcert.ConsensusValidatorSet, error)
|
||||
}
|
||||
|
||||
// ValidatorSet is a concrete ConsensusValidatorSet for one epoch: the committed
|
||||
// weighted-validator-set root plus the per-leg verification keys the cert legs
|
||||
// verify against (the classical BLS aggregate key for the Beam leg and the
|
||||
// Pulsar ML-DSA threshold group key for the Pulsar leg — the HYBRID_PQ pair).
|
||||
//
|
||||
// Production wiring (the activation seam): in production these fields are
|
||||
// populated from the P-Chain-pinned validator set (Root, Epoch) and the active
|
||||
// KeyEra group keys for the epoch. That population is era/rotation-coupled and
|
||||
// lands with the producer + KeyEra-registry wiring (see producer.go). Corona
|
||||
// (STRICT_DUAL_PQ) and Magnetar/P3Q (POLARIS / RECOVERY) group keys + the
|
||||
// weighted-sig-set config are populated by that same follow-on; until then this
|
||||
// set serves the HYBRID_PQ pair and reports "no key" for the other lanes.
|
||||
type ValidatorSet struct {
|
||||
root [48]byte
|
||||
epoch uint64
|
||||
blsAggKey []byte // classical BLS-12-381 aggregate pubkey (Beam leg)
|
||||
pulsarGroup []byte // Pulsar ML-DSA threshold group pubkey (Pulsar leg)
|
||||
}
|
||||
|
||||
var _ qcert.ConsensusValidatorSet = (*ValidatorSet)(nil)
|
||||
|
||||
// NewValidatorSet builds a committed validator set for one epoch from its root
|
||||
// and the HYBRID_PQ verification keys.
|
||||
func NewValidatorSet(root [48]byte, epoch uint64, blsAggKey, pulsarGroup []byte) *ValidatorSet {
|
||||
return &ValidatorSet{root: root, epoch: epoch, blsAggKey: blsAggKey, pulsarGroup: pulsarGroup}
|
||||
}
|
||||
|
||||
// Root returns the 48-byte weighted-validator-set commitment.
|
||||
func (v *ValidatorSet) Root() [48]byte { return v.root }
|
||||
|
||||
// Epoch returns the epoch this set was committed under.
|
||||
func (v *ValidatorSet) Epoch() uint64 { return v.epoch }
|
||||
|
||||
// WeightedConfig returns the QuorumVerifierConfig for the WeightedSigSet
|
||||
// evidence mode. HYBRID_PQ does not use weighted-sig-set legs; the zero config
|
||||
// is correct here and is populated by the POLARIS / RECOVERY follow-on.
|
||||
func (v *ValidatorSet) WeightedConfig() qcert.QuorumVerifierConfig {
|
||||
return qcert.QuorumVerifierConfig{}
|
||||
}
|
||||
|
||||
// WeightedEnvelope returns the round-digest posture axes for the inner
|
||||
// WeightedQuorumCert. Zero for HYBRID_PQ (no weighted-sig-set leg); populated by
|
||||
// the POLARIS / RECOVERY follow-on.
|
||||
func (v *ValidatorSet) WeightedEnvelope() qcert.QuorumMessageEnvelope {
|
||||
return qcert.QuorumMessageEnvelope{}
|
||||
}
|
||||
|
||||
// ThresholdGroupKey returns the threshold-signature group public key for a leg
|
||||
// kind. Serves the Pulsar (ML-DSA) lane; reports (zero, false) for the others
|
||||
// until their group keys are wired by the follow-on.
|
||||
func (v *ValidatorSet) ThresholdGroupKey(kind qcert.LegKind) (qcert.ThresholdGroupKey, bool) {
|
||||
if kind == qcert.LegPulsarMLDSA && len(v.pulsarGroup) > 0 {
|
||||
return qcert.ThresholdGroupKey{Kind: qcert.LegPulsarMLDSA, PulsarGroupKey: v.pulsarGroup}, true
|
||||
}
|
||||
return qcert.ThresholdGroupKey{}, false
|
||||
}
|
||||
|
||||
// ClassicalAggregateKey returns the classical aggregate verification key for a
|
||||
// scheme. Serves the BLS-12-381 Beam leg.
|
||||
func (v *ValidatorSet) ClassicalAggregateKey(scheme qcert.ClassicalScheme) ([]byte, bool) {
|
||||
if scheme == qcert.ClassicalSchemeBLS12381 && len(v.blsAggKey) > 0 {
|
||||
return v.blsAggKey, true
|
||||
}
|
||||
return nil, false
|
||||
}
|
||||
|
||||
// StaticValidatorSetProvider returns the same committed set for every (chain,
|
||||
// epoch). It is the single-era / test provider; the production provider resolves
|
||||
// per-epoch sets from the P-Chain validator manager + KeyEra registry.
|
||||
type StaticValidatorSetProvider struct{ Set qcert.ConsensusValidatorSet }
|
||||
|
||||
// ValidatorSet implements ValidatorSetProvider.
|
||||
func (p StaticValidatorSetProvider) ValidatorSet(_ uint32, _ uint64) (qcert.ConsensusValidatorSet, error) {
|
||||
if p.Set == nil {
|
||||
return nil, ErrValidatorSetUnavailable
|
||||
}
|
||||
return p.Set, nil
|
||||
}
|
||||
@@ -0,0 +1,378 @@
|
||||
// Copyright (C) 2019-2025, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
// Package zap provides ZAP (Zero-copy Agent Protocol) integration for Lux consensus.
|
||||
//
|
||||
// This package bridges ZAP's agentic consensus with Lux's Quasar threshold signatures,
|
||||
// enabling W3C DID-based validator identity and post-quantum secure finality.
|
||||
package zap
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"sync"
|
||||
|
||||
"github.com/luxfi/ids"
|
||||
"github.com/luxfi/log"
|
||||
"github.com/luxfi/node/consensus/quasar"
|
||||
)
|
||||
|
||||
var (
|
||||
// ErrNotInitialized is returned when the bridge is used before initialization
|
||||
ErrNotInitialized = errors.New("zap bridge not initialized")
|
||||
// ErrValidatorNotFound is returned when a validator is not in the set
|
||||
ErrValidatorNotFound = errors.New("validator not found")
|
||||
// ErrQueryNotFound is returned when a query ID is not known
|
||||
ErrQueryNotFound = errors.New("query not found")
|
||||
// ErrAlreadyVoted is returned when a validator tries to vote twice
|
||||
ErrAlreadyVoted = errors.New("already voted on this query")
|
||||
// ErrInvalidDID is returned when a DID is malformed
|
||||
ErrInvalidDID = errors.New("invalid DID format")
|
||||
)
|
||||
|
||||
// BridgeConfig configures the ZAP-Lux consensus bridge
|
||||
type BridgeConfig struct {
|
||||
// ConsensusThreshold is the fraction of votes needed (0.5 = majority)
|
||||
ConsensusThreshold float64
|
||||
// MinResponses is the minimum responses before checking consensus
|
||||
MinResponses int
|
||||
// MinVotes is the minimum votes before checking consensus
|
||||
MinVotes int
|
||||
// EnablePQCrypto enables post-quantum signatures (ML-DSA-65)
|
||||
EnablePQCrypto bool
|
||||
}
|
||||
|
||||
// DefaultBridgeConfig returns sensible defaults for the bridge
|
||||
func DefaultBridgeConfig() BridgeConfig {
|
||||
return BridgeConfig{
|
||||
ConsensusThreshold: 0.5,
|
||||
MinResponses: 1,
|
||||
MinVotes: 3,
|
||||
EnablePQCrypto: true,
|
||||
}
|
||||
}
|
||||
|
||||
// Bridge connects ZAP agentic consensus to Lux's Quasar finality
|
||||
type Bridge struct {
|
||||
log log.Logger
|
||||
config BridgeConfig
|
||||
quasar *quasar.CoronaCoordinator
|
||||
mu sync.RWMutex
|
||||
queries map[string]*QueryState // QueryID -> QueryState
|
||||
dids map[ids.NodeID]*DID // NodeID -> DID
|
||||
}
|
||||
|
||||
// NewBridge creates a new ZAP-Lux consensus bridge
|
||||
func NewBridge(log log.Logger, config BridgeConfig) *Bridge {
|
||||
return &Bridge{
|
||||
log: log,
|
||||
config: config,
|
||||
queries: make(map[string]*QueryState),
|
||||
dids: make(map[ids.NodeID]*DID),
|
||||
}
|
||||
}
|
||||
|
||||
// Initialize sets up the bridge with a Quasar coordinator
|
||||
func (b *Bridge) Initialize(coordinator *quasar.CoronaCoordinator) error {
|
||||
b.mu.Lock()
|
||||
defer b.mu.Unlock()
|
||||
|
||||
if coordinator == nil {
|
||||
return ErrNotInitialized
|
||||
}
|
||||
|
||||
b.quasar = coordinator
|
||||
b.log.Info("ZAP bridge initialized",
|
||||
log.Int("threshold", coordinator.Threshold()),
|
||||
log.Int("parties", coordinator.NumParties()),
|
||||
log.Bool("pqCrypto", b.config.EnablePQCrypto),
|
||||
)
|
||||
return nil
|
||||
}
|
||||
|
||||
// RegisterValidator associates a DID with a Lux NodeID
|
||||
func (b *Bridge) RegisterValidator(nodeID ids.NodeID, did *DID) error {
|
||||
if did == nil || !did.Valid() {
|
||||
return ErrInvalidDID
|
||||
}
|
||||
|
||||
b.mu.Lock()
|
||||
defer b.mu.Unlock()
|
||||
|
||||
b.dids[nodeID] = did
|
||||
b.log.Debug("Registered validator DID",
|
||||
log.Stringer("nodeID", nodeID),
|
||||
log.String("did", did.String()),
|
||||
)
|
||||
return nil
|
||||
}
|
||||
|
||||
// GetValidatorDID returns the DID for a NodeID
|
||||
func (b *Bridge) GetValidatorDID(nodeID ids.NodeID) (*DID, error) {
|
||||
b.mu.RLock()
|
||||
defer b.mu.RUnlock()
|
||||
|
||||
did, ok := b.dids[nodeID]
|
||||
if !ok {
|
||||
return nil, ErrValidatorNotFound
|
||||
}
|
||||
return did, nil
|
||||
}
|
||||
|
||||
// SubmitQuery creates a new agentic consensus query
|
||||
func (b *Bridge) SubmitQuery(ctx context.Context, queryID string, content []byte, submitter ids.NodeID) error {
|
||||
b.mu.Lock()
|
||||
defer b.mu.Unlock()
|
||||
|
||||
if b.quasar == nil {
|
||||
return ErrNotInitialized
|
||||
}
|
||||
|
||||
submitterDID, ok := b.dids[submitter]
|
||||
if !ok {
|
||||
return ErrValidatorNotFound
|
||||
}
|
||||
|
||||
b.queries[queryID] = &QueryState{
|
||||
ID: queryID,
|
||||
Content: content,
|
||||
Submitter: submitterDID,
|
||||
Responses: make(map[string]*Response),
|
||||
Votes: make(map[string][]*DID),
|
||||
}
|
||||
|
||||
b.log.Debug("Query submitted",
|
||||
log.String("queryID", queryID),
|
||||
log.String("submitter", submitterDID.String()),
|
||||
)
|
||||
return nil
|
||||
}
|
||||
|
||||
// SubmitResponse adds a response to a query
|
||||
func (b *Bridge) SubmitResponse(ctx context.Context, queryID, responseID string, content []byte, responder ids.NodeID) error {
|
||||
b.mu.Lock()
|
||||
defer b.mu.Unlock()
|
||||
|
||||
state, ok := b.queries[queryID]
|
||||
if !ok {
|
||||
return ErrQueryNotFound
|
||||
}
|
||||
|
||||
responderDID, ok := b.dids[responder]
|
||||
if !ok {
|
||||
return ErrValidatorNotFound
|
||||
}
|
||||
|
||||
if state.Finalized != "" {
|
||||
return errors.New("query already finalized")
|
||||
}
|
||||
|
||||
state.Responses[responseID] = &Response{
|
||||
ID: responseID,
|
||||
QueryID: queryID,
|
||||
Content: content,
|
||||
Responder: responderDID,
|
||||
}
|
||||
state.Votes[responseID] = []*DID{}
|
||||
|
||||
b.log.Debug("Response submitted",
|
||||
log.String("queryID", queryID),
|
||||
log.String("responseID", responseID),
|
||||
log.String("responder", responderDID.String()),
|
||||
)
|
||||
return nil
|
||||
}
|
||||
|
||||
// Vote casts a vote for a response
|
||||
func (b *Bridge) Vote(ctx context.Context, queryID, responseID string, voter ids.NodeID) error {
|
||||
b.mu.Lock()
|
||||
defer b.mu.Unlock()
|
||||
|
||||
state, ok := b.queries[queryID]
|
||||
if !ok {
|
||||
return ErrQueryNotFound
|
||||
}
|
||||
|
||||
if state.Finalized != "" {
|
||||
return errors.New("query already finalized")
|
||||
}
|
||||
|
||||
if _, ok := state.Responses[responseID]; !ok {
|
||||
return errors.New("response not found")
|
||||
}
|
||||
|
||||
voterDID, ok := b.dids[voter]
|
||||
if !ok {
|
||||
return ErrValidatorNotFound
|
||||
}
|
||||
|
||||
// Check for double voting (by DID URI)
|
||||
voterURI := voterDID.String()
|
||||
for _, voters := range state.Votes {
|
||||
for _, v := range voters {
|
||||
if v.String() == voterURI {
|
||||
return ErrAlreadyVoted
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
state.Votes[responseID] = append(state.Votes[responseID], voterDID)
|
||||
|
||||
// Check consensus
|
||||
b.checkConsensus(state)
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
func (b *Bridge) checkConsensus(state *QueryState) {
|
||||
if state.Finalized != "" {
|
||||
return
|
||||
}
|
||||
|
||||
if len(state.Responses) < b.config.MinResponses {
|
||||
return
|
||||
}
|
||||
|
||||
totalVotes := 0
|
||||
for _, voters := range state.Votes {
|
||||
totalVotes += len(voters)
|
||||
}
|
||||
|
||||
if totalVotes < b.config.MinVotes {
|
||||
return
|
||||
}
|
||||
|
||||
// Find best response
|
||||
var best struct {
|
||||
id string
|
||||
count int
|
||||
}
|
||||
|
||||
for responseID, voters := range state.Votes {
|
||||
count := len(voters)
|
||||
confidence := float64(count) / float64(totalVotes)
|
||||
|
||||
if confidence >= b.config.ConsensusThreshold {
|
||||
if best.id == "" || count > best.count {
|
||||
best.id = responseID
|
||||
best.count = count
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if best.id != "" {
|
||||
state.Finalized = best.id
|
||||
b.log.Info("Query finalized",
|
||||
log.String("queryID", state.ID),
|
||||
log.String("responseID", best.id),
|
||||
log.Int("votes", best.count),
|
||||
log.Int("total", totalVotes),
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// GetResult returns the consensus result for a query
|
||||
func (b *Bridge) GetResult(queryID string) (*ConsensusResult, error) {
|
||||
b.mu.RLock()
|
||||
defer b.mu.RUnlock()
|
||||
|
||||
state, ok := b.queries[queryID]
|
||||
if !ok {
|
||||
return nil, ErrQueryNotFound
|
||||
}
|
||||
|
||||
if state.Finalized == "" {
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
response := state.Responses[state.Finalized]
|
||||
votes := len(state.Votes[state.Finalized])
|
||||
|
||||
totalVoters := 0
|
||||
for _, v := range state.Votes {
|
||||
totalVoters += len(v)
|
||||
}
|
||||
|
||||
confidence := 0.0
|
||||
if totalVoters > 0 {
|
||||
confidence = float64(votes) / float64(totalVoters)
|
||||
}
|
||||
|
||||
return &ConsensusResult{
|
||||
Response: response,
|
||||
Votes: votes,
|
||||
TotalVoters: totalVoters,
|
||||
Confidence: confidence,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// IsFinalized checks if a query has reached consensus
|
||||
func (b *Bridge) IsFinalized(queryID string) bool {
|
||||
b.mu.RLock()
|
||||
defer b.mu.RUnlock()
|
||||
|
||||
state, ok := b.queries[queryID]
|
||||
return ok && state.Finalized != ""
|
||||
}
|
||||
|
||||
// SignWithQuasar signs a message using Quasar hybrid signatures
|
||||
func (b *Bridge) SignWithQuasar(msg []byte) (quasar.Signature, error) {
|
||||
b.mu.RLock()
|
||||
defer b.mu.RUnlock()
|
||||
|
||||
if b.quasar == nil {
|
||||
return nil, ErrNotInitialized
|
||||
}
|
||||
|
||||
return b.quasar.Sign(msg)
|
||||
}
|
||||
|
||||
// VerifyQuasar verifies a Quasar signature
|
||||
func (b *Bridge) VerifyQuasar(msg []byte, sig quasar.Signature) bool {
|
||||
b.mu.RLock()
|
||||
defer b.mu.RUnlock()
|
||||
|
||||
if b.quasar == nil {
|
||||
return false
|
||||
}
|
||||
|
||||
return b.quasar.Verify(msg, sig)
|
||||
}
|
||||
|
||||
// Stats returns bridge statistics
|
||||
func (b *Bridge) Stats() BridgeStats {
|
||||
b.mu.RLock()
|
||||
defer b.mu.RUnlock()
|
||||
|
||||
active := 0
|
||||
finalized := 0
|
||||
for _, state := range b.queries {
|
||||
if state.Finalized != "" {
|
||||
finalized++
|
||||
} else {
|
||||
active++
|
||||
}
|
||||
}
|
||||
|
||||
var quasarStats quasar.CoronaStats
|
||||
if b.quasar != nil {
|
||||
quasarStats = b.quasar.Stats()
|
||||
}
|
||||
|
||||
return BridgeStats{
|
||||
RegisteredValidators: len(b.dids),
|
||||
ActiveQueries: active,
|
||||
FinalizedQueries: finalized,
|
||||
QuasarInitialized: b.quasar != nil && b.quasar.IsInitialized(),
|
||||
QuasarStats: quasarStats,
|
||||
}
|
||||
}
|
||||
|
||||
// BridgeStats contains statistics about the bridge
|
||||
type BridgeStats struct {
|
||||
RegisteredValidators int
|
||||
ActiveQueries int
|
||||
FinalizedQueries int
|
||||
QuasarInitialized bool
|
||||
QuasarStats quasar.CoronaStats
|
||||
}
|
||||
@@ -0,0 +1,459 @@
|
||||
// Copyright (C) 2019-2025, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
package zap
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
|
||||
"github.com/luxfi/ids"
|
||||
"github.com/luxfi/log"
|
||||
"github.com/luxfi/node/consensus/quasar"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func TestParseDID(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
input string
|
||||
want *DID
|
||||
wantErr bool
|
||||
}{
|
||||
{
|
||||
name: "valid did:lux",
|
||||
input: "did:lux:z6MkhaXgBZDvotDkL5257faiztiGiC2QtKLGpbnnEGta2doK",
|
||||
want: &DID{
|
||||
Method: DIDMethodLux,
|
||||
ID: "z6MkhaXgBZDvotDkL5257faiztiGiC2QtKLGpbnnEGta2doK",
|
||||
},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "valid did:key",
|
||||
input: "did:key:z6MkhaXgBZDvotDkL5257faiztiGiC2QtKLGpbnnEGta2doK",
|
||||
want: &DID{
|
||||
Method: DIDMethodKey,
|
||||
ID: "z6MkhaXgBZDvotDkL5257faiztiGiC2QtKLGpbnnEGta2doK",
|
||||
},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "valid did:web",
|
||||
input: "did:web:example.com:users:alice",
|
||||
want: &DID{
|
||||
Method: DIDMethodWeb,
|
||||
ID: "example.com:users:alice",
|
||||
},
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "invalid - no did prefix",
|
||||
input: "lux:z6MkhaXgBZDvotDkL5257faiztiGiC2QtKLGpbnnEGta2doK",
|
||||
wantErr: true,
|
||||
},
|
||||
{
|
||||
name: "invalid - unknown method",
|
||||
input: "did:unknown:abc123",
|
||||
wantErr: true,
|
||||
},
|
||||
{
|
||||
name: "invalid - empty id",
|
||||
input: "did:lux:",
|
||||
wantErr: true,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
got, err := ParseDID(tt.input)
|
||||
if tt.wantErr {
|
||||
require.Error(t, err)
|
||||
return
|
||||
}
|
||||
require.NoError(t, err)
|
||||
require.Equal(t, tt.want.Method, got.Method)
|
||||
require.Equal(t, tt.want.ID, got.ID)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestDIDString(t *testing.T) {
|
||||
did := &DID{Method: DIDMethodLux, ID: "z6MkTest"}
|
||||
require.Equal(t, "did:lux:z6MkTest", did.String())
|
||||
}
|
||||
|
||||
func TestDIDValid(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
did *DID
|
||||
valid bool
|
||||
}{
|
||||
{
|
||||
name: "valid lux",
|
||||
did: &DID{Method: DIDMethodLux, ID: "z6MkTest"},
|
||||
valid: true,
|
||||
},
|
||||
{
|
||||
name: "valid key",
|
||||
did: &DID{Method: DIDMethodKey, ID: "z6MkTest"},
|
||||
valid: true,
|
||||
},
|
||||
{
|
||||
name: "valid web",
|
||||
did: &DID{Method: DIDMethodWeb, ID: "example.com"},
|
||||
valid: true,
|
||||
},
|
||||
{
|
||||
name: "nil did",
|
||||
did: nil,
|
||||
valid: false,
|
||||
},
|
||||
{
|
||||
name: "empty id",
|
||||
did: &DID{Method: DIDMethodLux, ID: ""},
|
||||
valid: false,
|
||||
},
|
||||
{
|
||||
name: "unknown method",
|
||||
did: &DID{Method: "unknown", ID: "test"},
|
||||
valid: false,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
require.Equal(t, tt.valid, tt.did.Valid())
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestDIDFromNodeID(t *testing.T) {
|
||||
nodeID := ids.GenerateTestNodeID()
|
||||
did := DIDFromNodeID(nodeID)
|
||||
|
||||
require.NotNil(t, did)
|
||||
require.Equal(t, DIDMethodLux, did.Method)
|
||||
require.True(t, did.Valid())
|
||||
require.Contains(t, did.String(), "did:lux:")
|
||||
}
|
||||
|
||||
func TestDIDFromWeb(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
domain string
|
||||
path string
|
||||
wantID string
|
||||
wantErr bool
|
||||
}{
|
||||
{
|
||||
name: "domain only",
|
||||
domain: "example.com",
|
||||
path: "",
|
||||
wantID: "example.com",
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "domain with path",
|
||||
domain: "example.com",
|
||||
path: "users/alice",
|
||||
wantID: "example.com:users:alice",
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "empty domain",
|
||||
domain: "",
|
||||
path: "",
|
||||
wantErr: true,
|
||||
},
|
||||
{
|
||||
name: "domain with slash",
|
||||
domain: "example/com",
|
||||
path: "",
|
||||
wantErr: true,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
got, err := DIDFromWeb(tt.domain, tt.path)
|
||||
if tt.wantErr {
|
||||
require.Error(t, err)
|
||||
return
|
||||
}
|
||||
require.NoError(t, err)
|
||||
require.Equal(t, DIDMethodWeb, got.Method)
|
||||
require.Equal(t, tt.wantID, got.ID)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestGenerateDocument(t *testing.T) {
|
||||
did := &DID{Method: DIDMethodLux, ID: "z6MkTest"}
|
||||
doc, err := did.GenerateDocument()
|
||||
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, doc)
|
||||
require.Equal(t, "did:lux:z6MkTest", doc.ID)
|
||||
require.Len(t, doc.Context, 2)
|
||||
require.Len(t, doc.VerificationMethod, 1)
|
||||
require.Len(t, doc.Authentication, 1)
|
||||
require.Len(t, doc.Service, 1)
|
||||
require.Equal(t, "ZapAgent", doc.Service[0].Type)
|
||||
}
|
||||
|
||||
func TestBridgeNew(t *testing.T) {
|
||||
logger := log.Noop()
|
||||
config := DefaultBridgeConfig()
|
||||
bridge := NewBridge(logger, config)
|
||||
|
||||
require.NotNil(t, bridge)
|
||||
require.Equal(t, 0.5, bridge.config.ConsensusThreshold)
|
||||
}
|
||||
|
||||
func TestBridgeInitialize(t *testing.T) {
|
||||
logger := log.Noop()
|
||||
bridge := NewBridge(logger, DefaultBridgeConfig())
|
||||
|
||||
// Test with nil coordinator
|
||||
err := bridge.Initialize(nil)
|
||||
require.ErrorIs(t, err, ErrNotInitialized)
|
||||
|
||||
// Test with valid coordinator
|
||||
coordinator, err := quasar.NewTestCoronaCoordinator(logger, quasar.CoronaConfig{
|
||||
NumParties: 4,
|
||||
Threshold: 3,
|
||||
PartyIndex: 0,
|
||||
})
|
||||
require.NoError(t, err)
|
||||
|
||||
err = bridge.Initialize(coordinator)
|
||||
require.NoError(t, err)
|
||||
}
|
||||
|
||||
func TestBridgeRegisterValidator(t *testing.T) {
|
||||
logger := log.Noop()
|
||||
bridge := NewBridge(logger, DefaultBridgeConfig())
|
||||
|
||||
nodeID := ids.GenerateTestNodeID()
|
||||
did := DIDFromNodeID(nodeID)
|
||||
|
||||
// Register validator
|
||||
err := bridge.RegisterValidator(nodeID, did)
|
||||
require.NoError(t, err)
|
||||
|
||||
// Get validator DID
|
||||
got, err := bridge.GetValidatorDID(nodeID)
|
||||
require.NoError(t, err)
|
||||
require.Equal(t, did.String(), got.String())
|
||||
|
||||
// Unknown validator
|
||||
_, err = bridge.GetValidatorDID(ids.GenerateTestNodeID())
|
||||
require.ErrorIs(t, err, ErrValidatorNotFound)
|
||||
|
||||
// Invalid DID
|
||||
err = bridge.RegisterValidator(nodeID, nil)
|
||||
require.ErrorIs(t, err, ErrInvalidDID)
|
||||
}
|
||||
|
||||
func TestBridgeConsensusFlow(t *testing.T) {
|
||||
logger := log.Noop()
|
||||
bridge := NewBridge(logger, BridgeConfig{
|
||||
ConsensusThreshold: 0.5,
|
||||
MinResponses: 1,
|
||||
MinVotes: 2,
|
||||
EnablePQCrypto: true,
|
||||
})
|
||||
|
||||
// Initialize with coordinator
|
||||
coordinator, err := quasar.NewTestCoronaCoordinator(logger, quasar.CoronaConfig{
|
||||
NumParties: 4,
|
||||
Threshold: 3,
|
||||
PartyIndex: 0,
|
||||
})
|
||||
require.NoError(t, err)
|
||||
require.NoError(t, bridge.Initialize(coordinator))
|
||||
|
||||
// Register validators
|
||||
submitter := ids.GenerateTestNodeID()
|
||||
responder := ids.GenerateTestNodeID()
|
||||
voter1 := ids.GenerateTestNodeID()
|
||||
voter2 := ids.GenerateTestNodeID()
|
||||
|
||||
require.NoError(t, bridge.RegisterValidator(submitter, DIDFromNodeID(submitter)))
|
||||
require.NoError(t, bridge.RegisterValidator(responder, DIDFromNodeID(responder)))
|
||||
require.NoError(t, bridge.RegisterValidator(voter1, DIDFromNodeID(voter1)))
|
||||
require.NoError(t, bridge.RegisterValidator(voter2, DIDFromNodeID(voter2)))
|
||||
|
||||
ctx := context.Background()
|
||||
|
||||
// Submit query
|
||||
queryID := "query123"
|
||||
err = bridge.SubmitQuery(ctx, queryID, []byte("What is 2+2?"), submitter)
|
||||
require.NoError(t, err)
|
||||
|
||||
// Submit response
|
||||
responseID := "response456"
|
||||
err = bridge.SubmitResponse(ctx, queryID, responseID, []byte("4"), responder)
|
||||
require.NoError(t, err)
|
||||
|
||||
// Vote
|
||||
require.NoError(t, bridge.Vote(ctx, queryID, responseID, voter1))
|
||||
require.False(t, bridge.IsFinalized(queryID)) // Not enough votes yet
|
||||
|
||||
require.NoError(t, bridge.Vote(ctx, queryID, responseID, voter2))
|
||||
require.True(t, bridge.IsFinalized(queryID)) // Now finalized
|
||||
|
||||
// Get result
|
||||
result, err := bridge.GetResult(queryID)
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, result)
|
||||
require.Equal(t, responseID, result.Response.ID)
|
||||
require.Equal(t, 2, result.Votes)
|
||||
require.Equal(t, 2, result.TotalVoters)
|
||||
require.Equal(t, 1.0, result.Confidence)
|
||||
}
|
||||
|
||||
func TestBridgeDoubleVote(t *testing.T) {
|
||||
logger := log.Noop()
|
||||
bridge := NewBridge(logger, DefaultBridgeConfig())
|
||||
|
||||
coordinator, _ := quasar.NewTestCoronaCoordinator(logger, quasar.CoronaConfig{
|
||||
NumParties: 2,
|
||||
Threshold: 1,
|
||||
})
|
||||
bridge.Initialize(coordinator)
|
||||
|
||||
submitter := ids.GenerateTestNodeID()
|
||||
voter := ids.GenerateTestNodeID()
|
||||
|
||||
bridge.RegisterValidator(submitter, DIDFromNodeID(submitter))
|
||||
bridge.RegisterValidator(voter, DIDFromNodeID(voter))
|
||||
|
||||
ctx := context.Background()
|
||||
queryID := "q1"
|
||||
responseID := "r1"
|
||||
|
||||
bridge.SubmitQuery(ctx, queryID, []byte("test"), submitter)
|
||||
bridge.SubmitResponse(ctx, queryID, responseID, []byte("answer"), submitter)
|
||||
|
||||
// First vote succeeds
|
||||
require.NoError(t, bridge.Vote(ctx, queryID, responseID, voter))
|
||||
|
||||
// Second vote fails
|
||||
err := bridge.Vote(ctx, queryID, responseID, voter)
|
||||
require.ErrorIs(t, err, ErrAlreadyVoted)
|
||||
}
|
||||
|
||||
func TestBridgeStats(t *testing.T) {
|
||||
logger := log.Noop()
|
||||
bridge := NewBridge(logger, DefaultBridgeConfig())
|
||||
|
||||
coordinator, _ := quasar.NewTestCoronaCoordinator(logger, quasar.CoronaConfig{
|
||||
NumParties: 4,
|
||||
Threshold: 3,
|
||||
})
|
||||
bridge.Initialize(coordinator)
|
||||
|
||||
nodeID := ids.GenerateTestNodeID()
|
||||
bridge.RegisterValidator(nodeID, DIDFromNodeID(nodeID))
|
||||
|
||||
stats := bridge.Stats()
|
||||
require.Equal(t, 1, stats.RegisteredValidators)
|
||||
require.Equal(t, 0, stats.ActiveQueries)
|
||||
require.Equal(t, 0, stats.FinalizedQueries)
|
||||
require.False(t, stats.QuasarInitialized) // Not initialized with validators
|
||||
}
|
||||
|
||||
func TestNewQuery(t *testing.T) {
|
||||
did := &DID{Method: DIDMethodLux, ID: "z6MkTest"}
|
||||
query := NewQuery([]byte("What is 2+2?"), did)
|
||||
|
||||
require.NotEmpty(t, query.ID)
|
||||
require.Equal(t, 64, len(query.ID)) // SHA-256 hex
|
||||
require.Equal(t, []byte("What is 2+2?"), query.Content)
|
||||
require.Equal(t, did, query.Submitter)
|
||||
require.Greater(t, query.Timestamp, int64(0))
|
||||
}
|
||||
|
||||
func TestNewResponse(t *testing.T) {
|
||||
did := &DID{Method: DIDMethodLux, ID: "z6MkTest"}
|
||||
queryID := "0000000000000000000000000000000000000000000000000000000000000000"
|
||||
response := NewResponse(queryID, []byte("4"), did)
|
||||
|
||||
require.NotEmpty(t, response.ID)
|
||||
require.Equal(t, 64, len(response.ID)) // SHA-256 hex
|
||||
require.Equal(t, queryID, response.QueryID)
|
||||
require.Equal(t, []byte("4"), response.Content)
|
||||
require.Equal(t, did, response.Responder)
|
||||
}
|
||||
|
||||
func TestInMemoryStakeRegistry(t *testing.T) {
|
||||
registry := NewInMemoryStakeRegistry()
|
||||
did := &DID{Method: DIDMethodLux, ID: "z6MkTest"}
|
||||
|
||||
// Initial stake is 0
|
||||
stake, err := registry.GetStake(did)
|
||||
require.NoError(t, err)
|
||||
require.Equal(t, uint64(0), stake)
|
||||
|
||||
// Set stake
|
||||
require.NoError(t, registry.SetStake(did, 1000))
|
||||
|
||||
// Get stake
|
||||
stake, err = registry.GetStake(did)
|
||||
require.NoError(t, err)
|
||||
require.Equal(t, uint64(1000), stake)
|
||||
|
||||
// Total stake
|
||||
require.Equal(t, uint64(1000), registry.TotalStake())
|
||||
|
||||
// Has sufficient stake
|
||||
has, err := registry.HasSufficientStake(did, 500)
|
||||
require.NoError(t, err)
|
||||
require.True(t, has)
|
||||
|
||||
has, err = registry.HasSufficientStake(did, 2000)
|
||||
require.NoError(t, err)
|
||||
require.False(t, has)
|
||||
|
||||
// Stake weight
|
||||
weight, err := registry.StakeWeight(did)
|
||||
require.NoError(t, err)
|
||||
require.Equal(t, 1.0, weight) // Only staker
|
||||
}
|
||||
|
||||
func TestAgentMessageType(t *testing.T) {
|
||||
require.Equal(t, "Query", AgentMessageTypeQuery.String())
|
||||
require.Equal(t, "Response", AgentMessageTypeResponse.String())
|
||||
require.Equal(t, "Vote", AgentMessageTypeVote.String())
|
||||
require.Equal(t, "Finality", AgentMessageTypeFinality.String())
|
||||
require.Equal(t, "Unknown", AgentMessageType(255).String())
|
||||
}
|
||||
|
||||
func TestBase58EncodeDecode(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
data []byte
|
||||
}{
|
||||
{"empty", []byte{}},
|
||||
{"single byte", []byte{0x01}},
|
||||
{"multiple bytes", []byte{0x01, 0x02, 0x03, 0x04}},
|
||||
{"leading zeros", []byte{0x00, 0x00, 0x01, 0x02}},
|
||||
{"all zeros", []byte{0x00, 0x00, 0x00}},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
encoded := base58Encode(tt.data)
|
||||
decoded, err := base58Decode(encoded)
|
||||
require.NoError(t, err)
|
||||
require.Equal(t, tt.data, decoded)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestBase58DecodeInvalidChar(t *testing.T) {
|
||||
_, err := base58Decode("0OIl") // Invalid chars
|
||||
require.Error(t, err)
|
||||
}
|
||||
@@ -0,0 +1,355 @@
|
||||
// Copyright (C) 2019-2025, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
package zap
|
||||
|
||||
import (
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"github.com/luxfi/ids"
|
||||
)
|
||||
|
||||
const (
|
||||
// MethodLux is the did:lux method for blockchain-anchored DIDs
|
||||
MethodLux = "lux"
|
||||
// MethodKey is the did:key method for self-certifying DIDs
|
||||
MethodKey = "key"
|
||||
// MethodWeb is the did:web method for DNS-based DIDs
|
||||
MethodWeb = "web"
|
||||
|
||||
// MLDSAPublicKeySize is the expected size of ML-DSA-65 public keys
|
||||
MLDSAPublicKeySize = 1952
|
||||
|
||||
// MultibaseBase58BTC is the multibase prefix for base58btc
|
||||
MultibaseBase58BTC = 'z'
|
||||
)
|
||||
|
||||
// MulticodecMLDSA65 is the provisional multicodec prefix for ML-DSA-65
|
||||
var MulticodecMLDSA65 = []byte{0x13, 0x09}
|
||||
|
||||
// DIDMethod represents a DID method identifier
|
||||
type DIDMethod string
|
||||
|
||||
const (
|
||||
DIDMethodLux DIDMethod = MethodLux
|
||||
DIDMethodKey DIDMethod = MethodKey
|
||||
DIDMethodWeb DIDMethod = MethodWeb
|
||||
)
|
||||
|
||||
// DID represents a W3C Decentralized Identifier
|
||||
type DID struct {
|
||||
Method DIDMethod
|
||||
ID string
|
||||
}
|
||||
|
||||
// NewDID creates a DID from method and identifier
|
||||
func NewDID(method DIDMethod, id string) *DID {
|
||||
return &DID{Method: method, ID: id}
|
||||
}
|
||||
|
||||
// ParseDID parses a DID from a string in format "did:method:id"
|
||||
func ParseDID(s string) (*DID, error) {
|
||||
if !strings.HasPrefix(s, "did:") {
|
||||
return nil, fmt.Errorf("%w: must start with 'did:'", ErrInvalidDID)
|
||||
}
|
||||
|
||||
rest := s[4:] // Skip "did:"
|
||||
colonIndex := strings.Index(rest, ":")
|
||||
if colonIndex == -1 {
|
||||
return nil, fmt.Errorf("%w: expected 'did:method:id'", ErrInvalidDID)
|
||||
}
|
||||
|
||||
methodStr := rest[:colonIndex]
|
||||
id := rest[colonIndex+1:]
|
||||
|
||||
if id == "" {
|
||||
return nil, fmt.Errorf("%w: identifier cannot be empty", ErrInvalidDID)
|
||||
}
|
||||
|
||||
var method DIDMethod
|
||||
switch methodStr {
|
||||
case MethodLux:
|
||||
method = DIDMethodLux
|
||||
case MethodKey:
|
||||
method = DIDMethodKey
|
||||
case MethodWeb:
|
||||
method = DIDMethodWeb
|
||||
default:
|
||||
return nil, fmt.Errorf("%w: unknown method '%s'", ErrInvalidDID, methodStr)
|
||||
}
|
||||
|
||||
return &DID{Method: method, ID: id}, nil
|
||||
}
|
||||
|
||||
// String returns the full DID URI
|
||||
func (d *DID) String() string {
|
||||
if d == nil {
|
||||
return ""
|
||||
}
|
||||
return fmt.Sprintf("did:%s:%s", d.Method, d.ID)
|
||||
}
|
||||
|
||||
// Valid checks if the DID is well-formed
|
||||
func (d *DID) Valid() bool {
|
||||
if d == nil || d.ID == "" {
|
||||
return false
|
||||
}
|
||||
switch d.Method {
|
||||
case DIDMethodLux, DIDMethodKey, DIDMethodWeb:
|
||||
return true
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
// DIDFromNodeID creates a did:lux from a Lux NodeID
|
||||
func DIDFromNodeID(nodeID ids.NodeID) *DID {
|
||||
// Use multibase-encoded NodeID bytes
|
||||
encoded := base58Encode(nodeID[:])
|
||||
return &DID{
|
||||
Method: DIDMethodLux,
|
||||
ID: string(MultibaseBase58BTC) + encoded,
|
||||
}
|
||||
}
|
||||
|
||||
// DIDFromPublicKey creates a did:key from an ML-DSA-65 public key
|
||||
func DIDFromPublicKey(publicKey []byte) (*DID, error) {
|
||||
if len(publicKey) != MLDSAPublicKeySize {
|
||||
return nil, fmt.Errorf("invalid ML-DSA public key size: expected %d, got %d",
|
||||
MLDSAPublicKeySize, len(publicKey))
|
||||
}
|
||||
|
||||
// Prefix with multicodec for ML-DSA-65
|
||||
prefixed := make([]byte, len(MulticodecMLDSA65)+len(publicKey))
|
||||
copy(prefixed, MulticodecMLDSA65)
|
||||
copy(prefixed[len(MulticodecMLDSA65):], publicKey)
|
||||
|
||||
// Encode with multibase (base58btc)
|
||||
encoded := base58Encode(prefixed)
|
||||
|
||||
return &DID{
|
||||
Method: DIDMethodKey,
|
||||
ID: string(MultibaseBase58BTC) + encoded,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// DIDFromWeb creates a did:web from a domain and optional path
|
||||
func DIDFromWeb(domain string, path string) (*DID, error) {
|
||||
if domain == "" {
|
||||
return nil, fmt.Errorf("%w: domain cannot be empty", ErrInvalidDID)
|
||||
}
|
||||
if strings.ContainsAny(domain, "/:") {
|
||||
return nil, fmt.Errorf("%w: domain cannot contain '/' or ':'", ErrInvalidDID)
|
||||
}
|
||||
|
||||
var id string
|
||||
if path != "" {
|
||||
// Replace '/' with ':' per did:web spec
|
||||
pathParts := strings.ReplaceAll(path, "/", ":")
|
||||
id = domain + ":" + pathParts
|
||||
} else {
|
||||
id = domain
|
||||
}
|
||||
|
||||
return &DID{Method: DIDMethodWeb, ID: id}, nil
|
||||
}
|
||||
|
||||
// ExtractKeyMaterial extracts raw key bytes from did:key or did:lux
|
||||
func (d *DID) ExtractKeyMaterial() ([]byte, error) {
|
||||
if d == nil || d.ID == "" {
|
||||
return nil, fmt.Errorf("%w: empty identifier", ErrInvalidDID)
|
||||
}
|
||||
|
||||
if d.ID[0] != MultibaseBase58BTC {
|
||||
return nil, fmt.Errorf("%w: unsupported multibase encoding", ErrInvalidDID)
|
||||
}
|
||||
|
||||
// Decode base58btc (skip multibase prefix)
|
||||
decoded, err := base58Decode(d.ID[1:])
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("%w: %v", ErrInvalidDID, err)
|
||||
}
|
||||
|
||||
if len(decoded) < 2 {
|
||||
return nil, fmt.Errorf("%w: identifier too short", ErrInvalidDID)
|
||||
}
|
||||
|
||||
// Skip multicodec prefix if it matches ML-DSA-65
|
||||
if len(decoded) >= 2 && decoded[0] == MulticodecMLDSA65[0] && decoded[1] == MulticodecMLDSA65[1] {
|
||||
return decoded[2:], nil
|
||||
}
|
||||
|
||||
return decoded, nil
|
||||
}
|
||||
|
||||
// Hash returns a 32-byte hash of the DID for indexing
|
||||
func (d *DID) Hash() ids.ID {
|
||||
h := sha256.Sum256([]byte(d.String()))
|
||||
var id ids.ID
|
||||
copy(id[:], h[:])
|
||||
return id
|
||||
}
|
||||
|
||||
// ToHex returns the DID hash as a hex string
|
||||
func (d *DID) ToHex() string {
|
||||
id := d.Hash()
|
||||
return hex.EncodeToString(id[:])
|
||||
}
|
||||
|
||||
// VerificationMethod represents a verification method in a DID Document
|
||||
type VerificationMethod struct {
|
||||
ID string
|
||||
Type string
|
||||
Controller string
|
||||
PublicKeyMultibase string
|
||||
BlockchainAccountID string
|
||||
}
|
||||
|
||||
// Service represents a service endpoint in a DID Document
|
||||
type Service struct {
|
||||
ID string
|
||||
Type string
|
||||
ServiceEndpoint string
|
||||
}
|
||||
|
||||
// DIDDocument represents a W3C DID Document
|
||||
type DIDDocument struct {
|
||||
Context []string
|
||||
ID string
|
||||
Controller string
|
||||
VerificationMethod []VerificationMethod
|
||||
Authentication []string
|
||||
AssertionMethod []string
|
||||
KeyAgreement []string
|
||||
CapabilityInvocation []string
|
||||
CapabilityDelegation []string
|
||||
Service []Service
|
||||
}
|
||||
|
||||
// GenerateDocument creates a DID Document for a DID
|
||||
func (d *DID) GenerateDocument() (*DIDDocument, error) {
|
||||
if !d.Valid() {
|
||||
return nil, fmt.Errorf("%w: invalid DID", ErrInvalidDID)
|
||||
}
|
||||
|
||||
uri := d.String()
|
||||
keyID := uri + "#keys-1"
|
||||
|
||||
var vm VerificationMethod
|
||||
switch d.Method {
|
||||
case DIDMethodLux, DIDMethodKey:
|
||||
vm = VerificationMethod{
|
||||
ID: keyID,
|
||||
Type: "JsonWebKey2020",
|
||||
Controller: uri,
|
||||
PublicKeyMultibase: d.ID,
|
||||
}
|
||||
if d.Method == DIDMethodLux {
|
||||
// Add blockchain account ID
|
||||
keyMaterial, err := d.ExtractKeyMaterial()
|
||||
if err == nil && len(keyMaterial) >= 20 {
|
||||
vm.BlockchainAccountID = "lux:" + hex.EncodeToString(keyMaterial[:20])
|
||||
}
|
||||
}
|
||||
case DIDMethodWeb:
|
||||
vm = VerificationMethod{
|
||||
ID: keyID,
|
||||
Type: "JsonWebKey2020",
|
||||
Controller: uri,
|
||||
}
|
||||
}
|
||||
|
||||
return &DIDDocument{
|
||||
Context: []string{
|
||||
"https://www.w3.org/ns/did/v1",
|
||||
"https://w3id.org/security/suites/jws-2020/v1",
|
||||
},
|
||||
ID: uri,
|
||||
VerificationMethod: []VerificationMethod{vm},
|
||||
Authentication: []string{keyID},
|
||||
AssertionMethod: []string{keyID},
|
||||
CapabilityInvocation: []string{keyID},
|
||||
Service: []Service{
|
||||
{
|
||||
ID: uri + "#zap-agent",
|
||||
Type: "ZapAgent",
|
||||
ServiceEndpoint: "zap://" + d.ID,
|
||||
},
|
||||
},
|
||||
}, nil
|
||||
}
|
||||
|
||||
// Base58 encoding/decoding (Bitcoin alphabet)
|
||||
const base58Alphabet = "123456789ABCDEFGHJKLMNPQRSTUVWXYZabcdefghijkmnopqrstuvwxyz"
|
||||
|
||||
func base58Encode(data []byte) string {
|
||||
// Convert to big integer
|
||||
result := make([]byte, 0, len(data)*2)
|
||||
for _, b := range data {
|
||||
carry := int(b)
|
||||
for i := 0; i < len(result); i++ {
|
||||
carry += int(result[i]) << 8
|
||||
result[i] = byte(carry % 58)
|
||||
carry /= 58
|
||||
}
|
||||
for carry > 0 {
|
||||
result = append(result, byte(carry%58))
|
||||
carry /= 58
|
||||
}
|
||||
}
|
||||
|
||||
// Handle leading zeros
|
||||
for _, b := range data {
|
||||
if b != 0 {
|
||||
break
|
||||
}
|
||||
result = append(result, 0)
|
||||
}
|
||||
|
||||
// Reverse and convert to alphabet
|
||||
output := make([]byte, len(result))
|
||||
for i := range result {
|
||||
output[len(result)-1-i] = base58Alphabet[result[i]]
|
||||
}
|
||||
|
||||
return string(output)
|
||||
}
|
||||
|
||||
func base58Decode(s string) ([]byte, error) {
|
||||
result := make([]byte, 0, len(s))
|
||||
for _, c := range s {
|
||||
index := strings.IndexRune(base58Alphabet, c)
|
||||
if index == -1 {
|
||||
return nil, fmt.Errorf("invalid base58 character: %c", c)
|
||||
}
|
||||
|
||||
carry := index
|
||||
for i := 0; i < len(result); i++ {
|
||||
carry += int(result[i]) * 58
|
||||
result[i] = byte(carry)
|
||||
carry >>= 8
|
||||
}
|
||||
for carry > 0 {
|
||||
result = append(result, byte(carry))
|
||||
carry >>= 8
|
||||
}
|
||||
}
|
||||
|
||||
// Handle leading ones
|
||||
for _, c := range s {
|
||||
if c != rune(base58Alphabet[0]) {
|
||||
break
|
||||
}
|
||||
result = append(result, 0)
|
||||
}
|
||||
|
||||
// Reverse
|
||||
for i, j := 0, len(result)-1; i < j; i, j = i+1, j-1 {
|
||||
result[i], result[j] = result[j], result[i]
|
||||
}
|
||||
|
||||
return result, nil
|
||||
}
|
||||
@@ -0,0 +1,260 @@
|
||||
// Copyright (C) 2019-2025, Lux Industries Inc. All rights reserved.
|
||||
// See the file LICENSE for licensing terms.
|
||||
|
||||
package zap
|
||||
|
||||
import (
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"time"
|
||||
)
|
||||
|
||||
// QueryState represents the state of a consensus query
|
||||
type QueryState struct {
|
||||
ID string
|
||||
Content []byte
|
||||
Submitter *DID
|
||||
Timestamp time.Time
|
||||
Responses map[string]*Response
|
||||
Votes map[string][]*DID // ResponseID -> list of voter DIDs
|
||||
Finalized string // ResponseID if finalized
|
||||
}
|
||||
|
||||
// Response represents a response to a query
|
||||
type Response struct {
|
||||
ID string
|
||||
QueryID string
|
||||
Content []byte
|
||||
Responder *DID
|
||||
Timestamp time.Time
|
||||
}
|
||||
|
||||
// ConsensusResult represents the outcome of consensus voting
|
||||
type ConsensusResult struct {
|
||||
Response *Response
|
||||
Votes int
|
||||
TotalVoters int
|
||||
Confidence float64
|
||||
}
|
||||
|
||||
// Query represents an agentic consensus query
|
||||
type Query struct {
|
||||
ID string
|
||||
Content []byte
|
||||
Submitter *DID
|
||||
Timestamp int64
|
||||
}
|
||||
|
||||
// NewQuery creates a query with auto-generated ID
|
||||
func NewQuery(content []byte, submitter *DID) *Query {
|
||||
timestamp := time.Now().Unix()
|
||||
data := append(content, []byte(submitter.String())...)
|
||||
data = append(data, int64ToBytes(timestamp)...)
|
||||
hash := sha256.Sum256(data)
|
||||
|
||||
return &Query{
|
||||
ID: hex.EncodeToString(hash[:]),
|
||||
Content: content,
|
||||
Submitter: submitter,
|
||||
Timestamp: timestamp,
|
||||
}
|
||||
}
|
||||
|
||||
// NewResponse creates a response with auto-generated ID
|
||||
func NewResponse(queryID string, content []byte, responder *DID) *Response {
|
||||
timestamp := time.Now()
|
||||
queryIDBytes, _ := hex.DecodeString(queryID)
|
||||
data := append(queryIDBytes, content...)
|
||||
data = append(data, []byte(responder.String())...)
|
||||
data = append(data, int64ToBytes(timestamp.Unix())...)
|
||||
hash := sha256.Sum256(data)
|
||||
|
||||
return &Response{
|
||||
ID: hex.EncodeToString(hash[:]),
|
||||
QueryID: queryID,
|
||||
Content: content,
|
||||
Responder: responder,
|
||||
Timestamp: timestamp,
|
||||
}
|
||||
}
|
||||
|
||||
// Vote represents a vote cast by a validator
|
||||
type Vote struct {
|
||||
QueryID string
|
||||
ResponseID string
|
||||
Voter *DID
|
||||
Timestamp int64
|
||||
Signature []byte // Optional post-quantum signature
|
||||
}
|
||||
|
||||
// NewVote creates a new vote
|
||||
func NewVote(queryID, responseID string, voter *DID) *Vote {
|
||||
return &Vote{
|
||||
QueryID: queryID,
|
||||
ResponseID: responseID,
|
||||
Voter: voter,
|
||||
Timestamp: time.Now().Unix(),
|
||||
}
|
||||
}
|
||||
|
||||
// FinalityProof represents proof of agentic consensus finality
|
||||
type FinalityProof struct {
|
||||
QueryID string
|
||||
ResponseID string
|
||||
Votes []Vote
|
||||
TotalVoters int
|
||||
Confidence float64
|
||||
Timestamp int64
|
||||
Signature []byte // Quasar hybrid signature
|
||||
}
|
||||
|
||||
// ValidatorWeight represents a validator's weight in consensus
|
||||
type ValidatorWeight struct {
|
||||
DID *DID
|
||||
Weight uint64
|
||||
Stake uint64
|
||||
Active bool
|
||||
}
|
||||
|
||||
// AgentMessage represents a ZAP protocol message for agentic consensus
|
||||
type AgentMessage struct {
|
||||
Type AgentMessageType
|
||||
Query *Query
|
||||
Response *Response
|
||||
Vote *Vote
|
||||
Signature []byte
|
||||
}
|
||||
|
||||
// AgentMessageType identifies the type of agent message
|
||||
type AgentMessageType uint8
|
||||
|
||||
const (
|
||||
AgentMessageTypeQuery AgentMessageType = iota
|
||||
AgentMessageTypeResponse
|
||||
AgentMessageTypeVote
|
||||
AgentMessageTypeFinality
|
||||
)
|
||||
|
||||
// String returns the message type name
|
||||
func (t AgentMessageType) String() string {
|
||||
switch t {
|
||||
case AgentMessageTypeQuery:
|
||||
return "Query"
|
||||
case AgentMessageTypeResponse:
|
||||
return "Response"
|
||||
case AgentMessageTypeVote:
|
||||
return "Vote"
|
||||
case AgentMessageTypeFinality:
|
||||
return "Finality"
|
||||
default:
|
||||
return "Unknown"
|
||||
}
|
||||
}
|
||||
|
||||
func int64ToBytes(n int64) []byte {
|
||||
b := make([]byte, 8)
|
||||
for i := 0; i < 8; i++ {
|
||||
b[i] = byte(n >> (i * 8))
|
||||
}
|
||||
return b
|
||||
}
|
||||
|
||||
// PQSignatureType identifies the post-quantum signature algorithm
|
||||
type PQSignatureType uint8
|
||||
|
||||
const (
|
||||
// PQSignatureTypeMLDSA65 is NIST FIPS 204 ML-DSA-65
|
||||
PQSignatureTypeMLDSA65 PQSignatureType = iota
|
||||
// PQSignatureTypeCorona is Ring-LWE based threshold signatures
|
||||
PQSignatureTypeCorona
|
||||
// PQSignatureTypeHybrid combines classical and post-quantum
|
||||
PQSignatureTypeHybrid
|
||||
)
|
||||
|
||||
// PQSignature wraps a post-quantum signature
|
||||
type PQSignature struct {
|
||||
Type PQSignatureType
|
||||
Signature []byte
|
||||
PublicKey []byte
|
||||
}
|
||||
|
||||
// PQKeypair represents a post-quantum keypair
|
||||
type PQKeypair struct {
|
||||
Type PQSignatureType
|
||||
PublicKey []byte
|
||||
PrivateKey []byte
|
||||
}
|
||||
|
||||
// Sign signs a message using the keypair (stub - real impl in pqcrypto)
|
||||
func (k *PQKeypair) Sign(message []byte) (*PQSignature, error) {
|
||||
// Stub: returns SHA-256 hash as fixed-size signature for testing
|
||||
hash := sha256.Sum256(append(message, k.PrivateKey...))
|
||||
return &PQSignature{
|
||||
Type: k.Type,
|
||||
Signature: hash[:],
|
||||
PublicKey: k.PublicKey,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// Verify verifies a signature (stub - real impl in pqcrypto)
|
||||
func (k *PQKeypair) Verify(message []byte, sig *PQSignature) bool {
|
||||
// Stub: accepts any non-empty signature
|
||||
return sig != nil && len(sig.Signature) > 0
|
||||
}
|
||||
|
||||
// StakeRegistry interface for validator stake tracking
|
||||
type StakeRegistry interface {
|
||||
GetStake(did *DID) (uint64, error)
|
||||
SetStake(did *DID, amount uint64) error
|
||||
TotalStake() uint64
|
||||
HasSufficientStake(did *DID, minimum uint64) (bool, error)
|
||||
StakeWeight(did *DID) (float64, error)
|
||||
}
|
||||
|
||||
// InMemoryStakeRegistry is a simple in-memory stake registry for testing
|
||||
type InMemoryStakeRegistry struct {
|
||||
stakes map[string]uint64
|
||||
}
|
||||
|
||||
// NewInMemoryStakeRegistry creates a new in-memory stake registry
|
||||
func NewInMemoryStakeRegistry() *InMemoryStakeRegistry {
|
||||
return &InMemoryStakeRegistry{
|
||||
stakes: make(map[string]uint64),
|
||||
}
|
||||
}
|
||||
|
||||
// GetStake returns the stake for a DID
|
||||
func (r *InMemoryStakeRegistry) GetStake(did *DID) (uint64, error) {
|
||||
return r.stakes[did.String()], nil
|
||||
}
|
||||
|
||||
// SetStake sets the stake for a DID
|
||||
func (r *InMemoryStakeRegistry) SetStake(did *DID, amount uint64) error {
|
||||
r.stakes[did.String()] = amount
|
||||
return nil
|
||||
}
|
||||
|
||||
// TotalStake returns the total stake across all validators
|
||||
func (r *InMemoryStakeRegistry) TotalStake() uint64 {
|
||||
var total uint64
|
||||
for _, stake := range r.stakes {
|
||||
total += stake
|
||||
}
|
||||
return total
|
||||
}
|
||||
|
||||
// HasSufficientStake checks if a DID has at least the minimum stake
|
||||
func (r *InMemoryStakeRegistry) HasSufficientStake(did *DID, minimum uint64) (bool, error) {
|
||||
stake, _ := r.GetStake(did)
|
||||
return stake >= minimum, nil
|
||||
}
|
||||
|
||||
// StakeWeight returns the stake weight as a fraction of total stake
|
||||
func (r *InMemoryStakeRegistry) StakeWeight(did *DID) (float64, error) {
|
||||
stake, _ := r.GetStake(did)
|
||||
total := r.TotalStake()
|
||||
if total == 0 {
|
||||
return 0.0, nil
|
||||
}
|
||||
return float64(stake) / float64(total), nil
|
||||
}
|
||||
+5
-5
@@ -133,12 +133,12 @@ deploy:
|
||||
|
||||
### Prometheus Metrics
|
||||
|
||||
The node exposes metrics at `http://localhost:9630/v1/metrics`
|
||||
The node exposes metrics at `http://localhost:9630/ext/metrics`
|
||||
|
||||
### Health Checks
|
||||
|
||||
- Liveness: `http://localhost:9630/v1/health`
|
||||
- Readiness: `http://localhost:9630/v1/info`
|
||||
- Liveness: `http://localhost:9630/ext/health`
|
||||
- Readiness: `http://localhost:9630/ext/info`
|
||||
|
||||
### Grafana Dashboard
|
||||
|
||||
@@ -178,12 +178,12 @@ docker exec -it luxd /bin/bash
|
||||
# Check if bootstrapped
|
||||
curl -X POST -H "Content-Type: application/json" \
|
||||
-d '{"jsonrpc":"2.0","id":1,"method":"info.isBootstrapped","params":{"chain":"C"}}' \
|
||||
http://localhost:9630/v1/info
|
||||
http://localhost:9630/ext/info
|
||||
|
||||
# Get block number
|
||||
curl -X POST -H "Content-Type: application/json" \
|
||||
-d '{"jsonrpc":"2.0","id":1,"method":"eth_blockNumber","params":[]}' \
|
||||
http://localhost:9630/v1/bc/C/rpc
|
||||
http://localhost:9630/ext/bc/C/rpc
|
||||
```
|
||||
|
||||
## CI/CD
|
||||
|
||||
@@ -287,10 +287,10 @@ echo "📝 Starting with command:"
|
||||
echo " $CMD"
|
||||
echo ""
|
||||
echo "📡 API Endpoints:"
|
||||
echo " - JSON-RPC: http://$HTTP_HOST:$HTTP_PORT/v1/bc/C/rpc"
|
||||
echo " - WebSocket: ws://$HTTP_HOST:$HTTP_PORT/v1/bc/C/ws"
|
||||
echo " - Health: http://$HTTP_HOST:$HTTP_PORT/v1/health"
|
||||
echo " - Info: http://$HTTP_HOST:$HTTP_PORT/v1/info"
|
||||
echo " - JSON-RPC: http://$HTTP_HOST:$HTTP_PORT/ext/bc/C/rpc"
|
||||
echo " - WebSocket: ws://$HTTP_HOST:$HTTP_PORT/ext/bc/C/ws"
|
||||
echo " - Health: http://$HTTP_HOST:$HTTP_PORT/ext/health"
|
||||
echo " - Info: http://$HTTP_HOST:$HTTP_PORT/ext/info"
|
||||
echo ""
|
||||
|
||||
# Execute
|
||||
|
||||
+1
-1
@@ -346,7 +346,7 @@ export default function HomePage() {
|
||||
</p>
|
||||
<pre className="mt-4 overflow-x-auto rounded-lg bg-black/50 p-4">
|
||||
<code className="text-sm text-green-400">
|
||||
curl -X POST --data '{"jsonrpc":"2.0","method":"health.health","id":1}' \{"\n"} -H 'content-type:application/json' \{"\n"} 127.0.0.1:9650/v1/health
|
||||
curl -X POST --data '{"jsonrpc":"2.0","method":"health.health","id":1}' \{"\n"} -H 'content-type:application/json' \{"\n"} 127.0.0.1:9650/ext/health
|
||||
</code>
|
||||
</pre>
|
||||
</div>
|
||||
|
||||
@@ -31,7 +31,7 @@ Or in configuration:
|
||||
## Endpoint
|
||||
|
||||
```
|
||||
http://localhost:9630/v1/admin
|
||||
http://localhost:9630/ext/admin
|
||||
```
|
||||
|
||||
## Methods
|
||||
@@ -54,7 +54,7 @@ curl -X POST --data '{
|
||||
"chain":"2S53R2ub94CV5vmSRAjqPYmRxvuiFunCb1gN2CAw3DQBfPWghX"
|
||||
},
|
||||
"id":1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/admin
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/admin
|
||||
```
|
||||
|
||||
---
|
||||
@@ -85,7 +85,7 @@ curl -X POST --data '{
|
||||
"chain":"2S53R2ub94CV5vmSRAjqPYmRxvuiFunCb1gN2CAw3DQBfPWghX"
|
||||
},
|
||||
"id":1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/admin
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/admin
|
||||
```
|
||||
|
||||
**Response:**
|
||||
@@ -114,7 +114,7 @@ curl -X POST --data '{
|
||||
"method":"admin.getLogLevel",
|
||||
"params":{},
|
||||
"id":1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/admin
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/admin
|
||||
```
|
||||
|
||||
**Response:**
|
||||
@@ -146,7 +146,7 @@ curl -X POST --data '{
|
||||
"logLevel":"debug"
|
||||
},
|
||||
"id":1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/admin
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/admin
|
||||
```
|
||||
|
||||
---
|
||||
@@ -181,7 +181,7 @@ curl -X POST --data '{
|
||||
"method":"admin.loadVMs",
|
||||
"params":{},
|
||||
"id":1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/admin
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/admin
|
||||
```
|
||||
|
||||
**Response:**
|
||||
@@ -214,7 +214,7 @@ curl -X POST --data '{
|
||||
"method":"admin.startCPUProfiler",
|
||||
"params":{},
|
||||
"id":1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/admin
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/admin
|
||||
```
|
||||
|
||||
---
|
||||
@@ -232,7 +232,7 @@ curl -X POST --data '{
|
||||
"method":"admin.stopCPUProfiler",
|
||||
"params":{},
|
||||
"id":1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/admin
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/admin
|
||||
```
|
||||
|
||||
---
|
||||
@@ -250,7 +250,7 @@ curl -X POST --data '{
|
||||
"method":"admin.memoryProfile",
|
||||
"params":{},
|
||||
"id":1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/admin
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/admin
|
||||
```
|
||||
|
||||
---
|
||||
@@ -276,7 +276,7 @@ curl -X POST --data '{
|
||||
"method":"admin.getConfig",
|
||||
"params":{},
|
||||
"id":1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/admin
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/admin
|
||||
```
|
||||
|
||||
**Response:**
|
||||
@@ -311,7 +311,7 @@ curl -X POST --data '{
|
||||
"method":"admin.shutdown",
|
||||
"params":{},
|
||||
"id":1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/admin
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/admin
|
||||
```
|
||||
|
||||
---
|
||||
@@ -330,7 +330,7 @@ curl -X POST --data '{
|
||||
"method":"admin.db.commit",
|
||||
"params":{},
|
||||
"id":1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/admin
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/admin
|
||||
```
|
||||
|
||||
## Profiling and Debugging
|
||||
@@ -344,7 +344,7 @@ curl -X POST --data '{
|
||||
"method":"admin.startCPUProfiler",
|
||||
"params":{},
|
||||
"id":1
|
||||
}' http://localhost:9630/v1/admin
|
||||
}' http://localhost:9630/ext/admin
|
||||
|
||||
# Let it run for 30 seconds
|
||||
sleep 30
|
||||
@@ -355,7 +355,7 @@ curl -X POST --data '{
|
||||
"method":"admin.stopCPUProfiler",
|
||||
"params":{},
|
||||
"id":2
|
||||
}' http://localhost:9630/v1/admin
|
||||
}' http://localhost:9630/ext/admin
|
||||
|
||||
# Analyze profile
|
||||
go tool pprof cpu.prof
|
||||
@@ -370,7 +370,7 @@ curl -X POST --data '{
|
||||
"method":"admin.memoryProfile",
|
||||
"params":{},
|
||||
"id":1
|
||||
}' http://localhost:9630/v1/admin
|
||||
}' http://localhost:9630/ext/admin
|
||||
|
||||
# Analyze
|
||||
go tool pprof mem.prof
|
||||
@@ -401,7 +401,7 @@ set_log_level() {
|
||||
\"method\":\"admin.setLogLevel\",
|
||||
\"params\":{\"logLevel\":\"$LEVEL\"},
|
||||
\"id\":1
|
||||
}" http://localhost:9630/v1/admin
|
||||
}" http://localhost:9630/ext/admin
|
||||
}
|
||||
|
||||
# Increase verbosity for debugging
|
||||
@@ -476,7 +476,7 @@ Enable audit logging for admin operations:
|
||||
# auto_restart.sh
|
||||
|
||||
check_health() {
|
||||
curl -s http://localhost:9630/v1/health | jq -r '.healthy'
|
||||
curl -s http://localhost:9630/ext/health | jq -r '.healthy'
|
||||
}
|
||||
|
||||
restart_node() {
|
||||
@@ -488,7 +488,7 @@ restart_node() {
|
||||
"method":"admin.shutdown",
|
||||
"params":{},
|
||||
"id":1
|
||||
}' http://localhost:9630/v1/admin
|
||||
}' http://localhost:9630/ext/admin
|
||||
|
||||
sleep 10
|
||||
|
||||
@@ -514,7 +514,7 @@ CURRENT=$(curl -s -X POST --data '{
|
||||
"method":"admin.getConfig",
|
||||
"params":{},
|
||||
"id":1
|
||||
}' http://localhost:9630/v1/admin)
|
||||
}' http://localhost:9630/ext/admin)
|
||||
|
||||
echo "Current configuration:"
|
||||
echo $CURRENT | jq .
|
||||
@@ -525,7 +525,7 @@ curl -X POST --data '{
|
||||
"method":"admin.setLogLevel",
|
||||
"params":{"logLevel":"debug"},
|
||||
"id":2
|
||||
}' http://localhost:9630/v1/admin
|
||||
}' http://localhost:9630/ext/admin
|
||||
```
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
@@ -10,7 +10,7 @@ The Health API provides endpoints to monitor the health and readiness of your Lu
|
||||
## Endpoint
|
||||
|
||||
```
|
||||
http://localhost:9630/v1/health
|
||||
http://localhost:9630/ext/health
|
||||
```
|
||||
|
||||
## Health Check Types
|
||||
@@ -20,7 +20,7 @@ http://localhost:9630/v1/health
|
||||
Get the overall health status of the node:
|
||||
|
||||
```bash
|
||||
curl http://localhost:9630/v1/health
|
||||
curl http://localhost:9630/ext/health
|
||||
```
|
||||
|
||||
**Response:**
|
||||
@@ -61,7 +61,7 @@ curl http://localhost:9630/v1/health
|
||||
Check if the node is ready to serve requests:
|
||||
|
||||
```bash
|
||||
curl http://localhost:9630/v1/health/readiness
|
||||
curl http://localhost:9630/ext/health/readiness
|
||||
```
|
||||
|
||||
Returns:
|
||||
@@ -73,7 +73,7 @@ Returns:
|
||||
Check if the node is alive and running:
|
||||
|
||||
```bash
|
||||
curl http://localhost:9630/v1/health/liveness
|
||||
curl http://localhost:9630/ext/health/liveness
|
||||
```
|
||||
|
||||
Returns:
|
||||
@@ -92,7 +92,7 @@ curl -X POST --data '{
|
||||
"id": 1,
|
||||
"method": "health.health",
|
||||
"params": {}
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/health
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/health
|
||||
```
|
||||
|
||||
**Response:**
|
||||
@@ -175,7 +175,7 @@ Configure per-chain health parameters:
|
||||
Export health metrics in Prometheus format:
|
||||
|
||||
```bash
|
||||
curl http://localhost:9630/v1/metrics | grep health
|
||||
curl http://localhost:9630/ext/metrics | grep health
|
||||
```
|
||||
|
||||
Metrics:
|
||||
@@ -208,13 +208,13 @@ spec:
|
||||
image: luxfi/node:latest
|
||||
livenessProbe:
|
||||
httpGet:
|
||||
path: /v1/health/liveness
|
||||
path: /ext/health/liveness
|
||||
port: 9630
|
||||
initialDelaySeconds: 30
|
||||
periodSeconds: 10
|
||||
readinessProbe:
|
||||
httpGet:
|
||||
path: /v1/health/readiness
|
||||
path: /ext/health/readiness
|
||||
port: 9630
|
||||
initialDelaySeconds: 60
|
||||
periodSeconds: 5
|
||||
@@ -229,7 +229,7 @@ spec:
|
||||
# health_monitor.sh
|
||||
|
||||
while true; do
|
||||
HEALTH=$(curl -s http://localhost:9630/v1/health | jq -r '.healthy')
|
||||
HEALTH=$(curl -s http://localhost:9630/ext/health | jq -r '.healthy')
|
||||
|
||||
if [ "$HEALTH" != "true" ]; then
|
||||
echo "ALERT: Node unhealthy at $(date)"
|
||||
@@ -248,7 +248,7 @@ done
|
||||
# health_analysis.sh
|
||||
|
||||
# Get detailed health
|
||||
RESPONSE=$(curl -s http://localhost:9630/v1/health)
|
||||
RESPONSE=$(curl -s http://localhost:9630/ext/health)
|
||||
|
||||
# Parse each chain
|
||||
for CHAIN in P X C Q; do
|
||||
@@ -271,7 +271,7 @@ done
|
||||
|
||||
```
|
||||
backend lux_nodes
|
||||
option httpchk GET /v1/health
|
||||
option httpchk GET /ext/health
|
||||
http-check expect status 200
|
||||
|
||||
server node1 192.168.1.10:9630 check
|
||||
@@ -289,7 +289,7 @@ upstream lux_nodes {
|
||||
}
|
||||
|
||||
location /health_check {
|
||||
proxy_pass http://lux_nodes/v1/health;
|
||||
proxy_pass http://lux_nodes/ext/health;
|
||||
proxy_connect_timeout 1s;
|
||||
proxy_read_timeout 1s;
|
||||
}
|
||||
|
||||
@@ -10,7 +10,7 @@ The Info API provides general information about the node and network status. Thi
|
||||
## Endpoint
|
||||
|
||||
```
|
||||
http://localhost:9630/v1/info
|
||||
http://localhost:9630/ext/info
|
||||
```
|
||||
|
||||
## Format
|
||||
@@ -23,7 +23,7 @@ curl -X POST --data '{
|
||||
"method": "info.<method>",
|
||||
"params": {...},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/info
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/info
|
||||
```
|
||||
|
||||
## Methods
|
||||
@@ -41,7 +41,7 @@ curl -X POST --data '{
|
||||
"method": "info.getNodeVersion",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/info
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/info
|
||||
```
|
||||
|
||||
**Example Response:**
|
||||
@@ -78,7 +78,7 @@ curl -X POST --data '{
|
||||
"method": "info.getNodeID",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/info
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/info
|
||||
```
|
||||
|
||||
**Example Response:**
|
||||
@@ -111,7 +111,7 @@ curl -X POST --data '{
|
||||
"method": "info.getNodeIP",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/info
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/info
|
||||
```
|
||||
|
||||
**Example Response:**
|
||||
@@ -140,7 +140,7 @@ curl -X POST --data '{
|
||||
"method": "info.getNetworkID",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/info
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/info
|
||||
```
|
||||
|
||||
**Example Response:**
|
||||
@@ -175,7 +175,7 @@ curl -X POST --data '{
|
||||
"method": "info.getNetworkName",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/info
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/info
|
||||
```
|
||||
|
||||
**Example Response:**
|
||||
@@ -207,7 +207,7 @@ curl -X POST --data '{
|
||||
"alias": "P"
|
||||
},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/info
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/info
|
||||
```
|
||||
|
||||
**Example Response:**
|
||||
@@ -239,7 +239,7 @@ curl -X POST --data '{
|
||||
"chain": "P"
|
||||
},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/info
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/info
|
||||
```
|
||||
|
||||
**Example Response:**
|
||||
@@ -269,7 +269,7 @@ curl -X POST --data '{
|
||||
"method": "info.peers",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/info
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/info
|
||||
```
|
||||
|
||||
**Example Response:**
|
||||
@@ -313,7 +313,7 @@ curl -X POST --data '{
|
||||
"method": "info.getTxFee",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/info
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/info
|
||||
```
|
||||
|
||||
**Example Response:**
|
||||
@@ -348,7 +348,7 @@ curl -X POST --data '{
|
||||
"method": "info.uptime",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/info
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/info
|
||||
```
|
||||
|
||||
**Example Response (Validator):**
|
||||
@@ -390,7 +390,7 @@ curl -X POST --data '{
|
||||
"method": "info.getVMs",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/info
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/info
|
||||
```
|
||||
|
||||
**Example Response:**
|
||||
@@ -424,7 +424,7 @@ curl -X POST --data '{
|
||||
"method": "info.getUpgrades",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/info
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/info
|
||||
```
|
||||
|
||||
**Example Response:**
|
||||
@@ -468,7 +468,7 @@ VERSION=$(curl -s -X POST --data '{
|
||||
"method": "info.getNodeVersion",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' $NODE_URL/v1/info | jq -r '.result.version')
|
||||
}' -H 'content-type:application/json;' $NODE_URL/ext/info | jq -r '.result.version')
|
||||
|
||||
echo "Version: $VERSION"
|
||||
|
||||
@@ -478,7 +478,7 @@ NODE_ID=$(curl -s -X POST --data '{
|
||||
"method": "info.getNodeID",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' $NODE_URL/v1/info | jq -r '.result.nodeID')
|
||||
}' -H 'content-type:application/json;' $NODE_URL/ext/info | jq -r '.result.nodeID')
|
||||
|
||||
echo "Node ID: $NODE_ID"
|
||||
|
||||
@@ -488,7 +488,7 @@ PEERS=$(curl -s -X POST --data '{
|
||||
"method": "info.peers",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' $NODE_URL/v1/info | jq -r '.result.numPeers')
|
||||
}' -H 'content-type:application/json;' $NODE_URL/ext/info | jq -r '.result.numPeers')
|
||||
|
||||
echo "Connected Peers: $PEERS"
|
||||
|
||||
@@ -499,7 +499,7 @@ for CHAIN in P X C Q; do
|
||||
\"method\": \"info.isBootstrapped\",
|
||||
\"params\": {\"chain\": \"$CHAIN\"},
|
||||
\"id\": 1
|
||||
}" -H 'content-type:application/json;' $NODE_URL/v1/info | jq -r '.result.isBootstrapped')
|
||||
}" -H 'content-type:application/json;' $NODE_URL/ext/info | jq -r '.result.isBootstrapped')
|
||||
|
||||
echo "$CHAIN-Chain Bootstrapped: $STATUS"
|
||||
done
|
||||
@@ -510,7 +510,7 @@ UPTIME=$(curl -s -X POST --data '{
|
||||
"method": "info.uptime",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' $NODE_URL/v1/info 2>/dev/null | jq -r '.result.rewardingStakePercentage' 2>/dev/null)
|
||||
}' -H 'content-type:application/json;' $NODE_URL/ext/info 2>/dev/null | jq -r '.result.rewardingStakePercentage' 2>/dev/null)
|
||||
|
||||
if [ "$UPTIME" != "null" ] && [ -n "$UPTIME" ]; then
|
||||
echo "Validator Uptime: $UPTIME%"
|
||||
@@ -531,7 +531,7 @@ PEERS=$(curl -s -X POST --data '{
|
||||
"method": "info.peers",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' $NODE_URL/v1/info | jq -r '.result.peers[]')
|
||||
}' -H 'content-type:application/json;' $NODE_URL/ext/info | jq -r '.result.peers[]')
|
||||
|
||||
echo "=== Peer Connection Quality ==="
|
||||
echo "Node ID | Latency (ms) | Uptime % | Version"
|
||||
@@ -560,7 +560,7 @@ while true; do
|
||||
\"method\": \"info.isBootstrapped\",
|
||||
\"params\": {\"chain\": \"$CHAIN\"},
|
||||
\"id\": 1
|
||||
}" -H 'content-type:application/json;' $NODE_URL/v1/info | jq -r '.result.isBootstrapped')
|
||||
}" -H 'content-type:application/json;' $NODE_URL/ext/info | jq -r '.result.isBootstrapped')
|
||||
|
||||
if [ "$STATUS" == "true" ]; then
|
||||
echo "✅ $CHAIN-Chain: Bootstrapped"
|
||||
@@ -575,7 +575,7 @@ while true; do
|
||||
"method": "info.peers",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' $NODE_URL/v1/info | jq -r '.result.numPeers')
|
||||
}' -H 'content-type:application/json;' $NODE_URL/ext/info | jq -r '.result.numPeers')
|
||||
|
||||
echo ""
|
||||
echo "Connected Peers: $PEERS"
|
||||
@@ -588,7 +588,7 @@ while true; do
|
||||
\"method\": \"info.isBootstrapped\",
|
||||
\"params\": {\"chain\": \"$CHAIN\"},
|
||||
\"id\": 1
|
||||
}" -H 'content-type:application/json;' $NODE_URL/v1/info | jq -r '.result.isBootstrapped')
|
||||
}" -H 'content-type:application/json;' $NODE_URL/ext/info | jq -r '.result.isBootstrapped')
|
||||
|
||||
if [ "$STATUS" != "true" ]; then
|
||||
ALL_BOOTSTRAPPED=false
|
||||
|
||||
@@ -10,7 +10,7 @@ The Platform Chain (P-Chain) is responsible for staking, validators, and chain m
|
||||
## Endpoint
|
||||
|
||||
```
|
||||
http://localhost:9630/v1/bc/P
|
||||
http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
## Format
|
||||
@@ -23,7 +23,7 @@ curl -X POST --data '{
|
||||
"method": "platform.<method>",
|
||||
"params": {...},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
## Methods
|
||||
@@ -41,7 +41,7 @@ curl -X POST --data '{
|
||||
"method": "platform.getHeight",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
**Example Response:**
|
||||
@@ -73,7 +73,7 @@ curl -X POST --data '{
|
||||
"addresses": ["P-lux1q8tgunsf7sxpl2qp9fz0xjs36kstm7vdjvpzcc"]
|
||||
},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
**Example Response:**
|
||||
@@ -114,7 +114,7 @@ curl -X POST --data '{
|
||||
"limit": 100
|
||||
},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
---
|
||||
@@ -134,7 +134,7 @@ curl -X POST --data '{
|
||||
"method": "platform.getCurrentValidators",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
**Example Response:**
|
||||
@@ -190,7 +190,7 @@ curl -X POST --data '{
|
||||
"method": "platform.getPendingValidators",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
---
|
||||
@@ -212,7 +212,7 @@ curl -X POST --data '{
|
||||
"height": 365000
|
||||
},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
---
|
||||
@@ -233,7 +233,7 @@ curl -X POST --data '{
|
||||
"height": 365000
|
||||
},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
---
|
||||
@@ -255,7 +255,7 @@ curl -X POST --data '{
|
||||
"addresses": ["P-lux1q8tgunsf7sxpl2qp9fz0xjs36kstm7vdjvpzcc"]
|
||||
},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
---
|
||||
@@ -274,7 +274,7 @@ curl -X POST --data '{
|
||||
"method": "platform.getMinStake",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
**Example Response:**
|
||||
@@ -305,7 +305,7 @@ curl -X POST --data '{
|
||||
"method": "platform.getTotalStake",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
---
|
||||
@@ -327,7 +327,7 @@ curl -X POST --data '{
|
||||
"txID": "2nmH8LithVbdjaXsxVQCQfXtzN9hBbmebrsaEYnLM9T32Uy3Y5"
|
||||
},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
---
|
||||
@@ -345,7 +345,7 @@ curl -X POST --data '{
|
||||
"method": "platform.getTimestamp",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
---
|
||||
@@ -363,7 +363,7 @@ curl -X POST --data '{
|
||||
"method": "platform.getBlockchains",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
**Example Response:**
|
||||
@@ -409,7 +409,7 @@ curl -X POST --data '{
|
||||
"blockID": "vXSY7FK7NR65Y8BrJDKQH4a6vBdJuqAMvVzWj3Zxcap5J4ZE3"
|
||||
},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
---
|
||||
@@ -431,7 +431,7 @@ curl -X POST --data '{
|
||||
"txID": "2nmH8LithVbdjaXsxVQCQfXtzN9hBbmebrsaEYnLM9T32Uy3Y5"
|
||||
},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
---
|
||||
@@ -452,7 +452,7 @@ curl -X POST --data '{
|
||||
"txID": "2nmH8LithVbdjaXsxVQCQfXtzN9hBbmebrsaEYnLM9T32Uy3Y5"
|
||||
},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
**Example Response:**
|
||||
@@ -488,7 +488,7 @@ curl -X POST --data '{
|
||||
"method": "platform.getCurrentSupply",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
---
|
||||
@@ -507,7 +507,7 @@ curl -X POST --data '{
|
||||
"method": "platform.getChains",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
---
|
||||
@@ -544,7 +544,7 @@ curl -X POST --data '{
|
||||
"password": "mypassword"
|
||||
},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
---
|
||||
@@ -694,7 +694,7 @@ curl -s -X POST --data "{
|
||||
\"nodeIDs\": [\"$NODE_ID\"]
|
||||
},
|
||||
\"id\": 1
|
||||
}" -H 'content-type:application/json;' http://localhost:9630/v1/bc/P | jq '.result.validators[0]'
|
||||
}" -H 'content-type:application/json;' http://localhost:9630/ext/bc/P | jq '.result.validators[0]'
|
||||
```
|
||||
|
||||
### Monitor Staking Rewards
|
||||
@@ -711,7 +711,7 @@ STAKE=$(curl -s -X POST --data "{
|
||||
\"addresses\": [\"$ADDRESS\"]
|
||||
},
|
||||
\"id\": 1
|
||||
}" -H 'content-type:application/json;' http://localhost:9630/v1/bc/P)
|
||||
}" -H 'content-type:application/json;' http://localhost:9630/ext/bc/P)
|
||||
|
||||
echo "Current stake: $(echo $STAKE | jq -r '.result.staked')"
|
||||
echo "Stakeable: $(echo $STAKE | jq -r '.result.stakeable')"
|
||||
|
||||
@@ -210,14 +210,14 @@ curl -X POST --data '{
|
||||
"jsonrpc":"2.0",
|
||||
"id": 1,
|
||||
"method": "health.health"
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/health
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/health
|
||||
|
||||
# Get node info
|
||||
curl -X POST --data '{
|
||||
"jsonrpc":"2.0",
|
||||
"id": 1,
|
||||
"method": "info.getNodeVersion"
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/info
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/info
|
||||
|
||||
# Check bootstrap status
|
||||
curl -X POST --data '{
|
||||
@@ -227,7 +227,7 @@ curl -X POST --data '{
|
||||
"params": {
|
||||
"chain": "P"
|
||||
}
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/info
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/info
|
||||
```
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
@@ -308,7 +308,7 @@ curl -X POST --data '{
|
||||
"jsonrpc":"2.0",
|
||||
"id": 1,
|
||||
"method": "health.health"
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/health
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/health
|
||||
```
|
||||
|
||||
### Bootstrap Status
|
||||
@@ -320,7 +320,7 @@ curl -X POST --data '{
|
||||
"id": 1,
|
||||
"method": "info.isBootstrapped",
|
||||
"params": {"chain": "P"}
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/info
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/info
|
||||
```
|
||||
|
||||
### Metrics
|
||||
@@ -328,7 +328,7 @@ curl -X POST --data '{
|
||||
Access Prometheus metrics:
|
||||
|
||||
```bash
|
||||
curl http://localhost:9630/v1/metrics
|
||||
curl http://localhost:9630/ext/metrics
|
||||
```
|
||||
|
||||
Key metrics to monitor:
|
||||
|
||||
@@ -58,7 +58,7 @@ curl -X POST --data '{
|
||||
"jsonrpc":"2.0",
|
||||
"id": 1,
|
||||
"method": "platform.getHeight"
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
|
||||
# Check bootstrap status for all chains
|
||||
curl -X POST --data '{
|
||||
@@ -66,7 +66,7 @@ curl -X POST --data '{
|
||||
"id": 1,
|
||||
"method": "info.isBootstrapped",
|
||||
"params": {"chain": "P"}
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/info
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/info
|
||||
```
|
||||
|
||||
### Configure Public IP
|
||||
@@ -90,7 +90,7 @@ curl -X POST --data '{
|
||||
"jsonrpc":"2.0",
|
||||
"id": 1,
|
||||
"method": "info.getNodeID"
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/info
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/info
|
||||
```
|
||||
|
||||
Response:
|
||||
@@ -155,7 +155,7 @@ curl -X POST --data '{
|
||||
"password": "mypassword"
|
||||
},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/keystore
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/keystore
|
||||
|
||||
# Create P-Chain address
|
||||
curl -X POST --data '{
|
||||
@@ -166,7 +166,7 @@ curl -X POST --data '{
|
||||
"password": "mypassword"
|
||||
},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
### Transfer LUX to P-Chain
|
||||
@@ -185,7 +185,7 @@ curl -X POST --data '{
|
||||
"amount": 2000000000000
|
||||
},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/X
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/X
|
||||
|
||||
# Import to P-Chain
|
||||
curl -X POST --data '{
|
||||
@@ -197,7 +197,7 @@ curl -X POST --data '{
|
||||
"sourceChain": "X"
|
||||
},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
## Step 4: Add as Validator
|
||||
@@ -221,7 +221,7 @@ curl -X POST --data '{
|
||||
"delegationFeeRate": 10
|
||||
},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
Parameters:
|
||||
@@ -243,7 +243,7 @@ curl -X POST --data '{
|
||||
"method": "platform.getPendingValidators",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
|
||||
# Get current validators
|
||||
curl -X POST --data '{
|
||||
@@ -251,7 +251,7 @@ curl -X POST --data '{
|
||||
"method": "platform.getCurrentValidators",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
## Step 5: Monitor Your Validator
|
||||
@@ -268,7 +268,7 @@ curl -X POST --data '{
|
||||
"nodeID": "NodeID-5KqnQfaFQxY9rsFBAa68377qWSherYLQ7"
|
||||
},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
### Monitor Performance Metrics
|
||||
@@ -277,10 +277,10 @@ Key metrics to track:
|
||||
|
||||
```bash
|
||||
# Node health
|
||||
curl http://localhost:9630/v1/health
|
||||
curl http://localhost:9630/ext/health
|
||||
|
||||
# Prometheus metrics
|
||||
curl http://localhost:9630/v1/metrics | grep -E "uptime|stake|validator"
|
||||
curl http://localhost:9630/ext/metrics | grep -E "uptime|stake|validator"
|
||||
```
|
||||
|
||||
Important metrics:
|
||||
@@ -308,7 +308,7 @@ NODE_URL="http://localhost:9630"
|
||||
WEBHOOK_URL="your-webhook-url"
|
||||
|
||||
# Check if node is responsive
|
||||
if ! curl -s "$NODE_URL/v1/health" > /dev/null; then
|
||||
if ! curl -s "$NODE_URL/ext/health" > /dev/null; then
|
||||
curl -X POST "$WEBHOOK_URL" -d '{"text":"ALERT: Node is not responding!"}'
|
||||
fi
|
||||
|
||||
@@ -318,7 +318,7 @@ UPTIME=$(curl -s -X POST --data '{
|
||||
"method":"platform.getValidator",
|
||||
"params":{"nodeID":"NodeID-xxx"},
|
||||
"id":1
|
||||
}' "$NODE_URL/v1/bc/P" | jq -r '.result.uptime')
|
||||
}' "$NODE_URL/ext/bc/P" | jq -r '.result.uptime')
|
||||
|
||||
if [ "$UPTIME" -lt "80" ]; then
|
||||
curl -X POST "$WEBHOOK_URL" -d "{\"text\":\"WARNING: Uptime is $UPTIME%\"}"
|
||||
@@ -349,7 +349,7 @@ curl -X POST --data '{
|
||||
"nodeID": "NodeID-5KqnQfaFQxY9rsFBAa68377qWSherYLQ7"
|
||||
},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
## Chain Validation
|
||||
@@ -373,7 +373,7 @@ curl -X POST --data '{
|
||||
"weight": 1000
|
||||
},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
### Chain Requirements
|
||||
|
||||
@@ -104,14 +104,14 @@ curl -X POST --data '{
|
||||
"jsonrpc":"2.0",
|
||||
"id": 1,
|
||||
"method": "info.getNetworkInfo"
|
||||
}' -H 'content-type:application/json;' http://localhost:9650/v1/info
|
||||
}' -H 'content-type:application/json;' http://localhost:9650/ext/info
|
||||
|
||||
# Get node ID
|
||||
curl -X POST --data '{
|
||||
"jsonrpc":"2.0",
|
||||
"id": 1,
|
||||
"method": "info.getNodeID"
|
||||
}' -H 'content-type:application/json;' http://localhost:9650/v1/info
|
||||
}' -H 'content-type:application/json;' http://localhost:9650/ext/info
|
||||
```
|
||||
|
||||
## Chain Management
|
||||
@@ -130,7 +130,7 @@ curl -X POST --data '{
|
||||
"genesisData": "...",
|
||||
"netID": "network-id"
|
||||
}
|
||||
}' -H 'content-type:application/json;' http://localhost:9650/v1/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9650/ext/P
|
||||
```
|
||||
|
||||
### Validator Management
|
||||
@@ -147,7 +147,7 @@ curl -X POST --data '{
|
||||
"endTime": ...,
|
||||
"stakeAmount": ...
|
||||
}
|
||||
}' -H 'content-type:application/json;' http://localhost:9650/v1/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9650/ext/P
|
||||
```
|
||||
|
||||
## Configuration Reference
|
||||
@@ -175,10 +175,10 @@ curl -X POST --data '{
|
||||
|
||||
```bash
|
||||
# Prometheus metrics endpoint
|
||||
curl http://localhost:9650/v1/metrics
|
||||
curl http://localhost:9650/ext/metrics
|
||||
|
||||
# Health check
|
||||
curl http://localhost:9650/v1/health
|
||||
curl http://localhost:9650/ext/health
|
||||
```
|
||||
|
||||
### Logs
|
||||
@@ -193,7 +193,7 @@ curl -X POST --data '{
|
||||
"id": 1,
|
||||
"method": "admin.setLogLevel",
|
||||
"params": {"logLevel": "debug"}
|
||||
}' http://localhost:9650/v1/admin
|
||||
}' http://localhost:9650/ext/admin
|
||||
```
|
||||
|
||||
## Testing
|
||||
|
||||
@@ -595,10 +595,10 @@ curl -X POST --data '{
|
||||
"jsonrpc":"2.0",
|
||||
"method":"info.peers",
|
||||
"id":1
|
||||
}' http://localhost:9630/v1/info
|
||||
}' http://localhost:9630/ext/info
|
||||
|
||||
# Check network metrics
|
||||
curl http://localhost:9630/v1/metrics | grep network
|
||||
curl http://localhost:9630/ext/metrics | grep network
|
||||
|
||||
# Monitor connections
|
||||
netstat -an | grep 9631
|
||||
|
||||
@@ -39,7 +39,7 @@ scrape_configs:
|
||||
- job_name: 'lux-node'
|
||||
static_configs:
|
||||
- targets: ['localhost:9630']
|
||||
metrics_path: '/v1/metrics'
|
||||
metrics_path: '/ext/metrics'
|
||||
|
||||
- job_name: 'node-exporter'
|
||||
static_configs:
|
||||
@@ -56,7 +56,7 @@ alerting:
|
||||
|
||||
### Available Metrics
|
||||
|
||||
The Lux node exposes 400+ Prometheus metrics at `http://localhost:9630/v1/metrics`.
|
||||
The Lux node exposes 400+ Prometheus metrics at `http://localhost:9630/ext/metrics`.
|
||||
|
||||
#### Key Metric Categories
|
||||
|
||||
@@ -419,11 +419,11 @@ groups:
|
||||
|
||||
```bash
|
||||
# Basic health check
|
||||
curl http://localhost:9630/v1/health
|
||||
curl http://localhost:9630/ext/health
|
||||
|
||||
# Detailed health with readiness/liveness
|
||||
curl http://localhost:9630/v1/health/readiness
|
||||
curl http://localhost:9630/v1/health/liveness
|
||||
curl http://localhost:9630/ext/health/readiness
|
||||
curl http://localhost:9630/ext/health/liveness
|
||||
```
|
||||
|
||||
### Custom Health Script
|
||||
@@ -443,7 +443,7 @@ send_alert() {
|
||||
}
|
||||
|
||||
# Check if node is responsive
|
||||
if ! curl -s "$NODE_URL/v1/health" > /dev/null; then
|
||||
if ! curl -s "$NODE_URL/ext/health" > /dev/null; then
|
||||
send_alert "Node is not responding!"
|
||||
exit 1
|
||||
fi
|
||||
@@ -455,7 +455,7 @@ for CHAIN in P X C Q; do
|
||||
\"method\": \"info.isBootstrapped\",
|
||||
\"params\": {\"chain\": \"$CHAIN\"},
|
||||
\"id\": 1
|
||||
}" -H 'content-type:application/json;' $NODE_URL/v1/info | jq -r '.result.isBootstrapped')
|
||||
}" -H 'content-type:application/json;' $NODE_URL/ext/info | jq -r '.result.isBootstrapped')
|
||||
|
||||
if [ "$STATUS" != "true" ]; then
|
||||
send_alert "$CHAIN-Chain is not bootstrapped!"
|
||||
@@ -468,7 +468,7 @@ PEERS=$(curl -s -X POST --data '{
|
||||
"method": "info.peers",
|
||||
"params": {},
|
||||
"id": 1
|
||||
}' -H 'content-type:application/json;' $NODE_URL/v1/info | jq -r '.result.numPeers')
|
||||
}' -H 'content-type:application/json;' $NODE_URL/ext/info | jq -r '.result.numPeers')
|
||||
|
||||
if [ "$PEERS" -lt 4 ]; then
|
||||
send_alert "Low peer count: $PEERS"
|
||||
@@ -619,7 +619,7 @@ Create runbooks for common issues:
|
||||
|
||||
```bash
|
||||
# Real-time metrics
|
||||
watch -n 1 'curl -s http://localhost:9630/v1/metrics | grep -E "chain_height|network_peers"'
|
||||
watch -n 1 'curl -s http://localhost:9630/ext/metrics | grep -E "chain_height|network_peers"'
|
||||
|
||||
# Log streaming
|
||||
tail -f ~/.luxd/logs/*.log | grep --line-buffered ERROR
|
||||
|
||||
@@ -26,7 +26,7 @@ else
|
||||
fi
|
||||
|
||||
# Check API responsiveness
|
||||
if curl -s http://localhost:9630/v1/health > /dev/null 2>&1; then
|
||||
if curl -s http://localhost:9630/ext/health > /dev/null 2>&1; then
|
||||
echo "✅ API is responsive"
|
||||
else
|
||||
echo "❌ API is NOT responsive"
|
||||
@@ -53,7 +53,7 @@ PEERS=$(curl -s -X POST --data '{
|
||||
"jsonrpc":"2.0",
|
||||
"method":"info.peers",
|
||||
"id":1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/info 2>/dev/null | jq -r '.result.numPeers')
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/info 2>/dev/null | jq -r '.result.numPeers')
|
||||
|
||||
if [ -n "$PEERS" ] && [ "$PEERS" -gt 0 ]; then
|
||||
echo "✅ Connected peers: $PEERS"
|
||||
@@ -225,7 +225,7 @@ curl -X POST --data '{
|
||||
"method":"platform.getCurrentValidators",
|
||||
"params":{"nodeIDs":["<your-node-id>"]},
|
||||
"id":1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
|
||||
# Verify staking transaction
|
||||
curl -X POST --data '{
|
||||
@@ -233,7 +233,7 @@ curl -X POST --data '{
|
||||
"method":"platform.getTx",
|
||||
"params":{"txID":"<staking-tx-id>"},
|
||||
"id":1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
#### Issue: Low uptime percentage
|
||||
@@ -275,7 +275,7 @@ curl -X POST --data '{
|
||||
"jsonrpc":"2.0",
|
||||
"method":"platform.getHeight",
|
||||
"id":1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
|
||||
# Reduce consensus participation
|
||||
./build/node --consensus-gossip-concurrent=2
|
||||
@@ -334,7 +334,7 @@ grep "api" ~/.luxd/configs/node-config.json
|
||||
./build/node --http-host=0.0.0.0
|
||||
|
||||
# Check for rate limiting
|
||||
curl -I http://localhost:9630/v1/info
|
||||
curl -I http://localhost:9630/ext/info
|
||||
```
|
||||
|
||||
### Staking Issues
|
||||
@@ -369,7 +369,7 @@ curl -X POST --data '{
|
||||
"method":"platform.getBalance",
|
||||
"params":{"addresses":["P-lux1..."]},
|
||||
"id":1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
|
||||
# Import funds from X-Chain
|
||||
curl -X POST --data '{
|
||||
@@ -381,7 +381,7 @@ curl -X POST --data '{
|
||||
"sourceChain":"X"
|
||||
},
|
||||
"id":1
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/v1/bc/P
|
||||
}' -H 'content-type:application/json;' http://localhost:9630/ext/bc/P
|
||||
```
|
||||
|
||||
## Log Analysis
|
||||
@@ -518,7 +518,7 @@ curl -X POST --data '{
|
||||
"jsonrpc":"2.0",
|
||||
"method":"info.getNodeID",
|
||||
"id":1
|
||||
}' http://localhost:9630/v1/info
|
||||
}' http://localhost:9630/ext/info
|
||||
```
|
||||
|
||||
### Support Channels
|
||||
|
||||
@@ -1,163 +0,0 @@
|
||||
# Postmortem: C-Chain accepted-head state GC eviction (mainnet freeze)
|
||||
|
||||
**Status:** Resolved. Mainnet recovered and accepted at 5/5 validators, C-Chain
|
||||
height 1085412, hash `0xd957eae6cb0bbef37174…`, all validators in agreement,
|
||||
explorer at tip, treasury and Genesis NFT state verified.
|
||||
|
||||
**Severity:** Critical (mainnet C-Chain unable to build blocks; no state loss).
|
||||
|
||||
## Accepted permanent invariant
|
||||
|
||||
> The accepted C-Chain head's state root must never be GC/pruning eligible —
|
||||
> across idle windows, duplicate empty-block state roots, cold snapshot/cache
|
||||
> layers, small state history, and restarts.
|
||||
>
|
||||
> accepted head ⇒ accepted head state root is pinned ⇒ GC/pruning cannot evict
|
||||
> the execution base for H+1.
|
||||
|
||||
The specific accepted-head GC eviction failure is **structurally prevented** by
|
||||
the head-state pin, and production evidence confirms the fleet no longer
|
||||
exhibits the prior failure signature. (A formal long-idle ritual was not
|
||||
completed to termination during the incident window; the sign-off rests on the
|
||||
structural invariant plus production evidence: a head idle for 3h04m was built
|
||||
on cleanly, multiple 5–15 minute zero-traffic windows passed with tip state
|
||||
readable, and zero eviction/materialize canaries appeared fleet-wide after the
|
||||
fix.)
|
||||
|
||||
## Failure mode (exact)
|
||||
|
||||
The C-Chain EVM (coreth-lineage, `luxfi/evm`) in pruning mode manages trie
|
||||
memory with `cappedMemoryTrieWriter` (`core/state_manager.go`):
|
||||
|
||||
- Accepted state roots are held in a `tipBuffer` (`BoundedBuffer`) of depth
|
||||
`state-history` (default **32**); as roots age out of the buffer they are
|
||||
`Dereference`d.
|
||||
- Dirty trie nodes are only committed to disk at `commit-interval` boundaries
|
||||
(default **4096** blocks), with optimistic `Cap` flushes near the boundary.
|
||||
- The insert-time `triedb.Reference(root, {})` in `writeBlockAndSetHead` is
|
||||
**refcount-balanced**: it is consumed as the block ages through the
|
||||
tipBuffer (or via `RejectTrie`). It therefore does not protect an idle head.
|
||||
|
||||
On an idle chain, consecutive empty blocks share identical state roots. The
|
||||
tipBuffer's aging `Dereference` for an old entry then lands on the *live head
|
||||
root* (duplicate key), dropping its reference count to zero. At any height that
|
||||
is not a commit boundary the head root has never been persisted, so the next
|
||||
`Cap`/flush evicts it from the dirty cache. The subsequent `BuildBlock` cannot
|
||||
open the parent (head) state:
|
||||
|
||||
```
|
||||
failed to materialize parent state for build: … StateAt: missing trie node
|
||||
<head state root> … is not available, not found
|
||||
```
|
||||
|
||||
and the chain wedges. RPC reads at `latest` fail with the same error (any
|
||||
`StateAt(root)` caller). Consensus is unaffected — all validators agree on the
|
||||
head *block*; only the local execution base for H+1 is gone. State is always
|
||||
deterministically re-derivable from durable blocks, so a restart re-executes
|
||||
and recovers — **restart is recovery evidence, not a fix**: the head could
|
||||
still be evicted again in the next idle window.
|
||||
|
||||
### Preconditions (all defaults on affected mainnet validators)
|
||||
|
||||
- `pruning-enabled: true`
|
||||
- `state-history: 32`
|
||||
- `commit-interval: 4096`
|
||||
- idle or bursty-then-idle traffic (heartbeat pause, low organic flow)
|
||||
- duplicate empty-block state roots at the tip
|
||||
- current height not at a commit boundary
|
||||
|
||||
Any C-Chain deployment matching these preconditions is exposed on affected
|
||||
versions — this bug class will recur wherever the EVM runs with pruning, small
|
||||
state history, long commit intervals, and idle traffic.
|
||||
|
||||
### Observed occurrences
|
||||
|
||||
1. Mainnet froze at height 1085200 after a ~10 minute heartbeat pause
|
||||
(2026-07-07). All five validators had the block, none could serve or build
|
||||
on its state.
|
||||
2. An earlier fleet-wide variant contributed to the 1082879→1085012 incident
|
||||
window (mixed with a separate proposervm/consensus issue documented in the
|
||||
consensus fault-recovery audit).
|
||||
|
||||
## Affected / fixed versions
|
||||
|
||||
| Component | Affected | Fixed |
|
||||
|---|---|---|
|
||||
| `luxfi/evm` (C-Chain plugin) | ≤ v1.104.6 (all pruning-mode deployments; the balanced insert-time Reference in v1.104.3-hotfix was insufficient) | **v1.104.7** |
|
||||
| `luxfi/node` image | v1.34.14 – v1.34.23 (carry affected EVM plugins) | **v1.34.24** (interim), **v1.34.25** (canonical: identical fix, proper semver, clean go.mod) |
|
||||
|
||||
Fix commits (`luxfi/evm`, branch `evm-main-bugb`): `9bab20f7a` (pin),
|
||||
`58b90490c` (non-fatal degradation), `e3781c35c` (vm v1.2.6 parity).
|
||||
|
||||
## The fix (structural)
|
||||
|
||||
`core/blockchain.go`: a dedicated **unbalanced** GC reference held on the
|
||||
accepted head's state root — `headStatePinRoot` + `pinAcceptedHead(root)`:
|
||||
|
||||
- Transferred head-to-head: `Reference` the new head root first, then
|
||||
`Dereference` the previous pinned root; exactly one live head pin exists at
|
||||
all times, on `lastAccepted.Root()`.
|
||||
- Established in `Accept`, in `SetLastAcceptedBlockDirect`, and on the loaded
|
||||
head at startup (`loadLastState`), so the invariant holds across restarts
|
||||
and the restart-then-idle path.
|
||||
- Skips when the root is unchanged (duplicate empty-block roots keep exactly
|
||||
one reference) and when state is not yet materialized (bootstrapping /
|
||||
state-sync; the first `Accept` then establishes it).
|
||||
- Deliberately **not** balanced against `InsertTrie`/`AcceptTrie`/`RejectTrie`
|
||||
— its lifetime is "is the accepted head", nothing else.
|
||||
- Non-fatal on backend error (pathdb `Reference`/`Dereference` are no-ops /
|
||||
"not supported"): a failed pin degrades to pre-fix behavior with a WARN
|
||||
rather than wedging Accept or startup.
|
||||
|
||||
No archive-mode workaround and no state-sync hack. Disk growth is unchanged
|
||||
(one extra referenced root).
|
||||
|
||||
## Recovery recipe (what actually worked)
|
||||
|
||||
1. **Wedged-at-tip (state evicted, DB otherwise consistent):** restart the
|
||||
node. Boot re-executes from the last committed root and re-materializes the
|
||||
head state deterministically. Valid as *recovery*; deploy the fixed version
|
||||
so it cannot recur.
|
||||
2. **proposervm/EVM height split** (`proposervm finality index … is BEHIND the
|
||||
inner VM tip`; produced here by crash-churn on affected versions — the
|
||||
fail-closed guard then correctly refuses to mount): restarts cannot heal a
|
||||
split. Restore the node's PVC from a `VolumeSnapshot` of a currently
|
||||
healthy peer. Per-ordinal staking keys are installed by `startup.sh` from
|
||||
the `luxd-staking` secret, so cross-node volume clones are safe (distinct
|
||||
NodeIDs).
|
||||
3. **PVC swaps must happen at StatefulSet `replicas=0`.** A live single-pod
|
||||
PVC delete/recreate always loses the race to the StatefulSet controller,
|
||||
which recreates a blank PVC first.
|
||||
4. If a fleet-consistent EVM rewind is needed instead (no healthy peer):
|
||||
`evm/cmd/repair-cchain` rewinds the standalone EVM `lastAccepted` to the
|
||||
proposervm floor; on boot the heightAhead branch self-heals (used in the
|
||||
1084996 recovery). Zero re-execution; never a re-genesis.
|
||||
5. Retain evidence snapshots before every destructive step.
|
||||
|
||||
## Residual follow-ups (non-blocking; restart/churn liveness, not consensus or state-loss)
|
||||
|
||||
1. **Proposer-preference restart loop:** after heavy sibling churn, a node's
|
||||
proposervm preference can reference a never-persisted outer block; every
|
||||
`BuildBlock` then fails `not found` in a tight loop and the node's voter
|
||||
goes mute (observed ~170 err/s). Restart clears it. Fix: fall back to
|
||||
last-accepted when the preferred parent is not fetchable.
|
||||
**FIXED (commit `8001bc5179`, branch `ship/node-v1.34.24`, ships in
|
||||
v1.34.26):** `vms/proposervm/vm.go` `BuildBlock` now builds the child on
|
||||
last-accepted (always held — committed state) when `vm.preferred` is
|
||||
unfetchable, instead of hard-erroring; it surfaces the original error only
|
||||
when last-accepted is itself the unfetchable id. Build-side companion to the
|
||||
already-shipped defect #1 `SetPreference` validate-before-assign hardening.
|
||||
Tests: `vms/proposervm/vm_buildblock_fallback_test.go`.
|
||||
2. **Ancestor-fetch liveness:** the finality guard refuses certs with
|
||||
"ancestor … is not tracked (behind; fetch and retry)" but the fetch never
|
||||
fires, so the node loops instead of catching up. Restart clears it. Fix:
|
||||
actually schedule the ancestor fetch on this path.
|
||||
|
||||
## Monitoring (keep active)
|
||||
|
||||
- Eviction/materialize canaries: `STATE-MATERIALIZE`, `missing trie node`,
|
||||
`ACCEPT-BACKSTOP` log lines — expect zero.
|
||||
- Accepted height/hash equality across all validators.
|
||||
- Explorer tip parity with chain head.
|
||||
- `is BEHIND the inner` (heightBehind) occurrences — expect zero.
|
||||
- Do not treat the heartbeat as a safety mechanism; it is a liveness nicety.
|
||||
@@ -75,9 +75,9 @@ func NewMultiNetworkNode() *MultiNetworkNode {
|
||||
|
||||
// StartRPCServer starts the unified RPC server
|
||||
func (n *MultiNetworkNode) StartRPCServer(port int) {
|
||||
http.HandleFunc("/v1/crossnet/status", n.handleCrossNetStatus)
|
||||
http.HandleFunc("/v1/crossnet/validators", n.handleCrossNetValidators)
|
||||
http.HandleFunc("/v1/network/", n.handleNetworkSpecific)
|
||||
http.HandleFunc("/ext/crossnet/status", n.handleCrossNetStatus)
|
||||
http.HandleFunc("/ext/crossnet/validators", n.handleCrossNetValidators)
|
||||
http.HandleFunc("/ext/network/", n.handleNetworkSpecific)
|
||||
|
||||
fmt.Printf("🌐 Multi-Network RPC Server starting on port %d\n", port)
|
||||
log.Fatal(http.ListenAndServe(fmt.Sprintf(":%d", port), nil))
|
||||
@@ -172,7 +172,7 @@ func (n *MultiNetworkNode) handleCrossNetValidators(w http.ResponseWriter, r *ht
|
||||
|
||||
// handleNetworkSpecific routes to network-specific handlers
|
||||
func (n *MultiNetworkNode) handleNetworkSpecific(w http.ResponseWriter, r *http.Request) {
|
||||
// Parse network ID from path: /v1/network/{networkID}/...
|
||||
// Parse network ID from path: /ext/network/{networkID}/...
|
||||
// This would route to the appropriate network's chain manager
|
||||
|
||||
response := fmt.Sprintf(`{
|
||||
@@ -251,9 +251,9 @@ func main() {
|
||||
}
|
||||
|
||||
fmt.Println("\n🌐 Starting Multi-Network RPC Server...")
|
||||
fmt.Println(" • Cross-network status: http://localhost:9650/v1/crossnet/status")
|
||||
fmt.Println(" • Cross-network validators: http://localhost:9650/v1/crossnet/validators")
|
||||
fmt.Println(" • Network-specific: http://localhost:9650/v1/network/{networkID}/...")
|
||||
fmt.Println(" • Cross-network status: http://localhost:9650/ext/crossnet/status")
|
||||
fmt.Println(" • Cross-network validators: http://localhost:9650/ext/crossnet/validators")
|
||||
fmt.Println(" • Network-specific: http://localhost:9650/ext/network/{networkID}/...")
|
||||
|
||||
// This would be replaced with actual RPC server
|
||||
node.StartRPCServer(9650)
|
||||
|
||||
Executable
BIN
Binary file not shown.
+13
-23
@@ -54,8 +54,7 @@ var (
|
||||
QChainAliases = AliasesFor("Q")
|
||||
AChainAliases = AliasesFor("A")
|
||||
BChainAliases = AliasesFor("B")
|
||||
MChainAliases = AliasesFor("M")
|
||||
FChainAliases = AliasesFor("F")
|
||||
TChainAliases = AliasesFor("T")
|
||||
ZChainAliases = AliasesFor("Z")
|
||||
GChainAliases = AliasesFor("G")
|
||||
KChainAliases = AliasesFor("K")
|
||||
@@ -605,8 +604,7 @@ func FromConfig(config *genesiscfg.Config) ([]byte, ids.ID, error) {
|
||||
{[]byte(config.CChainGenesis), constants.EVMID, "C-Chain", nil},
|
||||
{[]byte(config.DChainGenesis), constants.DexVMID, "D-Chain", nil},
|
||||
{[]byte(config.BChainGenesis), constants.BridgeVMID, "B-Chain", nil},
|
||||
{[]byte(config.MChainGenesis), constants.MPCVMID, "M-Chain", nil},
|
||||
{[]byte(config.FChainGenesis), constants.FHEVMID, "F-Chain", nil},
|
||||
{[]byte(config.TChainGenesis), constants.ThresholdVMID, "T-Chain", nil},
|
||||
{[]byte(config.QChainGenesis), constants.QuantumVMID, "Q-Chain", nil},
|
||||
{[]byte(config.ZChainGenesis), constants.ZKVMID, "Z-Chain", nil},
|
||||
// A/G/K carry genesis blobs and are deterministic genesis chains
|
||||
@@ -728,10 +726,10 @@ func UTXOAssetIDFromGenesisBytes(genesisBytes []byte) (ids.ID, bool, error) {
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
if uChain.VMID() != constants.XVMID {
|
||||
if uChain.VMID != constants.XVMID {
|
||||
continue
|
||||
}
|
||||
id, err := xvmgenesis.AssetIDFromBytes(uChain.GenesisData())
|
||||
id, err := xvmgenesis.AssetIDFromBytes(uChain.GenesisData)
|
||||
if err != nil {
|
||||
return ids.Empty, false, fmt.Errorf("derive X-Chain asset ID from genesis data: %w", err)
|
||||
}
|
||||
@@ -749,7 +747,7 @@ func VMGenesis(genesisBytes []byte, vmID ids.ID) (*pchaintxs.Tx, error) {
|
||||
}
|
||||
for _, chain := range gen.Chains {
|
||||
uChain := chain.Unsigned.(*pchaintxs.CreateChainTx)
|
||||
if uChain.VMID() == vmID {
|
||||
if uChain.VMID == vmID {
|
||||
return chain, nil
|
||||
}
|
||||
}
|
||||
@@ -778,7 +776,7 @@ func Aliases(genesisBytes []byte) (map[string][]string, map[ids.ID][]string, err
|
||||
uChain := chain.Unsigned.(*pchaintxs.CreateChainTx)
|
||||
chainID := chain.ID()
|
||||
endpoint := path.Join(constants.ChainAliasPrefix, chainID.String())
|
||||
switch uChain.VMID() {
|
||||
switch uChain.VMID {
|
||||
case constants.XVMID:
|
||||
apiAliases[endpoint] = []string{
|
||||
"X",
|
||||
@@ -834,24 +832,16 @@ func Aliases(genesisBytes []byte) (map[string][]string, map[ids.ID][]string, err
|
||||
path.Join(constants.ChainAliasPrefix, "bridge"),
|
||||
}
|
||||
chainAliases[chainID] = BChainAliases
|
||||
case constants.MPCVMID:
|
||||
case constants.ThresholdVMID:
|
||||
apiAliases[endpoint] = []string{
|
||||
"M",
|
||||
"T",
|
||||
"threshold",
|
||||
"thresholdvm",
|
||||
"mpc",
|
||||
"mpcvm",
|
||||
path.Join(constants.ChainAliasPrefix, "M"),
|
||||
path.Join(constants.ChainAliasPrefix, "mpc"),
|
||||
path.Join(constants.ChainAliasPrefix, "T"),
|
||||
path.Join(constants.ChainAliasPrefix, "threshold"),
|
||||
}
|
||||
chainAliases[chainID] = MChainAliases
|
||||
case constants.FHEVMID:
|
||||
apiAliases[endpoint] = []string{
|
||||
"F",
|
||||
"fhe",
|
||||
"fhevm",
|
||||
path.Join(constants.ChainAliasPrefix, "F"),
|
||||
path.Join(constants.ChainAliasPrefix, "fhe"),
|
||||
}
|
||||
chainAliases[chainID] = FChainAliases
|
||||
chainAliases[chainID] = TChainAliases
|
||||
case constants.ZKVMID:
|
||||
apiAliases[endpoint] = []string{
|
||||
"Z",
|
||||
|
||||
@@ -11,6 +11,7 @@ import (
|
||||
"github.com/stretchr/testify/require"
|
||||
|
||||
"github.com/luxfi/node/vms/platformvm/genesis"
|
||||
pchaintxs "github.com/luxfi/node/vms/platformvm/txs"
|
||||
)
|
||||
|
||||
// TestUTXOAssetIDFromGenesisBytes_Sovereign asserts the canonical
|
||||
@@ -62,7 +63,7 @@ func TestUTXOAssetIDFromGenesisBytes_POnly(t *testing.T) {
|
||||
require := require.New(t)
|
||||
|
||||
pOnly := &genesis.Genesis{Chains: nil}
|
||||
pOnlyBytes, err := pOnly.Bytes()
|
||||
pOnlyBytes, err := genesis.Codec.Marshal(pchaintxs.CodecVersion, pOnly)
|
||||
require.NoError(err)
|
||||
|
||||
id, ok, err := UTXOAssetIDFromGenesisBytes(pOnlyBytes)
|
||||
@@ -92,7 +93,7 @@ func TestVMGenesisOptInChains(t *testing.T) {
|
||||
pOnly := &genesis.Genesis{
|
||||
Chains: nil,
|
||||
}
|
||||
pOnlyBytes, err := pOnly.Bytes()
|
||||
pOnlyBytes, err := genesis.Codec.Marshal(pchaintxs.CodecVersion, pOnly)
|
||||
require.NoError(err)
|
||||
|
||||
for name, vmID := range map[string]ids.ID{
|
||||
|
||||
@@ -314,15 +314,15 @@ func TestFromConfigExplicitStakers(t *testing.T) {
|
||||
for i, vdrTx := range parsed.Validators {
|
||||
switch ut := vdrTx.Unsigned.(type) {
|
||||
case *txs.AddValidatorTx:
|
||||
require.Equal(stakers[i].Weight, ut.Weight(),
|
||||
require.Equal(stakers[i].Weight, ut.Wght,
|
||||
"validator %d weight mismatch", i)
|
||||
t.Logf("Validator %d: NodeID=%s Weight=%d StakeOuts=%d",
|
||||
i, ut.Validator().NodeID, ut.Weight(), len(ut.StakeOuts()))
|
||||
i, ut.Validator.NodeID, ut.Wght, len(ut.StakeOuts))
|
||||
case *txs.AddPermissionlessValidatorTx:
|
||||
require.Equal(stakers[i].Weight, ut.Weight(),
|
||||
require.Equal(stakers[i].Weight, ut.Wght,
|
||||
"validator %d weight mismatch", i)
|
||||
t.Logf("Validator %d: NodeID=%s Weight=%d StakeOuts=%d",
|
||||
i, ut.Validator().NodeID, ut.Weight(), len(ut.StakeOuts()))
|
||||
i, ut.Validator.NodeID, ut.Wght, len(ut.StakeOuts))
|
||||
default:
|
||||
t.Fatalf("unexpected validator tx type: %T", ut)
|
||||
}
|
||||
@@ -396,15 +396,15 @@ func TestFromConfigExplicitStakersNoStakedFunds(t *testing.T) {
|
||||
for i, vdrTx := range parsed.Validators {
|
||||
switch ut := vdrTx.Unsigned.(type) {
|
||||
case *txs.AddValidatorTx:
|
||||
require.Greater(ut.Weight(), uint64(0), "validator %d weight must be non-zero", i)
|
||||
require.Greater(len(ut.StakeOuts()), 0, "validator %d must have stake outputs", i)
|
||||
require.Greater(ut.Wght, uint64(0), "validator %d weight must be non-zero", i)
|
||||
require.Greater(len(ut.StakeOuts), 0, "validator %d must have stake outputs", i)
|
||||
t.Logf("Validator %d: NodeID=%s Weight=%d StakeOuts=%d",
|
||||
i, ut.Validator().NodeID, ut.Weight(), len(ut.StakeOuts()))
|
||||
i, ut.Validator.NodeID, ut.Wght, len(ut.StakeOuts))
|
||||
case *txs.AddPermissionlessValidatorTx:
|
||||
require.Greater(ut.Weight(), uint64(0), "validator %d weight must be non-zero", i)
|
||||
require.Greater(len(ut.StakeOuts()), 0, "validator %d must have stake outputs", i)
|
||||
require.Greater(ut.Wght, uint64(0), "validator %d weight must be non-zero", i)
|
||||
require.Greater(len(ut.StakeOuts), 0, "validator %d must have stake outputs", i)
|
||||
t.Logf("Validator %d: NodeID=%s Weight=%d StakeOuts=%d",
|
||||
i, ut.Validator().NodeID, ut.Weight(), len(ut.StakeOuts()))
|
||||
i, ut.Validator.NodeID, ut.Wght, len(ut.StakeOuts))
|
||||
default:
|
||||
t.Fatalf("unexpected validator tx type: %T", ut)
|
||||
}
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user