mirror of
https://github.com/tendermint/tendermint.git
synced 2026-09-06 16:17:11 +00:00
ci: add markdown linter (#146)
This commit is contained in:
@@ -1,11 +1,13 @@
|
||||
# Blockchain Reactor v1
|
||||
|
||||
### Data Structures
|
||||
## Data Structures
|
||||
|
||||
The data structures used are illustrated below.
|
||||
|
||||

|
||||
|
||||
#### BlockchainReactor
|
||||
### BlockchainReactor
|
||||
|
||||
- is a `p2p.BaseReactor`.
|
||||
- has a `store.BlockStore` for persistence.
|
||||
- executes blocks using an `sm.BlockExecutor`.
|
||||
@@ -17,33 +19,34 @@ The data structures used are illustrated below.
|
||||
|
||||
```go
|
||||
type BlockchainReactor struct {
|
||||
p2p.BaseReactor
|
||||
p2p.BaseReactor
|
||||
|
||||
initialState sm.State // immutable
|
||||
state sm.State
|
||||
initialState sm.State // immutable
|
||||
state sm.State
|
||||
|
||||
blockExec *sm.BlockExecutor
|
||||
store *store.BlockStore
|
||||
blockExec *sm.BlockExecutor
|
||||
store *store.BlockStore
|
||||
|
||||
fastSync bool
|
||||
fastSync bool
|
||||
|
||||
fsm *BcReactorFSM
|
||||
blocksSynced int
|
||||
fsm *BcReactorFSM
|
||||
blocksSynced int
|
||||
|
||||
// Receive goroutine forwards messages to this channel to be processed in the context of the poolRoutine.
|
||||
messagesForFSMCh chan bcReactorMessage
|
||||
// Receive goroutine forwards messages to this channel to be processed in the context of the poolRoutine.
|
||||
messagesForFSMCh chan bcReactorMessage
|
||||
|
||||
// Switch goroutine may send RemovePeer to the blockchain reactor. This is an error message that is relayed
|
||||
// to this channel to be processed in the context of the poolRoutine.
|
||||
errorsForFSMCh chan bcReactorMessage
|
||||
// Switch goroutine may send RemovePeer to the blockchain reactor. This is an error message that is relayed
|
||||
// to this channel to be processed in the context of the poolRoutine.
|
||||
errorsForFSMCh chan bcReactorMessage
|
||||
|
||||
// This channel is used by the FSM and indirectly the block pool to report errors to the blockchain reactor and
|
||||
// the switch.
|
||||
eventsFromFSMCh chan bcFsmMessage
|
||||
// This channel is used by the FSM and indirectly the block pool to report errors to the blockchain reactor and
|
||||
// the switch.
|
||||
eventsFromFSMCh chan bcFsmMessage
|
||||
}
|
||||
```
|
||||
|
||||
#### BcReactorFSM
|
||||
|
||||
- implements a simple finite state machine.
|
||||
- has a state and a state timer.
|
||||
- has a `BlockPool` to keep track of block requests sent to peers and blocks received from peers.
|
||||
@@ -51,49 +54,53 @@ type BlockchainReactor struct {
|
||||
|
||||
```go
|
||||
type BcReactorFSM struct {
|
||||
logger log.Logger
|
||||
mtx sync.Mutex
|
||||
logger log.Logger
|
||||
mtx sync.Mutex
|
||||
|
||||
startTime time.Time
|
||||
startTime time.Time
|
||||
|
||||
state *bcReactorFSMState
|
||||
stateTimer *time.Timer
|
||||
pool *BlockPool
|
||||
state *bcReactorFSMState
|
||||
stateTimer *time.Timer
|
||||
pool *BlockPool
|
||||
|
||||
// interface used to call the Blockchain reactor to send StatusRequest, BlockRequest, reporting errors, etc.
|
||||
toBcR bcReactor
|
||||
// interface used to call the Blockchain reactor to send StatusRequest, BlockRequest, reporting errors, etc.
|
||||
toBcR bcReactor
|
||||
}
|
||||
```
|
||||
|
||||
#### BlockPool
|
||||
|
||||
- maintains a peer set, implemented as a map of peer ID to `BpPeer`.
|
||||
- maintains a set of requests made to peers, implemented as a map of block request heights to peer IDs.
|
||||
- maintains a list of future block requests needed to advance the fast-sync. This is a list of block heights.
|
||||
- maintains a list of future block requests needed to advance the fast-sync. This is a list of block heights.
|
||||
- keeps track of the maximum height of the peers in the set.
|
||||
- uses an interface to send requests and report errors to the reactor (via FSM).
|
||||
|
||||
```go
|
||||
type BlockPool struct {
|
||||
logger log.Logger
|
||||
// Set of peers that have sent status responses, with height bigger than pool.Height
|
||||
peers map[p2p.ID]*BpPeer
|
||||
// Set of block heights and the corresponding peers from where a block response is expected or has been received.
|
||||
blocks map[int64]p2p.ID
|
||||
logger log.Logger
|
||||
// Set of peers that have sent status responses, with height bigger than pool.Height
|
||||
peers map[p2p.ID]*BpPeer
|
||||
// Set of block heights and the corresponding peers from where a block response is expected or has been received.
|
||||
blocks map[int64]p2p.ID
|
||||
|
||||
plannedRequests map[int64]struct{} // list of blocks to be assigned peers for blockRequest
|
||||
nextRequestHeight int64 // next height to be added to plannedRequests
|
||||
plannedRequests map[int64]struct{} // list of blocks to be assigned peers for blockRequest
|
||||
nextRequestHeight int64 // next height to be added to plannedRequests
|
||||
|
||||
Height int64 // height of next block to execute
|
||||
MaxPeerHeight int64 // maximum height of all peers
|
||||
toBcR bcReactor
|
||||
Height int64 // height of next block to execute
|
||||
MaxPeerHeight int64 // maximum height of all peers
|
||||
toBcR bcReactor
|
||||
}
|
||||
```
|
||||
|
||||
Some reasons for the `BlockPool` data structure content:
|
||||
|
||||
1. If a peer is removed by the switch fast access is required to the peer and the block requests made to that peer in order to redo them.
|
||||
2. When block verification fails fast access is required from the block height to the peer and the block requests made to that peer in order to redo them.
|
||||
3. The `BlockchainReactor` main routine decides when the block pool is running low and asks the `BlockPool` (via FSM) to make more requests. The `BlockPool` creates a list of requests and triggers the sending of the block requests (via the interface). The reason it maintains a list of requests is the redo operations that may occur during error handling. These are redone when the `BlockchainReactor` requires more blocks.
|
||||
|
||||
#### BpPeer
|
||||
|
||||
- keeps track of a single peer, with height bigger than the initial height.
|
||||
- maintains the block requests made to the peer and the blocks received from the peer until they are executed.
|
||||
- monitors the peer speed when there are pending requests.
|
||||
@@ -101,17 +108,17 @@ Some reasons for the `BlockPool` data structure content:
|
||||
|
||||
```go
|
||||
type BpPeer struct {
|
||||
logger log.Logger
|
||||
ID p2p.ID
|
||||
logger log.Logger
|
||||
ID p2p.ID
|
||||
|
||||
Height int64 // the peer reported height
|
||||
NumPendingBlockRequests int // number of requests still waiting for block responses
|
||||
blocks map[int64]*types.Block // blocks received or expected to be received from this peer
|
||||
blockResponseTimer *time.Timer
|
||||
recvMonitor *flow.Monitor
|
||||
params *BpPeerParams // parameters for timer and monitor
|
||||
Height int64 // the peer reported height
|
||||
NumPendingBlockRequests int // number of requests still waiting for block responses
|
||||
blocks map[int64]*types.Block // blocks received or expected to be received from this peer
|
||||
blockResponseTimer *time.Timer
|
||||
recvMonitor *flow.Monitor
|
||||
params *BpPeerParams // parameters for timer and monitor
|
||||
|
||||
onErr func(err error, peerID p2p.ID) // function to call on error
|
||||
onErr func(err error, peerID p2p.ID) // function to call on error
|
||||
}
|
||||
```
|
||||
|
||||
@@ -120,61 +127,73 @@ type BpPeer struct {
|
||||
The diagram below shows the goroutines (depicted by the gray blocks), timers (shown on the left with their values) and channels (colored rectangles). The FSM box shows some of the functionality and it is not a separate goroutine.
|
||||
|
||||
The interface used by the FSM is shown in light red with the `IF` block. This is used to:
|
||||
- send block requests
|
||||
|
||||
- send block requests
|
||||
- report peer errors to the switch - this results in the reactor calling `switch.StopPeerForError()` and, if triggered by the peer timeout routine, a `removePeerEv` is sent to the FSM and action is taken from the context of the `poolRoutine()`
|
||||
- ask the reactor to reset the state timers. The timers are owned by the FSM while the timeout routine is defined by the reactor. This was done in order to avoid running timers in tests and will change in the next revision.
|
||||
|
||||
|
||||
There are two main goroutines implemented by the blockchain reactor. All I/O operations are performed from the `poolRoutine()` context while the CPU intensive operations related to the block execution are performed from the context of the `executeBlocksRoutine()`. All goroutines are detailed in the next sections.
|
||||
|
||||

|
||||
|
||||
#### Receive()
|
||||
|
||||
Fast-sync messages from peers are received by this goroutine. It performs basic validation and:
|
||||
|
||||
- in helper mode (i.e. for request message) it replies immediately. This is different than the proposal in adr-040 that specifies having the FSM handling these.
|
||||
- forwards response messages to the `poolRoutine()`.
|
||||
|
||||
#### poolRoutine()
|
||||
(named kept as in the previous reactor).
|
||||
|
||||
(named kept as in the previous reactor).
|
||||
It starts the `executeBlocksRoutine()` and the FSM. It then waits in a loop for events. These are received from the following channels:
|
||||
|
||||
- `sendBlockRequestTicker.C` - every 10msec the reactor asks FSM to make more block requests up to a maximum. Note: currently this value is constant but could be changed based on low/ high watermark thresholds for the number of blocks received and waiting to be processed, the number of blockResponse messages waiting in messagesForFSMCh, etc.
|
||||
- `statusUpdateTicker.C` - every 10 seconds the reactor broadcasts status requests to peers. While adr-040 specifies this to run within the FSM, at this point this functionality is kept in the reactor.
|
||||
- `messagesForFSMCh` - the `Receive()` goroutine sends status and block response messages to this channel and the reactor calls FSM to handle them.
|
||||
- `errorsForFSMCh` - this channel receives the following events:
|
||||
- `errorsForFSMCh` - this channel receives the following events:
|
||||
- peer remove - when the switch removes a peer
|
||||
- sate timeout event - when FSM state timers trigger
|
||||
The reactor forwards this messages to the FSM.
|
||||
- `eventsFromFSMCh` - there are two type of events sent over this channel:
|
||||
- `syncFinishedEv` - triggered when FSM enters `finished` state and calls the switchToConsensus() interface function.
|
||||
- `peerErrorEv`- peer timer expiry goroutine sends this event over the channel for processing from poolRoutine() context.
|
||||
|
||||
#### executeBlocksRoutine()
|
||||
Started by the `poolRoutine()`, it retrieves blocks from the pool and executes them:
|
||||
- `processReceivedBlockTicker.C` - a ticker event is received over the channel every 10msec and its handling results in a signal being sent to the doProcessBlockCh channel.
|
||||
- doProcessBlockCh - events are received on this channel as described as above and upon processing blocks are retrieved from the pool and executed.
|
||||
|
||||
#### executeBlocksRoutine()
|
||||
|
||||
Started by the `poolRoutine()`, it retrieves blocks from the pool and executes them:
|
||||
|
||||
- `processReceivedBlockTicker.C` - a ticker event is received over the channel every 10msec and its handling results in a signal being sent to the doProcessBlockCh channel.
|
||||
- doProcessBlockCh - events are received on this channel as described as above and upon processing blocks are retrieved from the pool and executed.
|
||||
|
||||
### FSM
|
||||
|
||||

|
||||
|
||||
#### States
|
||||
|
||||
##### init (aka unknown)
|
||||
|
||||
The FSM is created in `unknown` state. When started, by the reactor (`startFSMEv`), it broadcasts Status requests and transitions to `waitForPeer` state.
|
||||
|
||||
##### waitForPeer
|
||||
|
||||
In this state, the FSM waits for a Status responses from a "tall" peer. A timer is running in this state to allow the FSM to finish if there are no useful peers.
|
||||
|
||||
If the timer expires, it moves to `finished` state and calls the reactor to switch to consensus.
|
||||
If a Status response is received from a peer within the timeout, the FSM transitions to `waitForBlock` state.
|
||||
|
||||
##### waitForBlock
|
||||
|
||||
In this state the FSM makes Block requests (triggered by a ticker in reactor) and waits for Block responses. There is a timer running in this state to detect if a peer is not sending the block at current processing height. If the timer expires, the FSM removes the peer where the request was sent and all requests made to that peer are redone.
|
||||
|
||||
As blocks are received they are stored by the pool. Block execution is independently performed by the reactor and the result reported to the FSM:
|
||||
|
||||
- if there are no errors, the FSM increases the pool height and resets the state timer.
|
||||
- if there are errors, the peers that delivered the two blocks (at height and height+1) are removed and the requests redone.
|
||||
|
||||
In this state the FSM may receive peer remove events in any of the following scenarios:
|
||||
In this state the FSM may receive peer remove events in any of the following scenarios:
|
||||
|
||||
- the switch is removing a peer
|
||||
- a peer is penalized because it has not responded to some block requests for a long time
|
||||
- a peer is penalized for being slow
|
||||
@@ -183,6 +202,7 @@ When processing of the last block (the one with height equal to the highest peer
|
||||
If after a peer update or removal the pool height is same as maxPeerHeight, the FSM transitions to `finished` state.
|
||||
|
||||
##### finished
|
||||
|
||||
When entering this state, the FSM calls the reactor to switch to consensus and performs cleanup.
|
||||
|
||||
#### Events
|
||||
@@ -191,18 +211,19 @@ The following events are handled by the FSM:
|
||||
|
||||
```go
|
||||
const (
|
||||
startFSMEv = iota + 1
|
||||
statusResponseEv
|
||||
blockResponseEv
|
||||
processedBlockEv
|
||||
makeRequestsEv
|
||||
stopFSMEv
|
||||
peerRemoveEv = iota + 256
|
||||
stateTimeoutEv
|
||||
startFSMEv = iota + 1
|
||||
statusResponseEv
|
||||
blockResponseEv
|
||||
processedBlockEv
|
||||
makeRequestsEv
|
||||
stopFSMEv
|
||||
peerRemoveEv = iota + 256
|
||||
stateTimeoutEv
|
||||
)
|
||||
```
|
||||
|
||||
### Examples of Scenarios and Termination Handling
|
||||
|
||||
A few scenarios are covered in this section together with the current/ proposed handling.
|
||||
In general, the scenarios involving faulty peers are made worse by the fact that they may quickly be re-added.
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
## Blockchain Reactor v0 Modules
|
||||
# Blockchain Reactor v0 Module
|
||||
|
||||
### Blockchain Reactor
|
||||
## Blockchain Reactor
|
||||
|
||||
- coordinates the pool for syncing
|
||||
- coordinates the store for persistence
|
||||
@@ -10,35 +10,34 @@
|
||||
- starts the pool.Start() and its poolRoutine()
|
||||
- registers all the concrete types and interfaces for serialisation
|
||||
|
||||
#### poolRoutine
|
||||
### poolRoutine
|
||||
|
||||
- listens to these channels:
|
||||
- pool requests blocks from a specific peer by posting to requestsCh, block reactor then sends
|
||||
- pool requests blocks from a specific peer by posting to requestsCh, block reactor then sends
|
||||
a &bcBlockRequestMessage for a specific height
|
||||
- pool signals timeout of a specific peer by posting to timeoutsCh
|
||||
- switchToConsensusTicker to periodically try and switch to consensus
|
||||
- trySyncTicker to periodically check if we have fallen behind and then catch-up sync
|
||||
- if there aren't any new blocks available on the pool it skips syncing
|
||||
- pool signals timeout of a specific peer by posting to timeoutsCh
|
||||
- switchToConsensusTicker to periodically try and switch to consensus
|
||||
- trySyncTicker to periodically check if we have fallen behind and then catch-up sync
|
||||
- if there aren't any new blocks available on the pool it skips syncing
|
||||
- tries to sync the app by taking downloaded blocks from the pool, gives them to the app and stores
|
||||
them on disk
|
||||
- implements Receive which is called by the switch/peer
|
||||
- calls AddBlock on the pool when it receives a new block from a peer
|
||||
- calls AddBlock on the pool when it receives a new block from a peer
|
||||
|
||||
### Block Pool
|
||||
## Block Pool
|
||||
|
||||
- responsible for downloading blocks from peers
|
||||
- makeRequestersRoutine()
|
||||
- removes timeout peers
|
||||
- starts new requesters by calling makeNextRequester()
|
||||
- removes timeout peers
|
||||
- starts new requesters by calling makeNextRequester()
|
||||
- requestRoutine():
|
||||
- picks a peer and sends the request, then blocks until:
|
||||
- pool is stopped by listening to pool.Quit
|
||||
- requester is stopped by listening to Quit
|
||||
- request is redone
|
||||
- we receive a block
|
||||
- gotBlockCh is strange
|
||||
- picks a peer and sends the request, then blocks until:
|
||||
- pool is stopped by listening to pool.Quit
|
||||
- requester is stopped by listening to Quit
|
||||
- request is redone
|
||||
- we receive a block
|
||||
- gotBlockCh is strange
|
||||
|
||||
|
||||
### Go Routines in Blockchain Reactor
|
||||
## Go Routines in Blockchain Reactor
|
||||
|
||||

|
||||
|
||||
@@ -186,7 +186,7 @@ fetchBlock(height, pool):
|
||||
mtx.Lock()
|
||||
pool.numPending++
|
||||
redo = true
|
||||
mtx.UnLock()
|
||||
mtx.UnLock()
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -251,23 +251,23 @@ main(pool):
|
||||
|
||||
while true do
|
||||
select {
|
||||
upon receiving BlockRequest(Height, Peer) on pool.requestsChannel:
|
||||
try to send bcBlockRequestMessage(Height) to Peer
|
||||
upon receiving BlockRequest(Height, Peer) on pool.requestsChannel:
|
||||
try to send bcBlockRequestMessage(Height) to Peer
|
||||
|
||||
upon receiving error(peer) on errorsChannel:
|
||||
stop peer for error
|
||||
upon receiving error(peer) on errorsChannel:
|
||||
stop peer for error
|
||||
|
||||
upon receiving message on statusUpdateTickerChannel:
|
||||
broadcast bcStatusRequestMessage(bcR.store.Height) // message sent in a separate routine
|
||||
upon receiving message on statusUpdateTickerChannel:
|
||||
broadcast bcStatusRequestMessage(bcR.store.Height) // message sent in a separate routine
|
||||
|
||||
upon receiving message on switchToConsensusTickerChannel:
|
||||
pool.mtx.Lock()
|
||||
receivedBlockOrTimedOut = pool.height > 0 || (time.Now() - pool.startTime) > 5 Seconds
|
||||
ourChainIsLongestAmongPeers = pool.maxPeerHeight == 0 || pool.height >= pool.maxPeerHeight
|
||||
haveSomePeers = size of pool.peers > 0
|
||||
pool.mtx.Unlock()
|
||||
if haveSomePeers && receivedBlockOrTimedOut && ourChainIsLongestAmongPeers then
|
||||
switch to consensus mode
|
||||
upon receiving message on switchToConsensusTickerChannel:
|
||||
pool.mtx.Lock()
|
||||
receivedBlockOrTimedOut = pool.height > 0 || (time.Now() - pool.startTime) > 5 Seconds
|
||||
ourChainIsLongestAmongPeers = pool.maxPeerHeight == 0 || pool.height >= pool.maxPeerHeight
|
||||
haveSomePeers = size of pool.peers > 0
|
||||
pool.mtx.Unlock()
|
||||
if haveSomePeers && receivedBlockOrTimedOut && ourChainIsLongestAmongPeers then
|
||||
switch to consensus mode
|
||||
|
||||
upon receiving message on trySyncTickerChannel:
|
||||
for i = 0; i < 10; i++ do
|
||||
@@ -294,7 +294,7 @@ main(pool):
|
||||
redoRequestsForPeer(pool, peerId):
|
||||
for each requester in pool.requesters do
|
||||
if requester.getPeerID() == peerID
|
||||
enqueue msg on redoChannel for requester
|
||||
enqueue msg on redoChannel for requester
|
||||
```
|
||||
|
||||
## Channels
|
||||
|
||||
@@ -42,19 +42,19 @@ received votes and last commit and last validators set.
|
||||
|
||||
```go
|
||||
type RoundState struct {
|
||||
Height int64
|
||||
Round int
|
||||
Step RoundStepType
|
||||
Validators ValidatorSet
|
||||
Proposal Proposal
|
||||
ProposalBlock Block
|
||||
ProposalBlockParts PartSet
|
||||
LockedRound int
|
||||
LockedBlock Block
|
||||
LockedBlockParts PartSet
|
||||
Votes HeightVoteSet
|
||||
LastCommit VoteSet
|
||||
LastValidators ValidatorSet
|
||||
Height int64
|
||||
Round int
|
||||
Step RoundStepType
|
||||
Validators ValidatorSet
|
||||
Proposal Proposal
|
||||
ProposalBlock Block
|
||||
ProposalBlockParts PartSet
|
||||
LockedRound int
|
||||
LockedBlock Block
|
||||
LockedBlockParts PartSet
|
||||
Votes HeightVoteSet
|
||||
LastCommit VoteSet
|
||||
LastValidators ValidatorSet
|
||||
}
|
||||
```
|
||||
|
||||
@@ -77,20 +77,20 @@ Consensus Reactor and by the gossip routines upon sending a message to the peer.
|
||||
|
||||
```golang
|
||||
type PeerRoundState struct {
|
||||
Height int64 // Height peer is at
|
||||
Round int // Round peer is at, -1 if unknown.
|
||||
Step RoundStepType // Step peer is at
|
||||
Proposal bool // True if peer has proposal for this round
|
||||
ProposalBlockPartsHeader PartSetHeader
|
||||
ProposalBlockParts BitArray
|
||||
ProposalPOLRound int // Proposal's POL round. -1 if none.
|
||||
ProposalPOL BitArray // nil until ProposalPOLMessage received.
|
||||
Prevotes BitArray // All votes peer has for this round
|
||||
Precommits BitArray // All precommits peer has for this round
|
||||
LastCommitRound int // Round of commit for last height. -1 if none.
|
||||
LastCommit BitArray // All commit precommits of commit for last height.
|
||||
CatchupCommitRound int // Round that we have commit for. Not necessarily unique. -1 if none.
|
||||
CatchupCommit BitArray // All commit precommits peer has for this height & CatchupCommitRound
|
||||
Height int64 // Height peer is at
|
||||
Round int // Round peer is at, -1 if unknown.
|
||||
Step RoundStepType // Step peer is at
|
||||
Proposal bool // True if peer has proposal for this round
|
||||
ProposalBlockPartsHeader PartSetHeader
|
||||
ProposalBlockParts BitArray
|
||||
ProposalPOLRound int // Proposal's POL round. -1 if none.
|
||||
ProposalPOL BitArray // nil until ProposalPOLMessage received.
|
||||
Prevotes BitArray // All votes peer has for this round
|
||||
Precommits BitArray // All precommits peer has for this round
|
||||
LastCommitRound int // Round of commit for last height. -1 if none.
|
||||
LastCommit BitArray // All commit precommits of commit for last height.
|
||||
CatchupCommitRound int // Round that we have commit for. Not necessarily unique. -1 if none.
|
||||
CatchupCommit BitArray // All commit precommits peer has for this height & CatchupCommitRound
|
||||
}
|
||||
```
|
||||
|
||||
@@ -106,7 +106,7 @@ respectively.
|
||||
|
||||
### NewRoundStepMessage handler
|
||||
|
||||
```
|
||||
```go
|
||||
handleMessage(msg):
|
||||
if msg is from smaller height/round/step then return
|
||||
// Just remember these values.
|
||||
@@ -123,17 +123,17 @@ handleMessage(msg):
|
||||
if prs.Height has been updated then
|
||||
if prsHeight+1 == msg.Height && prsRound == msg.LastCommitRound then
|
||||
prs.LastCommitRound = msg.LastCommitRound
|
||||
prs.LastCommit = prs.Precommits
|
||||
prs.LastCommit = prs.Precommits
|
||||
} else {
|
||||
prs.LastCommitRound = msg.LastCommitRound
|
||||
prs.LastCommit = nil
|
||||
prs.LastCommit = nil
|
||||
}
|
||||
Reset prs.CatchupCommitRound and prs.CatchupCommit
|
||||
```
|
||||
|
||||
### NewValidBlockMessage handler
|
||||
|
||||
```
|
||||
```go
|
||||
handleMessage(msg):
|
||||
if prs.Height != msg.Height then return
|
||||
|
||||
@@ -148,7 +148,7 @@ protect the node against DOS attacks.
|
||||
|
||||
### HasVoteMessage handler
|
||||
|
||||
```
|
||||
```go
|
||||
handleMessage(msg):
|
||||
if prs.Height == msg.Height then
|
||||
prs.setHasVote(msg.Height, msg.Round, msg.Type, msg.Index)
|
||||
@@ -156,7 +156,7 @@ handleMessage(msg):
|
||||
|
||||
### VoteSetMaj23Message handler
|
||||
|
||||
```
|
||||
```go
|
||||
handleMessage(msg):
|
||||
if prs.Height == msg.Height then
|
||||
Record in rs that a peer claim to have ⅔ majority for msg.BlockID
|
||||
@@ -165,7 +165,7 @@ handleMessage(msg):
|
||||
|
||||
### ProposalMessage handler
|
||||
|
||||
```
|
||||
```go
|
||||
handleMessage(msg):
|
||||
if prs.Height != msg.Height || prs.Round != msg.Round || prs.Proposal then return
|
||||
prs.Proposal = true
|
||||
@@ -178,7 +178,7 @@ handleMessage(msg):
|
||||
|
||||
### ProposalPOLMessage handler
|
||||
|
||||
```
|
||||
```go
|
||||
handleMessage(msg):
|
||||
if prs.Height != msg.Height or prs.ProposalPOLRound != msg.ProposalPOLRound then return
|
||||
prs.ProposalPOL = msg.ProposalPOL
|
||||
@@ -189,7 +189,7 @@ node against DOS attacks.
|
||||
|
||||
### BlockPartMessage handler
|
||||
|
||||
```
|
||||
```go
|
||||
handleMessage(msg):
|
||||
if prs.Height != msg.Height || prs.Round != msg.Round then return
|
||||
Record in prs that peer has block part msg.Part.Index
|
||||
@@ -198,7 +198,7 @@ handleMessage(msg):
|
||||
|
||||
### VoteMessage handler
|
||||
|
||||
```
|
||||
```go
|
||||
handleMessage(msg):
|
||||
Record in prs that a peer knows vote with index msg.vote.ValidatorIndex for particular height and round
|
||||
Send msg trough internal peerMsgQueue to ConsensusState service
|
||||
@@ -206,7 +206,7 @@ handleMessage(msg):
|
||||
|
||||
### VoteSetBitsMessage handler
|
||||
|
||||
```
|
||||
```go
|
||||
handleMessage(msg):
|
||||
Update prs for the bit-array of votes peer claims to have for the msg.BlockID
|
||||
```
|
||||
@@ -220,12 +220,12 @@ It is used to send the following messages to the peer: `BlockPartMessage`, `Prop
|
||||
`ProposalPOLMessage` on the DataChannel. The gossip data routine is based on the local RoundState (`rs`)
|
||||
and the known PeerRoundState (`prs`). The routine repeats forever the logic shown below:
|
||||
|
||||
```
|
||||
```go
|
||||
1a) if rs.ProposalBlockPartsHeader == prs.ProposalBlockPartsHeader and the peer does not have all the proposal parts then
|
||||
Part = pick a random proposal block part the peer does not have
|
||||
Send BlockPartMessage(rs.Height, rs.Round, Part) to the peer on the DataChannel
|
||||
if send returns true, record that the peer knows the corresponding block Part
|
||||
Continue
|
||||
Continue
|
||||
|
||||
1b) if (0 < prs.Height) and (prs.Height < rs.Height) then
|
||||
help peer catch up using gossipDataForCatchup function
|
||||
@@ -239,8 +239,8 @@ and the known PeerRoundState (`prs`). The routine repeats forever the logic show
|
||||
1d) if (rs.Proposal != nil and !prs.Proposal) then
|
||||
Send ProposalMessage(rs.Proposal) to the peer
|
||||
if send returns true, record that the peer knows Proposal
|
||||
if 0 <= rs.Proposal.POLRound then
|
||||
polRound = rs.Proposal.POLRound
|
||||
if 0 <= rs.Proposal.POLRound then
|
||||
polRound = rs.Proposal.POLRound
|
||||
prevotesBitArray = rs.Votes.Prevotes(polRound).BitArray()
|
||||
Send ProposalPOLMessage(rs.Height, polRound, prevotesBitArray)
|
||||
Continue
|
||||
@@ -253,16 +253,18 @@ and the known PeerRoundState (`prs`). The routine repeats forever the logic show
|
||||
This function is responsible for helping peer catch up if it is at the smaller height (prs.Height < rs.Height).
|
||||
The function executes the following logic:
|
||||
|
||||
```go
|
||||
if peer does not have all block parts for prs.ProposalBlockPart then
|
||||
blockMeta = Load Block Metadata for height prs.Height from blockStore
|
||||
if (!blockMeta.BlockID.PartsHeader == prs.ProposalBlockPartsHeader) then
|
||||
Sleep PeerGossipSleepDuration
|
||||
return
|
||||
return
|
||||
Part = pick a random proposal block part the peer does not have
|
||||
Send BlockPartMessage(prs.Height, prs.Round, Part) to the peer on the DataChannel
|
||||
if send returns true, record that the peer knows the corresponding block Part
|
||||
return
|
||||
else Sleep PeerGossipSleepDuration
|
||||
```
|
||||
|
||||
## Gossip Votes Routine
|
||||
|
||||
@@ -270,7 +272,7 @@ It is used to send the following message: `VoteMessage` on the VoteChannel.
|
||||
The gossip votes routine is based on the local RoundState (`rs`)
|
||||
and the known PeerRoundState (`prs`). The routine repeats forever the logic shown below:
|
||||
|
||||
```
|
||||
```go
|
||||
1a) if rs.Height == prs.Height then
|
||||
if prs.Step == RoundStepNewHeight then
|
||||
vote = random vote from rs.LastCommit the peer does not have
|
||||
@@ -284,7 +286,7 @@ and the known PeerRoundState (`prs`). The routine repeats forever the logic show
|
||||
if send returns true, continue
|
||||
|
||||
if prs.Step <= RoundStepPrecommit and prs.Round != -1 and prs.Round <= rs.Round then
|
||||
Precommits = rs.Votes.Precommits(prs.Round)
|
||||
Precommits = rs.Votes.Precommits(prs.Round)
|
||||
vote = random vote from Precommits the peer does not have
|
||||
Send VoteMessage(vote) to the peer
|
||||
if send returns true, continue
|
||||
@@ -315,7 +317,7 @@ It is used to send the following message: `VoteSetMaj23Message`. `VoteSetMaj23Me
|
||||
BlockID has seen +2/3 votes. This routine is based on the local RoundState (`rs`) and the known PeerRoundState
|
||||
(`prs`). The routine repeats forever the logic shown below.
|
||||
|
||||
```
|
||||
```go
|
||||
1a) if rs.Height == prs.Height then
|
||||
Prevotes = rs.Votes.Prevotes(prs.Round)
|
||||
if there is a ⅔ majority for some blockId in Prevotes then
|
||||
|
||||
@@ -16,7 +16,7 @@ explained in a forthcoming document.
|
||||
For efficiency reasons, validators in Tendermint consensus protocol do not agree directly on the
|
||||
block as the block size is big, i.e., they don't embed the block inside `Proposal` and
|
||||
`VoteMessage`. Instead, they reach agreement on the `BlockID` (see `BlockID` definition in
|
||||
[Blockchain](https://github.com/tendermint/spec/blob/master/spec/core/data_structures.md#blockid) section)
|
||||
[Blockchain](https://github.com/tendermint/spec/blob/master/spec/core/data_structures.md#blockid) section)
|
||||
that uniquely identifies each block. The block itself is
|
||||
disseminated to validator processes using peer-to-peer gossiping protocol. It starts by having a
|
||||
proposer first splitting a block into a number of block parts, that are then gossiped between
|
||||
@@ -49,7 +49,7 @@ type ProposalMessage struct {
|
||||
|
||||
Proposal contains height and round for which this proposal is made, BlockID as a unique identifier
|
||||
of proposed block, timestamp, and POLRound (a so-called Proof-of-Lock (POL) round) that is needed for
|
||||
termination of the consensus. If POLRound >= 0, then BlockID corresponds to the block that
|
||||
termination of the consensus. If POLRound >= 0, then BlockID corresponds to the block that
|
||||
is locked in POLRound. The message is signed by the validator private key.
|
||||
|
||||
```go
|
||||
@@ -66,8 +66,8 @@ type Proposal struct {
|
||||
## VoteMessage
|
||||
|
||||
VoteMessage is sent to vote for some block (or to inform others that a process does not vote in the
|
||||
current round). Vote is defined in the
|
||||
[Blockchain](https://github.com/tendermint/spec/blob/master/spec/core/data_structures.md#blockidd)
|
||||
current round). Vote is defined in the
|
||||
[Blockchain](https://github.com/tendermint/spec/blob/master/spec/core/data_structures.md#blockidd)
|
||||
section and contains validator's
|
||||
information (validator address and index), height and round for which the vote is sent, vote type,
|
||||
blockID if process vote for some block (`nil` otherwise) and a timestamp when the vote is sent. The
|
||||
@@ -110,9 +110,9 @@ type NewRoundStepMessage struct {
|
||||
|
||||
## NewValidBlockMessage
|
||||
|
||||
NewValidBlockMessage is sent when a validator observes a valid block B in some round r,
|
||||
NewValidBlockMessage is sent when a validator observes a valid block B in some round r,
|
||||
i.e., there is a Proposal for block B and 2/3+ prevotes for the block B in the round r.
|
||||
It contains height and round in which valid block is observed, block parts header that describes
|
||||
It contains height and round in which valid block is observed, block parts header that describes
|
||||
the valid block and is used to obtain all
|
||||
block parts, and a bit array of the block parts a process currently has, so its peers can know what
|
||||
parts it is missing so they can send them.
|
||||
@@ -121,7 +121,7 @@ In case the block is also committed, then IsCommit flag is set to true.
|
||||
```go
|
||||
type NewValidBlockMessage struct {
|
||||
Height int64
|
||||
Round int
|
||||
Round int
|
||||
BlockPartsHeader PartSetHeader
|
||||
BlockParts BitArray
|
||||
IsCommit bool
|
||||
|
||||
@@ -4,41 +4,46 @@ This document specifies the Proposer Selection Procedure that is used in Tenderm
|
||||
As Tendermint is “leader-based protocol”, the proposer selection is critical for its correct functioning.
|
||||
|
||||
At a given block height, the proposer selection algorithm runs with the same validator set at each round .
|
||||
Between heights, an updated validator set may be specified by the application as part of the ABCIResponses' EndBlock.
|
||||
Between heights, an updated validator set may be specified by the application as part of the ABCIResponses' EndBlock.
|
||||
|
||||
## Requirements for Proposer Selection
|
||||
|
||||
This sections covers the requirements with Rx being mandatory and Ox optional requirements.
|
||||
The following requirements must be met by the Proposer Selection procedure:
|
||||
|
||||
#### R1: Determinism
|
||||
### R1: Determinism
|
||||
|
||||
Given a validator set `V`, and two honest validators `p` and `q`, for each height `h` and each round `r` the following must hold:
|
||||
|
||||
`proposer_p(h,r) = proposer_q(h,r)`
|
||||
|
||||
where `proposer_p(h,r)` is the proposer returned by the Proposer Selection Procedure at process `p`, at height `h` and round `r`.
|
||||
|
||||
#### R2: Fairness
|
||||
### R2: Fairness
|
||||
|
||||
Given a validator set with total voting power P and a sequence S of elections. In any sub-sequence of S with length C*P, a validator v must be elected as proposer P/VP(v) times, i.e. with frequency:
|
||||
|
||||
f(v) ~ VP(v) / P
|
||||
|
||||
where C is a tolerance factor for validator set changes with following values:
|
||||
|
||||
- C == 1 if there are no validator set changes
|
||||
- C ~ k when there are validator changes
|
||||
- C ~ k when there are validator changes
|
||||
|
||||
*[this needs more work]*
|
||||
|
||||
### Basic Algorithm
|
||||
## Basic Algorithm
|
||||
|
||||
At its core, the proposer selection procedure uses a weighted round-robin algorithm.
|
||||
|
||||
A model that gives a good intuition on how/ why the selection algorithm works and it is fair is that of a priority queue. The validators move ahead in this queue according to their voting power (the higher the voting power the faster a validator moves towards the head of the queue). When the algorithm runs the following happens:
|
||||
|
||||
- all validators move "ahead" according to their powers: for each validator, increase the priority by the voting power
|
||||
- first in the queue becomes the proposer: select the validator with highest priority
|
||||
- first in the queue becomes the proposer: select the validator with highest priority
|
||||
- move the proposer back in the queue: decrease the proposer's priority by the total voting power
|
||||
|
||||
Notation:
|
||||
|
||||
- vset - the validator set
|
||||
- n - the number of validators
|
||||
- VP(i) - voting power of validator i
|
||||
@@ -49,7 +54,7 @@ Notation:
|
||||
|
||||
Simple view at the Selection Algorithm:
|
||||
|
||||
```
|
||||
```md
|
||||
def ProposerSelection (vset):
|
||||
|
||||
// compute priorities and elect proposer
|
||||
@@ -59,16 +64,16 @@ Simple view at the Selection Algorithm:
|
||||
A(prop) -= P
|
||||
```
|
||||
|
||||
### Stable Set
|
||||
## Stable Set
|
||||
|
||||
Consider the validator set:
|
||||
|
||||
Validator | p1| p2
|
||||
Validator | p1| p2
|
||||
----------|---|---
|
||||
VP | 1 | 3
|
||||
|
||||
Assuming no validator changes, the following table shows the proposer priority computation over a few runs. Four runs of the selection procedure are shown, starting with the 5th the same values are computed.
|
||||
Each row shows the priority queue and the process place in it. The proposer is the closest to the head, the rightmost validator. As priorities are updated, the validators move right in the queue. The proposer moves left as its priority is reduced after election.
|
||||
Each row shows the priority queue and the process place in it. The proposer is the closest to the head, the rightmost validator. As priorities are updated, the validators move right in the queue. The proposer moves left as its priority is reduced after election.
|
||||
|
||||
|Priority Run | -2| -1| 0 | 1| 2 | 3 | 4 | 5 | Alg step
|
||||
|--------------- |---|---|---- |---|---- |---|---|---|--------
|
||||
@@ -83,20 +88,23 @@ Each row shows the priority queue and the process place in it. The proposer is t
|
||||
| | | |p1,p2| | | | | |A(p2)-= P
|
||||
|
||||
It can be shown that:
|
||||
- At the end of each run k+1 the sum of the priorities is the same as at end of run k. If a new set's priorities are initialized to 0 then the sum of priorities will be 0 at each run while there are no changes.
|
||||
- The max distance between priorites is (n-1) * P. *[formal proof not finished]*
|
||||
|
||||
### Validator Set Changes
|
||||
- At the end of each run k+1 the sum of the priorities is the same as at end of run k. If a new set's priorities are initialized to 0 then the sum of priorities will be 0 at each run while there are no changes.
|
||||
- The max distance between priorites is (n-1) *P.*[formal proof not finished]*
|
||||
|
||||
## Validator Set Changes
|
||||
|
||||
Between proposer selection runs the validator set may change. Some changes have implications on the proposer election.
|
||||
|
||||
#### Voting Power Change
|
||||
### Voting Power Change
|
||||
|
||||
Consider again the earlier example and assume that the voting power of p1 is changed to 4:
|
||||
|
||||
Validator | p1| p2
|
||||
Validator | p1| p2
|
||||
----------|---| ---
|
||||
VP | 4 | 3
|
||||
|
||||
Let's also assume that before this change the proposer priorites were as shown in first row (last run). As it can be seen, the selection could run again, without changes, as before.
|
||||
Let's also assume that before this change the proposer priorites were as shown in first row (last run). As it can be seen, the selection could run again, without changes, as before.
|
||||
|
||||
|Priority Run| -2 | -1 | 0 | 1 | 2 | Comment
|
||||
|--------------| ---|--- |------|--- |--- |--------
|
||||
@@ -107,20 +115,22 @@ Let's also assume that before this change the proposer priorites were as shown i
|
||||
However, when a validator changes power from a high to a low value, some other validator remain far back in the queue for a long time. This scenario is considered again in the Proposer Priority Range section.
|
||||
|
||||
As before:
|
||||
|
||||
- At the end of each run k+1 the sum of the priorities is the same as at run k.
|
||||
- The max distance between priorites is (n-1) * P.
|
||||
|
||||
#### Validator Removal
|
||||
### Validator Removal
|
||||
|
||||
Consider a new example with set:
|
||||
|
||||
Validator | p1 | p2 | p3 |
|
||||
--------- |--- |--- |--- |
|
||||
VP | 1 | 2 | 3 |
|
||||
|
||||
Let's assume that after the last run the proposer priorities were as shown in first row with their sum being 0. After p2 is removed, at the end of next proposer selection run (penultimate row) the sum of priorities is -2 (minus the priority of the removed process).
|
||||
Let's assume that after the last run the proposer priorities were as shown in first row with their sum being 0. After p2 is removed, at the end of next proposer selection run (penultimate row) the sum of priorities is -2 (minus the priority of the removed process).
|
||||
|
||||
The procedure could continue without modifications. However, after a sufficiently large number of modifications in validator set, the priority values would migrate towards maximum or minimum allowed values causing truncations due to overflow detection.
|
||||
For this reason, the selection procedure adds another __new step__ that centers the current priority values such that the priority sum remains close to 0.
|
||||
For this reason, the selection procedure adds another __new step__ that centers the current priority values such that the priority sum remains close to 0.
|
||||
|
||||
|Priority Run |-3 | -2 | -1 | 0 | 1 | 2 | 4 |Comment
|
||||
|--------------- |--- | ---|--- |--- |--- |--- |---|--------
|
||||
@@ -132,6 +142,7 @@ For this reason, the selection procedure adds another __new step__ that centers
|
||||
|
||||
The modified selection algorithm is:
|
||||
|
||||
```md
|
||||
def ProposerSelection (vset):
|
||||
|
||||
// center priorities around zero
|
||||
@@ -144,18 +155,23 @@ The modified selection algorithm is:
|
||||
A(i) += VP(i)
|
||||
prop = max(A)
|
||||
A(prop) -= P
|
||||
```
|
||||
|
||||
Observations:
|
||||
|
||||
- The sum of priorities is now close to 0. Due to integer division the sum is an integer in (-n, n), where n is the number of validators.
|
||||
|
||||
#### New Validator
|
||||
### New Validator
|
||||
|
||||
When a new validator is added, same problem as the one described for removal appears, the sum of priorities in the new set is not zero. This is fixed with the centering step introduced above.
|
||||
|
||||
One other issue that needs to be addressed is the following. A validator V that has just been elected is moved to the end of the queue. If the validator set is large and/ or other validators have significantly higher power, V will have to wait many runs to be elected. If V removes and re-adds itself to the set, it would make a significant (albeit unfair) "jump" ahead in the queue.
|
||||
One other issue that needs to be addressed is the following. A validator V that has just been elected is moved to the end of the queue. If the validator set is large and/ or other validators have significantly higher power, V will have to wait many runs to be elected. If V removes and re-adds itself to the set, it would make a significant (albeit unfair) "jump" ahead in the queue.
|
||||
|
||||
In order to prevent this, when a new validator is added, its initial priority is set to:
|
||||
|
||||
```md
|
||||
A(V) = -1.125 * P
|
||||
```
|
||||
|
||||
where P is the total voting power of the set including V.
|
||||
|
||||
@@ -169,7 +185,9 @@ VP | 1 | 3 | 8
|
||||
|
||||
then p3 will start with proposer priority:
|
||||
|
||||
```md
|
||||
A(p3) = -1.125 * (1 + 3 + 8) ~ -13
|
||||
```
|
||||
|
||||
Note that since current computation uses integer division there is penalty loss when sum of the voting power is less than 8.
|
||||
|
||||
@@ -183,7 +201,8 @@ In the next run, p3 will still be ahead in the queue, elected as proposer and mo
|
||||
| | | | | | p3 | | | | p2| | p1|A(i)+=VP(i)
|
||||
| | | | p1 | | p3 | | | | p2| | |A(p1)-=P
|
||||
|
||||
### Proposer Priority Range
|
||||
## Proposer Priority Range
|
||||
|
||||
With the introduction of centering, some interesting cases occur. Low power validators that bind early in a set that includes high power validator(s) benefit from subsequent additions to the set. This is because these early validators run through more right shift operations during centering, operations that increase their priority.
|
||||
|
||||
As an example, consider the set where p2 is added after p1, with priority -1.125 * 80k = -90k. After the selection procedure runs once:
|
||||
@@ -198,83 +217,90 @@ Then execute the following steps:
|
||||
|
||||
1. Add a new validator p3:
|
||||
|
||||
Validator | p1 | p2 | p3
|
||||
----------|-----|--- |----
|
||||
VP | 80k | 10 | 10
|
||||
Validator | p1 | p2 | p3
|
||||
----------|-----|--- |----
|
||||
VP | 80k | 10 | 10
|
||||
|
||||
2. Run selection once. The notation '..p'/'p..' means very small deviations compared to column priority.
|
||||
|
||||
|Priority Run | -90k..| -60k | -45k | -15k| 0 | 45k | 75k | 155k | Comment
|
||||
|--------------|------ |----- |------- |---- |---|---- |----- |------- |---------
|
||||
| last run | p3 | | p2 | | | p1 | | | __added p3__
|
||||
| next run
|
||||
| *right_shift*| | p3 | | p2 | | | p1 | | A(i) -= avg,avg=-30k
|
||||
| | | ..p3| | ..p2| | | | p1 | A(i)+=VP(i)
|
||||
| | | ..p3| | ..p2| | | p1.. | | A(p1)-=P, P=80k+20
|
||||
|
||||
|Priority Run | -90k..| -60k | -45k | -15k| 0 | 45k | 75k | 155k | Comment
|
||||
|--------------|------ |----- |------- |---- |---|---- |----- |------- |---------
|
||||
| last run | p3 | | p2 | | | p1 | | | __added p3__
|
||||
| next run
|
||||
| *right_shift*| | p3 | | p2 | | | p1 | | A(i) -= avg,avg=-30k
|
||||
| | | ..p3| | ..p2| | | | p1 | A(i)+=VP(i)
|
||||
| | | ..p3| | ..p2| | | p1.. | | A(p1)-=P, P=80k+20
|
||||
|
||||
3. Remove p1 and run selection once:
|
||||
|
||||
Validator | p3 | p2 | Comment
|
||||
----------|----- |---- |--------
|
||||
VP | 10 | 10 |
|
||||
A |-60k |-15k |
|
||||
A |-22.5k|22.5k| __run selection__
|
||||
Validator | p3 | p2 | Comment
|
||||
----------|----- |---- |--------
|
||||
VP | 10 | 10 |
|
||||
A |-60k |-15k |
|
||||
A |-22.5k|22.5k| __run selection__
|
||||
|
||||
At this point, while the total voting power is 20, the distance between priorities is 45k. It will take 4500 runs for p3 to catch up with p2.
|
||||
|
||||
In order to prevent these types of scenarios, the selection algorithm performs scaling of priorities such that the difference between min and max values is smaller than two times the total voting power.
|
||||
In order to prevent these types of scenarios, the selection algorithm performs scaling of priorities such that the difference between min and max values is smaller than two times the total voting power.
|
||||
|
||||
The modified selection algorithm is:
|
||||
|
||||
```md
|
||||
def ProposerSelection (vset):
|
||||
|
||||
// scale the priority values
|
||||
diff = max(A)-min(A)
|
||||
threshold = 2 * P
|
||||
if diff > threshold:
|
||||
if diff > threshold:
|
||||
scale = diff/threshold
|
||||
for each validator i in vset:
|
||||
A(i) = A(i)/scale
|
||||
A(i) = A(i)/scale
|
||||
|
||||
// center priorities around zero
|
||||
avg = sum(A(i) for i in vset)/len(vset)
|
||||
for each validator i in vset:
|
||||
A(i) -= avg
|
||||
|
||||
|
||||
// compute priorities and elect proposer
|
||||
for each validator i in vset:
|
||||
A(i) += VP(i)
|
||||
prop = max(A)
|
||||
A(prop) -= P
|
||||
```
|
||||
|
||||
Observations:
|
||||
|
||||
- With this modification, the maximum distance between priorites becomes 2 * P.
|
||||
|
||||
Note also that even during steady state the priority range may increase beyond 2 * P. The scaling introduced here helps to keep the range bounded.
|
||||
Note also that even during steady state the priority range may increase beyond 2 * P. The scaling introduced here helps to keep the range bounded.
|
||||
|
||||
### Wrinkles
|
||||
## Wrinkles
|
||||
|
||||
### Validator Power Overflow Conditions
|
||||
|
||||
#### Validator Power Overflow Conditions
|
||||
The validator voting power is a positive number stored as an int64. When a validator is added the `1.125 * P` computation must not overflow. As a consequence the code handling validator updates (add and update) checks for overflow conditions making sure the total voting power is never larger than the largest int64 `MAX`, with the property that `1.125 * MAX` is still in the bounds of int64. Fatal error is return when overflow condition is detected.
|
||||
|
||||
#### Proposer Priority Overflow/ Underflow Handling
|
||||
### Proposer Priority Overflow/ Underflow Handling
|
||||
|
||||
The proposer priority is stored as an int64. The selection algorithm performs additions and subtractions to these values and in the case of overflows and underflows it limits the values to:
|
||||
|
||||
```go
|
||||
MaxInt64 = 1 << 63 - 1
|
||||
MinInt64 = -1 << 63
|
||||
```
|
||||
|
||||
### Requirement Fulfillment Claims
|
||||
__[R1]__
|
||||
## Requirement Fulfillment Claims
|
||||
|
||||
The proposer algorithm is deterministic giving consistent results across executions with same transactions and validator set modifications.
|
||||
__[R1]__
|
||||
|
||||
The proposer algorithm is deterministic giving consistent results across executions with same transactions and validator set modifications.
|
||||
[WIP - needs more detail]
|
||||
|
||||
__[R2]__
|
||||
__[R2]__
|
||||
|
||||
Given a set of processes with the total voting power P, during a sequence of elections of length P, the number of times any process is selected as proposer is equal to its voting power. The sequence of the P proposers then repeats. If we consider the validator set:
|
||||
|
||||
Validator | p1| p2
|
||||
Validator | p1| p2
|
||||
----------|---|---
|
||||
VP | 1 | 3
|
||||
|
||||
@@ -286,6 +312,8 @@ Assigning priorities to each validator based on the voting power and updating th
|
||||
|
||||
Intuitively, a process v jumps ahead in the queue at most (max(A) - min(A))/VP(v) times until it reaches the head and is elected. The frequency is then:
|
||||
|
||||
```md
|
||||
f(v) ~ VP(v)/(max(A)-min(A)) = 1/k * VP(v)/P
|
||||
```
|
||||
|
||||
For current implementation, this means v should be proposer at least VP(v) times out of k * P runs, with scaling factor k=2.
|
||||
|
||||
@@ -12,7 +12,7 @@ Environment: `TM_MEMPOOL_RECHECK=false`
|
||||
|
||||
Config:
|
||||
|
||||
```
|
||||
```toml
|
||||
[mempool]
|
||||
recheck = false
|
||||
```
|
||||
|
||||
@@ -35,7 +35,7 @@ What guarantees does it need from the ABCI app?
|
||||
|
||||
The implementation within this library also implements a tx cache.
|
||||
This is so that signatures don't have to be reverified if the tx has
|
||||
already been seen before.
|
||||
already been seen before.
|
||||
However, we only store valid txs in the cache, not invalid ones.
|
||||
This is because invalid txs could become good later.
|
||||
Txs that are included in a block aren't removed from the cache,
|
||||
|
||||
@@ -7,7 +7,7 @@ See [this issue](https://github.com/tendermint/tendermint/issues/1503)
|
||||
Mempool maintains a cache of the last 10000 transactions to prevent
|
||||
replaying old transactions (plus transactions coming from other
|
||||
validators, who are continually exchanging transactions). Read [Replay
|
||||
Protection](https://github.com/tendermint/tendermint/blob/master/docs/app-dev/app-development.md#replay-protection)
|
||||
Protection](https://github.com/tendermint/tendermint/blob/8cdaa7f515a9d366bbc9f0aff2a263a1a6392ead/docs/app-dev/app-development.md#replay-protection)
|
||||
for details.
|
||||
|
||||
Sending incorrectly encoded data or data exceeding `maxMsgSize` will result
|
||||
|
||||
@@ -70,13 +70,13 @@ when calculating a bucket.
|
||||
|
||||
When placing a peer into a new bucket:
|
||||
|
||||
```
|
||||
```md
|
||||
hash(key + sourcegroup + int64(hash(key + group + sourcegroup)) % bucket_per_group) % num_new_buckets
|
||||
```
|
||||
|
||||
When placing a peer into an old bucket:
|
||||
|
||||
```
|
||||
```md
|
||||
hash(key + group + int64(hash(key + addr)) % buckets_per_group) % num_old_buckets
|
||||
```
|
||||
|
||||
|
||||
@@ -5,14 +5,14 @@ and restoring state machine snapshots. For more information, see the [state sync
|
||||
|
||||
The state sync reactor has two main responsibilites:
|
||||
|
||||
* Serving state machine snapshots taken by the local ABCI application to new nodes joining the
|
||||
* Serving state machine snapshots taken by the local ABCI application to new nodes joining the
|
||||
network.
|
||||
|
||||
* Discovering existing snapshots and fetching snapshot chunks for an empty local application
|
||||
being bootstrapped.
|
||||
|
||||
The state sync process for bootstrapping a new node is described in detail in the section linked
|
||||
above. While technically part of the reactor (see `statesync/syncer.go` and related components),
|
||||
above. While technically part of the reactor (see `statesync/syncer.go` and related components),
|
||||
this document will only cover the P2P reactor component.
|
||||
|
||||
For details on the ABCI methods and data types, see the [ABCI documentation](../../abci/abci.md).
|
||||
@@ -26,16 +26,16 @@ available snapshots:
|
||||
type snapshotsRequestMessage struct{}
|
||||
```
|
||||
|
||||
The receiver will query the local ABCI application via `ListSnapshots`, and send a message
|
||||
The receiver will query the local ABCI application via `ListSnapshots`, and send a message
|
||||
containing snapshot metadata (limited to 4 MB) for each of the 10 most recent snapshots:
|
||||
|
||||
```go
|
||||
type snapshotsResponseMessage struct {
|
||||
Height uint64
|
||||
Format uint32
|
||||
Chunks uint32
|
||||
Hash []byte
|
||||
Metadata []byte
|
||||
Height uint64
|
||||
Format uint32
|
||||
Chunks uint32
|
||||
Hash []byte
|
||||
Metadata []byte
|
||||
}
|
||||
```
|
||||
|
||||
@@ -45,9 +45,9 @@ is accepted, the state syncer will request snapshot chunks from appropriate peer
|
||||
|
||||
```go
|
||||
type chunkRequestMessage struct {
|
||||
Height uint64
|
||||
Format uint32
|
||||
Index uint32
|
||||
Height uint64
|
||||
Format uint32
|
||||
Index uint32
|
||||
}
|
||||
```
|
||||
|
||||
@@ -56,16 +56,16 @@ and respond with it (limited to 16 MB):
|
||||
|
||||
```go
|
||||
type chunkResponseMessage struct {
|
||||
Height uint64
|
||||
Format uint32
|
||||
Index uint32
|
||||
Chunk []byte
|
||||
Missing bool
|
||||
Height uint64
|
||||
Format uint32
|
||||
Index uint32
|
||||
Chunk []byte
|
||||
Missing bool
|
||||
}
|
||||
```
|
||||
|
||||
Here, `Missing` is used to signify that the chunk was not found on the peer, since an empty
|
||||
chunk is a valid (although unlikely) response.
|
||||
chunk is a valid (although unlikely) response.
|
||||
|
||||
The returned chunk is given to the ABCI application via `ApplySnapshotChunk` until the snapshot
|
||||
is restored. If a chunk response is not returned within some time, it will be re-requested,
|
||||
@@ -73,5 +73,5 @@ possibly from a different peer.
|
||||
|
||||
The ABCI application is able to request peer bans and chunk refetching as part of the ABCI protocol.
|
||||
|
||||
If no state sync is in progress (i.e. during normal operation), any unsolicited response messages
|
||||
are discarded.
|
||||
If no state sync is in progress (i.e. during normal operation), any unsolicited response messages
|
||||
are discarded.
|
||||
|
||||
Reference in New Issue
Block a user