Line data Source code
1 : #ifndef HEADER_fd_src_discof_rotor_fd_rotor_tile_h
2 : #define HEADER_fd_src_discof_rotor_fd_rotor_tile_h
3 :
4 : #include "../../disco/tiles.h"
5 : #include "../../disco/shred/fd_shred_tile.h"
6 :
7 : /* Rotor OUTPUTS
8 :
9 : Rotor tile forwards FECs in replay order to replay tile with sig
10 : ROTOR_SIG_FEC_REPLAY.
11 :
12 : However since rotor is a direct drop in for repair tile, we have to make
13 : sure the sigs do not clobber the repair tile's sigs.
14 :
15 : See fd_repair_tile.h
16 : #define REPAIR_SIG_FEC_INVALID (2UL)
17 : #define REPAIR_SIG_FEC_LEADER (1UL)
18 : #define REPAIR_SIG_FEC (0UL)
19 :
20 : FECs are delivered in replay order. Blocks that were not repaired
21 : through verified (i.e., received through turbine) means are delivered
22 : with block_id = {0} until the last FEC in the block, in which the
23 : block_id is set to the computed DMR of the previously delivered FECs.
24 :
25 : Blocks that were repaired through verified means (i.e. using ag block
26 : id repair from a votor event) know the block_id immediately before
27 : the first FEC is delivered, so the block_id on the FECs of this slot
28 : is set to the correct value starting from fec 0. For these blocks,
29 : the verified bit is 1.
30 :
31 : There is a race in the case no equivocation occurred, but we
32 : suffered some network disconnection and are slow to complete the
33 : block (but we get a votor event for the block_id). Then we would be
34 : simultaneously completing the same block through turbine and ag
35 : block_id repair, and the turbine copy's slot-complete FEC would
36 : re-key its replay bank from {slot, 0} to a {slot, block_id} that the
37 : verified copy's bank already occupies.
38 :
39 : To prevent that, the chainer ABANDONS the turbine version of a slot
40 : the moment a votor-driven version of it is created while the turbine
41 : block_id is still unknown (see fd_chainer.h): the abandoned version
42 : keeps absorbing turbine shreds (they fill the FECs the verified
43 : version shares) but never delivers another FEC and never finalizes a
44 : block_id.
45 :
46 : Consider this case:
47 : Slot A (started receiving through turbine): received FEC 0, 1, and 5
48 : shreds of FEC 2. FEC 0 and 1 are delivered to replay with {verified=0, block_id=null}
49 :
50 : *blip*
51 :
52 : Get a notar fallback for slot A'. No equivocation occurred, but we
53 : can't tell, so we also start repairing A' using ag block id repair,
54 : and the turbine version of the slot is abandoned. Slot A' is
55 : immediately able to complete FEC 0 and 1 (the shreds are local), and
56 : they are re-delivered to replay with {verified=1, block_id=A'}.
57 : Remaining shreds of FEC 2 -- whether they arrive through turbine or
58 : ShredForBlockId repair -- fill the shared FEC, and FEC 2 is delivered
59 : once, under A', with {verified=1, block_id=A'}.
60 :
61 : The effect is that in time of network blips, replay ends up
62 : allocating up to two banks for the same slot/block: the turbine bank
63 : keyed {slot, 0} receives only a prefix of the block, never completes,
64 : never gets re-keyed (so it can never collide with the verified bank
65 : keyed {slot, block_id}), and is eventually evicted or pruned.
66 :
67 : INPUTS: REPLAY
68 :
69 : Rotor tile consumes from replay tile the sigs
70 : REPLAY_SIG_ROOT_ADVANCED and REPLAY_SIG_MISSING_FEC.
71 :
72 : Pruning:
73 :
74 : Rooting is not done off of votor rooting messages, but by replay root
75 : advance updates. Consider the following case:
76 :
77 : The cluster is having trouble rooting, and so we have a long chain of
78 : unfinalized slots. Banks begins to evict arbitrarily. It evicts slot
79 : N and begins executing down a different fork, but then soon after a
80 : finalization arrives for slot N.
81 :
82 : Replay tile updates its consensus root, but can't advance to it yet,
83 : because the bank for it has not been executed. Rotor has no
84 : eviction, and thus could root from the finalized message immediately.
85 : This is clearly a problem; replay needs the consensus root data
86 : re-delivered for execution, so rotor cannot immediately prune based
87 : on the finalized message.
88 :
89 : Instead replay already does its own bookkeeping. It has a highest
90 : known consensus root, a storage root that is the earliest slot data
91 : maintained, and a notified root that is the highest consensus root
92 : that replay verifies is live and can't be evicted.
93 :
94 : Rotor can safely assume anything below the notified root is no longer
95 : needed.
96 :
97 : Eviction:
98 :
99 : Replay can evict leaf banks at will, without rotor's knowledge. This
100 : means rotor may continue deliverying down a lineage that banks cannot
101 : immediately replay, since it has evicted an ancestor. When that
102 : occurs, replay should drain the rotor in-link dcache and send a
103 : REPLAY_SIG_MISSING_FEC to rotor. Rotor then sends it's next FEC set
104 : with the full lineage starting from the chainer root. It does this
105 : only for the next FEC set to deliver. If repeated evictions occur,
106 : rotor can expect repeated REPLAY_SIG_MISSING_FEC messages to arrive,
107 : and many redundant FECs to be delivered.
108 :
109 : We assume currently that chainer will not require eviction. The
110 : default size is bounded to the Agave cap on future certs it tracks.
111 : We can bound rotor even tighter once dynamic vote timeouts are
112 : implemented, and thus rotor should always have all the data replay
113 : needs.
114 :
115 : INPUTS: NET
116 :
117 : Alpenglow introduces three new repair types: ShredForBlockId,
118 : ParentAndFecCount, and FecRoot. ShredForBlockId response type is a
119 : regular shred, similar to the legacy repair types, so those responses
120 : get routed through the shred tile and are matched by nonce for
121 : verification.
122 :
123 : ParentAndFecCount and FecRoot response types are metadata, not
124 : regular shreds. Thus the net tile routes them directly to the rotor
125 : tile. The routing is done entirely by packet size, so rotor tile
126 : filters and validates aggressively. The responses are matches by
127 : nonce and verified before being ingested by the chainer. */
128 :
129 : /* keep in line with repair tile sigs */
130 0 : #define REPAIR_SIG_FEC (0UL)
131 0 : #define REPAIR_SIG_FEC_LEADER (1UL)
132 : #define REPAIR_SIG_FEC_INVALID (2UL)
133 : /* alpenglow type - replayable fec */
134 0 : #define ROTOR_SIG_FEC_REPLAY (3UL)
135 :
136 : struct fd_rotor_replay_fec {
137 : ulong slot;
138 : uint fec_set_idx;
139 : fd_hash_t mr;
140 :
141 : /* conditional fields */
142 :
143 : ulong parent_slot; /* only present if fec_set_idx is 0 or has
144 : parentUpdate. TBD, could also just have
145 : replay do parent reparsing */
146 : fd_hash_t parent_block_id;
147 :
148 : int slot_complete;
149 : int data_complete;
150 : int is_leader;
151 :
152 : /* known_id. This is not the same as slot_complete = 1. known_id
153 : should be set always to 1 if the block id was known from the
154 : start, i.e. these FECs were recovered through block_id repair of a
155 : votor event. known_id should be 0 for blocks that were received
156 : through turbine, until the last FEC is received, which should
157 : complete knowledge of the block_id.
158 :
159 : In other words, known_id is a keying instruction, not a statement about whether the block id is known:
160 : - known_id set: Replay keys the block by {slot, block_id} starting at FEC 0 and looks up its parent element by that key for every later FEC.
161 : - known_id clear: the block is a turbine version. Replay keys it by {slot, 0} until it processes the slot-complete FEC, then re-keys it to {slot, dmr}.
162 : This holds for every FEC of the block, including redelivered copies, so all FECs of one block always resolve to the same element.
163 :
164 : Redelivery from root never changes known_id. It only affects block_id */
165 : int known_id;
166 : fd_hash_t block_id; /* always populated if known_id is 1, or if slot_complete is 1. Otherwise could be populated on redelivery or as soon as the block_id is computed. */
167 : };
168 : typedef struct fd_rotor_replay_fec fd_rotor_replay_fec_t;
169 :
170 : #endif /* HEADER_fd_src_discof_rotor_fd_rotor_tile_h */
|