| 12345678910111213141516171819202122232425262728293031323334353637383940414243444546474849505152535455565758596061626364656667686970717273747576777879808182838485868788899091929394959697989910010110210310410510610710810911011111211311411511611711811912012112212312412512612712812913013113213313413513613713813914014114214314414514614714814915015115215315415515615715815916016116216316416516616716816917017117217317417517617717817918018118218318418518618718818919019119219319419519619719819920020120220320420520620720820921021121221321421521621721821922022122222322422522622722822923023123223323423523623723823924024124224324424524624724824925025125225325425525625725825926026126226326426526626726826927027127227327427527627727827928028128228328428528628728828929029129229329429529629729829930030130230330430530630730830931031131231331431531631731831932032132232332432532632732832933033133233333433533633733833934034134234334434534634734834935035135235335435535635735835936036136236336436536636736836937037137237337437537637737837938038138238338438538638738838939039139239339439539639739839940040140240340440540640740840941041141241341441541641741841942042142242342442542642742842943043143243343443543643743843944044144244344444544644744844945045145245345445545645745845946046146246346446546646746846947047147247347447547647747847948048148248348448548648748848949049149249349449549649749849950050150250350450550650750850951051151251351451551651751851952052152252352452552652752852953053153253353453553653753853954054154254354454554654754854955055155255355455555655755855956056156256356456556656756856957057157257357457557657757857958058158258358458558658758858959059159259359459559659759859960060160260360460560660760860961061161261361461561661761861962062162262362462562662762862963063163263363463563663763863964064164264364464564664764864965065165265365465565665765865966066166266366466566666766866967067167267367467567667767867968068168268368468568668768868969069169269369469569669769869970070170270370470570670770870971071171271371471571671771871972072172272372472572672772872973073173273373473573673773873974074174274374474574674774874975075175275375475575675775875976076176276376476576676776876977077177277377477577677777877978078178278378478578678778878979079179279379479579679779879980080180280380480580680780880981081181281381481581681781881982082182282382482582682782882983083183283383483583683783883984084184284384484584684784884985085185285385485585685785885986086186286386486586686786886987087187287387487587687787887988088188288388488588688788888989089189289389489589689789889990090190290390490590690790890991091191291391491591691791891992092192292392492592692792892993093193293393493593693793893994094194294394494594694794894995095195295395495595695795895996096196296396496596696796896997097197297397497597697797897998098198298398498598698798898999099199299399499599699799899910001001100210031004100510061007100810091010101110121013101410151016101710181019102010211022102310241025102610271028102910301031103210331034103510361037103810391040104110421043104410451046104710481049105010511052105310541055105610571058105910601061106210631064106510661067106810691070107110721073107410751076107710781079108010811082108310841085108610871088108910901091109210931094109510961097109810991100110111021103110411051106110711081109111011111112111311141115111611171118111911201121112211231124112511261127112811291130113111321133113411351136113711381139114011411142114311441145114611471148114911501151115211531154115511561157115811591160116111621163116411651166116711681169117011711172117311741175117611771178117911801181118211831184118511861187118811891190119111921193119411951196119711981199120012011202120312041205120612071208120912101211121212131214121512161217121812191220122112221223122412251226122712281229123012311232123312341235123612371238123912401241124212431244124512461247124812491250125112521253125412551256125712581259126012611262126312641265126612671268126912701271127212731274127512761277127812791280128112821283128412851286128712881289129012911292129312941295129612971298129913001301130213031304130513061307130813091310131113121313131413151316131713181319132013211322132313241325132613271328132913301331133213331334133513361337133813391340134113421343134413451346134713481349135013511352135313541355135613571358135913601361136213631364136513661367136813691370137113721373137413751376137713781379138013811382138313841385138613871388138913901391139213931394139513961397139813991400140114021403140414051406140714081409141014111412141314141415141614171418141914201421142214231424142514261427142814291430143114321433143414351436143714381439144014411442144314441445144614471448144914501451145214531454145514561457145814591460146114621463146414651466146714681469147014711472147314741475147614771478147914801481148214831484148514861487148814891490149114921493149414951496149714981499150015011502150315041505150615071508150915101511151215131514151515161517151815191520152115221523152415251526152715281529153015311532153315341535153615371538153915401541154215431544154515461547154815491550155115521553155415551556155715581559156015611562156315641565156615671568156915701571157215731574157515761577157815791580158115821583158415851586158715881589159015911592159315941595159615971598159916001601160216031604160516061607160816091610161116121613161416151616161716181619162016211622162316241625162616271628162916301631163216331634163516361637163816391640164116421643164416451646164716481649165016511652165316541655165616571658165916601661166216631664166516661667166816691670167116721673167416751676167716781679168016811682168316841685168616871688168916901691169216931694169516961697169816991700170117021703170417051706170717081709171017111712171317141715171617171718171917201721172217231724172517261727172817291730173117321733173417351736173717381739174017411742174317441745174617471748174917501751175217531754175517561757175817591760176117621763176417651766176717681769177017711772177317741775177617771778177917801781178217831784178517861787178817891790179117921793179417951796179717981799180018011802180318041805180618071808180918101811181218131814181518161817181818191820182118221823182418251826182718281829183018311832183318341835183618371838183918401841184218431844184518461847184818491850185118521853185418551856185718581859186018611862186318641865186618671868186918701871187218731874187518761877187818791880188118821883188418851886188718881889189018911892189318941895189618971898189919001901190219031904190519061907190819091910191119121913191419151916191719181919192019211922192319241925192619271928192919301931193219331934193519361937193819391940194119421943194419451946194719481949195019511952195319541955195619571958195919601961196219631964196519661967196819691970197119721973197419751976197719781979198019811982198319841985198619871988198919901991199219931994199519961997199819992000200120022003200420052006200720082009201020112012201320142015201620172018201920202021202220232024202520262027202820292030203120322033203420352036203720382039204020412042204320442045204620472048204920502051205220532054205520562057205820592060206120622063206420652066206720682069207020712072207320742075207620772078207920802081208220832084208520862087208820892090209120922093209420952096209720982099210021012102210321042105210621072108210921102111211221132114211521162117211821192120212121222123212421252126212721282129213021312132213321342135213621372138213921402141214221432144214521462147214821492150215121522153215421552156215721582159216021612162216321642165216621672168216921702171217221732174217521762177217821792180218121822183218421852186218721882189219021912192219321942195219621972198219922002201220222032204220522062207220822092210221122122213221422152216221722182219222022212222222322242225222622272228222922302231223222332234223522362237223822392240224122422243224422452246224722482249225022512252225322542255225622572258225922602261226222632264226522662267226822692270227122722273227422752276227722782279228022812282228322842285228622872288228922902291229222932294229522962297229822992300230123022303230423052306230723082309231023112312231323142315231623172318231923202321232223232324232523262327232823292330233123322333233423352336233723382339234023412342234323442345234623472348234923502351235223532354235523562357235823592360236123622363236423652366236723682369237023712372237323742375237623772378237923802381238223832384238523862387238823892390239123922393239423952396239723982399240024012402 |
- /* This file is part of DarkFi (https://dark.fi)
- *
- * Copyright (C) 2020-2026 Dyne.org foundation
- *
- * This program is free software: you can redistribute it and/or modify
- * it under the terms of the GNU Affero General Public License as
- * published by the Free Software Foundation, either version 3 of the
- * License, or (at your option) any later version.
- *
- * This program is distributed in the hope that it will be useful,
- * but WITHOUT ANY WARRANTY; without even the implied warranty of
- * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
- * GNU Affero General Public License for more details.
- *
- * You should have received a copy of the GNU Affero General Public License
- * along with this program. If not, see <https://www.gnu.org/licenses/>.
- */
- //! Multi-DAG Event Graph with bidirectional sync, RLN rate limiting,
- //! and periodic DAG rotation.
- use std::{
- collections::{BTreeMap, BTreeSet, HashMap, HashSet, VecDeque},
- path::PathBuf,
- str::FromStr,
- sync::{
- atomic::{AtomicBool, Ordering},
- Arc,
- },
- };
- use darkfi_sdk::{crypto::pasta_prelude::PrimeField, pasta::pallas};
- use darkfi_serial::{deserialize_async, deserialize_async_partial, serialize_async};
- use futures::{stream::FuturesUnordered, StreamExt};
- use sled_overlay::{sled, SledTreeOverlay};
- use smol::{
- lock::{OnceCell, RwLock},
- Executor,
- };
- use tracing::{error, info, warn};
- use url::Url;
- use crate::{
- net::{channel::Channel, P2pPtr},
- system::{msleep, Publisher, PublisherPtr, StoppableTask, StoppableTaskPtr, Subscription},
- Error, Result,
- };
- pub mod event;
- pub use event::{display_order, Event, Header};
- pub mod proto;
- use proto::{EventRep, EventReq, HeaderRep, HeaderReq, StaticPut, SyncDirection, TipRep, TipReq};
- pub mod rln;
- use rln::{IdentityState, RlnState, ZkKeys};
- pub mod util;
- use util::{
- generate_genesis, millis_until_next_rotation, next_hour_timestamp, next_rotation_timestamp,
- replayer_log,
- };
- pub mod deg;
- use deg::DegEvent;
- #[cfg(test)]
- mod tests;
- #[cfg(test)]
- mod tests_rln;
- #[cfg(test)]
- mod test_helpers;
- /// Number of parent references each event carries.
- pub const N_EVENT_PARENTS: usize = 5;
- /// Allowed timestamp drift in milliseconds.
- const EVENT_TIME_DRIFT: u64 = 60_000;
- /// The null event ID (32 zero bytes).
- pub const NULL_ID: blake3::Hash = blake3::Hash::from_bytes([0x00; blake3::OUT_LEN]);
- /// Array of null parents (used by genesis events).
- pub const NULL_PARENTS: [blake3::Hash; N_EVENT_PARENTS] = [NULL_ID; N_EVENT_PARENTS];
- /// Maximum number of static-DAG events `static_sync` will pull in
- /// one invocation. Defends against malicious deep-ancestry chains.
- const SYNC_MAX_STATIC_EVENTS: usize = 100_000;
- /// Runtime configuration for an Event Graph instance.
- #[derive(Clone, Debug)]
- pub struct EventGraphConfig {
- /// Epoch origin timestamp in millis.
- /// All rotation boundaries are computed as offsets from this point.
- /// Should be UTC midnight for clean hourly alignment.
- pub initial_genesis: u64,
- /// How often the DAG rotates, in hours. 0 = no rotation.
- pub hours_rotation: u64,
- /// Unique payload embedded in genesis events.
- /// Different protocols must use different values.
- pub genesis_contents: Vec<u8>,
- /// Maximum number of DAGs to keep in the rolling window.
- ///
- /// * `Some(n)` - keep n rotation periods.
- /// When the n+1 period is created, the oldest is permanently
- /// deleted from sled. This is the normal mode for end-user nodes.
- /// * `None` - never prune. Every DAG ever created is kept in sled
- /// and loaded at startup. This is archive mode for nodes that want
- /// complete history.
- ///
- /// With `hours_rotation = 1` and `max_dags = Some(24)`, events
- /// older than 24 hours are lost. With `hours_rotation = 6` and
- /// `max_dags = Some(24)`, the window is 6 days.
- pub max_dags: Option<usize>,
- }
- pub type EventGraphPtr = Arc<EventGraph>;
- /// Unreferenced tips grouped by layer.
- pub type LayerUTips = BTreeMap<u64, HashSet<blake3::Hash>>;
- /// Bidirectional timestamp -> event-ID index.
- #[derive(Clone, Debug, Default)]
- pub struct TimeIndex {
- index: BTreeMap<u64, Vec<blake3::Hash>>,
- count: usize,
- }
- impl TimeIndex {
- pub fn new() -> Self {
- Self::default()
- }
- pub async fn from_header_dag(tree: &sled::Tree) -> Self {
- let mut idx = Self::new();
- for item in tree.iter() {
- let (id, hdr) = item.unwrap();
- let id = blake3::Hash::from_bytes((&id as &[u8]).try_into().unwrap());
- let hdr: Header = deserialize_async(&hdr).await.unwrap();
- idx.insert(hdr.timestamp, id);
- }
- idx
- }
- pub fn insert(&mut self, ts: u64, id: blake3::Hash) {
- self.index.entry(ts).or_default().push(id);
- self.count += 1;
- }
- pub fn newest(&self, n: usize) -> Vec<blake3::Hash> {
- self.rev(u64::MAX, n)
- }
- pub fn oldest(&self, n: usize) -> Vec<blake3::Hash> {
- self.fwd(0, n)
- }
- pub fn before(&self, cursor: u64, n: usize) -> Vec<blake3::Hash> {
- self.rev(cursor.saturating_sub(1), n)
- }
- pub fn after(&self, cursor: u64, n: usize) -> Vec<blake3::Hash> {
- self.fwd(cursor.saturating_add(1), n)
- }
- fn rev(&self, start: u64, n: usize) -> Vec<blake3::Hash> {
- let mut out = Vec::with_capacity(n);
- for (_, ids) in self.index.range(..=start).rev() {
- for id in ids {
- out.push(*id);
- if out.len() >= n {
- return out
- }
- }
- }
- out
- }
- fn fwd(&self, start: u64, n: usize) -> Vec<blake3::Hash> {
- let mut out = Vec::with_capacity(n);
- for (_, ids) in self.index.range(start..) {
- for id in ids {
- out.push(*id);
- if out.len() >= n {
- return out
- }
- }
- }
- out
- }
- pub fn len(&self) -> usize {
- self.count
- }
- pub fn is_empty(&self) -> bool {
- self.count == 0
- }
- }
- /// All per-DAG state: trees, tips, and the timestamp index.
- pub struct DagSlot {
- pub header_tree: sled::Tree,
- pub main_tree: sled::Tree,
- pub tips: LayerUTips,
- pub time_index: TimeIndex,
- }
- /// Full-scan tip computation.
- /// Compute unreferenced tips - events that exist in the DAG but are
- /// not referenced as a parent by any other event - grouped by layer.
- pub(crate) async fn compute_unreferenced_tips(dag: &sled::Tree) -> LayerUTips {
- let mut candidates: HashMap<blake3::Hash, u64> = HashMap::new();
- let mut referenced: HashSet<blake3::Hash> = HashSet::new();
- for item in dag.iter() {
- let (id_bytes, val_bytes) = item.unwrap();
- let id = blake3::Hash::from_bytes((&id_bytes as &[u8]).try_into().unwrap());
- let ev: Event = deserialize_async(&val_bytes).await.unwrap();
- candidates.insert(id, ev.header.layer);
- for p in ev.header.parents.iter() {
- if *p != NULL_ID {
- referenced.insert(*p);
- }
- }
- }
- // Bucket the unreferenced candidates by their layer
- let mut map: LayerUTips = BTreeMap::new();
- for (id, layer) in candidates {
- if !referenced.contains(&id) {
- map.entry(layer).or_default().insert(id);
- }
- }
- map
- }
- /// Pick up to N_EVENT_PARENTS tips from the highest layers.
- fn select_parents_from_tips(tips: &LayerUTips) -> (u64, [blake3::Hash; N_EVENT_PARENTS]) {
- let mut parents = [NULL_ID; N_EVENT_PARENTS];
- let mut i = 0;
- 'outer: for (_, layer_tips) in tips.iter().rev() {
- for t in layer_tips {
- parents[i] = *t;
- i += 1;
- if i >= N_EVENT_PARENTS {
- break 'outer
- }
- }
- }
- (tips.last_key_value().unwrap().0 + 1, parents)
- }
- /// Storage layer for all rotating DAGs.
- pub struct DagStore {
- db: sled::Db,
- dags: BTreeMap<u64, DagSlot>,
- }
- impl DagStore {
- /// Create or open DAG slots.
- ///
- /// * **Bounded mode** (`max_dags = Some(n)`): create a rolling
- /// window of the most recent `n` DAGs. Old trees already in
- /// sled outside this window are left untouched (they're just
- /// not loaded into memory).
- /// * **Archive mode** (`max_dags = None`): discover *all*
- /// existing DAG trees in sled and load them, plus ensure the
- /// recent window exists. Nothing is ever dropped.
- pub async fn new(sled_db: sled::Db, config: &EventGraphConfig) -> Self {
- let mut dags = BTreeMap::new();
- if config.hours_rotation == 0 {
- let genesis = generate_genesis(config);
- dags.insert(genesis.header.timestamp, Self::create_slot(&sled_db, &genesis).await);
- return Self { db: sled_db, dags }
- }
- // Determine how many recent DAGs to create/ensure exist.
- let window = config.max_dags.unwrap_or(24);
- // In archive mode, first discover and load any existing DAG
- // trees that are already in sled from previous runs.
- //
- // A DAG is stored across two trees: `<timestamp>` for events
- // and `headers_<timestamp>` for headers. We walk every tree
- // name in sled and pick out the ones whose name is a valid u64
- // timestamp.
- if config.max_dags.is_none() {
- for name in sled_db.tree_names() {
- let name_str = String::from_utf8_lossy(&name);
- if let Ok(ts) = name_str.parse::<u64>() {
- // Reconstruct the genesis for this timestamp
- let hdr = Header {
- timestamp: ts,
- parents: NULL_PARENTS,
- layer: 0,
- content_hash: blake3::hash(&config.genesis_contents),
- };
- let genesis = Event { header: hdr, content: config.genesis_contents.clone() };
- let slot = Self::create_slot(&sled_db, &genesis).await;
- dags.insert(ts, slot);
- }
- }
- }
- // Ensure the recent window of DAGs exists.
- // Creates them if they're not already loaded from the discovery step.
- for i in 1..=window {
- let ts = next_hour_timestamp((i as i64) - (window as i64));
- if dags.contains_key(&ts) {
- // Already loaded from sled discovery
- continue
- }
- let hdr = Header {
- timestamp: ts,
- parents: NULL_PARENTS,
- layer: 0,
- content_hash: blake3::hash(&config.genesis_contents),
- };
- let genesis = Event { header: hdr, content: config.genesis_contents.clone() };
- dags.insert(ts, Self::create_slot(&sled_db, &genesis).await);
- }
- Self { db: sled_db, dags }
- }
- async fn create_slot(db: &sled::Db, genesis: &Event) -> DagSlot {
- let name = genesis.header.timestamp.to_string();
- let ht = db.open_tree(format!("headers_{name}")).unwrap();
- let mt = db.open_tree(&name).unwrap();
- for (tree, data) in
- [(&ht, serialize_async(&genesis.header).await), (&mt, serialize_async(genesis).await)]
- {
- if tree.is_empty() {
- let mut ov = SledTreeOverlay::new(tree);
- ov.insert(genesis.id().as_bytes(), &data).unwrap();
- if let Some(b) = ov.aggregate() {
- tree.apply_batch(b).unwrap();
- }
- }
- }
- DagSlot {
- tips: compute_unreferenced_tips(&mt).await,
- time_index: TimeIndex::from_header_dag(&ht).await,
- header_tree: ht,
- main_tree: mt,
- }
- }
- /// Add a new DAG on rotation. In bounded mode, drops the oldest DAG
- /// when the limit is reached. In archive mode, never drops.
- pub async fn add_dag(&mut self, genesis: &Event, max_dags: Option<usize>) {
- if let Some(limit) = max_dags {
- if self.dags.len() >= limit {
- let (_, old) = self.dags.pop_first().unwrap();
- self.db.drop_tree(old.header_tree.name()).unwrap();
- self.db.drop_tree(old.main_tree.name()).unwrap();
- }
- }
- let slot = Self::create_slot(&self.db, genesis).await;
- self.dags.insert(genesis.header.timestamp, slot);
- }
- pub fn get_slot(&self, ts: &u64) -> Option<&DagSlot> {
- self.dags.get(ts)
- }
- pub fn get_slot_mut(&mut self, ts: &u64) -> Option<&mut DagSlot> {
- self.dags.get_mut(ts)
- }
- pub fn get_header_tree(&self, dag_name: &str) -> sled::Tree {
- self.db.open_tree(format!("headers_{dag_name}")).unwrap()
- }
- pub fn dag_timestamps(&self) -> Vec<u64> {
- self.dags.keys().cloned().collect()
- }
- }
- enum PeerStatus {
- Free,
- Busy,
- Failed,
- }
- /// The main Event Graph instance.
- ///
- /// Manages a rolling window of DAGs (one per rotation period), a
- /// static DAG for long-lived state (RLN identities), and the P2P
- /// protocol for syncing with peers.
- ///
- /// # Sync model
- ///
- /// Headers are synced eagerly (complete DAG skeleton in seconds).
- /// Event content is fetched lazily in the direction the application
- /// needs.
- ///
- /// The [`TimeIndex`] in each [`DagSlot`] enables O(log n)
- /// bidirectional pagination that crosses DAG boundaries
- /// transparently - the caller sees a flat chronological stream.
- pub struct EventGraph {
- pub(crate) p2p: P2pPtr,
- pub(crate) dag_store: RwLock<DagStore>,
- /// Side-table mapping `event_id -> original RLN signal blob` for
- /// rotating-DAG events. Mirror of [`Self::static_dag_blobs`] but
- /// for the rotating DAGs.
- ///
- /// Populated by `handle_event_put` after successful RLN
- /// verification, and by `dag_insert_with_blobs` during sync when
- /// the serving peer included the blob in its `EventRep`. Read
- /// by `handle_event_req` to forward blobs to syncing peers.
- /// Pruned by `dag_prune` when the corresponding rotating DAG
- /// rolls out of the retention window.
- pub(crate) dag_blobs: sled::Tree,
- /// Historical SMT roots, in canonical apply order.
- ///
- /// Key: `(layer:u64_be, event_id:32) = 40 bytes`. Value:
- /// `(root:32, timestamp:u64_be:8) = 40 bytes`.
- ///
- /// Big-endian layer encoding makes lexicographic byte order
- /// match canonical apply order, so `Tree::range` iterates
- /// chronologically and `Tree::get_lt` / `get_gt` give cheap
- /// neighbor lookups (used to find the timestamp interval during
- /// which a given root was the live root).
- ///
- /// See [`Self::apply_rln_static_event`] for the canonical-order
- /// rationale and [`Self::is_root_valid_at`] for how this is
- /// consulted during signal verification.
- pub(crate) rln_historical_roots_ordered: sled::Tree,
- /// Reverse index: `root:32 -> (layer:u64_be, event_id:32) = 40 bytes`.
- ///
- /// Lets us answer "is this root historical?" with a single
- /// `Tree::get(root)`, then chase the returned key into
- /// `rln_historical_roots_ordered` to get the timestamp interval.
- pub(crate) rln_historical_roots_by_value: sled::Tree,
- pub(crate) static_dag: sled::Tree,
- /// Side-table mapping `event_id -> original RLN blob` for static
- /// events. Used by [`Self::static_sync`] to re-verify the ZK
- /// proof of historical events at sync time. Every static-DAG
- /// event MUST have a corresponding entry - `static_sync` rejects
- /// events whose blob isn't available rather than falling through.
- pub(crate) static_dag_blobs: sled::Tree,
- datastore: PathBuf,
- replay_mode: bool,
- pub(crate) broadcasted_ids: RwLock<HashSet<blake3::Hash>>,
- pub prune_task: OnceCell<StoppableTaskPtr>,
- pub event_pub: PublisherPtr<Event>,
- pub static_pub: PublisherPtr<Event>,
- pub current_genesis: RwLock<Event>,
- pub config: EventGraphConfig,
- pub synced: AtomicBool,
- pub deg_enabled: AtomicBool,
- deg_publisher: PublisherPtr<DegEvent>,
- pub sled_db: sled::Db,
- pub zk_keys: Arc<ZkKeys>,
- pub identity_state: RwLock<IdentityState>,
- pub rln_state: RwLock<RlnState>,
- /// App identifier mixed into the RLN external nullifier. Derived
- /// from `config.genesis_contents` so two deployments using the
- /// same circuit cannot collide on internal_nullifiers.
- rln_app_id: rln::RlnAppId,
- }
- impl EventGraph {
- /// Create a new Event Graph.
- pub async fn new(
- p2p: P2pPtr,
- sled_db: sled::Db,
- datastore: PathBuf,
- replay_mode: bool,
- config: EventGraphConfig,
- ex: Arc<Executor<'_>>,
- ) -> Result<EventGraphPtr> {
- let zk_keys = Arc::new(ZkKeys::build_and_load(&sled_db)?);
- Self::with_zk_keys(p2p, sled_db, datastore, replay_mode, config, zk_keys, ex).await
- }
- /// Same as [`Self::new`] but accepts a pre-built [`ZkKeys`].
- ///
- /// Production always wants `Self::new`, which builds keys once
- /// against its own sled DB. Tests use this variant to share a
- /// single [`Arc<ZkKeys>`] across many `EventGraph` instances -
- /// proving keys are large (hundreds of MB each) and copying
- /// them per-test would blow out RAM and `/dev/shm`.
- pub async fn with_zk_keys(
- p2p: P2pPtr,
- sled_db: sled::Db,
- datastore: PathBuf,
- replay_mode: bool,
- config: EventGraphConfig,
- zk_keys: Arc<ZkKeys>,
- ex: Arc<Executor<'_>>,
- ) -> Result<EventGraphPtr> {
- let identity_state = IdentityState::new(&sled_db)?;
- let rln_app_id = rln::RlnAppId::from_genesis(&config.genesis_contents);
- let current_genesis = generate_genesis(&config);
- let dag_store = DagStore::new(sled_db.clone(), &config).await;
- let static_dag = Self::static_new(&sled_db, &config).await?;
- let static_dag_blobs = sled_db.open_tree("static-dag-blobs")?;
- let dag_blobs = sled_db.open_tree("dag-blobs")?;
- // Historical-roots side-tables. See the design comment on
- // `EventGraph::apply_rln_static_event` for the full rationale.
- // In short: every static-DAG mutation produces a new SMT root,
- // and we need to recognize *any* historical root for sync-time
- // signal verification, not just the most recent N. The
- // `ordered` tree gives us canonical replay (and successor
- // lookup for the time-window check), the `by_value` tree
- // gives us O(log n) "is this root historical?" queries.
- let rln_historical_roots_ordered = sled_db.open_tree("rln-historical-roots-ordered")?;
- let rln_historical_roots_by_value = sled_db.open_tree("rln-historical-roots-by-value")?;
- // Check whether the current genesis event is already in the
- // store. If not, we need to prune (create a fresh slot).
- let dag_ts = current_genesis.header.timestamp;
- let need_prune = dag_store
- .get_slot(&dag_ts)
- .map(|s| !s.main_tree.contains_key(current_genesis.id().as_bytes()).unwrap_or(false))
- .unwrap_or(true);
- let self_ = Arc::new(Self {
- p2p,
- sled_db: sled_db.clone(),
- dag_store: RwLock::new(dag_store),
- static_dag,
- static_dag_blobs,
- dag_blobs,
- rln_historical_roots_ordered,
- rln_historical_roots_by_value,
- datastore,
- replay_mode,
- broadcasted_ids: RwLock::new(HashSet::new()),
- prune_task: OnceCell::new(),
- event_pub: Publisher::new(),
- static_pub: Publisher::new(),
- current_genesis: RwLock::new(current_genesis.clone()),
- config: config.clone(),
- synced: AtomicBool::new(false),
- deg_enabled: AtomicBool::new(false),
- deg_publisher: Publisher::new(),
- zk_keys,
- identity_state: RwLock::new(identity_state),
- rln_state: RwLock::new(RlnState::new()),
- rln_app_id,
- });
- if need_prune {
- info!(
- target: "event_graph::new",
- "[EVENTGRAPH] Pruning: current genesis not found",
- );
- self_.dag_prune(current_genesis).await?;
- }
- // Consistency check: if the static DAG has events but the
- // historical-roots tables are empty, rebuild them by
- // replaying the static DAG in canonical order. This handles
- // the case where the operator manually deleted the
- // historical-roots trees, or where this is the first startup
- // after upgrading from a version that didn't track them.
- //
- // Without this, signal verification would fail for any root
- // beyond the in-memory `recent_roots` window.
- self_.rebuild_historical_roots_if_needed().await?;
- if config.hours_rotation > 0 {
- let task = StoppableTask::new();
- let _ = self_.prune_task.set(task.clone()).await;
- task.clone().start(
- self_.clone().dag_prune_task(),
- |res| async move {
- if let Err(e) = res {
- if !matches!(e, Error::DetachedTaskStopped) {
- error!("Prune: {e}");
- }
- }
- },
- Error::DetachedTaskStopped,
- ex,
- );
- }
- Ok(self_)
- }
- /// Rebuild the historical-roots side-tables from the static DAG.
- ///
- /// Called once at startup. No-op if the historical-roots tables
- /// already match the static-DAG event count. Otherwise replays
- /// every static-DAG event in canonical `(layer, event_id)` order
- /// and re-records the post-mutation root for each one.
- ///
- /// **Side effect.** Resets the in-memory SMT to empty, then
- /// rebuilds it leaf-by-leaf in canonical order, so the SMT and
- /// the historical-roots tables come out consistent. The
- /// `rln-identity-leaves` tree (which `IdentityState::new`
- /// originally read) is implicitly re-derived; we don't read it
- /// during rebuild because we want to honor any slashes in the
- /// static DAG even if the leaves tree is stale.
- async fn rebuild_historical_roots_if_needed(self: &Arc<Self>) -> Result<()> {
- // Walk the static DAG once, computing both:
- // * static_count: total non-genesis events
- // * expected_leaves: registrations - slashes (the number
- // of identities that should currently be in the SMT)
- // We need the second one to detect a state where leaves and
- // historical-roots happen to share counts but the leaves
- // don't actually correspond to the static-DAG events. That
- // can happen across schema changes or when older code paths
- // wrote to leaves without going through `apply_rln_static_event`.
- let mut static_count: u64 = 0;
- let mut registrations: i64 = 0;
- let mut slashes: i64 = 0;
- for item in self.static_dag.iter() {
- let (_, val) = item?;
- let ev: Event = deserialize_async(&val).await?;
- if ev.header.parents == NULL_PARENTS {
- continue
- }
- static_count += 1;
- // Try to classify this event. We tolerate failed parses
- // here because the rebuild path is best-effort: if an
- // event's content is unparseable, we just don't count it
- // toward expected_leaves. The replay loop below skips
- // it for the same reason.
- if let Ok((node, _)) = deserialize_async_partial::<rln::RLNNode>(ev.content()).await {
- match node {
- rln::RLNNode::Registration(_) => registrations += 1,
- rln::RLNNode::Slashing(_) => slashes += 1,
- }
- }
- }
- let expected_leaves = (registrations - slashes).max(0) as usize;
- let recorded_count = self.rln_historical_roots_ordered.len() as u64;
- let actual_leaves = self.identity_state.read().await.leaves_count();
- let counts_consistent = recorded_count == static_count;
- let leaves_consistent = actual_leaves == expected_leaves;
- info!(
- target: "event_graph::new",
- "[EVENTGRAPH] RLN state audit: static_count={} recorded_count={} \
- actual_leaves={} expected_leaves={} consistent={}",
- static_count, recorded_count, actual_leaves, expected_leaves,
- counts_consistent && leaves_consistent,
- );
- if counts_consistent && leaves_consistent {
- // Already consistent across all three sources (static
- // DAG, historical-roots table, leaves tree).
- return Ok(())
- }
- info!(
- target: "event_graph::new",
- "[EVENTGRAPH] Rebuilding historical-roots: {} static events, \
- {} recorded roots, {} leaves (expected {})",
- static_count, recorded_count, actual_leaves, expected_leaves,
- );
- // Reset the historical-roots tables to a known-empty state.
- self.rln_historical_roots_ordered.clear()?;
- self.rln_historical_roots_by_value.clear()?;
- // Reset the in-memory SMT and the leaves tree so the replay
- // below builds it correctly from the canonical static-DAG
- // sequence (including any slashes).
- {
- let mut state = self.identity_state.write().await;
- state.clear_for_rebuild()?;
- }
- // Collect static-DAG events and sort canonically.
- let mut events: Vec<Event> = vec![];
- for item in self.static_dag.iter() {
- let (_, val) = item?;
- let ev: Event = deserialize_async(&val).await?;
- if ev.header.parents != NULL_PARENTS {
- events.push(ev);
- }
- }
- events.sort_by(|a, b| {
- a.header
- .layer
- .cmp(&b.header.layer)
- .then_with(|| a.id().as_bytes().cmp(b.id().as_bytes()))
- });
- // Replay each event through the canonical apply path.
- for ev in events {
- let rln_node: rln::RLNNode = match deserialize_async_partial(ev.content()).await {
- Ok((v, _)) => v,
- Err(_) => continue,
- };
- let _ = self.apply_rln_static_event(&ev, &rln_node).await?;
- }
- info!(
- target: "event_graph::new",
- "[EVENTGRAPH] Historical-roots rebuild complete",
- );
- Ok(())
- }
- /// After header sync, event content can be fetched lazily via
- /// [`fetch_page`] or peer [`RangeReq`] - the application pulls
- /// the events it actually wants to display or process, without
- /// downloading the entire content on every sync.
- pub async fn dag_sync_headers(&self, dag_ts: u64) -> Result<()> {
- self.sync_impl(dag_ts, false).await
- }
- /// Full sync: headers plus all event content currently in the DAG.
- ///
- /// Use this when the application wants the complete historical
- /// content (e.g. an archive node, or a node rebuilding local state
- /// from the full event stream).
- pub async fn dag_sync(&self, dag_ts: u64) -> Result<()> {
- self.sync_impl(dag_ts, true).await
- }
- async fn sync_impl(&self, dag_ts: u64, fetch_content: bool) -> Result<()> {
- let dag_name = dag_ts.to_string();
- let channels = self.p2p.hosts().peers();
- // We need at least one peer to ask
- if channels.is_empty() {
- return Err(Error::DagSyncFailed)
- }
- let timeout = self.p2p.settings().read().await.outbound_connect_timeout_max();
- // Parallel tip collection
- let mut futs = FuturesUnordered::new();
- for ch in channels.iter() {
- futs.push(request_tips(ch, dag_name.clone(), timeout));
- }
- let mut tips: HashMap<blake3::Hash, (u64, usize)> = HashMap::new();
- let mut responded = 0usize;
- while let Some(res) = futs.next().await {
- if let Ok(peer_tips) = res {
- responded += 1;
- for (layer, hashes) in &peer_tips {
- for h in hashes {
- tips.entry(*h).and_modify(|e| e.1 += 1).or_insert((*layer, 1));
- }
- }
- }
- }
- if tips.is_empty() {
- return Err(Error::DagSyncFailed)
- }
- // 2/3 quorum
- let threshold = (responded * 2).div_ceil(3);
- let accepted: HashSet<blake3::Hash> = tips
- .iter()
- .filter(|(h, (_, n))| **h != NULL_ID && *n >= threshold)
- .map(|(h, _)| *h)
- .collect();
- let store = self.dag_store.read().await;
- let slot = store.get_slot(&dag_ts).ok_or(Error::DagSyncFailed)?;
- let missing: HashSet<blake3::Hash> = accepted
- .iter()
- .filter(|h| !slot.main_tree.contains_key(h.as_bytes()).unwrap_or(true))
- .cloned()
- .collect();
- if missing.is_empty() {
- return Ok(())
- }
- let our_tips = slot.tips.clone();
- drop(store);
- // Parallel header sync
- let mut hfuts = FuturesUnordered::new();
- for ch in channels.iter() {
- hfuts.push(request_header(ch, dag_name.clone(), our_tips.clone(), timeout));
- }
- while let Some(res) = hfuts.next().await {
- if let Ok(hdrs) = res {
- self.header_dag_insert(hdrs, &dag_name).await?;
- }
- }
- if fetch_content {
- self.fetch_missing_events(dag_ts, &dag_name, timeout).await?;
- }
- Ok(())
- }
- async fn fetch_missing_events(&self, dag_ts: u64, dag_name: &str, timeout: u64) -> Result<()> {
- let store = self.dag_store.read().await;
- let slot = store.get_slot(&dag_ts).ok_or(Error::DagSyncFailed)?;
- let mut sorted = vec![];
- for item in slot.header_tree.iter() {
- let (hb, val) = item.unwrap();
- let hdr: Header = deserialize_async(&val).await.unwrap();
- if hdr.parents != NULL_PARENTS && !slot.main_tree.contains_key(hb)? {
- sorted.push(hdr);
- }
- }
- sorted.sort_by_key(|h| h.layer);
- drop(store);
- if sorted.is_empty() {
- return Ok(())
- }
- let batch = 20;
- let mut chunks: BTreeMap<usize, Vec<blake3::Hash>> = BTreeMap::new();
- for (i, c) in sorted.chunks(batch).enumerate() {
- chunks.insert(i, c.iter().map(|h| h.id()).collect());
- }
- let mut remaining: BTreeSet<usize> = chunks.keys().cloned().collect();
- let mut peer_st: HashMap<Url, PeerStatus> = HashMap::new();
- let mut count = 0;
- let mut fs = FuturesUnordered::new();
- // Each received chunk is (events, blobs) - blobs aligned
- // index-wise with events. Empty `Vec<u8>` entries mean
- // "this event has no blob from the serving peer".
- let mut received: BTreeMap<usize, (Vec<Event>, Vec<Vec<u8>>)> = BTreeMap::new();
- while count < sorted.len() {
- let mut free = vec![];
- let mut busy = 0;
- self.p2p.hosts().peers().iter().for_each(|ch| match peer_st.get(ch.address()) {
- Some(PeerStatus::Free) | None => {
- free.push(ch.clone());
- }
- Some(PeerStatus::Busy) => {
- busy += 1;
- }
- _ => {}
- });
- if free.is_empty() && busy == 0 {
- return Err(Error::DagSyncFailed)
- }
- let n = std::cmp::min(free.len(), remaining.len());
- let ids: Vec<usize> = remaining.iter().take(n).copied().collect();
- for (i, cid) in ids.iter().enumerate() {
- fs.push(request_event(free[i].clone(), chunks[cid].clone(), *cid, timeout));
- remaining.remove(cid);
- peer_st.insert(free[i].address().clone(), PeerStatus::Busy);
- }
- if let Some((evts, cid, ch)) = fs.next().await {
- if let Ok((e, blobs)) = evts {
- count += e.len();
- received.insert(cid, (e, blobs));
- peer_st.insert(ch.address().clone(), PeerStatus::Free);
- } else {
- remaining.insert(cid);
- peer_st.insert(ch.address().clone(), PeerStatus::Failed);
- }
- }
- }
- for (_, (events_chunk, blobs_chunk)) in received {
- // dag_insert_with_blobs handles RLN re-verification per
- // event when blobs are present, and falls through to the
- // trust-the-quorum path when they're not. See
- // dag_insert_with_blobs's docstring for the policy.
- self.dag_insert_with_blobs(&events_chunk, &blobs_chunk, dag_name).await?;
- }
- Ok(())
- }
- /// Sync the `count` most recent DAGs (full content).
- ///
- /// Iterates oldest-first so that later syncs build on earlier
- /// ones (parent events exist before children reference them).
- pub async fn sync_selected(&self, count: usize) -> Result<()> {
- let ts: Vec<u64> =
- self.dag_store.read().await.dag_timestamps().into_iter().rev().take(count).collect();
- for t in ts.into_iter().rev() {
- self.dag_sync(t).await?;
- }
- self.synced.store(true, Ordering::Release);
- Ok(())
- }
- /// Sync only headers for the `count` most recent DAGs.
- ///
- /// Fast variant - gives a full DAG skeleton without downloading
- /// event bodies. Pair with [`fetch_page`] to pull content on-demand.
- pub async fn sync_selected_headers(&self, count: usize) -> Result<()> {
- let ts: Vec<u64> =
- self.dag_store.read().await.dag_timestamps().into_iter().rev().take(count).collect();
- for t in ts.into_iter().rev() {
- self.dag_sync_headers(t).await?;
- }
- self.synced.store(true, Ordering::Release);
- Ok(())
- }
- /// Sync the static DAG from peers.
- ///
- /// The static DAG holds RLN identity events (registrations and
- /// slashes). It is *persistent* across rotation windows - unlike
- /// rotating DAGs, events are never pruned - and has no separate
- /// `header_tree`, so it uses a different sync strategy:
- ///
- /// 1. Ask every peer for their `"static-dag"` tips.
- /// 2. Take the tips that reach a 2/3 quorum.
- /// 3. BFS-fetch the events and their ancestors directly via
- /// `EventReq` until the entire reachable subgraph is local.
- ///
- /// Peers serve static-DAG event requests even when the IDs are
- /// not in their `broadcasted_ids` set (see the relaxation in
- /// `handle_event_req`), because static-DAG state is public
- /// consensus information. Registration-event proof verification,
- /// duplicate detection, and commitment-tree updates are all done
- /// through the normal `StaticPut` ingestion path
- /// (`handle_static_put`) - but `static_sync` uses direct-insert
- /// via [`Self::static_insert`] plus on-the-fly identity-state
- /// application, because we're catching up rather than processing
- /// broadcasts.
- ///
- /// Note: for security, this method ONLY applies events whose
- /// blob/RLN verification passes. We do not trust peers blindly
- /// on historical state - proofs are re-verified locally for
- /// every single event before its effect is merged into the
- /// identity tree. This is the same discipline `handle_static_put`
- /// uses; see [`Self::rln_verify_static_event`].
- pub async fn static_sync(&self) -> Result<()> {
- static DAG_NAME: &str = "static-dag";
- let channels = self.p2p.hosts().peers();
- if channels.is_empty() {
- return Err(Error::DagSyncFailed)
- }
- let timeout = self.p2p.settings().read().await.outbound_connect_timeout_max();
- // Step 1: gather tips from every peer in parallel.
- let mut tip_futs = FuturesUnordered::new();
- for ch in channels.iter() {
- tip_futs.push(request_tips(ch, DAG_NAME.to_string(), timeout));
- }
- let mut tip_counts: HashMap<blake3::Hash, usize> = HashMap::new();
- let mut responded = 0usize;
- while let Some(res) = tip_futs.next().await {
- if let Ok(peer_tips) = res {
- responded += 1;
- for hashes in peer_tips.values() {
- for h in hashes {
- *tip_counts.entry(*h).or_insert(0) += 1;
- }
- }
- }
- }
- // If no peer answered we have nothing to do. An empty
- // network-side static DAG is a valid state (brand new app
- // deployment), so we return Ok rather than error.
- if responded == 0 {
- info!(
- target: "event_graph::static_sync",
- "[STATIC_SYNC] no peer responded to TipReq; nothing to sync"
- );
- return Ok(())
- }
- // Step 2: take tips at 2/3 quorum. This matches the
- // threshold used in `sync_impl`.
- let threshold = (responded * 2).div_ceil(3);
- let total_distinct_tips = tip_counts.len();
- let tip_ids: HashSet<blake3::Hash> = tip_counts
- .into_iter()
- .filter(|(h, n)| *h != NULL_ID && *n >= threshold)
- .map(|(h, _)| h)
- .collect();
- // What's already local?
- let mut known: HashSet<blake3::Hash> = HashSet::new();
- for item in self.static_dag.iter() {
- let (k, _) = item?;
- if let Ok(bytes) = <[u8; 32]>::try_from(&k as &[u8]) {
- known.insert(blake3::Hash::from_bytes(bytes));
- }
- }
- info!(
- target: "event_graph::static_sync",
- "[STATIC_SYNC] peers_responded={} threshold={} distinct_tips_seen={} \
- tip_ids_quorum={} known_local={}",
- responded, threshold, total_distinct_tips, tip_ids.len(), known.len(),
- );
- // Step 3: BFS from the quorum tips, fetching events we
- // don't have. Any event we pull in may reference ancestors
- // we ALSO don't have; enqueue them and keep going until the
- // frontier is empty.
- //
- // Bounded at SYNC_MAX_STATIC_EVENTS (defined at module level)
- // to defend against a malicious peer who serves a fabricated
- // deep-ancestry chain. In practice static DAGs are small (one
- // event per registration / slash), so this bound is
- // comfortably above any real deployment's size.
- let mut want: HashSet<blake3::Hash> = tip_ids.difference(&known).copied().collect();
- // Events fetched during BFS, paired with their blobs (empty
- // Vec if the peer didn't have the blob - see EventRep
- // docstring). Index alignment is preserved through the
- // entire pipeline up to the apply loop.
- let mut fetched: Vec<(Event, Vec<u8>)> = vec![];
- while !want.is_empty() {
- if fetched.len() >= SYNC_MAX_STATIC_EVENTS {
- error!(
- target: "event_graph::static_sync",
- "[STATIC_SYNC] reached {} event cap; aborting",
- SYNC_MAX_STATIC_EVENTS,
- );
- return Err(Error::DagSyncFailed)
- }
- let batch: Vec<blake3::Hash> = want.iter().copied().collect();
- want.clear();
- // Race the batch against all peers; first to respond
- // with valid events wins. A peer that returns events we
- // didn't ask for is striked via its protocol handler,
- // not here - this is a best-effort pull.
- let mut req_futs = FuturesUnordered::new();
- for (i, ch) in channels.iter().enumerate() {
- req_futs.push(request_event(ch.clone(), batch.clone(), i, timeout));
- }
- let mut got_any = false;
- while let Some((res, _, _)) = req_futs.next().await {
- let Ok((evs, blobs)) = res else { continue };
- if evs.is_empty() {
- continue
- }
- got_any = true;
- for (i, ev) in evs.into_iter().enumerate() {
- let eid = ev.id();
- if !batch.contains(&eid) {
- // Peer sent something we didn't ask for;
- // ignore the rest of this reply.
- break
- }
- if known.insert(eid) {
- // New parents to chase next round.
- for p in ev.header.parents.iter() {
- if *p != NULL_ID && !known.contains(p) {
- want.insert(*p);
- }
- }
- // Pair the event with its blob (or empty if
- // the peer didn't supply one - that's not an
- // error, see EventRep doc and the fall-through
- // in the apply loop below).
- let blob = blobs.get(i).cloned().unwrap_or_default();
- fetched.push((ev, blob));
- }
- }
- break
- }
- if !got_any {
- // Nobody responded usefully. Give up so we don't
- // loop forever on an unreachable ancestor.
- error!(
- target: "event_graph::static_sync",
- "[STATIC_SYNC] no peer served requested events; aborting",
- );
- return Err(Error::DagSyncFailed)
- }
- }
- // Step 4: canonical-order the fetched events so all nodes
- // produce the same intermediate SMT roots. Primary key:
- // layer (matches DAG topology). Secondary key: event_id
- // (32-byte hash, lexicographic byte order is total). Without
- // the tie-breaker, two events at the same layer could be
- // applied in different orders on different nodes, producing
- // different intermediate roots and breaking sync-time signal
- // verification. See the design comment on
- // `apply_rln_static_event` for the full rationale.
- fetched.sort_by(|(a, _), (b, _)| {
- a.header
- .layer
- .cmp(&b.header.layer)
- .then_with(|| a.id().as_bytes().cmp(b.id().as_bytes()))
- });
- // Track the apply-loop outcome for the summary log.
- let mut applied = 0usize;
- let mut already_present = 0usize;
- let mut blob_missing = 0usize;
- let mut rejected = 0usize;
- let mut structural_invalid = 0usize;
- let mut content_unparseable = 0usize;
- let total_to_consider = fetched.len();
- for (ev, blob) in fetched {
- // Skip if someone else inserted it concurrently.
- if self.static_dag.contains_key(ev.id().as_bytes())? {
- already_present += 1;
- continue
- }
- // Structural validation always runs. Static-DAG events
- // are persistent and may be far older than the 60s drift
- // window allowed by `validate_new`; use the static
- // sibling that omits the freshness check while keeping
- // the structural ones.
- if !ev.validate_new_static() {
- structural_invalid += 1;
- continue
- }
- let rln_node: rln::RLNNode = match deserialize_async_partial(ev.content()).await {
- Ok((v, _)) => v,
- Err(_) => {
- content_unparseable += 1;
- continue
- }
- };
- // RLN verification is mandatory. A non-genesis static
- // event without a blob during sync is treated as
- // misbehavior: either the serving peer is buggy or
- // adversarial, or the originator never persisted the blob
- // (which itself is a protocol violation). Skip with a
- // loud log - we don't strike here because static_sync
- // doesn't have a single peer to attribute the failure
- // to (the quorum collected blobs from multiple peers).
- if blob.is_empty() {
- blob_missing += 1;
- error!(
- target: "event_graph::static_sync",
- "[STATIC_SYNC] no blob available for static event {}; skipping. \
- Every static-DAG event must carry an RLN blob.",
- ev.id(),
- );
- continue
- }
- let outcome = self.rln_verify_static_event(&rln_node, &blob, ev.header.timestamp).await;
- match outcome {
- rln::StaticEventCheck::AcceptedRegistration(_) |
- rln::StaticEventCheck::AcceptedSlash(_) => {
- // apply_rln_static_event handles both Registration
- // and Slashing branches and also records the
- // post-mutation root in the historical-roots
- // side-tables.
- let _ = self.apply_rln_static_event(&ev, &rln_node).await;
- self.static_blob_store(&ev.id(), &blob)?;
- self.static_insert(&ev).await?;
- applied += 1;
- }
- rln::StaticEventCheck::Rejected | rln::StaticEventCheck::Malicious => {
- // A historical event whose blob fails
- // re-verification despite being held by the 2/3
- // quorum is a serious finding - either the blob
- // was tampered with, the quorum was compromised,
- // or our verifying keys diverged. Log loudly and
- // skip.
- rejected += 1;
- error!(
- target: "event_graph::static_sync",
- "[STATIC_SYNC] historical blob FAILED re-verification for event {}: {:?}; \
- skipping event despite quorum inclusion",
- ev.id(),
- outcome,
- );
- }
- }
- }
- info!(
- target: "event_graph::static_sync",
- "[STATIC_SYNC] complete: fetched={} applied={} already_present={} \
- blob_missing={} verification_rejected={} structural_invalid={} \
- unparseable={}",
- total_to_consider, applied, already_present, blob_missing, rejected,
- structural_invalid, content_unparseable,
- );
- Ok(())
- }
- /// Fetch a page of events, crossing DAG boundaries transparently.
- pub async fn fetch_page(
- &self,
- cursor_ts: u64,
- dir: SyncDirection,
- limit: usize,
- ) -> Result<Vec<Event>> {
- let mut out = vec![];
- let store = self.dag_store.read().await;
- let slots: Vec<_> = match dir {
- SyncDirection::Forward => store.dags.iter().collect(),
- SyncDirection::Backward => store.dags.iter().rev().collect(),
- };
- for (_, slot) in slots {
- if out.len() >= limit {
- break
- }
- let rem = limit - out.len();
- let ids = match dir {
- SyncDirection::Forward => slot.time_index.after(cursor_ts, rem),
- SyncDirection::Backward => slot.time_index.before(cursor_ts, rem),
- };
- for id in ids {
- if let Some(bytes) = slot.main_tree.get(id.as_bytes())? {
- out.push(deserialize_async(&bytes).await?);
- }
- }
- }
- out.truncate(limit);
- Ok(out)
- }
- async fn dag_prune(&self, genesis: Event) -> Result<()> {
- let mut bcast = self.broadcasted_ids.write().await;
- let mut cur = self.current_genesis.write().await;
- // Before the DAG store evicts the oldest DAG (which would
- // drop its main_tree), enumerate the about-to-be-dropped
- // event IDs so we can remove their blobs from `dag_blobs`.
- // Without this, blob entries would orphan and accumulate
- // forever - the side-table is not bounded by the rotation
- // window on its own.
- if let Some(limit) = self.config.max_dags {
- let store = self.dag_store.read().await;
- if store.dags.len() >= limit {
- if let Some((_, oldest)) = store.dags.iter().next() {
- for item in oldest.main_tree.iter() {
- let (eid, _) = match item {
- Ok(v) => v,
- Err(_) => continue,
- };
- let _ = self.dag_blobs.remove(&eid);
- }
- }
- }
- }
- self.dag_store.write().await.add_dag(&genesis, self.config.max_dags).await;
- *cur = genesis;
- *bcast = HashSet::new();
- Ok(())
- }
- async fn dag_prune_task(self: Arc<Self>) -> Result<()> {
- loop {
- let next =
- next_rotation_timestamp(self.config.initial_genesis, self.config.hours_rotation);
- let hdr = Header {
- timestamp: next,
- parents: NULL_PARENTS,
- layer: 0,
- content_hash: blake3::hash(&self.config.genesis_contents),
- };
- let genesis = Event { header: hdr, content: self.config.genesis_contents.clone() };
- msleep(millis_until_next_rotation(next)).await;
- self.dag_prune(genesis).await?;
- }
- }
- /// Insert events into a rotating DAG **without RLN verification**.
- ///
- /// This is the post-verification entry point for callers that
- /// have already verified the proof separately. Two legitimate
- /// callers in production:
- ///
- /// * `handle_event_put` - already ran `rln_verify_signal` and
- /// recorded the share. Calling `dag_insert_with_blobs` would
- /// trigger the duplicate-share rejection.
- /// * The IRC client's own outbound flow - same shape.
- pub async fn dag_insert(&self, events: &[Event], dag_name: &str) -> Result<Vec<blake3::Hash>> {
- // Implementation just runs the structural-insert path;
- // dag_insert_with_blobs reaches the same shared inner code
- // when called with a `skip_verify=true` shortcut, which is
- // what an empty `blobs` slice now means after the strictness
- // tightening below - but only via this private wrapper.
- self.dag_insert_inner(events, &[], /* require_blobs */ false, dag_name).await
- }
- /// Insert events into a rotating DAG, with mandatory RLN
- /// verification.
- ///
- /// `blobs` is index-aligned with `events`. Every non-genesis
- /// event MUST have a non-empty `blobs[i]`; events that don't
- /// (whether `blobs` is empty, shorter, or has an empty entry
- /// at position `i`) are rejected with a loud log. This is the
- /// strict policy required for sync paths - a peer that serves
- /// an event without its blob is buggy or adversarial.
- ///
- /// On `Slashable`, this method does NOT broadcast a slash -
- /// that's the protocol layer's job (see
- /// `proto::handle_event_put::verify_rln_signal`). Sync-time
- /// detection of a slashable conflict simply skips the event.
- /// We don't want a node coming online to flood the network
- /// with stale slash broadcasts.
- pub async fn dag_insert_with_blobs(
- &self,
- events: &[Event],
- blobs: &[Vec<u8>],
- dag_name: &str,
- ) -> Result<Vec<blake3::Hash>> {
- self.dag_insert_inner(events, blobs, /* require_blobs */ true, dag_name).await
- }
- /// Inner implementation shared by both insert paths. The
- /// `require_blobs` flag selects strict (sync) vs. lenient
- /// (post-verified) semantics.
- async fn dag_insert_inner(
- &self,
- events: &[Event],
- blobs: &[Vec<u8>],
- require_blobs: bool,
- dag_name: &str,
- ) -> Result<Vec<blake3::Hash>> {
- if events.is_empty() {
- return Ok(vec![])
- }
- // Pre-flight RLN verification. Done BEFORE acquiring the
- // DAG-store write lock so a slow proof verification doesn't
- // hold up other inserts.
- //
- // Events we already have are skipped without verification.
- // This matters because `rln_verify_signal` records the share
- // on `Accepted`, and re-running it for an already-seen event
- // would trip its duplicate-share check (returning `Rejected`)
- // - which would be incorrect: the event is legitimate, we
- // just already know about it.
- let dag_ts = u64::from_str(dag_name)?;
- let already_have: Vec<bool> = {
- let store = self.dag_store.read().await;
- let slot = store.get_slot(&dag_ts);
- events
- .iter()
- .map(|ev| match slot {
- Some(s) => s.main_tree.contains_key(ev.id().as_bytes()).unwrap_or(false),
- None => false,
- })
- .collect()
- };
- let mut accepted: Vec<usize> = Vec::with_capacity(events.len());
- for (i, ev) in events.iter().enumerate() {
- // Already-known events go through structurally (the
- // downstream `contains_key` check will skip them) but
- // skip the RLN verifier to avoid double-recording the
- // share for the same (epoch, internal_nullifier, x, y)
- // tuple.
- if already_have[i] {
- accepted.push(i);
- continue
- }
- // Genesis-shaped events have no blob and no proof -
- // they're consensus inputs, not user signals.
- if ev.header.parents == NULL_PARENTS {
- accepted.push(i);
- continue
- }
- let blob = blobs.get(i).cloned().unwrap_or_default();
- if blob.is_empty() {
- if require_blobs {
- error!(
- target: "event_graph::dag_insert",
- "[DAG_INSERT] sync event {} arrived without an RLN blob; rejecting. \
- Every non-genesis rotating-DAG event must carry a blob.",
- ev.id(),
- );
- continue
- }
- // Lenient path: caller pre-verified. Accept the
- // event structurally without running the RLN
- // verifier on it.
- accepted.push(i);
- continue
- }
- match self.rln_verify_signal(ev, &blob).await {
- rln::SignalCheck::Accepted => accepted.push(i),
- rln::SignalCheck::Rejected => {
- error!(
- target: "event_graph::dag_insert",
- "[DAG_INSERT] sync event {} failed RLN re-verification; skipping",
- ev.id(),
- );
- }
- rln::SignalCheck::Slashable(_) => {
- // The conflicting share is recorded inside
- // `rln_verify_signal` ONLY on `Accepted`. On
- // `Slashable` it returns the conflicting shares
- // *without* mutating metadata, so we don't
- // double-record. We don't broadcast a slash
- // here - that's the live broadcast handler's
- // job. We just skip the event.
- error!(
- target: "event_graph::dag_insert",
- "[DAG_INSERT] sync event {} is slashable (slot reuse); skipping",
- ev.id(),
- );
- }
- }
- }
- let mut bcast = self.broadcasted_ids.write().await;
- let mut store = self.dag_store.write().await;
- let slot = store.get_slot_mut(&dag_ts).ok_or(Error::DagSyncFailed)?;
- let mut ids = Vec::with_capacity(accepted.len());
- let mut overlay = SledTreeOverlay::new(&slot.main_tree);
- for &i in &accepted {
- let ev = &events[i];
- let eid = ev.id();
- if ev.header.parents == NULL_PARENTS {
- continue
- }
- if slot.main_tree.contains_key(eid.as_bytes())? {
- continue
- }
- if !slot.header_tree.contains_key(eid.as_bytes())? {
- continue
- }
- if !ev.dag_validate(&slot.header_tree, &self.config).await? {
- return Err(Error::EventIsInvalid)
- }
- let se = serialize_async(ev).await;
- overlay.insert(eid.as_bytes(), &se)?;
- if self.replay_mode {
- replayer_log(&self.datastore, "insert".into(), se)?;
- }
- // Persist the blob alongside the event for future
- // sync-time re-verification by other late-joiners.
- if let Some(blob) = blobs.get(i) {
- if !blob.is_empty() {
- let _ = self.dag_blob_store(&eid, blob);
- }
- }
- ids.push(eid);
- }
- if let Some(b) = overlay.aggregate() {
- slot.main_tree.apply_batch(b).unwrap();
- } else {
- return Ok(vec![])
- }
- for &i in &accepted {
- let ev = &events[i];
- let eid = ev.id();
- if ev.header.parents == NULL_PARENTS {
- continue
- }
- for pid in ev.header.parents.iter() {
- if *pid != NULL_ID {
- for (layer, tips) in slot.tips.iter_mut() {
- if *layer < ev.header.layer {
- tips.remove(pid);
- }
- }
- bcast.insert(*pid);
- }
- }
- slot.tips.retain(|_, t| !t.is_empty());
- slot.tips.entry(ev.header.layer).or_default().insert(eid);
- self.event_pub.notify(ev.clone()).await;
- }
- Ok(ids)
- }
- pub async fn header_dag_insert(&self, headers: Vec<Header>, dag_name: &str) -> Result<()> {
- let dag_ts = u64::from_str(dag_name)?;
- // The genesis ID we expect any layer-1 header in this slot
- // to reference. Computed locally from config - two networks
- // with different `genesis_contents` (or any other config
- // mismatch) produce different genesis ids, so a peer whose
- // layer-1 headers reference something else is on a different
- // network. Catching this explicitly here is strictly a
- // defense-in-depth and diagnostics improvement: the existing
- // parent-existence check in `Header::validate` already
- // rejects these (genesis headers are filtered from
- // `header_tree` on insert, so a foreign genesis id never
- // lands in the local tree). The explicit boundary check just
- // turns "HeaderIsInvalid" into a logged, named condition, so
- // an operator debugging a misconfigured deployment sees
- // "peer is on a different network" instead of a generic
- // header rejection.
- //
- // Why layer 1 is sufficient: `select_parents_from_tips` puts
- // an event at layer N+1 where N is the highest layer with
- // tips. For layer = 1, the highest tip layer must be 0, and
- // the only layer-0 entry in any slot is the genesis (the
- // single event placed by `DagStore::create_slot`). So every
- // layer-1 event's non-NULL parents are equal to that slot's
- // genesis id. Higher layers don't need the check because
- // their parent chains transitively pass through layer 1; if
- // the layer-1 events get rejected, layer-2+ events lose
- // their referenced parents and fail the existing parent-
- // existence check.
- let local_genesis_id = Header {
- timestamp: dag_ts,
- parents: NULL_PARENTS,
- layer: 0,
- content_hash: blake3::hash(&self.config.genesis_contents),
- }
- .id();
- let mut store = self.dag_store.write().await;
- let slot = store.get_slot_mut(&dag_ts).ok_or(Error::DagSyncFailed)?;
- let mut overlay = SledTreeOverlay::new(&slot.header_tree);
- let mut hdrs = headers;
- hdrs.sort_by_key(|h| h.layer);
- for hdr in &hdrs {
- if hdr.parents == NULL_PARENTS {
- continue
- }
- // Cross-network detection at the layer-1 boundary.
- if hdr.layer == 1 {
- for pid in hdr.parents.iter() {
- if *pid != NULL_ID && *pid != local_genesis_id {
- error!(
- target: "event_graph::header_dag_insert",
- "[HEADER_DAG_INSERT] layer-1 header for dag {dag_ts} \
- references foreign genesis: claimed parent {pid:?}, \
- local genesis is {local_genesis_id:?}. Peer is on a \
- different network.",
- );
- return Err(Error::HeaderIsInvalid)
- }
- }
- }
- let hid = hdr.id();
- if !hdr.validate(&slot.header_tree, &self.config, Some(&overlay)).await? {
- return Err(Error::HeaderIsInvalid)
- }
- overlay.insert(hid.as_bytes(), &serialize_async(hdr).await)?;
- slot.time_index.insert(hdr.timestamp, hid);
- }
- if let Some(b) = overlay.aggregate() {
- slot.header_tree.apply_batch(b).unwrap();
- }
- Ok(())
- }
- pub async fn fetch_event_from_dags(&self, eid: &blake3::Hash) -> Result<Option<Event>> {
- for (_, slot) in self.dag_store.read().await.dags.iter() {
- if let Some(b) = slot.main_tree.get(eid.as_bytes())? {
- return Ok(Some(deserialize_async(&b).await?))
- }
- }
- // Also check the static DAG. Static events (RLN registrations
- // and slashes) are public consensus state, so they're served
- // alongside rotating-DAG events through the same EventReq
- // path. This is what lets a fresh peer's `static_sync` walk
- // ancestry through EventReq after discovering tips.
- if let Some(b) = self.static_dag.get(eid.as_bytes())? {
- return Ok(Some(deserialize_async(&b).await?))
- }
- Ok(None)
- }
- pub(crate) async fn get_next_layer_with_parents(
- &self,
- dag_ts: &u64,
- ) -> (u64, [blake3::Hash; N_EVENT_PARENTS]) {
- select_parents_from_tips(&self.dag_store.read().await.get_slot(dag_ts).unwrap().tips)
- }
- pub(crate) async fn get_next_layer_with_parents_static(
- &self,
- ) -> (u64, [blake3::Hash; N_EVENT_PARENTS]) {
- select_parents_from_tips(&compute_unreferenced_tips(&self.static_dag).await)
- }
- pub async fn order_events(&self) -> Vec<Event> {
- let mut all = vec![];
- for (_, slot) in self.dag_store.read().await.dags.iter() {
- for item in slot.main_tree.iter() {
- let (_, b) = item.unwrap();
- let ev: Event = deserialize_async(&b).await.unwrap();
- if ev.header.parents != NULL_PARENTS {
- all.push(ev);
- }
- }
- }
- all.sort_unstable_by(display_order);
- all
- }
- pub async fn fetch_headers_with_tips(
- &self,
- dag_name: &str,
- tips: &LayerUTips,
- ) -> Result<Vec<Header>> {
- let dag_ts = u64::from_str(dag_name)?;
- let store = self.dag_store.read().await;
- let slot = store.get_slot(&dag_ts).ok_or(Error::DagSyncFailed)?;
- let mut ancestors = HashSet::new();
- for hashes in tips.values() {
- for h in hashes {
- ancestors.insert(*h);
- if let Some(v) = slot.header_tree.get(h.as_bytes())? {
- self.get_ancestors(
- &mut ancestors,
- deserialize_async(&v).await?,
- &slot.header_tree,
- )
- .await?;
- }
- }
- }
- let mut out = vec![];
- for item in slot.header_tree.iter() {
- let (id, v) = item?;
- let h = blake3::Hash::from_bytes((&id as &[u8]).try_into()?);
- if !ancestors.contains(&h) {
- out.push(deserialize_async(&v).await?);
- }
- }
- out.sort_unstable_by_key(|h: &Header| h.layer);
- Ok(out)
- }
- pub(crate) async fn get_ancestors(
- &self,
- visited: &mut HashSet<blake3::Hash>,
- hdr: Header,
- tree: &sled::Tree,
- ) -> Result<()> {
- let mut stack = VecDeque::new();
- stack.push_back(hdr);
- while let Some(h) = stack.pop_back() {
- for p in h.parents {
- if p != NULL_ID && visited.insert(p) {
- if let Some(v) = tree.get(p.as_bytes())? {
- stack.push_back(deserialize_async(&v).await?);
- }
- }
- }
- }
- Ok(())
- }
- async fn static_new(sled_db: &sled::Db, config: &EventGraphConfig) -> Result<sled::Tree> {
- let tree = sled_db.open_tree("static-dag")?;
- let genesis = generate_genesis(&EventGraphConfig { hours_rotation: 0, ..config.clone() });
- let mut ov = SledTreeOverlay::new(&tree);
- ov.insert(genesis.id().as_bytes(), &serialize_async(&genesis).await).unwrap();
- if let Some(b) = ov.aggregate() {
- tree.apply_batch(b).unwrap();
- }
- Ok(tree)
- }
- pub async fn static_broadcast(&self, ev: Event, blob: Vec<u8>) -> Result<()> {
- self.p2p.broadcast(&StaticPut(ev, blob)).await;
- Ok(())
- }
- pub async fn static_insert(&self, ev: &Event) -> Result<()> {
- let mut ov = SledTreeOverlay::new(&self.static_dag);
- ov.insert(ev.id().as_bytes(), &serialize_async(ev).await).unwrap();
- if let Some(b) = ov.aggregate() {
- self.static_dag.apply_batch(b).unwrap();
- }
- self.static_pub.notify(ev.clone()).await;
- Ok(())
- }
- pub async fn static_fetch(&self, eid: &blake3::Hash) -> Result<Option<Event>> {
- Ok(match self.static_dag.get(eid.as_bytes())? {
- Some(b) => Some(deserialize_async(&b).await?),
- None => None,
- })
- }
- pub async fn static_unreferenced_tips(&self) -> LayerUTips {
- compute_unreferenced_tips(&self.static_dag).await
- }
- /// Persist the original RLN blob for a static-DAG event. The
- /// blob is the wire payload from the originating `StaticPut` -
- /// Persist the original RLN blob for a static-DAG event. The
- /// blob is the wire payload from the originating `StaticPut` -
- /// proof + public inputs + attestation - needed to re-verify
- /// the proof at sync time by late-joining peers.
- ///
- /// Writing the same `(eid, blob)` repeatedly is safe.
- pub fn static_blob_store(&self, eid: &blake3::Hash, blob: &[u8]) -> Result<()> {
- self.static_dag_blobs.insert(eid.as_bytes(), blob)?;
- Ok(())
- }
- /// Look up the original RLN blob for a static-DAG event.
- ///
- /// Returns `Ok(None)` only for legitimate "not stored" cases -
- /// a peer that's never seen the event, or an event that pre-dates
- /// the side-table. The verification path in `static_sync` treats
- /// missing blobs on non-genesis events as a sync failure, not as
- /// a fall-through.
- pub fn static_blob_fetch(&self, eid: &blake3::Hash) -> Result<Option<Vec<u8>>> {
- Ok(self.static_dag_blobs.get(eid.as_bytes())?.map(|ivec| ivec.to_vec()))
- }
- /// Persist the original RLN signal blob for a rotating-DAG event.
- ///
- /// Mirror of [`Self::static_blob_store`] but for rotating-DAG
- /// events. Idempotent. Called after successful RLN verification
- /// in `handle_event_put`, and during sync by
- /// `dag_insert_with_blobs` when the peer included a blob in
- /// `EventRep`.
- pub fn dag_blob_store(&self, eid: &blake3::Hash, blob: &[u8]) -> Result<()> {
- self.dag_blobs.insert(eid.as_bytes(), blob)?;
- Ok(())
- }
- /// Look up the original RLN signal blob for a rotating-DAG
- /// event. Returns `Ok(None)` if not stored - see
- /// [`Self::static_blob_fetch`] for the exhaustive list of
- /// reasons a blob may legitimately be missing.
- ///
- /// Note: rotating-DAG blobs are pruned alongside their DAGs
- /// (see `dag_blobs_prune`). Older-than-window events therefore
- /// don't accumulate blobs in this side-table.
- pub fn dag_blob_fetch(&self, eid: &blake3::Hash) -> Result<Option<Vec<u8>>> {
- Ok(self.dag_blobs.get(eid.as_bytes())?.map(|ivec| ivec.to_vec()))
- }
- /// Apply a static-DAG event (registration or slash) to the
- /// identity-state SMT, and record the resulting root in the
- /// historical-roots side-tables.
- ///
- /// **This is the single canonical entry point** for SMT
- /// mutation. All callers - live broadcast (`handle_static_put`),
- /// originator (`nickserv.rs::handle_register`), and sync
- /// (`static_sync` apply loop) - go through here. Bypassing it
- /// will desynchronize the SMT from the historical-roots tables,
- /// which silently breaks signal verification.
- ///
- /// **Canonical order requirement.** Two nodes processing the
- /// same set of static events must produce the same sequence of
- /// intermediate roots. SMTs are commutative under set-of-leaves
- /// (final root is order-independent) but the *intermediate*
- /// roots produced during application are order-dependent. We
- /// pin the order with `(layer, event_id)`: layer is the primary
- /// key (defined by the event's parent links and consensus-agreed),
- /// event_id is the tie-breaker within a layer (32-byte hash,
- /// total-ordered lexicographically).
- ///
- /// In live broadcast and originator paths, events arrive one at
- /// a time; the canonical-order requirement is automatically
- /// satisfied because each event's layer is greater than its
- /// parents'. In sync, the caller must sort by `(layer, event_id)`
- /// before invoking this method (see `static_sync`).
- ///
- /// **Returns** the post-mutation SMT root, or an error if the
- /// SMT mutation itself fails. A duplicate-registration or
- /// slash-of-nonexistent are both treated as soft no-ops at the
- /// SMT layer, but we still record the root (which equals the
- /// pre-call root in that case) - this preserves the invariant
- /// that "every static-DAG event has a corresponding entry in
- /// rln-historical-roots-ordered" without complicating the
- /// caller's logic.
- pub async fn apply_rln_static_event(
- &self,
- ev: &Event,
- node: &rln::RLNNode,
- ) -> Result<pallas::Base> {
- let mut state = self.identity_state.write().await;
- match node {
- rln::RLNNode::Registration(commitment) => {
- // Soft-fail on duplicate (race with another peer).
- let _ = state.register(*commitment);
- }
- rln::RLNNode::Slashing(commitment) => {
- // Idempotent - slashing a non-present identity is
- // a no-op.
- let _ = state.slash(*commitment);
- }
- }
- let new_root = state.root();
- drop(state);
- // Record the root in both side-tables. We do this even if
- // the SMT mutation was a no-op (duplicate register, slash of
- // missing) so the historical-roots table has one entry per
- // static-DAG event. This makes canonical-order replay simple:
- // every event has exactly one entry, no conditional skips.
- let key = encode_historical_root_key(ev.header.layer, &ev.id());
- let value = encode_historical_root_value(&new_root, ev.header.timestamp);
- self.rln_historical_roots_ordered.insert(key, value.as_slice())?;
- self.rln_historical_roots_by_value.insert(new_root.to_repr(), key.as_slice())?;
- Ok(new_root)
- }
- /// Check whether `root` is a valid SMT root for a signal whose
- /// `signal_timestamp` is given (in millis-since-epoch).
- ///
- /// A root is valid if it was the live root at any time in the
- /// drift window `[signal_timestamp - DRIFT, signal_timestamp +
- /// DRIFT]`. The drift symmetry handles two distinct concerns:
- ///
- /// * **Forward drift** (signal sees a slightly stale root): the
- /// originator built a proof against root `R_n`, then someone
- /// else registered, producing `R_{n+1}`, before the signal
- /// reached the verifier. The signal's claimed `R_n` is older
- /// than the verifier's current root by an amount up to the
- /// propagation delay. Accept if R_n was current within DRIFT
- /// of the signal's timestamp.
- ///
- /// * **Backward drift** (signal arrives before its root): rare
- /// but possible if the static-DAG broadcast is racing the
- /// rotating-DAG broadcast. The originator's machine knew about
- /// a registration that hadn't fully propagated yet. We tolerate
- /// up to DRIFT of backward skew.
- ///
- /// The check uses `rln_historical_roots_by_value` to find the
- /// canonical position of `root`, then `rln_historical_roots_ordered`
- /// to bracket the time interval during which `root` was live.
- /// The interval starts at the timestamp of the event that
- /// produced `root` and ends just before the next event's
- /// timestamp (or `u64::MAX` if `root` is currently live).
- ///
- /// This subsumes the old `recent_roots` window as a special case
- /// (live verification = signal_timestamp ~= now). The in-memory
- /// `recent_roots` cache in `IdentityState` remains as a fast
- /// path for the live-broadcast hot loop; this method is consulted
- /// when the cache misses.
- pub fn is_root_valid_at(&self, root: &pallas::Base, signal_timestamp: u64) -> Result<bool> {
- // Reverse lookup: where in the canonical sequence is `root`?
- let Some(key_bytes) = self.rln_historical_roots_by_value.get(root.to_repr())? else {
- return Ok(false)
- };
- // Read the entry that produced this root.
- let Some(value_bytes) = self.rln_historical_roots_ordered.get(&key_bytes)? else {
- // Inconsistency between the two tables (shouldn't happen
- // under normal operation; might happen if a write was
- // partially applied during a crash). Treat as not-found.
- return Ok(false)
- };
- let (recorded_root, root_timestamp) = decode_historical_root_value(&value_bytes)?;
- if &recorded_root != root {
- // Same key collision - should be impossible because the
- // by_value index is keyed on the root itself, but defend
- // against future code changes.
- return Ok(false)
- }
- // The interval during which `root` was live ends at the
- // timestamp of the next event in canonical order, or
- // u64::MAX if `root` is the current live root.
- //
- // We need the strictly-greater key. sled's range-from-exclusive
- // pattern is range((Excluded(key), Unbounded)).next().
- let next_timestamp: u64 = {
- use std::ops::Bound::{Excluded, Unbounded};
- match self
- .rln_historical_roots_ordered
- .range::<&[u8], _>((Excluded(key_bytes.as_ref()), Unbounded))
- .next()
- {
- Some(Ok((_, val))) => decode_historical_root_value(&val)?.1,
- Some(Err(_)) | None => u64::MAX,
- }
- };
- // Drift window. Reuse EVENT_TIME_DRIFT for consistency with
- // event-graph timestamp validation.
- let drift = EVENT_TIME_DRIFT;
- let lo = signal_timestamp.saturating_sub(drift);
- let hi = signal_timestamp.saturating_add(drift);
- // `root` was live during [root_timestamp, next_timestamp).
- // Acceptable iff this interval intersects [lo, hi].
- Ok(root_timestamp <= hi && next_timestamp > lo)
- }
- }
- fn encode_historical_root_key(layer: u64, event_id: &blake3::Hash) -> [u8; 40] {
- let mut buf = [0u8; 40];
- buf[..8].copy_from_slice(&layer.to_be_bytes());
- buf[8..].copy_from_slice(event_id.as_bytes());
- buf
- }
- fn encode_historical_root_value(root: &pallas::Base, timestamp: u64) -> [u8; 40] {
- let mut buf = [0u8; 40];
- buf[..32].copy_from_slice(&root.to_repr());
- buf[32..].copy_from_slice(×tamp.to_be_bytes());
- buf
- }
- fn decode_historical_root_value(bytes: &[u8]) -> Result<(pallas::Base, u64)> {
- if bytes.len() != 40 {
- return Err(Error::Custom(format!(
- "historical-root value must be 40 bytes, got {}",
- bytes.len()
- )))
- }
- let mut root_repr = [0u8; 32];
- root_repr.copy_from_slice(&bytes[..32]);
- let root: pallas::Base = match pallas::Base::from_repr(root_repr).into() {
- Some(r) => r,
- None => return Err(Error::Custom("invalid root encoding".into())),
- };
- let mut ts_bytes = [0u8; 8];
- ts_bytes.copy_from_slice(&bytes[32..]);
- Ok((root, u64::from_be_bytes(ts_bytes)))
- }
- impl EventGraph {
- /// Return a JSON-RPC response representing the current state of
- /// the event graph.
- ///
- /// Shape (matches [`util::recreate_from_replayer_log`] so clients
- /// can reuse their parsers):
- ///
- /// ```json
- /// {
- /// "eventgraph_info": {
- /// "dag": {
- /// "<event-id-hex>": <event>,
- /// ...
- /// }
- /// }
- /// }
- /// ```
- ///
- /// Walks every event currently held in every rotating DAG *and*
- /// every event in the static DAG. Genesis events are included.
- #[cfg(feature = "rpc")]
- pub async fn eventgraph_info(
- &self,
- id: u16,
- _params: crate::rpc::util::JsonValue,
- ) -> crate::rpc::jsonrpc::JsonResult {
- use crate::rpc::{
- jsonrpc::{JsonResponse, JsonResult},
- util::{json_map, JsonValue},
- };
- let mut dag = HashMap::new();
- // Walk every rotating DAG.
- for (_, slot) in self.dag_store.read().await.dags.iter() {
- for item in slot.main_tree.iter() {
- let (eid, val) = match item {
- Ok(v) => v,
- Err(_) => continue,
- };
- let Ok(ev) = deserialize_async::<Event>(&val).await else { continue };
- let key = blake3::Hash::from_bytes(match (&eid as &[u8]).try_into() {
- Ok(b) => b,
- Err(_) => continue,
- });
- dag.insert(key.to_string(), JsonValue::from(ev));
- }
- }
- // And the static DAG.
- for item in self.static_dag.iter() {
- let (eid, val) = match item {
- Ok(v) => v,
- Err(_) => continue,
- };
- let Ok(ev) = deserialize_async::<Event>(&val).await else { continue };
- let key = blake3::Hash::from_bytes(match (&eid as &[u8]).try_into() {
- Ok(b) => b,
- Err(_) => continue,
- });
- dag.insert(key.to_string(), JsonValue::from(ev));
- }
- let values = json_map([("dag", JsonValue::Object(dag))]);
- let result = JsonValue::Object(HashMap::from([("eventgraph_info".into(), values)]));
- JsonResult::Response(JsonResponse::new(result, id))
- }
- pub fn deg_enable(&self) {
- self.deg_enabled.store(true, Ordering::Release);
- }
- pub fn deg_disable(&self) {
- self.deg_enabled.store(false, Ordering::Release);
- }
- pub fn is_deg_enabled(&self) -> bool {
- self.deg_enabled.load(Ordering::Acquire)
- }
- pub fn is_synced(&self) -> bool {
- self.synced.load(Ordering::Acquire)
- }
- pub async fn deg_subscribe(&self) -> Subscription<DegEvent> {
- self.deg_publisher.clone().subscribe().await
- }
- pub async fn deg_notify(&self, ev: DegEvent) {
- self.deg_publisher.notify(ev).await;
- }
- /// Subscribe to rotating-DAG event insertions.
- ///
- /// Each subscriber receives a clone of every [`Event`] that
- /// passes validation and is committed via `dag_insert`. The
- /// publisher fires *after* state mutation, so subscribers can
- /// rely on the event being durably present in the DAG by the
- /// time they observe it.
- ///
- /// Used to build live JSON-RPC subscription endpoints (e.g.
- /// the Gource-feeding endpoint in DarkIRC). Bridge a
- /// [`Subscription`] to a `JsonSubscriber` in the application
- /// layer; this method itself contains no JSON-RPC logic.
- pub async fn event_subscribe(&self) -> Subscription<Event> {
- self.event_pub.clone().subscribe().await
- }
- /// Subscribe to static-DAG event insertions (RLN registrations
- /// and slashes). Mirrors [`Self::event_subscribe`] but for the
- /// static DAG. See that method for semantics.
- pub async fn static_subscribe(&self) -> Subscription<Event> {
- self.static_pub.clone().subscribe().await
- }
- /// The app identifier mixed into RLN external nullifiers.
- /// Exposed so the proto layer (and clients constructing signal
- /// proofs) can use the same value the verifier uses.
- pub fn rln_app_id(&self) -> rln::RlnAppId {
- self.rln_app_id
- }
- /// Build a membership proof for an identity commitment, plus the
- /// current root.
- ///
- /// This is the *only* sanctioned path for clients to produce a
- /// signal proof - they should not be holding their own copy of
- /// the SMT (the previous `event_graph.rln_identity_tree`
- /// pattern). Centralising this keeps client and verifier in
- /// agreement on the root they're proving against.
- pub async fn rln_membership_path(
- &self,
- commitment: &darkfi_sdk::pasta::pallas::Base,
- ) -> (darkfi_sdk::pasta::pallas::Base, darkfi_sdk::crypto::smt::PathFp) {
- let s = self.identity_state.read().await;
- (s.root(), s.prove_membership(commitment))
- }
- /// True if the given identity commitment is registered.
- pub async fn rln_contains(&self, commitment: &darkfi_sdk::pasta::pallas::Base) -> bool {
- self.identity_state.read().await.contains(commitment)
- }
- /// Verify an RLN signal blob against this event graph's state.
- ///
- /// The returned variant tells the caller what to do:
- /// * `Accepted` - proof valid, no conflict, share recorded.
- /// * `Rejected` - drop silently (bad proof, bad bounds, bad
- /// root, exact duplicate).
- /// * `Slashable(shares)` - different `(x, y)` for the same
- /// internal nullifier; the caller (protocol layer) should
- /// build and broadcast a slash from these shares.
- ///
- /// Critical invariant: `Rejected` and `Slashable` outcomes never
- /// mutate `metadata`. `Accepted` is the only mutating outcome.
- /// This is what prevents share-poisoning by an adversary who
- /// observes an honest internal_nullifier and tries to forge a
- /// share against it.
- pub async fn rln_verify_signal(&self, event: &Event, blob: &[u8]) -> rln::SignalCheck {
- use darkfi_sdk::{crypto::poseidon_hash, pasta::pallas};
- use rln::{epoch_of, hash_event, Blob, SignalCheck, MAX_MSG_LIMIT};
- let rcvd: Blob = match deserialize_async_partial(blob).await {
- Ok((v, _)) => v,
- Err(_) => return SignalCheck::Rejected,
- };
- // Defensive bounds. The proof PI binds these too, but
- // checking up front lets us skip an expensive verify() call
- // for trivially malformed blobs.
- if rcvd.user_msg_limit == 0 || rcvd.user_msg_limit > MAX_MSG_LIMIT {
- return SignalCheck::Rejected
- }
- let epoch_n = epoch_of(event.header.timestamp);
- let epoch_field = pallas::Base::from(epoch_n);
- let app_id = self.rln_app_id();
- let ext_null = poseidon_hash([epoch_field, app_id.as_field()]);
- let x = hash_event(event);
- // 1) The merkle root must be valid for a signal at this
- // timestamp. We accept any root that was the live SMT
- // root at any time within EVENT_TIME_DRIFT of the signal's
- // timestamp. This subsumes the old "recent_roots window"
- // behaviour as a special case (live verification, signal
- // timestamp ~= now) AND supports sync of historical signals
- // (signal timestamp = signing time, root corresponds to
- // that historical state).
- //
- // Hot-path optimization: check the in-memory recent_roots
- // cache first. For live broadcasts (the overwhelming
- // majority of verifications) the root will be in the
- // cache and we skip the sled lookup.
- {
- let id_state = self.identity_state.read().await;
- if !id_state.is_known_root(&rcvd.merkle_root) {
- drop(id_state);
- match self.is_root_valid_at(&rcvd.merkle_root, event.header.timestamp) {
- Ok(true) => {}
- Ok(false) => {
- // Useful diagnostic: this rejection path
- // catches both "garbage root" (attacker
- // submitted a forged root) and "slashed-user
- // replay" (slashed identity claiming a
- // pre-slash root after the propagation
- // window expired). The two are
- // indistinguishable from the verifier's
- // perspective by design - RLN-V2 privacy
- // guarantees prevent identifying the
- // signer. But observed in aggregate, a
- // burst of these from a single peer is a
- // strong signal of replay-after-slash
- // misbehavior, useful for operators
- // debugging "why are my messages being
- // rejected" or investigating peer abuse.
- let local_root = self.identity_state.read().await.root();
- let historical_count = self.rln_historical_roots_ordered.len();
- warn!(
- target: "event_graph::rln_verify_signal",
- "[RLN] Signal rejected: merkle_root not valid at signal \
- timestamp {}. Possible causes: (1) forged or out-of-sync \
- root, (2) slashed identity replaying against a pre-slash \
- root after the propagation window expired. event_id={}, \
- received_root={:?}, local_current_root={:?}, \
- historical_root_count={}",
- event.header.timestamp,
- event.id(),
- rcvd.merkle_root,
- local_root,
- historical_count,
- );
- return SignalCheck::Rejected
- }
- Err(e) => {
- error!(
- target: "event_graph::rln_verify_signal",
- "[RLN] is_root_valid_at lookup failed for event {}: {e}",
- event.id(),
- );
- return SignalCheck::Rejected
- }
- }
- }
- }
- // 2) Verify the ZK proof. PI order MUST match
- // constrain_instance() in rlnv2-diff-signal.zk:
- // root, external_nullifier, user_message_limit, x, y, internal_nullifier
- let pi = vec![
- rcvd.merkle_root,
- ext_null,
- pallas::Base::from(rcvd.user_msg_limit),
- x,
- rcvd.y,
- rcvd.internal_nullifier,
- ];
- if rcvd.proof.verify(&self.zk_keys.signal_vk, &pi).is_err() {
- return SignalCheck::Rejected
- }
- // 3) Now consult the metadata table. Any share we look at
- // here is guaranteed to be from a valid proof.
- let mut state = self.rln_state.write().await;
- // Prune metadata relative to THIS signal's epoch, not
- // wall-clock. The retention window is conceptually
- // "epochs near the signal we're processing", and the only
- // entries that matter for reuse detection are siblings
- // within `METADATA_RETAIN_EPOCHS` of `epoch_n`.
- //
- // In production, signals carry roughly-current wall-clock
- // timestamps, so `epoch_n ~= current_epoch()` and this is
- // equivalent to the previous `current_epoch()`-based prune.
- // The change matters in three places:
- //
- // * Sync of historical signals: a late-arriving signal
- // keeps its epoch's metadata visible long enough for
- // the verifier to detect reuse. With wall-clock prune,
- // a signal old enough to be outside the retention
- // window would always silently lose its sibling shares
- // before they could be matched.
- //
- // * Tests with deterministic event timestamps: the
- // verifier and the metadata stay consistent regardless
- // of when the test runs.
- //
- // * Minor DoS surface: a peer with a far-future system
- // clock no longer causes mass-wipe of real metadata
- // before consultation.
- state.metadata.prune_old(epoch_n);
- if state.metadata.is_duplicate(epoch_n, &rcvd.internal_nullifier, &x, &rcvd.y) {
- return SignalCheck::Rejected
- }
- if state.metadata.is_reused(epoch_n, &rcvd.internal_nullifier) {
- let mut shares = state.metadata.get_shares(epoch_n, &rcvd.internal_nullifier);
- shares.push((x, rcvd.y));
- return SignalCheck::Slashable(shares)
- }
- state.metadata.add_share(epoch_n, rcvd.internal_nullifier, x, rcvd.y);
- SignalCheck::Accepted
- }
- /// Verify a static-DAG event (RLN registration or slashing)
- /// against this event graph's state.
- ///
- /// This is the testable core of the protocol-layer
- /// `handle_static_put`. It performs all checks that the
- /// protocol layer does - bounds, attestation, proof, root
- /// recency - and returns a [`rln::StaticEventCheck`] outcome
- /// telling the caller what to do next.
- ///
- /// **This method does NOT mutate state.** The caller is
- /// responsible for invoking `IdentityState::register` /
- /// `slash` on `Accepted*` outcomes, and for striking the peer
- /// on `Malicious`. Separating decision from action makes the
- /// behaviour fully testable and lets the protocol layer keep
- /// its mutation under a single locked critical section.
- pub async fn rln_verify_static_event(
- &self,
- rln_node: &rln::RLNNode,
- blob: &[u8],
- event_timestamp: u64,
- ) -> rln::StaticEventCheck {
- use darkfi_sdk::{crypto::poseidon_hash, pasta::pallas};
- use rln::{RLNNode, RegistrationBlob, SlashBlob, StaticEventCheck, MAX_MSG_LIMIT};
- match rln_node {
- RLNNode::Registration(commitment) => {
- let reg: RegistrationBlob = match deserialize_async_partial(blob).await {
- Ok((v, _)) => v,
- Err(_) => return StaticEventCheck::Rejected,
- };
- // Bounds. Out-of-range limits are unambiguous misbehavior.
- if reg.user_message_limit == 0 ||
- reg.user_message_limit > MAX_MSG_LIMIT ||
- reg.max_message_limit != MAX_MSG_LIMIT
- {
- return StaticEventCheck::Malicious
- }
- // Attestation must permit the claimed limit.
- // (Free-tier cap; `Staked` rejected until the stake
- // contract is online.)
- if !reg.attestation.permits(reg.user_message_limit) {
- return StaticEventCheck::Malicious
- }
- // Duplicate registration is a soft Reject (we may
- // have raced a peer), checked here so we don't
- // pay the proof-verification cost for known leaves.
- if self.identity_state.read().await.contains(commitment) {
- return StaticEventCheck::Rejected
- }
- // Proof.
- let pi = vec![
- *commitment,
- pallas::Base::from(reg.user_message_limit),
- pallas::Base::from(reg.max_message_limit),
- ];
- if reg.proof.verify(&self.zk_keys.register_vk, &pi).is_err() {
- return StaticEventCheck::Rejected
- }
- StaticEventCheck::AcceptedRegistration(*commitment)
- }
- RLNNode::Slashing(commitment) => {
- let sl: SlashBlob = match deserialize_async_partial(blob).await {
- Ok((v, _)) => v,
- Err(_) => return StaticEventCheck::Rejected,
- };
- let pi = vec![sl.identity_secret_hash, sl.merkle_root];
- if sl.proof.verify(&self.zk_keys.slash_vk, &pi).is_err() {
- return StaticEventCheck::Rejected
- }
- // Slash event names a specific commitment; verify
- // that the recovered identity_secret_hash actually
- // maps to it. Mismatch is unambiguous misbehavior.
- let rebuilt = poseidon_hash([sl.identity_secret_hash]);
- if *commitment != rebuilt {
- return StaticEventCheck::Malicious
- }
- // The proof's root must be a valid SMT root at the
- // slash event's timestamp. Same logic as signal
- // verification: use the time-window check, which
- // accepts any root that was live within DRIFT of the
- // slash timestamp.
- {
- let id_state = self.identity_state.read().await;
- if !id_state.is_known_root(&sl.merkle_root) {
- drop(id_state);
- match self.is_root_valid_at(&sl.merkle_root, event_timestamp) {
- Ok(true) => {}
- Ok(false) | Err(_) => return StaticEventCheck::Rejected,
- }
- }
- }
- StaticEventCheck::AcceptedSlash(rebuilt)
- }
- }
- }
- }
- async fn request_tips(
- peer: &Channel,
- dag: String,
- timeout: u64,
- ) -> Result<BTreeMap<u64, HashSet<blake3::Hash>>> {
- let sub = peer.subscribe_msg::<TipRep>().await?;
- peer.send(&TipReq(dag)).await?;
- let r = sub
- .receive_with_timeout(timeout)
- .await
- .map_err(|_| Error::EventNotFound("tip timeout".into()))?;
- sub.unsubscribe().await;
- Ok(r.0.clone())
- }
- async fn request_header(
- peer: &Channel,
- name: String,
- tips: LayerUTips,
- timeout: u64,
- ) -> Result<Vec<Header>> {
- let sub = peer.subscribe_msg::<HeaderRep>().await?;
- peer.send(&HeaderReq(name, tips)).await?;
- let r = sub
- .receive_with_timeout(timeout)
- .await
- .map_err(|_| Error::EventNotFound("hdr timeout".into()))?;
- sub.unsubscribe().await;
- Ok(r.0.to_vec())
- }
- async fn request_event(
- peer: Arc<Channel>,
- ids: Vec<blake3::Hash>,
- cid: usize,
- timeout: u64,
- ) -> (Result<(Vec<Event>, Vec<Vec<u8>>)>, usize, Arc<Channel>) {
- let sub = match peer.subscribe_msg::<EventRep>().await {
- Ok(s) => s,
- Err(e) => return (Err(e), cid, peer),
- };
- if let Err(e) = peer.send(&EventReq(ids)).await {
- return (Err(e), cid, peer)
- }
- match sub.receive_with_timeout(timeout).await {
- Ok(r) => {
- sub.unsubscribe().await;
- (Ok((r.0.clone(), r.1.clone())), cid, peer)
- }
- Err(_) => (Err(Error::EventNotFound("ev timeout".into())), cid, peer),
- }
- }
|